diff --git a/backend/src/models/runtime-model-catalog.ts b/backend/src/models/runtime-model-catalog.ts index fa28ebb0..e77c0e05 100644 --- a/backend/src/models/runtime-model-catalog.ts +++ b/backend/src/models/runtime-model-catalog.ts @@ -41,8 +41,11 @@ const runtimeModelSchema = z.object({ supportsReasoningEffort: z.boolean(), supportsStore: z.boolean(), maxTokensField: z.string().optional(), + thinkingFormat: z.enum(["qwen", "qwen-chat-template"]).optional(), }).strict().optional(), - }).strict().optional(), + }).strict().refine((session) => !session.compatibility?.thinkingFormat || session.reasoning, { + message: "thinkingFormat requires reasoning: true", + }).optional(), metadataGeneration: z.object({ disableThinking: z.boolean() }).strict().optional(), }).strict(); diff --git a/backend/test/metadata-generation-models.test.ts b/backend/test/metadata-generation-models.test.ts index 6f5d00b0..f1288350 100644 --- a/backend/test/metadata-generation-models.test.ts +++ b/backend/test/metadata-generation-models.test.ts @@ -95,6 +95,35 @@ test("returns empty catalogs when no runtime projection is configured", () => { expect(loadMetadataGenerationModels({}).catalog()).toEqual({ models: [], default: null }); }); +test.each([ + ["qwen-chat-template", true, true], + ["qwen", true, true], + ["qwen-typo", true, false], + ["qwen-chat-template", false, false], +])("validates Qwen thinking format %s with reasoning=%s", (thinkingFormat, reasoning, valid) => { + const { catalogFile } = runtimeCatalog({ + defaultInteraction: "local/qwen", + models: [{ + id: "local/qwen", provider: "local", model: "qwen", label: "Qwen", + upstreamModel: "qwen", endpoint: { baseUrl: "http://localhost:8000/v1" }, + authentication: { mode: "none" }, sessionAdapter: { mode: "openai_compatible" }, + session: { + reasoning, contextWindow: 32768, maxTokens: 8192, + compatibility: { + supportsDeveloperRole: false, supportsReasoningEffort: false, + supportsStore: false, maxTokensField: "max_tokens", thinkingFormat, + }, + }, + }], + }); + if (valid) { + expect(loadRuntimeModelCatalog(catalogFile).sessionModels()[0].session?.compatibility) + .toMatchObject({ thinkingFormat }); + } else { + expect(() => loadRuntimeModelCatalog(catalogFile)).toThrow("runtime model catalog is invalid"); + } +}); + test("rejects a drifted default and an unprotected projection", () => { const drifted = runtimeCatalog({ defaultInteraction: "zai/missing" }); expect(() => loadRuntimeModelCatalog(drifted.catalogFile)).toThrow("interaction default is invalid"); diff --git a/docs/general/pi-configuration.md b/docs/general/pi-configuration.md index 9db489af..bb33c5a2 100644 --- a/docs/general/pi-configuration.md +++ b/docs/general/pi-configuration.md @@ -128,6 +128,51 @@ endpoint expects a model name different from the catalog key. Provider integrations remain declarative. Do not register providers from `harness/.pi/extensions/`; those extensions implement the workflow and human gates only. +### Qwen 3.6 sessions and thinking controls + +For Qwen served through a vLLM-compatible chat template, declare the following inside the +model's `session` block, alongside its context and output limits: + +```yaml +reasoning: true +compatibility: + supportsDeveloperRole: false + supportsReasoningEffort: false + supportsStore: false + maxTokensField: max_tokens + thinkingFormat: qwen-chat-template +``` + +This makes Pi send `chat_template_kwargs.enable_thinking` from the selected thinking level, +with `preserve_thinking: true`. Choose **off** to explicitly disable thinking. The alternative +`thinkingFormat: qwen` is for endpoints expecting top-level `enable_thinking`. Both formats +require `reasoning: true`; declaring `reasoning: false` does not tell the server to disable +thinking. Omit `thinkingFormat` to preserve Pi's default behavior for other providers. + +Regenerate projections with the updated host CLI and recreate the local core container after +rebuilding it. Do not add these fields directly to generated Pi files. These controls do not +force tool calls or certify the workflow; perform the operator verification above. + +```sh +tht --installation /absolute/path/thothii-installation.yaml installation generate +``` + +Use the model identifier exposed by your endpoint, such as `qwen3.6-35b-a3b`, and limits +supported by that deployment. The thinking format configures the Pi session adapter; +metadata generation continues to use its separate LiteLLM settings. + +If a session displays text such as `{"type":"bash","command":"tht session show … --json"}` +and never opens a review widget, that text is not an executed tool call. A verified cause +was the Evidence JSON extension being loaded into interactive sessions and forcing +`response_format: {type: "json_object"}`. Upgrade to the core image containing the fix: +the extension belongs in `.pi/evidence-extensions/` and is loaded explicitly only by +Evidence authoring. It must not also remain in the automatically loaded `.pi/extensions/` +directory. Regenerating model configuration alone does not remove an extension from an old image. + +After upgrading, reload the browser and resume the session. Verify that Pi executes +`tht session show` and opens a review widget. This fix does not require changing the Qwen +server, forcing every turn to call a tool, or teaching the model to print tool-call JSON. + ## Authentication Every provider chooses one explicit mode: diff --git a/docs/research/2026-09-21-pi-qwen-compatibility.md b/docs/research/2026-09-21-pi-qwen-compatibility.md new file mode 100644 index 00000000..42db6742 --- /dev/null +++ b/docs/research/2026-09-21-pi-qwen-compatibility.md @@ -0,0 +1,57 @@ +# Qwen 3.6: session tool-call failure and correction + +Date: 2026-09-21. Verified runtime: Pi coding agent and Pi AI 0.80.3. + +## Cause and correction + +Interactive sessions returned text resembling a bash call and stopped before the +first review widget. The Evidence-only extension `tht-evidence-json-mode.ts` lived +under `.pi/extensions`, so Pi automatically loaded it into interactive sessions. +Its `before_provider_request` hook imposed `response_format: {type: "json_object"}` +and temperature zero on every request. + +Replaying the captured startup request, with all original hooks preserved, isolated +the cause. With JSON response format, Qwen returned the command as text and finish +reason `stop`. Removing only that field produced a native `bash` call and finish +reason `tool_calls`. Both requests contained the same 15 tools and used High thinking. + +The extension now lives in `.pi/evidence-extensions` and is loaded explicitly by +`PiEvidenceRestructurer`. Evidence retains JSON output. Interactive sessions retain +native tool calls. No changes to the remote Qwen server were required. + +Earlier SDK probes replaced `session.agent.onPayload`, inadvertently bypassing the +extension hooks. Their success did not reproduce the application path and did not +establish a model or server fault. + +## Model configuration + +The installation catalog now accepts optional `session.compatibility.thinkingFormat` +values `qwen` and `qwen-chat-template`, requiring `reasoning: true`. Go validation, +the backend schema, and generated catalog/Pi projections preserve this setting. +It controls thinking; it was not the cause or correction of the tool-call failure. + +For the verified endpoint, `qwen-chat-template` sends `enable_thinking` and +`preserve_thinking` inside `chat_template_kwargs`. Selecting Off explicitly disables +thinking. Declaring `reasoning: false` alone does not disable thinking on the server. +See the [operator configuration](../general/pi-configuration.md#qwen-36-sessions-and-thinking-controls) +and [Pi 0.80.3 model documentation](https://github.com/earendil-works/pi/blob/v0.80.3/packages/coding-agent/docs/models.md#openai-compatibility). + +## Validation and local delivery + +- Catalog and projection regression tests failed before the compatibility change, + then passed. Go config/modelprojection/CLI tests, 84 targeted backend tests, + TypeScript checking, and the strict documentation build passed. +- The extension-isolation regression failed before relocation. All 46 Evidence + authoring/restructurer tests and Ruff checks on changed Python files passed. +- The rebuilt local core image has digest + `sha256:9cd593d7362dcbefe177f1b9eb4b9ffdf3010ced7bd7efc8fc3fdcd288f24579`. + Core/frontend were recreated, healthy, and returned HTTP 200. +- A real `pi --mode rpc --no-session` probe with Qwen and High thinking executed + `tht session show` successfully and reached the first clarification widget. + The probe allowed only that read and `reviewer_select`; it stopped without + submitting a human response or recording decisions. Its temporary configuration + referenced the mounted credential because the original runtime lease had expired. +- The user subsequently confirmed that the application now works. + +This verifies recovery from the startup failure. It does not certify every workflow +phase or the separate LiteLLM metadata-generation path. diff --git a/harness/.pi/extensions/tht-evidence-json-mode.ts b/harness/.pi/evidence-extensions/tht-evidence-json-mode.ts similarity index 69% rename from harness/.pi/extensions/tht-evidence-json-mode.ts rename to harness/.pi/evidence-extensions/tht-evidence-json-mode.ts index f5cfa038..c680ff16 100644 --- a/harness/.pi/extensions/tht-evidence-json-mode.ts +++ b/harness/.pi/evidence-extensions/tht-evidence-json-mode.ts @@ -1,5 +1,8 @@ import type { ExtensionAPI } from "@earendil-works/pi-coding-agent"; +// Loaded explicitly by PiEvidenceRestructurer. Keep outside .pi/extensions: +// JSON-only output prevents interactive sessions from emitting native tool calls. + export default function (pi: ExtensionAPI) { pi.on("before_provider_request", (event) => { if (typeof event.payload !== "object" || event.payload === null) { diff --git a/harness/tests/test_evidence_pi_restructurer.py b/harness/tests/test_evidence_pi_restructurer.py index feaae1a3..4d7f71c9 100644 --- a/harness/tests/test_evidence_pi_restructurer.py +++ b/harness/tests/test_evidence_pi_restructurer.py @@ -59,7 +59,7 @@ def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeyp ] assert "--no-skills" in argv assert argv[argv.index("--extension") + 1].endswith( - "/extensions/tht-evidence-json-mode.ts" + "/evidence-extensions/tht-evidence-json-mode.ts" ) assert "--system-prompt" in argv assert argv[argv.index("--system-prompt") + 1] == "Evidence-only system prompt" @@ -120,7 +120,7 @@ def test_evidence_authoring_skill_is_loadable_and_declares_the_wire_schema(): def test_evidence_authoring_json_mode_extension_only_rewrites_the_provider_payload(): extension = ( - Path(__file__).parents[1] / ".pi" / "extensions" / "tht-evidence-json-mode.ts" + Path(__file__).parents[1] / ".pi" / "evidence-extensions" / "tht-evidence-json-mode.ts" ).read_text(encoding="utf-8") assert 'pi.on("before_provider_request"' in extension @@ -129,6 +129,19 @@ def test_evidence_authoring_json_mode_extension_only_rewrites_the_provider_paylo assert "registerTool" not in extension +def test_evidence_json_mode_is_not_auto_loaded_into_interactive_sessions(): + from tht.evidence.authoring import authoring_skill_path + + skill_path = authoring_skill_path() + restructurer = PiEvidenceRestructurer("pi", skill_path) + extension_path = restructurer._json_mode_extension + auto_extensions = skill_path.parents[2] / "extensions" + + assert extension_path.is_file() + assert not extension_path.is_relative_to(auto_extensions) + assert not (auto_extensions / extension_path.name).exists() + + @pytest.mark.parametrize("response", [_candidate(), [_candidate()]]) def test_pi_restructurer_normalizes_bounded_candidate_envelopes( tmp_path, monkeypatch, response, diff --git a/harness/tht/evidence/authoring.py b/harness/tht/evidence/authoring.py index fedf1a1b..9ceb5692 100644 --- a/harness/tht/evidence/authoring.py +++ b/harness/tht/evidence/authoring.py @@ -209,7 +209,7 @@ class PiEvidenceRestructurer: self._pi_executable = pi_executable self._skill_path = skill_path self._json_mode_extension = ( - skill_path.parents[2] / "extensions" / "tht-evidence-json-mode.ts" + skill_path.parents[2] / "evidence-extensions" / "tht-evidence-json-mode.ts" ) self._timeout_seconds = timeout_seconds diff --git a/tools/tht/internal/config/installation_model_catalog_test.go b/tools/tht/internal/config/installation_model_catalog_test.go index b7f8dd49..b9a1601d 100644 --- a/tools/tht/internal/config/installation_model_catalog_test.go +++ b/tools/tht/internal/config/installation_model_catalog_test.go @@ -28,6 +28,38 @@ func TestLoadAcceptsInstallationModelCatalog(t *testing.T) { } } +func TestLoadQwenThinkingCompatibility(t *testing.T) { + for _, test := range []struct { + name, format string + reasoning bool + wantError bool + }{ + {"vllm template", "qwen-chat-template", true, false}, + {"top level Qwen", "qwen", true, false}, + {"unknown format", "qwen-typo", true, true}, + {"thinking disabled in model capabilities", "qwen-chat-template", false, true}, + } { + t.Run(test.name, func(t *testing.T) { + path, _, _, _ := writeInstallation(t, "local") + reasoning := "false" + if test.reasoning { + reasoning = "true" + } + catalog := strings.Replace(validModelCatalogYAML(), " contextWindow: 32768", + " reasoning: "+reasoning+"\n compatibility:\n thinkingFormat: "+test.format+"\n contextWindow: 32768", 1) + rewriteInstallationCatalog(t, path, catalog) + _, err := Load(path) + if test.wantError { + if err == nil || !strings.Contains(err.Error(), "thinkingFormat") { + t.Fatalf("Load() error = %v, want thinkingFormat error", err) + } + } else if err != nil { + t.Fatalf("Load() error = %v", err) + } + }) + } +} + func TestLoadAcceptsSharedDeepSeekBuiltinWithBundleAuthentication(t *testing.T) { path, _, envFile, _ := writeInstallation(t, "local") catalog := strings.Replace(validModelCatalogYAML(), "interaction: zai/glm-5.3", "interaction: deepseek/deepseek-v4-pro", 1) diff --git a/tools/tht/internal/config/model_catalog.go b/tools/tht/internal/config/model_catalog.go index b1fbdb27..fb264fcc 100644 --- a/tools/tht/internal/config/model_catalog.go +++ b/tools/tht/internal/config/model_catalog.go @@ -90,6 +90,7 @@ type ModelCompatibility struct { SupportsReasoningEffort bool `yaml:"supportsReasoningEffort" json:"supportsReasoningEffort"` SupportsStore bool `yaml:"supportsStore" json:"supportsStore"` MaxTokensField string `yaml:"maxTokensField" json:"maxTokensField,omitempty"` + ThinkingFormat string `yaml:"thinkingFormat,omitempty" json:"thinkingFormat,omitempty"` } type SessionModel struct { @@ -270,6 +271,15 @@ func validateCatalogProvider(id string, provider ModelProvider, hasSession, hasM if model.Session != nil && (model.Session.ContextWindow <= 0 || model.Session.MaxTokens <= 0) { return fmt.Errorf("modelCatalog model %q/%q requires contextWindow and maxTokens", id, modelID) } + if model.Session != nil && model.Session.Compatibility != nil { + format := model.Session.Compatibility.ThinkingFormat + if format != "" && format != "qwen" && format != "qwen-chat-template" { + return fmt.Errorf("modelCatalog model %q/%q compatibility.thinkingFormat is invalid", id, modelID) + } + if format != "" && !model.Session.Reasoning { + return fmt.Errorf("modelCatalog model %q/%q compatibility.thinkingFormat requires reasoning: true", id, modelID) + } + } } } } else if provider.Session != nil { diff --git a/tools/tht/internal/modelprojection/projection_test.go b/tools/tht/internal/modelprojection/projection_test.go index bc098095..7e5ca61b 100644 --- a/tools/tht/internal/modelprojection/projection_test.go +++ b/tools/tht/internal/modelprojection/projection_test.go @@ -97,6 +97,30 @@ func TestDeepSeekBuiltinAndMetadataShareCatalogIdentityAndSecretReference(t *tes } } +func TestQwenThinkingFormatSurvivesCatalogAndPiProjection(t *testing.T) { + i := projectionFixture(t) + model := i.ModelCatalog.Providers["local"].Models["qwen"] + if err := json.Unmarshal([]byte(`{"reasoning":true,"contextWindow":32768,"maxTokens":8192,"compatibility":{"supportsDeveloperRole":false,"supportsReasoningEffort":false,"supportsStore":false,"maxTokensField":"max_tokens","thinkingFormat":"qwen-chat-template"}}`), model.Session); err != nil { + t.Fatal(err) + } + files, err := Render(i) + if err != nil { + t.Fatal(err) + } + for _, path := range []string{CatalogFile, PiModelsFile} { + if !bytes.Contains(files[path], []byte(`"thinkingFormat": "qwen-chat-template"`)) { + t.Fatalf("%s dropped Qwen thinking compatibility", path) + } + } + var models piModels + if err := json.Unmarshal(files[PiModelsFile], &models); err != nil { + t.Fatal(err) + } + if !models.Providers["local"].Models[0].Reasoning { + t.Fatal("Pi must know the model supports explicit thinking control") + } +} + func TestProjectionKeepsSingleUseInventoryOutOfSharedPiSelection(t *testing.T) { i := projectionFixture(t) p := i.ModelCatalog.Providers["zai"]