Fix Qwen session tool calls and expose thinking compatibility
Publish documentation / publish (push) Successful in 30s
Publish documentation / publish (push) Successful in 30s
This commit is contained in:
@@ -41,8 +41,11 @@ const runtimeModelSchema = z.object({
|
|||||||
supportsReasoningEffort: z.boolean(),
|
supportsReasoningEffort: z.boolean(),
|
||||||
supportsStore: z.boolean(),
|
supportsStore: z.boolean(),
|
||||||
maxTokensField: z.string().optional(),
|
maxTokensField: z.string().optional(),
|
||||||
|
thinkingFormat: z.enum(["qwen", "qwen-chat-template"]).optional(),
|
||||||
}).strict().optional(),
|
}).strict().optional(),
|
||||||
}).strict().optional(),
|
}).strict().refine((session) => !session.compatibility?.thinkingFormat || session.reasoning, {
|
||||||
|
message: "thinkingFormat requires reasoning: true",
|
||||||
|
}).optional(),
|
||||||
metadataGeneration: z.object({ disableThinking: z.boolean() }).strict().optional(),
|
metadataGeneration: z.object({ disableThinking: z.boolean() }).strict().optional(),
|
||||||
}).strict();
|
}).strict();
|
||||||
|
|
||||||
|
|||||||
@@ -95,6 +95,35 @@ test("returns empty catalogs when no runtime projection is configured", () => {
|
|||||||
expect(loadMetadataGenerationModels({}).catalog()).toEqual({ models: [], default: null });
|
expect(loadMetadataGenerationModels({}).catalog()).toEqual({ models: [], default: null });
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test.each([
|
||||||
|
["qwen-chat-template", true, true],
|
||||||
|
["qwen", true, true],
|
||||||
|
["qwen-typo", true, false],
|
||||||
|
["qwen-chat-template", false, false],
|
||||||
|
])("validates Qwen thinking format %s with reasoning=%s", (thinkingFormat, reasoning, valid) => {
|
||||||
|
const { catalogFile } = runtimeCatalog({
|
||||||
|
defaultInteraction: "local/qwen",
|
||||||
|
models: [{
|
||||||
|
id: "local/qwen", provider: "local", model: "qwen", label: "Qwen",
|
||||||
|
upstreamModel: "qwen", endpoint: { baseUrl: "http://localhost:8000/v1" },
|
||||||
|
authentication: { mode: "none" }, sessionAdapter: { mode: "openai_compatible" },
|
||||||
|
session: {
|
||||||
|
reasoning, contextWindow: 32768, maxTokens: 8192,
|
||||||
|
compatibility: {
|
||||||
|
supportsDeveloperRole: false, supportsReasoningEffort: false,
|
||||||
|
supportsStore: false, maxTokensField: "max_tokens", thinkingFormat,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}],
|
||||||
|
});
|
||||||
|
if (valid) {
|
||||||
|
expect(loadRuntimeModelCatalog(catalogFile).sessionModels()[0].session?.compatibility)
|
||||||
|
.toMatchObject({ thinkingFormat });
|
||||||
|
} else {
|
||||||
|
expect(() => loadRuntimeModelCatalog(catalogFile)).toThrow("runtime model catalog is invalid");
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
test("rejects a drifted default and an unprotected projection", () => {
|
test("rejects a drifted default and an unprotected projection", () => {
|
||||||
const drifted = runtimeCatalog({ defaultInteraction: "zai/missing" });
|
const drifted = runtimeCatalog({ defaultInteraction: "zai/missing" });
|
||||||
expect(() => loadRuntimeModelCatalog(drifted.catalogFile)).toThrow("interaction default is invalid");
|
expect(() => loadRuntimeModelCatalog(drifted.catalogFile)).toThrow("interaction default is invalid");
|
||||||
|
|||||||
@@ -128,6 +128,51 @@ endpoint expects a model name different from the catalog key.
|
|||||||
Provider integrations remain declarative. Do not register providers from
|
Provider integrations remain declarative. Do not register providers from
|
||||||
`harness/.pi/extensions/`; those extensions implement the workflow and human gates only.
|
`harness/.pi/extensions/`; those extensions implement the workflow and human gates only.
|
||||||
|
|
||||||
|
### Qwen 3.6 sessions and thinking controls
|
||||||
|
|
||||||
|
For Qwen served through a vLLM-compatible chat template, declare the following inside the
|
||||||
|
model's `session` block, alongside its context and output limits:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
reasoning: true
|
||||||
|
compatibility:
|
||||||
|
supportsDeveloperRole: false
|
||||||
|
supportsReasoningEffort: false
|
||||||
|
supportsStore: false
|
||||||
|
maxTokensField: max_tokens
|
||||||
|
thinkingFormat: qwen-chat-template
|
||||||
|
```
|
||||||
|
|
||||||
|
This makes Pi send `chat_template_kwargs.enable_thinking` from the selected thinking level,
|
||||||
|
with `preserve_thinking: true`. Choose **off** to explicitly disable thinking. The alternative
|
||||||
|
`thinkingFormat: qwen` is for endpoints expecting top-level `enable_thinking`. Both formats
|
||||||
|
require `reasoning: true`; declaring `reasoning: false` does not tell the server to disable
|
||||||
|
thinking. Omit `thinkingFormat` to preserve Pi's default behavior for other providers.
|
||||||
|
|
||||||
|
Regenerate projections with the updated host CLI and recreate the local core container after
|
||||||
|
rebuilding it. Do not add these fields directly to generated Pi files. These controls do not
|
||||||
|
force tool calls or certify the workflow; perform the operator verification above.
|
||||||
|
|
||||||
|
```sh
|
||||||
|
tht --installation /absolute/path/thothii-installation.yaml installation generate
|
||||||
|
```
|
||||||
|
|
||||||
|
Use the model identifier exposed by your endpoint, such as `qwen3.6-35b-a3b`, and limits
|
||||||
|
supported by that deployment. The thinking format configures the Pi session adapter;
|
||||||
|
metadata generation continues to use its separate LiteLLM settings.
|
||||||
|
|
||||||
|
If a session displays text such as `{"type":"bash","command":"tht session show … --json"}`
|
||||||
|
and never opens a review widget, that text is not an executed tool call. A verified cause
|
||||||
|
was the Evidence JSON extension being loaded into interactive sessions and forcing
|
||||||
|
`response_format: {type: "json_object"}`. Upgrade to the core image containing the fix:
|
||||||
|
the extension belongs in `.pi/evidence-extensions/` and is loaded explicitly only by
|
||||||
|
Evidence authoring. It must not also remain in the automatically loaded `.pi/extensions/`
|
||||||
|
directory. Regenerating model configuration alone does not remove an extension from an old image.
|
||||||
|
|
||||||
|
After upgrading, reload the browser and resume the session. Verify that Pi executes
|
||||||
|
`tht session show` and opens a review widget. This fix does not require changing the Qwen
|
||||||
|
server, forcing every turn to call a tool, or teaching the model to print tool-call JSON.
|
||||||
|
|
||||||
## Authentication
|
## Authentication
|
||||||
|
|
||||||
Every provider chooses one explicit mode:
|
Every provider chooses one explicit mode:
|
||||||
|
|||||||
@@ -0,0 +1,57 @@
|
|||||||
|
# Qwen 3.6: session tool-call failure and correction
|
||||||
|
|
||||||
|
Date: 2026-09-21. Verified runtime: Pi coding agent and Pi AI 0.80.3.
|
||||||
|
|
||||||
|
## Cause and correction
|
||||||
|
|
||||||
|
Interactive sessions returned text resembling a bash call and stopped before the
|
||||||
|
first review widget. The Evidence-only extension `tht-evidence-json-mode.ts` lived
|
||||||
|
under `.pi/extensions`, so Pi automatically loaded it into interactive sessions.
|
||||||
|
Its `before_provider_request` hook imposed `response_format: {type: "json_object"}`
|
||||||
|
and temperature zero on every request.
|
||||||
|
|
||||||
|
Replaying the captured startup request, with all original hooks preserved, isolated
|
||||||
|
the cause. With JSON response format, Qwen returned the command as text and finish
|
||||||
|
reason `stop`. Removing only that field produced a native `bash` call and finish
|
||||||
|
reason `tool_calls`. Both requests contained the same 15 tools and used High thinking.
|
||||||
|
|
||||||
|
The extension now lives in `.pi/evidence-extensions` and is loaded explicitly by
|
||||||
|
`PiEvidenceRestructurer`. Evidence retains JSON output. Interactive sessions retain
|
||||||
|
native tool calls. No changes to the remote Qwen server were required.
|
||||||
|
|
||||||
|
Earlier SDK probes replaced `session.agent.onPayload`, inadvertently bypassing the
|
||||||
|
extension hooks. Their success did not reproduce the application path and did not
|
||||||
|
establish a model or server fault.
|
||||||
|
|
||||||
|
## Model configuration
|
||||||
|
|
||||||
|
The installation catalog now accepts optional `session.compatibility.thinkingFormat`
|
||||||
|
values `qwen` and `qwen-chat-template`, requiring `reasoning: true`. Go validation,
|
||||||
|
the backend schema, and generated catalog/Pi projections preserve this setting.
|
||||||
|
It controls thinking; it was not the cause or correction of the tool-call failure.
|
||||||
|
|
||||||
|
For the verified endpoint, `qwen-chat-template` sends `enable_thinking` and
|
||||||
|
`preserve_thinking` inside `chat_template_kwargs`. Selecting Off explicitly disables
|
||||||
|
thinking. Declaring `reasoning: false` alone does not disable thinking on the server.
|
||||||
|
See the [operator configuration](../general/pi-configuration.md#qwen-36-sessions-and-thinking-controls)
|
||||||
|
and [Pi 0.80.3 model documentation](https://github.com/earendil-works/pi/blob/v0.80.3/packages/coding-agent/docs/models.md#openai-compatibility).
|
||||||
|
|
||||||
|
## Validation and local delivery
|
||||||
|
|
||||||
|
- Catalog and projection regression tests failed before the compatibility change,
|
||||||
|
then passed. Go config/modelprojection/CLI tests, 84 targeted backend tests,
|
||||||
|
TypeScript checking, and the strict documentation build passed.
|
||||||
|
- The extension-isolation regression failed before relocation. All 46 Evidence
|
||||||
|
authoring/restructurer tests and Ruff checks on changed Python files passed.
|
||||||
|
- The rebuilt local core image has digest
|
||||||
|
`sha256:9cd593d7362dcbefe177f1b9eb4b9ffdf3010ced7bd7efc8fc3fdcd288f24579`.
|
||||||
|
Core/frontend were recreated, healthy, and returned HTTP 200.
|
||||||
|
- A real `pi --mode rpc --no-session` probe with Qwen and High thinking executed
|
||||||
|
`tht session show` successfully and reached the first clarification widget.
|
||||||
|
The probe allowed only that read and `reviewer_select`; it stopped without
|
||||||
|
submitting a human response or recording decisions. Its temporary configuration
|
||||||
|
referenced the mounted credential because the original runtime lease had expired.
|
||||||
|
- The user subsequently confirmed that the application now works.
|
||||||
|
|
||||||
|
This verifies recovery from the startup failure. It does not certify every workflow
|
||||||
|
phase or the separate LiteLLM metadata-generation path.
|
||||||
+3
@@ -1,5 +1,8 @@
|
|||||||
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
||||||
|
|
||||||
|
// Loaded explicitly by PiEvidenceRestructurer. Keep outside .pi/extensions:
|
||||||
|
// JSON-only output prevents interactive sessions from emitting native tool calls.
|
||||||
|
|
||||||
export default function (pi: ExtensionAPI) {
|
export default function (pi: ExtensionAPI) {
|
||||||
pi.on("before_provider_request", (event) => {
|
pi.on("before_provider_request", (event) => {
|
||||||
if (typeof event.payload !== "object" || event.payload === null) {
|
if (typeof event.payload !== "object" || event.payload === null) {
|
||||||
@@ -59,7 +59,7 @@ def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeyp
|
|||||||
]
|
]
|
||||||
assert "--no-skills" in argv
|
assert "--no-skills" in argv
|
||||||
assert argv[argv.index("--extension") + 1].endswith(
|
assert argv[argv.index("--extension") + 1].endswith(
|
||||||
"/extensions/tht-evidence-json-mode.ts"
|
"/evidence-extensions/tht-evidence-json-mode.ts"
|
||||||
)
|
)
|
||||||
assert "--system-prompt" in argv
|
assert "--system-prompt" in argv
|
||||||
assert argv[argv.index("--system-prompt") + 1] == "Evidence-only system prompt"
|
assert argv[argv.index("--system-prompt") + 1] == "Evidence-only system prompt"
|
||||||
@@ -120,7 +120,7 @@ def test_evidence_authoring_skill_is_loadable_and_declares_the_wire_schema():
|
|||||||
|
|
||||||
def test_evidence_authoring_json_mode_extension_only_rewrites_the_provider_payload():
|
def test_evidence_authoring_json_mode_extension_only_rewrites_the_provider_payload():
|
||||||
extension = (
|
extension = (
|
||||||
Path(__file__).parents[1] / ".pi" / "extensions" / "tht-evidence-json-mode.ts"
|
Path(__file__).parents[1] / ".pi" / "evidence-extensions" / "tht-evidence-json-mode.ts"
|
||||||
).read_text(encoding="utf-8")
|
).read_text(encoding="utf-8")
|
||||||
|
|
||||||
assert 'pi.on("before_provider_request"' in extension
|
assert 'pi.on("before_provider_request"' in extension
|
||||||
@@ -129,6 +129,19 @@ def test_evidence_authoring_json_mode_extension_only_rewrites_the_provider_paylo
|
|||||||
assert "registerTool" not in extension
|
assert "registerTool" not in extension
|
||||||
|
|
||||||
|
|
||||||
|
def test_evidence_json_mode_is_not_auto_loaded_into_interactive_sessions():
|
||||||
|
from tht.evidence.authoring import authoring_skill_path
|
||||||
|
|
||||||
|
skill_path = authoring_skill_path()
|
||||||
|
restructurer = PiEvidenceRestructurer("pi", skill_path)
|
||||||
|
extension_path = restructurer._json_mode_extension
|
||||||
|
auto_extensions = skill_path.parents[2] / "extensions"
|
||||||
|
|
||||||
|
assert extension_path.is_file()
|
||||||
|
assert not extension_path.is_relative_to(auto_extensions)
|
||||||
|
assert not (auto_extensions / extension_path.name).exists()
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.parametrize("response", [_candidate(), [_candidate()]])
|
@pytest.mark.parametrize("response", [_candidate(), [_candidate()]])
|
||||||
def test_pi_restructurer_normalizes_bounded_candidate_envelopes(
|
def test_pi_restructurer_normalizes_bounded_candidate_envelopes(
|
||||||
tmp_path, monkeypatch, response,
|
tmp_path, monkeypatch, response,
|
||||||
|
|||||||
@@ -209,7 +209,7 @@ class PiEvidenceRestructurer:
|
|||||||
self._pi_executable = pi_executable
|
self._pi_executable = pi_executable
|
||||||
self._skill_path = skill_path
|
self._skill_path = skill_path
|
||||||
self._json_mode_extension = (
|
self._json_mode_extension = (
|
||||||
skill_path.parents[2] / "extensions" / "tht-evidence-json-mode.ts"
|
skill_path.parents[2] / "evidence-extensions" / "tht-evidence-json-mode.ts"
|
||||||
)
|
)
|
||||||
self._timeout_seconds = timeout_seconds
|
self._timeout_seconds = timeout_seconds
|
||||||
|
|
||||||
|
|||||||
@@ -28,6 +28,38 @@ func TestLoadAcceptsInstallationModelCatalog(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestLoadQwenThinkingCompatibility(t *testing.T) {
|
||||||
|
for _, test := range []struct {
|
||||||
|
name, format string
|
||||||
|
reasoning bool
|
||||||
|
wantError bool
|
||||||
|
}{
|
||||||
|
{"vllm template", "qwen-chat-template", true, false},
|
||||||
|
{"top level Qwen", "qwen", true, false},
|
||||||
|
{"unknown format", "qwen-typo", true, true},
|
||||||
|
{"thinking disabled in model capabilities", "qwen-chat-template", false, true},
|
||||||
|
} {
|
||||||
|
t.Run(test.name, func(t *testing.T) {
|
||||||
|
path, _, _, _ := writeInstallation(t, "local")
|
||||||
|
reasoning := "false"
|
||||||
|
if test.reasoning {
|
||||||
|
reasoning = "true"
|
||||||
|
}
|
||||||
|
catalog := strings.Replace(validModelCatalogYAML(), " contextWindow: 32768",
|
||||||
|
" reasoning: "+reasoning+"\n compatibility:\n thinkingFormat: "+test.format+"\n contextWindow: 32768", 1)
|
||||||
|
rewriteInstallationCatalog(t, path, catalog)
|
||||||
|
_, err := Load(path)
|
||||||
|
if test.wantError {
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "thinkingFormat") {
|
||||||
|
t.Fatalf("Load() error = %v, want thinkingFormat error", err)
|
||||||
|
}
|
||||||
|
} else if err != nil {
|
||||||
|
t.Fatalf("Load() error = %v", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestLoadAcceptsSharedDeepSeekBuiltinWithBundleAuthentication(t *testing.T) {
|
func TestLoadAcceptsSharedDeepSeekBuiltinWithBundleAuthentication(t *testing.T) {
|
||||||
path, _, envFile, _ := writeInstallation(t, "local")
|
path, _, envFile, _ := writeInstallation(t, "local")
|
||||||
catalog := strings.Replace(validModelCatalogYAML(), "interaction: zai/glm-5.3", "interaction: deepseek/deepseek-v4-pro", 1)
|
catalog := strings.Replace(validModelCatalogYAML(), "interaction: zai/glm-5.3", "interaction: deepseek/deepseek-v4-pro", 1)
|
||||||
|
|||||||
@@ -90,6 +90,7 @@ type ModelCompatibility struct {
|
|||||||
SupportsReasoningEffort bool `yaml:"supportsReasoningEffort" json:"supportsReasoningEffort"`
|
SupportsReasoningEffort bool `yaml:"supportsReasoningEffort" json:"supportsReasoningEffort"`
|
||||||
SupportsStore bool `yaml:"supportsStore" json:"supportsStore"`
|
SupportsStore bool `yaml:"supportsStore" json:"supportsStore"`
|
||||||
MaxTokensField string `yaml:"maxTokensField" json:"maxTokensField,omitempty"`
|
MaxTokensField string `yaml:"maxTokensField" json:"maxTokensField,omitempty"`
|
||||||
|
ThinkingFormat string `yaml:"thinkingFormat,omitempty" json:"thinkingFormat,omitempty"`
|
||||||
}
|
}
|
||||||
|
|
||||||
type SessionModel struct {
|
type SessionModel struct {
|
||||||
@@ -270,6 +271,15 @@ func validateCatalogProvider(id string, provider ModelProvider, hasSession, hasM
|
|||||||
if model.Session != nil && (model.Session.ContextWindow <= 0 || model.Session.MaxTokens <= 0) {
|
if model.Session != nil && (model.Session.ContextWindow <= 0 || model.Session.MaxTokens <= 0) {
|
||||||
return fmt.Errorf("modelCatalog model %q/%q requires contextWindow and maxTokens", id, modelID)
|
return fmt.Errorf("modelCatalog model %q/%q requires contextWindow and maxTokens", id, modelID)
|
||||||
}
|
}
|
||||||
|
if model.Session != nil && model.Session.Compatibility != nil {
|
||||||
|
format := model.Session.Compatibility.ThinkingFormat
|
||||||
|
if format != "" && format != "qwen" && format != "qwen-chat-template" {
|
||||||
|
return fmt.Errorf("modelCatalog model %q/%q compatibility.thinkingFormat is invalid", id, modelID)
|
||||||
|
}
|
||||||
|
if format != "" && !model.Session.Reasoning {
|
||||||
|
return fmt.Errorf("modelCatalog model %q/%q compatibility.thinkingFormat requires reasoning: true", id, modelID)
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} else if provider.Session != nil {
|
} else if provider.Session != nil {
|
||||||
|
|||||||
@@ -97,6 +97,30 @@ func TestDeepSeekBuiltinAndMetadataShareCatalogIdentityAndSecretReference(t *tes
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestQwenThinkingFormatSurvivesCatalogAndPiProjection(t *testing.T) {
|
||||||
|
i := projectionFixture(t)
|
||||||
|
model := i.ModelCatalog.Providers["local"].Models["qwen"]
|
||||||
|
if err := json.Unmarshal([]byte(`{"reasoning":true,"contextWindow":32768,"maxTokens":8192,"compatibility":{"supportsDeveloperRole":false,"supportsReasoningEffort":false,"supportsStore":false,"maxTokensField":"max_tokens","thinkingFormat":"qwen-chat-template"}}`), model.Session); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
files, err := Render(i)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
for _, path := range []string{CatalogFile, PiModelsFile} {
|
||||||
|
if !bytes.Contains(files[path], []byte(`"thinkingFormat": "qwen-chat-template"`)) {
|
||||||
|
t.Fatalf("%s dropped Qwen thinking compatibility", path)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
var models piModels
|
||||||
|
if err := json.Unmarshal(files[PiModelsFile], &models); err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if !models.Providers["local"].Models[0].Reasoning {
|
||||||
|
t.Fatal("Pi must know the model supports explicit thinking control")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestProjectionKeepsSingleUseInventoryOutOfSharedPiSelection(t *testing.T) {
|
func TestProjectionKeepsSingleUseInventoryOutOfSharedPiSelection(t *testing.T) {
|
||||||
i := projectionFixture(t)
|
i := projectionFixture(t)
|
||||||
p := i.ModelCatalog.Providers["zai"]
|
p := i.ModelCatalog.Providers["zai"]
|
||||||
|
|||||||
Reference in New Issue
Block a user