// model-matrix.mjs — workstream G Tier 1: clean-room cross-model behavior matrix. // // Drives `pi --mode rpc` exactly as the backend does (set_model -> set_thinking_level -> // prompt) for each (model x scenario) and classifies the FIRST turn. Decisive signal: a // `toolcall_start`/`tool_execution_start` means the model chained into work in-turn; its // absence + quiescence = the narrate-and-stop stall. // // Scenarios (first-turn only, cheap ~15-30s each): // new -> prompt `/nuova-domanda "kickoff"` (THT_SESSION set => PROVIDED kickoff; first // tool should be `read` of SKILL.md) // resume -> prompt `/riprendi-sessione ` (first tools should be `tht session show` + // `read SKILL.md` in-turn) — workstream A robustness across models. // // Usage: // node harness/scripts/model-matrix.mjs [provider/model,...] [thinking] [outDir] // Defaults: the named primaries + one weak/local model; thinking=medium. // // NOTE: spawns real models against the configured workspace. Use a THROWAWAY session and // delete it afterwards (psd is real client data). Kills each child after the first decisive // signal, so mutation is minimal (reads only; no gate is answered). import { spawn } from "node:child_process"; import fs from "node:fs"; import path from "node:path"; import { fileURLToPath } from "node:url"; const harnessDir = path.resolve(path.dirname(fileURLToPath(import.meta.url)), ".."); const sessionId = process.argv[2]; if (!sessionId) { console.error("usage: node model-matrix.mjs [provider/model,...] [thinking] [outDir]"); process.exit(2); } const models = (process.argv[3] ?? "zai/glm-5.2,deepseek/deepseek-v4-pro,aritmolab/qwen3.6-35b-a3b") .split(",").map((s) => s.trim()).filter(Boolean); const thinking = process.argv[4] ?? "medium"; const outDir = process.argv[5] ?? "/tmp/model-matrix"; fs.mkdirSync(outDir, { recursive: true }); const scenarios = ["new", "resume"]; const DEADLINE = 90000; // hard cap per cell const QUIESCENCE = 14000; // turn ended idle if this long with no new events and no toolcall const sleep = (n) => new Promise((r) => setTimeout(r, n)); function runCell(provider, modelId, scenario, rawOut) { return new Promise((resolve) => { const env = { ...process.env, THT_SESSION: sessionId, THT_AUTHOR: "matrix@local", PATH: `${harnessDir}/.venv/bin:${process.env.PATH ?? ""}`, }; const child = spawn("pi", ["--mode", "rpc"], { cwd: harnessDir, env }); const raw = fs.createWriteStream(rawOut); child.stderr.resume(); const t0 = Date.now(); const ms = () => Date.now() - t0; const counts = {}; let firstTool = null; let assistantText = ""; let modelError = null; // a 404/endpoint error is "model unavailable", not a behavioral stall let lastEventAtMs = 0; let promptAtMs = Infinity; let buf = ""; child.stdout.on("data", (d) => { raw.write(d); buf += d.toString(); let i; while ((i = buf.indexOf("\n")) >= 0) { const line = buf.slice(0, i); buf = buf.slice(i + 1); if (!line.trim()) continue; lastEventAtMs = ms(); let e; try { e = JSON.parse(line); } catch { continue; } const type = e.type ?? "?"; counts[type] = (counts[type] || 0) + 1; if ((type === "toolcall_start" || type === "tool_execution_start") && !firstTool) { firstTool = { atMs: ms(), name: e.toolName ?? e.name ?? e.tool?.name ?? e.input?.command ?? "?", }; } if (type === "message_update") { const delta = e.assistantMessageEvent?.delta; if (typeof delta === "string") assistantText += delta; } if (!modelError) { const m = e.message; if (m && (m.errorMessage || m.stopReason === "error")) { modelError = m.errorMessage || "stopReason=error"; } } } }); function send(obj) { try { child.stdin.write(JSON.stringify(obj) + "\n"); } catch {} } function finish(verdict) { clearInterval(timer); try { child.kill("SIGKILL"); } catch {} resolve({ provider, modelId, scenario, verdict: modelError ? "MODEL_ERROR" : verdict, firstToolAtMs: firstTool?.atMs ?? null, firstToolName: firstTool?.name ?? null, error: modelError, events: counts, textTail: assistantText.slice(-240).replace(/\s+/g, " ").trim(), }); } (async () => { await sleep(700); send({ type: "set_model", provider, modelId }); await sleep(1800); send({ type: "set_thinking_level", level: thinking }); await sleep(1000); const message = scenario === "resume" ? `/riprendi-sessione ${sessionId}` : `/nuova-domanda "kickoff"`; send({ type: "prompt", message }); promptAtMs = ms(); })(); const timer = setInterval(() => { if (modelError) return finish("MODEL_ERROR"); if (firstTool) return finish("CHAINED"); const sawTurn = lastEventAtMs > promptAtMs && (counts.agent_start || counts.turn_start || counts.message_start); const quiet = ms() - lastEventAtMs > QUIESCENCE; if (sawTurn && quiet) return finish("STALLED"); if (ms() > DEADLINE) return finish(sawTurn ? "STALLED" : "NO_TURN"); }, 1000); child.on("exit", () => { if (!firstTool) finish("CHILD_EXIT"); }); }); } const results = []; for (const m of models) { const [provider, modelId] = m.split("/"); for (const scenario of scenarios) { const tag = `${provider}_${modelId}_${scenario}`.replace(/[^a-z0-9_.-]/gi, "-"); const r = await runCell(provider, modelId, scenario, path.join(outDir, `${tag}.jsonl`)); results.push(r); console.error(`done ${m} ${scenario}: ${r.verdict} (${r.firstToolName ?? "-"} @ ${r.firstToolAtMs ?? "-"}ms)`); } } // Markdown summary table + raw JSON for the record. const rows = results.map((r) => `| ${r.provider}/${r.modelId} | ${r.scenario} | ${r.verdict} | ${r.firstToolName ?? "-"} | ${r.firstToolAtMs ?? "-"} | ${r.error ?? ""} |`, ); console.log("\n| model | scenario | verdict | first tool | t(ms) | error |"); console.log("|---|---|---|---|---|---|"); console.log(rows.join("\n")); console.log("\nJSON " + JSON.stringify(results));