fix(evidence): stabilize real Pi authoring

This commit is contained in:
2026-08-25 15:21:59 +02:00
parent 610ae8c85a
commit 8ba87b68dc
7 changed files with 486 additions and 23 deletions
@@ -0,0 +1,14 @@
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
export default function (pi: ExtensionAPI) {
pi.on("before_provider_request", (event) => {
if (typeof event.payload !== "object" || event.payload === null) {
return undefined;
}
return {
...event.payload,
temperature: 0,
response_format: { type: "json_object" },
};
});
}
@@ -1,7 +1,13 @@
---
name: tht-evidence-authoring
description: Restructure exactly one normalized Thoth Source Evidence request into strict, typed Evidence candidate JSON for deterministic host-side review.
---
# Evidence authoring response contract # Evidence authoring response contract
You receive exactly one normalized Source Evidence request. Return one JSON object with You receive exactly one normalized Source Evidence request. Return one JSON object with
only a `candidates` array, without Markdown fences or explanatory text. only a `candidates` array, without Markdown fences, comments, or explanatory text. The
host strictly rejects unknown or missing fields.
Use only facts present in `normalized_text`. Never merge, cite, or infer facts from Use only facts present in `normalized_text`. Never merge, cite, or infer facts from
another source. Reuse an `existing_id` only when it was supplied in `previous_units`; another source. Reuse an `existing_id` only when it was supplied in `previous_units`;
@@ -9,7 +15,78 @@ otherwise omit it. Preserve prior reviewed wording when it is still supported. W
prior unit is no longer supported, omit its `existing_id` and add a prior unit is no longer supported, omit its `existing_id` and add a
`source_no_longer_supports_unit` review item to the related candidate when applicable. `source_no_longer_supports_unit` review item to the related candidate when applicable.
Every candidate must match schema version 1, use one of the eight declared Evidence Each candidate has exactly this shape:
kinds, include one to five short exact excerpts copied from `normalized_text`, and add
review items for ambiguities. Do not assign a new canonical ID; the host does that ```json
deterministically. Do not use tools or alter any repository state. {
"schema_version": 1,
"existing_id": "evidence:optional-existing-id",
"title": "Short human title",
"kind": "glossary",
"purposes": ["disambiguation"],
"applies_to": {
"concepts": ["concept"],
"tables": ["schema.table"],
"columns": ["schema.table.column"]
},
"language": "it",
"supporting_excerpts": ["One short exact excerpt copied from normalized_text."],
"review_items": [
{"code": "specific_ambiguity", "message": "What a human must decide.", "field": "payload"}
],
"payload": {"definition": "Typed payload described below.", "synonyms": [], "variants": []}
}
```
Omit `existing_id` for new candidates. `applies_to` must contain only `concepts`,
`tables`, and `columns`; use empty arrays when the source does not establish a value.
Every table identifier must be `schema.table`, and every column identifier must be
`schema.table.column`. Include a table or column identifier only when that fully
qualified literal already appears in `normalized_text`. Never qualify an unqualified
name yourself. If the source contains only names such as `fact_ilr` or `cod_paz`, leave
the corresponding `tables` or `columns` array empty, keep the names in prose, and add a
review item when qualification matters. `review_items` may be empty. A review item
contains only `code`, `message`, and optional `field`.
Allowed `purposes` are `disambiguation`, `rewriting`, `schema_linking`, and
`sql_generation`. Allowed `kind` values and their exact `payload` shapes are:
- `glossary`: `{"definition": string, "synonyms": [string], "variants": [string]}`
- `domain`: `{"rule": string}`
- `enum`: `{"column": "schema.table.column", "values": {"stored value": "meaning"}}`
- `example`: `{"question": string, "interpretation": string}`
- `mapping`: `{"concept": string, "tables": ["schema.table"], "columns": ["schema.table.column"]}`
- `normalization`: `{"input": string, "output": string, "rule": string}`
- `formula`: `{"concept": string, "columns": ["schema.table.column"], "sql": "one PostgreSQL expression"}`
- `reference`: `{"url": "https://...", "label": string, "description": string}`
Return exactly one candidate: the source's primary independent, reviewable Evidence
Unit. Preserve the source's secondary facts in that unit's typed rule, definition, or
interpretation instead of emitting extra candidates; do not atomize individual
sentences. The candidate must
include one to five nonempty exact `supporting_excerpts`, each at most 1000 characters.
Copy each excerpt as one continuous, character-for-character substring of
`normalized_text`, including its original Markdown punctuation. Prefer copying one
complete source line. Never paraphrase, normalize whitespace, remove backticks, or
change quotation marks inside an excerpt.
Before returning JSON, check every excerpt with the equivalent of
`excerpt in normalized_text`; replace any excerpt that would fail with an exact complete
line from the source. Also check that there is exactly one candidate, every object
has only the declared fields, every kind has the exact payload shape above, and stdout
contains only the JSON object. Add a review item whenever the source leaves a material
ambiguity; never silently guess a table, column, enum meaning, formula, or URL.
Use the source path as a kind hint: `00-glossario` normally yields `glossary` or
`domain`; `10-domini-clinici` normally yields `domain`; `20-valori-enum` normally yields
`enum`; `30-esempi-nlq` normally yields `example`; `40-mapping-semantico` normally yields
`mapping` or `domain`; and `50-metadati-normalizzazione` normally yields
`normalization`. Depart from the hinted kind only when the source explicitly provides
the complete typed payload for another kind. Emit `formula` only when the source states
one complete PostgreSQL expression and all referenced columns are fully qualified. If
an `enum`, `mapping`, or `formula` payload would require an identifier that is not
already fully qualified in the source, emit a `domain` candidate instead and record the
missing qualification as a review item.
Do not assign a new canonical ID; the host does that deterministically. Do not use tools
or alter any repository state.
+35
View File
@@ -1,4 +1,5 @@
import hashlib import hashlib
import threading
import unicodedata import unicodedata
import pytest import pytest
@@ -348,6 +349,40 @@ def test_prepare_changed_source_uses_one_model_call_and_applies_a_valid_batch(tm
assert validate_workspace_evidence(tmp_path).publishable is True assert validate_workspace_evidence(tmp_path).publishable is True
def test_prepare_can_issue_independent_source_calls_concurrently(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text)
source_root = tmp_path / "evidence" / "source" / "domain"
changed_text = source_text + "\nRegola revisionata.\n"
(source_root / "patient.md").write_text(changed_text, encoding="utf-8")
(source_root / "second.md").write_text(changed_text, encoding="utf-8")
barrier = threading.Barrier(2)
class ConcurrentRestructurer:
def __init__(self):
self.requests = []
def restructure(self, request):
self.requests.append(request)
barrier.wait(timeout=2)
return (_candidate(title=f"Unit {request.source_file}"),)
restructurer = ConcurrentRestructurer()
report = prepare_workspace_evidence(
tmp_path,
restructurer=restructurer,
git_status=lambda _: (),
max_workers=2,
)
assert report.model_calls == 2
assert sorted(request.source_file for request in restructurer.requests) == [
"source/domain/patient.md",
"source/domain/second.md",
]
def test_prepare_unchanged_source_skips_model_and_does_not_write(tmp_path): def test_prepare_unchanged_source_skips_model_and_does_not_write(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici." source_text = "I pazienti sotto i 18 anni sono pediatrici."
_write_workspace(tmp_path, _evidence(source_text), source_text) _write_workspace(tmp_path, _evidence(source_text), source_text)
+26
View File
@@ -14,6 +14,32 @@ def test_evidence_authoring_commands_are_distinct_from_runtime_preprocessing():
assert "validate" in result.output assert "validate" in result.output
def test_evidence_prepare_failure_identifies_the_source_file(monkeypatch, tmp_path):
from tht.cli import evidence_cmd
from tht.evidence import EvidencePreparationError
monkeypatch.setattr(evidence_cmd, "_canonical_worktree", lambda root: root)
monkeypatch.setattr(
evidence_cmd,
"prepare_workspace_evidence",
lambda *args, **kwargs: (_ for _ in ()).throw(
EvidencePreparationError("pi_restructure_invalid", "source/domain/patient.md")
),
)
result = CliRunner().invoke(app, ["evidence", "prepare", str(tmp_path), "--json"])
assert result.exit_code == 1
assert result.stderr == ""
assert json.loads(result.stdout) == {
"code": "pi_restructure_invalid",
"operation": "evidence_prepare",
"schemaVersion": 1,
"sourceFile": "source/domain/patient.md",
"status": "failed",
}
def test_evidence_resolve_json_is_pristine(monkeypatch, tmp_path): def test_evidence_resolve_json_is_pristine(monkeypatch, tmp_path):
from tht.cli import evidence_cmd from tht.cli import evidence_cmd
from tht.evidence.authoring import EvidenceResolutionReport from tht.evidence.authoring import EvidenceResolutionReport
+158 -3
View File
@@ -1,4 +1,5 @@
import json import json
from pathlib import Path
from types import SimpleNamespace from types import SimpleNamespace
import pytest import pytest
@@ -35,6 +36,9 @@ def _candidate():
def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeypatch): def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeypatch):
from tht.evidence import authoring from tht.evidence import authoring
monkeypatch.setenv("PI_PROVIDER", "zai")
monkeypatch.setenv("PI_MODEL", "glm-5.2")
monkeypatch.setenv("PI_THINKING", "medium")
calls = [] calls = []
def run(argv, **kwargs): def run(argv, **kwargs):
@@ -43,7 +47,9 @@ def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeyp
return SimpleNamespace(returncode=0) return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run) monkeypatch.setattr(authoring.subprocess, "run", run)
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md", timeout_seconds=12) skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
restructurer = PiEvidenceRestructurer("pi-test", skill_path, timeout_seconds=12)
candidates = restructurer.restructure(_request()) candidates = restructurer.restructure(_request())
@@ -52,7 +58,22 @@ def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeyp
assert argv[:8] == [ assert argv[:8] == [
"pi-test", "--mode", "text", "--print", "--no-session", "--no-tools", "--no-extensions", "--no-context-files", "pi-test", "--mode", "text", "--print", "--no-session", "--no-tools", "--no-extensions", "--no-context-files",
] ]
assert "--skill" in argv assert "--no-skills" in argv
assert argv[argv.index("--extension") + 1].endswith(
"/extensions/tht-evidence-json-mode.ts"
)
assert "--system-prompt" in argv
assert argv[argv.index("--system-prompt") + 1] == "Evidence-only system prompt"
assert "--skill" not in argv
assert "--append-system-prompt" not in argv
assert "/skill:tht-evidence-authoring" not in argv
assert argv[argv.index("--provider") + 1] == "zai"
assert argv[argv.index("--model") + 1] == "glm-5.2"
assert argv[argv.index("--thinking") + 1] == "medium"
assert any(
"must not use kind enum or formula" in argument
for argument in argv
)
assert any(argument.startswith("@") for argument in argv) assert any(argument.startswith("@") for argument in argv)
assert kwargs["timeout"] == 12 assert kwargs["timeout"] == 12
assert kwargs["shell"] is False assert kwargs["shell"] is False
@@ -71,10 +92,144 @@ def test_pi_restructurer_returns_a_bounded_error_without_model_output(tmp_path,
return result return result
monkeypatch.setattr(authoring.subprocess, "run", run) monkeypatch.setattr(authoring.subprocess, "run", run)
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md") skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
restructurer = PiEvidenceRestructurer("pi-test", skill_path)
with pytest.raises(EvidencePreparationError) as error: with pytest.raises(EvidencePreparationError) as error:
restructurer.restructure(_request()) restructurer.restructure(_request())
assert error.value.code in {"pi_restructure_failed", "pi_restructure_invalid"} assert error.value.code in {"pi_restructure_failed", "pi_restructure_invalid"}
assert "secret" not in str(error.value) assert "secret" not in str(error.value)
def test_evidence_authoring_skill_is_loadable_and_declares_the_wire_schema():
skill = (
Path(__file__).parents[1] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
).read_text(encoding="utf-8")
assert skill.startswith("---\nname: tht-evidence-authoring\ndescription:")
for required_field in (
"schema_version",
"existing_id",
"supporting_excerpts",
"review_items",
"payload",
):
assert required_field in skill
def test_evidence_authoring_json_mode_extension_only_rewrites_the_provider_payload():
extension = (
Path(__file__).parents[1] / ".pi" / "extensions" / "tht-evidence-json-mode.ts"
).read_text(encoding="utf-8")
assert 'pi.on("before_provider_request"' in extension
assert 'response_format: { type: "json_object" }' in extension
assert "temperature: 0" in extension
assert "registerTool" not in extension
@pytest.mark.parametrize("response", [_candidate(), [_candidate()]])
def test_pi_restructurer_normalizes_bounded_candidate_envelopes(
tmp_path, monkeypatch, response,
):
from tht.evidence import authoring
def run(argv, **kwargs):
kwargs["stdout"].write(json.dumps(response))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(_request())
assert [candidate.title for candidate in candidates] == ["Fascia pediatrica"]
def test_pi_restructurer_restores_one_unique_markdown_source_line(tmp_path, monkeypatch):
from tht.evidence import authoring
candidate = _candidate() | {
"supporting_excerpts": ["Pazienti sotto i 18 anni sono pediatrici."],
}
def run(argv, **kwargs):
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
request = _request().model_copy(update={
"normalized_text": "- **Pazienti** sotto i 18 anni sono pediatrici.\n",
})
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
assert candidates[0].supporting_excerpts == (
"- **Pazienti** sotto i 18 anni sono pediatrici.",
)
def test_pi_restructurer_flags_one_unique_fuzzy_source_line_for_review(tmp_path, monkeypatch):
from tht.evidence import authoring
candidate = _candidate() | {
"supporting_excerpts": ["Pazienti sotto 18 anni sono pediatrici."],
}
def run(argv, **kwargs):
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
request = _request().model_copy(update={
"normalized_text": "Pazienti sotto i 18 anni sono pediatrici.\nAdulti sopra i 65 anni.\n",
})
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
assert candidates[0].supporting_excerpts == (
"Pazienti sotto i 18 anni sono pediatrici.",
)
assert [item.code for item in candidates[0].review_items] == [
"supporting_excerpt_reconciled",
]
def test_pi_restructurer_restores_one_contiguous_multiline_source_excerpt(
tmp_path, monkeypatch,
):
from tht.evidence import authoring
candidate = _candidate() | {
"supporting_excerpts": [
(
"pivot/denormalization (solo per le FACT): legge le righe correlate "
"nella tabella source_table, estrae source_column usando le join_keys."
)
],
}
def run(argv, **kwargs):
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
return SimpleNamespace(returncode=0)
monkeypatch.setattr(authoring.subprocess, "run", run)
skill_path = tmp_path / "skill.md"
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
exact = (
"- **pivot/denormalization** (solo per le FACT):\n"
" - legge le righe correlate nella tabella `source_table`,\n"
" - estrae `source_column` usando le `join_keys`."
)
request = _request().model_copy(update={"normalized_text": exact + "\n"})
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
assert candidates[0].supporting_excerpts == (exact,)
+19 -2
View File
@@ -117,9 +117,26 @@ def prepare_cmd(
skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md" skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path) restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path)
try: try:
report = prepare_workspace_evidence(root, restructurer=restructurer, upgrade=upgrade) try:
max_workers = int(os.environ.get("THT_EVIDENCE_AUTHORING_WORKERS", "1"))
except ValueError as error:
raise EvidencePreparationError("authoring_workers_invalid") from error
report = prepare_workspace_evidence(
root,
restructurer=restructurer,
upgrade=upgrade,
max_workers=max_workers,
)
except EvidencePreparationError as error: except EvidencePreparationError as error:
_emit({"schemaVersion": 1, "operation": "evidence_prepare", "status": "failed", "code": error.code}, json_output) payload = {
"schemaVersion": 1,
"operation": "evidence_prepare",
"status": "failed",
"code": error.code,
}
if error.source_file is not None:
payload["sourceFile"] = error.source_file
_emit(payload, json_output)
raise typer.Exit(code=1) from error raise typer.Exit(code=1) from error
_emit(_preparation_payload(report), json_output) _emit(_preparation_payload(report), json_output)
if report.findings: if report.findings:
+152 -13
View File
@@ -11,7 +11,9 @@ import subprocess
import tempfile import tempfile
import unicodedata import unicodedata
from collections.abc import Callable from collections.abc import Callable
from concurrent.futures import ThreadPoolExecutor
from dataclasses import dataclass from dataclasses import dataclass
from difflib import SequenceMatcher
from pathlib import Path from pathlib import Path
from typing import Literal, Protocol from typing import Literal, Protocol
@@ -103,6 +105,90 @@ class EvidenceRestructurer(Protocol):
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ... def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ...
def _excerpt_signature(value: str) -> str:
value = "\n".join(
re.sub(r"^\s*(?:(?:[-+*>]|#+)\s+)", "", line)
for line in value.splitlines()
)
value = re.sub(r"[*_`]", "", value)
return " ".join(value.split())
def _request_specific_constraints(source_text: str) -> str:
qualified_column = re.search(
r"(?<![A-Za-z0-9_])[A-Za-z_][A-Za-z0-9_]*"
r"\.[A-Za-z_][A-Za-z0-9_]*\.[A-Za-z_][A-Za-z0-9_]*"
r"(?![A-Za-z0-9_])",
source_text,
)
if qualified_column is None:
return (
"This request contains no literal schema.table.column identifier, so you "
"must not use kind enum or formula. Use domain instead and add a review "
"item when a missing qualification prevents the more specific kind."
)
return "Apply the Evidence authoring response contract exactly."
def _restore_exact_source_excerpts(
candidate: RestructureCandidate,
source_text: str,
) -> RestructureCandidate:
source_lines = source_text.splitlines()
source_spans: list[str] = []
for start, first_line in enumerate(source_lines):
if not first_line:
continue
for width in range(1, 6):
selected = source_lines[start:start + width]
if len(selected) != width or not selected[-1]:
continue
span = "\n".join(selected)
if len(span) <= 1000:
source_spans.append(span)
restored: list[str] = []
reconciled = False
for excerpt in candidate.supporting_excerpts:
if excerpt in source_text:
restored.append(excerpt)
continue
signature = _excerpt_signature(excerpt)
matches = tuple(span for span in source_spans if _excerpt_signature(span) == signature)
if len(matches) == 1:
restored.append(matches[0])
continue
ranked = sorted(
(
(SequenceMatcher(None, signature, _excerpt_signature(line)).ratio(), line)
for line in source_spans
),
reverse=True,
)
best_score = ranked[0][0] if ranked else 0.0
runner_up_score = ranked[1][0] if len(ranked) > 1 else 0.0
if best_score >= 0.88 and best_score - runner_up_score >= 0.08:
restored.append(ranked[0][1])
reconciled = True
else:
restored.append(excerpt)
review_items = candidate.review_items
if reconciled and not any(
item.code == "supporting_excerpt_reconciled" for item in review_items
):
review_items = (*review_items, ReviewItem(
code="supporting_excerpt_reconciled",
message=(
"A model excerpt was reconciled to a unique exact source line; "
"confirm that the restored quotation supports this unit."
),
field="supporting_excerpts",
))
return candidate.model_copy(update={
"supporting_excerpts": tuple(restored),
"review_items": review_items,
})
class PiEvidenceRestructurer: class PiEvidenceRestructurer:
"""Invoke Pi once, without tools or session state, for one changed source.""" """Invoke Pi once, without tools or session state, for one changed source."""
@@ -111,13 +197,20 @@ class PiEvidenceRestructurer:
pi_executable: str, pi_executable: str,
skill_path: Path, skill_path: Path,
*, *,
timeout_seconds: int = 120, timeout_seconds: int = 300,
) -> None: ) -> None:
self._pi_executable = pi_executable self._pi_executable = pi_executable
self._skill_path = skill_path self._skill_path = skill_path
self._json_mode_extension = (
skill_path.parents[2] / "extensions" / "tht-evidence-json-mode.ts"
)
self._timeout_seconds = timeout_seconds self._timeout_seconds = timeout_seconds
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]:
try:
system_prompt = self._skill_path.read_text(encoding="utf-8")
except (OSError, UnicodeError) as error:
raise EvidencePreparationError("pi_skill_invalid", request.source_file) from error
with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary: with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary:
request_path = Path(temporary) / "request.json" request_path = Path(temporary) / "request.json"
request_path.write_text( request_path.write_text(
@@ -132,10 +225,25 @@ class PiEvidenceRestructurer:
"--no-tools", "--no-tools",
"--no-extensions", "--no-extensions",
"--no-context-files", "--no-context-files",
"--skill", str(self._skill_path), "--no-skills",
f"@{request_path}", "--extension", str(self._json_mode_extension),
"Return only the JSON object required by the Evidence authoring skill.",
] ]
for option, environment_name in (
("--provider", "PI_PROVIDER"),
("--model", "PI_MODEL"),
("--thinking", "PI_THINKING"),
):
value = os.environ.get(environment_name)
if value:
argv.extend((option, value))
argv.extend((
"--system-prompt", system_prompt,
f"@{request_path}",
(
"Return only the JSON object required by the Evidence authoring skill. "
+ _request_specific_constraints(request.normalized_text)
),
))
try: try:
with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response: with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response:
result = subprocess.run( result = subprocess.run(
@@ -156,9 +264,21 @@ class PiEvidenceRestructurer:
raise EvidencePreparationError("pi_restructure_failed", request.source_file) raise EvidencePreparationError("pi_restructure_failed", request.source_file)
try: try:
raw = json.loads(stdout) raw = json.loads(stdout)
if not isinstance(raw, dict) or set(raw) != {"candidates"} or not isinstance(raw["candidates"], list): if isinstance(raw, dict) and set(raw) == {"candidates"}:
candidates = raw["candidates"]
elif isinstance(raw, list):
candidates = raw
elif isinstance(raw, dict):
candidates = [raw]
else:
raise ValueError("response shape") raise ValueError("response shape")
return tuple(RestructureCandidate.model_validate(candidate) for candidate in raw["candidates"]) if not isinstance(candidates, list):
raise TypeError("response shape")
parsed = tuple(RestructureCandidate.model_validate(candidate) for candidate in candidates)
return tuple(
_restore_exact_source_excerpts(candidate, request.normalized_text)
for candidate in parsed
)
except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error: except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error:
raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error
@@ -418,6 +538,7 @@ def prepare_workspace_evidence(
restructurer: EvidenceRestructurer, restructurer: EvidenceRestructurer,
git_status: Callable[[Path], tuple[str, ...]] | None = None, git_status: Callable[[Path], tuple[str, ...]] | None = None,
upgrade: bool = False, upgrade: bool = False,
max_workers: int = 1,
) -> EvidencePreparationReport: ) -> EvidencePreparationReport:
"""Prepare all changed Source Evidence without publishing or committing it. """Prepare all changed Source Evidence without publishing or committing it.
@@ -425,6 +546,8 @@ def prepare_workspace_evidence(
one call through ``restructurer``; all model results validate before the staged one call through ``restructurer``; all model results validate before the staged
authoring tree replaces the current one. authoring tree replaces the current one.
""" """
if not 1 <= max_workers <= 8:
raise EvidencePreparationError("authoring_workers_invalid")
workspace_root = workspace_root.resolve() workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence" evidence_root = workspace_root / "evidence"
_reject_dirty_authoring_state(workspace_root, git_status or _git_status) _reject_dirty_authoring_state(workspace_root, git_status or _git_status)
@@ -454,7 +577,7 @@ def prepare_workspace_evidence(
) )
_preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned) _preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned)
reserved_ids = set(documents_by_id) requests: list[tuple[str, RestructureRequest]] = []
for source_file in sorted(source_texts): for source_file in sorted(source_texts):
source_text = source_texts[source_file] source_text = source_texts[source_file]
source_hash = _source_hash(source_text) source_hash = _source_hash(source_text)
@@ -476,12 +599,28 @@ def prepare_workspace_evidence(
normalized_text=source_text, normalized_text=source_text,
previous_units=previous, previous_units=previous,
) )
try: requests.append((source_file, request))
candidates = restructurer.restructure(request)
except EvidencePreparationError: candidates_by_source: dict[str, tuple[RestructureCandidate, ...]] = {}
raise with ThreadPoolExecutor(max_workers=max_workers) as executor:
except Exception as error: futures = {
raise EvidencePreparationError("restructuring_failed", source_file) from error source_file: executor.submit(restructurer.restructure, request)
for source_file, request in requests
}
for source_file, _request in requests:
try:
candidates_by_source[source_file] = futures[source_file].result()
except EvidencePreparationError:
raise
except Exception as error:
raise EvidencePreparationError("restructuring_failed", source_file) from error
reserved_ids = set(documents_by_id)
for source_file, request in requests:
source_text = source_texts[source_file]
source_hash = request.source_sha256
previous = request.previous_units
candidates = candidates_by_source[source_file]
model_calls += 1 model_calls += 1
selected_existing_ids: set[str] = set() selected_existing_ids: set[str] = set()
generated_ids: list[str] = [] generated_ids: list[str] = []