fix(evidence): stabilize real Pi authoring

This commit is contained in:
2026-08-25 15:21:59 +02:00
parent 610ae8c85a
commit 8ba87b68dc
7 changed files with 486 additions and 23 deletions
+19 -2
View File
@@ -117,9 +117,26 @@ def prepare_cmd(
skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path)
try:
report = prepare_workspace_evidence(root, restructurer=restructurer, upgrade=upgrade)
try:
max_workers = int(os.environ.get("THT_EVIDENCE_AUTHORING_WORKERS", "1"))
except ValueError as error:
raise EvidencePreparationError("authoring_workers_invalid") from error
report = prepare_workspace_evidence(
root,
restructurer=restructurer,
upgrade=upgrade,
max_workers=max_workers,
)
except EvidencePreparationError as error:
_emit({"schemaVersion": 1, "operation": "evidence_prepare", "status": "failed", "code": error.code}, json_output)
payload = {
"schemaVersion": 1,
"operation": "evidence_prepare",
"status": "failed",
"code": error.code,
}
if error.source_file is not None:
payload["sourceFile"] = error.source_file
_emit(payload, json_output)
raise typer.Exit(code=1) from error
_emit(_preparation_payload(report), json_output)
if report.findings:
+152 -13
View File
@@ -11,7 +11,9 @@ import subprocess
import tempfile
import unicodedata
from collections.abc import Callable
from concurrent.futures import ThreadPoolExecutor
from dataclasses import dataclass
from difflib import SequenceMatcher
from pathlib import Path
from typing import Literal, Protocol
@@ -103,6 +105,90 @@ class EvidenceRestructurer(Protocol):
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ...
def _excerpt_signature(value: str) -> str:
value = "\n".join(
re.sub(r"^\s*(?:(?:[-+*>]|#+)\s+)", "", line)
for line in value.splitlines()
)
value = re.sub(r"[*_`]", "", value)
return " ".join(value.split())
def _request_specific_constraints(source_text: str) -> str:
qualified_column = re.search(
r"(?<![A-Za-z0-9_])[A-Za-z_][A-Za-z0-9_]*"
r"\.[A-Za-z_][A-Za-z0-9_]*\.[A-Za-z_][A-Za-z0-9_]*"
r"(?![A-Za-z0-9_])",
source_text,
)
if qualified_column is None:
return (
"This request contains no literal schema.table.column identifier, so you "
"must not use kind enum or formula. Use domain instead and add a review "
"item when a missing qualification prevents the more specific kind."
)
return "Apply the Evidence authoring response contract exactly."
def _restore_exact_source_excerpts(
candidate: RestructureCandidate,
source_text: str,
) -> RestructureCandidate:
source_lines = source_text.splitlines()
source_spans: list[str] = []
for start, first_line in enumerate(source_lines):
if not first_line:
continue
for width in range(1, 6):
selected = source_lines[start:start + width]
if len(selected) != width or not selected[-1]:
continue
span = "\n".join(selected)
if len(span) <= 1000:
source_spans.append(span)
restored: list[str] = []
reconciled = False
for excerpt in candidate.supporting_excerpts:
if excerpt in source_text:
restored.append(excerpt)
continue
signature = _excerpt_signature(excerpt)
matches = tuple(span for span in source_spans if _excerpt_signature(span) == signature)
if len(matches) == 1:
restored.append(matches[0])
continue
ranked = sorted(
(
(SequenceMatcher(None, signature, _excerpt_signature(line)).ratio(), line)
for line in source_spans
),
reverse=True,
)
best_score = ranked[0][0] if ranked else 0.0
runner_up_score = ranked[1][0] if len(ranked) > 1 else 0.0
if best_score >= 0.88 and best_score - runner_up_score >= 0.08:
restored.append(ranked[0][1])
reconciled = True
else:
restored.append(excerpt)
review_items = candidate.review_items
if reconciled and not any(
item.code == "supporting_excerpt_reconciled" for item in review_items
):
review_items = (*review_items, ReviewItem(
code="supporting_excerpt_reconciled",
message=(
"A model excerpt was reconciled to a unique exact source line; "
"confirm that the restored quotation supports this unit."
),
field="supporting_excerpts",
))
return candidate.model_copy(update={
"supporting_excerpts": tuple(restored),
"review_items": review_items,
})
class PiEvidenceRestructurer:
"""Invoke Pi once, without tools or session state, for one changed source."""
@@ -111,13 +197,20 @@ class PiEvidenceRestructurer:
pi_executable: str,
skill_path: Path,
*,
timeout_seconds: int = 120,
timeout_seconds: int = 300,
) -> None:
self._pi_executable = pi_executable
self._skill_path = skill_path
self._json_mode_extension = (
skill_path.parents[2] / "extensions" / "tht-evidence-json-mode.ts"
)
self._timeout_seconds = timeout_seconds
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]:
try:
system_prompt = self._skill_path.read_text(encoding="utf-8")
except (OSError, UnicodeError) as error:
raise EvidencePreparationError("pi_skill_invalid", request.source_file) from error
with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary:
request_path = Path(temporary) / "request.json"
request_path.write_text(
@@ -132,10 +225,25 @@ class PiEvidenceRestructurer:
"--no-tools",
"--no-extensions",
"--no-context-files",
"--skill", str(self._skill_path),
f"@{request_path}",
"Return only the JSON object required by the Evidence authoring skill.",
"--no-skills",
"--extension", str(self._json_mode_extension),
]
for option, environment_name in (
("--provider", "PI_PROVIDER"),
("--model", "PI_MODEL"),
("--thinking", "PI_THINKING"),
):
value = os.environ.get(environment_name)
if value:
argv.extend((option, value))
argv.extend((
"--system-prompt", system_prompt,
f"@{request_path}",
(
"Return only the JSON object required by the Evidence authoring skill. "
+ _request_specific_constraints(request.normalized_text)
),
))
try:
with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response:
result = subprocess.run(
@@ -156,9 +264,21 @@ class PiEvidenceRestructurer:
raise EvidencePreparationError("pi_restructure_failed", request.source_file)
try:
raw = json.loads(stdout)
if not isinstance(raw, dict) or set(raw) != {"candidates"} or not isinstance(raw["candidates"], list):
if isinstance(raw, dict) and set(raw) == {"candidates"}:
candidates = raw["candidates"]
elif isinstance(raw, list):
candidates = raw
elif isinstance(raw, dict):
candidates = [raw]
else:
raise ValueError("response shape")
return tuple(RestructureCandidate.model_validate(candidate) for candidate in raw["candidates"])
if not isinstance(candidates, list):
raise TypeError("response shape")
parsed = tuple(RestructureCandidate.model_validate(candidate) for candidate in candidates)
return tuple(
_restore_exact_source_excerpts(candidate, request.normalized_text)
for candidate in parsed
)
except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error:
raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error
@@ -418,6 +538,7 @@ def prepare_workspace_evidence(
restructurer: EvidenceRestructurer,
git_status: Callable[[Path], tuple[str, ...]] | None = None,
upgrade: bool = False,
max_workers: int = 1,
) -> EvidencePreparationReport:
"""Prepare all changed Source Evidence without publishing or committing it.
@@ -425,6 +546,8 @@ def prepare_workspace_evidence(
one call through ``restructurer``; all model results validate before the staged
authoring tree replaces the current one.
"""
if not 1 <= max_workers <= 8:
raise EvidencePreparationError("authoring_workers_invalid")
workspace_root = workspace_root.resolve()
evidence_root = workspace_root / "evidence"
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
@@ -454,7 +577,7 @@ def prepare_workspace_evidence(
)
_preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned)
reserved_ids = set(documents_by_id)
requests: list[tuple[str, RestructureRequest]] = []
for source_file in sorted(source_texts):
source_text = source_texts[source_file]
source_hash = _source_hash(source_text)
@@ -476,12 +599,28 @@ def prepare_workspace_evidence(
normalized_text=source_text,
previous_units=previous,
)
try:
candidates = restructurer.restructure(request)
except EvidencePreparationError:
raise
except Exception as error:
raise EvidencePreparationError("restructuring_failed", source_file) from error
requests.append((source_file, request))
candidates_by_source: dict[str, tuple[RestructureCandidate, ...]] = {}
with ThreadPoolExecutor(max_workers=max_workers) as executor:
futures = {
source_file: executor.submit(restructurer.restructure, request)
for source_file, request in requests
}
for source_file, _request in requests:
try:
candidates_by_source[source_file] = futures[source_file].result()
except EvidencePreparationError:
raise
except Exception as error:
raise EvidencePreparationError("restructuring_failed", source_file) from error
reserved_ids = set(documents_by_id)
for source_file, request in requests:
source_text = source_texts[source_file]
source_hash = request.source_sha256
previous = request.previous_units
candidates = candidates_by_source[source_file]
model_calls += 1
selected_existing_ids: set[str] = set()
generated_ids: list[str] = []