fix(evidence): stabilize real Pi authoring
This commit is contained in:
@@ -117,9 +117,26 @@ def prepare_cmd(
|
||||
skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
||||
restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path)
|
||||
try:
|
||||
report = prepare_workspace_evidence(root, restructurer=restructurer, upgrade=upgrade)
|
||||
try:
|
||||
max_workers = int(os.environ.get("THT_EVIDENCE_AUTHORING_WORKERS", "1"))
|
||||
except ValueError as error:
|
||||
raise EvidencePreparationError("authoring_workers_invalid") from error
|
||||
report = prepare_workspace_evidence(
|
||||
root,
|
||||
restructurer=restructurer,
|
||||
upgrade=upgrade,
|
||||
max_workers=max_workers,
|
||||
)
|
||||
except EvidencePreparationError as error:
|
||||
_emit({"schemaVersion": 1, "operation": "evidence_prepare", "status": "failed", "code": error.code}, json_output)
|
||||
payload = {
|
||||
"schemaVersion": 1,
|
||||
"operation": "evidence_prepare",
|
||||
"status": "failed",
|
||||
"code": error.code,
|
||||
}
|
||||
if error.source_file is not None:
|
||||
payload["sourceFile"] = error.source_file
|
||||
_emit(payload, json_output)
|
||||
raise typer.Exit(code=1) from error
|
||||
_emit(_preparation_payload(report), json_output)
|
||||
if report.findings:
|
||||
|
||||
@@ -11,7 +11,9 @@ import subprocess
|
||||
import tempfile
|
||||
import unicodedata
|
||||
from collections.abc import Callable
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from dataclasses import dataclass
|
||||
from difflib import SequenceMatcher
|
||||
from pathlib import Path
|
||||
from typing import Literal, Protocol
|
||||
|
||||
@@ -103,6 +105,90 @@ class EvidenceRestructurer(Protocol):
|
||||
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ...
|
||||
|
||||
|
||||
def _excerpt_signature(value: str) -> str:
|
||||
value = "\n".join(
|
||||
re.sub(r"^\s*(?:(?:[-+*>]|#+)\s+)", "", line)
|
||||
for line in value.splitlines()
|
||||
)
|
||||
value = re.sub(r"[*_`]", "", value)
|
||||
return " ".join(value.split())
|
||||
|
||||
|
||||
def _request_specific_constraints(source_text: str) -> str:
|
||||
qualified_column = re.search(
|
||||
r"(?<![A-Za-z0-9_])[A-Za-z_][A-Za-z0-9_]*"
|
||||
r"\.[A-Za-z_][A-Za-z0-9_]*\.[A-Za-z_][A-Za-z0-9_]*"
|
||||
r"(?![A-Za-z0-9_])",
|
||||
source_text,
|
||||
)
|
||||
if qualified_column is None:
|
||||
return (
|
||||
"This request contains no literal schema.table.column identifier, so you "
|
||||
"must not use kind enum or formula. Use domain instead and add a review "
|
||||
"item when a missing qualification prevents the more specific kind."
|
||||
)
|
||||
return "Apply the Evidence authoring response contract exactly."
|
||||
|
||||
|
||||
def _restore_exact_source_excerpts(
|
||||
candidate: RestructureCandidate,
|
||||
source_text: str,
|
||||
) -> RestructureCandidate:
|
||||
source_lines = source_text.splitlines()
|
||||
source_spans: list[str] = []
|
||||
for start, first_line in enumerate(source_lines):
|
||||
if not first_line:
|
||||
continue
|
||||
for width in range(1, 6):
|
||||
selected = source_lines[start:start + width]
|
||||
if len(selected) != width or not selected[-1]:
|
||||
continue
|
||||
span = "\n".join(selected)
|
||||
if len(span) <= 1000:
|
||||
source_spans.append(span)
|
||||
restored: list[str] = []
|
||||
reconciled = False
|
||||
for excerpt in candidate.supporting_excerpts:
|
||||
if excerpt in source_text:
|
||||
restored.append(excerpt)
|
||||
continue
|
||||
signature = _excerpt_signature(excerpt)
|
||||
matches = tuple(span for span in source_spans if _excerpt_signature(span) == signature)
|
||||
if len(matches) == 1:
|
||||
restored.append(matches[0])
|
||||
continue
|
||||
ranked = sorted(
|
||||
(
|
||||
(SequenceMatcher(None, signature, _excerpt_signature(line)).ratio(), line)
|
||||
for line in source_spans
|
||||
),
|
||||
reverse=True,
|
||||
)
|
||||
best_score = ranked[0][0] if ranked else 0.0
|
||||
runner_up_score = ranked[1][0] if len(ranked) > 1 else 0.0
|
||||
if best_score >= 0.88 and best_score - runner_up_score >= 0.08:
|
||||
restored.append(ranked[0][1])
|
||||
reconciled = True
|
||||
else:
|
||||
restored.append(excerpt)
|
||||
review_items = candidate.review_items
|
||||
if reconciled and not any(
|
||||
item.code == "supporting_excerpt_reconciled" for item in review_items
|
||||
):
|
||||
review_items = (*review_items, ReviewItem(
|
||||
code="supporting_excerpt_reconciled",
|
||||
message=(
|
||||
"A model excerpt was reconciled to a unique exact source line; "
|
||||
"confirm that the restored quotation supports this unit."
|
||||
),
|
||||
field="supporting_excerpts",
|
||||
))
|
||||
return candidate.model_copy(update={
|
||||
"supporting_excerpts": tuple(restored),
|
||||
"review_items": review_items,
|
||||
})
|
||||
|
||||
|
||||
class PiEvidenceRestructurer:
|
||||
"""Invoke Pi once, without tools or session state, for one changed source."""
|
||||
|
||||
@@ -111,13 +197,20 @@ class PiEvidenceRestructurer:
|
||||
pi_executable: str,
|
||||
skill_path: Path,
|
||||
*,
|
||||
timeout_seconds: int = 120,
|
||||
timeout_seconds: int = 300,
|
||||
) -> None:
|
||||
self._pi_executable = pi_executable
|
||||
self._skill_path = skill_path
|
||||
self._json_mode_extension = (
|
||||
skill_path.parents[2] / "extensions" / "tht-evidence-json-mode.ts"
|
||||
)
|
||||
self._timeout_seconds = timeout_seconds
|
||||
|
||||
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]:
|
||||
try:
|
||||
system_prompt = self._skill_path.read_text(encoding="utf-8")
|
||||
except (OSError, UnicodeError) as error:
|
||||
raise EvidencePreparationError("pi_skill_invalid", request.source_file) from error
|
||||
with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary:
|
||||
request_path = Path(temporary) / "request.json"
|
||||
request_path.write_text(
|
||||
@@ -132,10 +225,25 @@ class PiEvidenceRestructurer:
|
||||
"--no-tools",
|
||||
"--no-extensions",
|
||||
"--no-context-files",
|
||||
"--skill", str(self._skill_path),
|
||||
f"@{request_path}",
|
||||
"Return only the JSON object required by the Evidence authoring skill.",
|
||||
"--no-skills",
|
||||
"--extension", str(self._json_mode_extension),
|
||||
]
|
||||
for option, environment_name in (
|
||||
("--provider", "PI_PROVIDER"),
|
||||
("--model", "PI_MODEL"),
|
||||
("--thinking", "PI_THINKING"),
|
||||
):
|
||||
value = os.environ.get(environment_name)
|
||||
if value:
|
||||
argv.extend((option, value))
|
||||
argv.extend((
|
||||
"--system-prompt", system_prompt,
|
||||
f"@{request_path}",
|
||||
(
|
||||
"Return only the JSON object required by the Evidence authoring skill. "
|
||||
+ _request_specific_constraints(request.normalized_text)
|
||||
),
|
||||
))
|
||||
try:
|
||||
with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response:
|
||||
result = subprocess.run(
|
||||
@@ -156,9 +264,21 @@ class PiEvidenceRestructurer:
|
||||
raise EvidencePreparationError("pi_restructure_failed", request.source_file)
|
||||
try:
|
||||
raw = json.loads(stdout)
|
||||
if not isinstance(raw, dict) or set(raw) != {"candidates"} or not isinstance(raw["candidates"], list):
|
||||
if isinstance(raw, dict) and set(raw) == {"candidates"}:
|
||||
candidates = raw["candidates"]
|
||||
elif isinstance(raw, list):
|
||||
candidates = raw
|
||||
elif isinstance(raw, dict):
|
||||
candidates = [raw]
|
||||
else:
|
||||
raise ValueError("response shape")
|
||||
return tuple(RestructureCandidate.model_validate(candidate) for candidate in raw["candidates"])
|
||||
if not isinstance(candidates, list):
|
||||
raise TypeError("response shape")
|
||||
parsed = tuple(RestructureCandidate.model_validate(candidate) for candidate in candidates)
|
||||
return tuple(
|
||||
_restore_exact_source_excerpts(candidate, request.normalized_text)
|
||||
for candidate in parsed
|
||||
)
|
||||
except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error:
|
||||
raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error
|
||||
|
||||
@@ -418,6 +538,7 @@ def prepare_workspace_evidence(
|
||||
restructurer: EvidenceRestructurer,
|
||||
git_status: Callable[[Path], tuple[str, ...]] | None = None,
|
||||
upgrade: bool = False,
|
||||
max_workers: int = 1,
|
||||
) -> EvidencePreparationReport:
|
||||
"""Prepare all changed Source Evidence without publishing or committing it.
|
||||
|
||||
@@ -425,6 +546,8 @@ def prepare_workspace_evidence(
|
||||
one call through ``restructurer``; all model results validate before the staged
|
||||
authoring tree replaces the current one.
|
||||
"""
|
||||
if not 1 <= max_workers <= 8:
|
||||
raise EvidencePreparationError("authoring_workers_invalid")
|
||||
workspace_root = workspace_root.resolve()
|
||||
evidence_root = workspace_root / "evidence"
|
||||
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
||||
@@ -454,7 +577,7 @@ def prepare_workspace_evidence(
|
||||
)
|
||||
_preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned)
|
||||
|
||||
reserved_ids = set(documents_by_id)
|
||||
requests: list[tuple[str, RestructureRequest]] = []
|
||||
for source_file in sorted(source_texts):
|
||||
source_text = source_texts[source_file]
|
||||
source_hash = _source_hash(source_text)
|
||||
@@ -476,12 +599,28 @@ def prepare_workspace_evidence(
|
||||
normalized_text=source_text,
|
||||
previous_units=previous,
|
||||
)
|
||||
try:
|
||||
candidates = restructurer.restructure(request)
|
||||
except EvidencePreparationError:
|
||||
raise
|
||||
except Exception as error:
|
||||
raise EvidencePreparationError("restructuring_failed", source_file) from error
|
||||
requests.append((source_file, request))
|
||||
|
||||
candidates_by_source: dict[str, tuple[RestructureCandidate, ...]] = {}
|
||||
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
||||
futures = {
|
||||
source_file: executor.submit(restructurer.restructure, request)
|
||||
for source_file, request in requests
|
||||
}
|
||||
for source_file, _request in requests:
|
||||
try:
|
||||
candidates_by_source[source_file] = futures[source_file].result()
|
||||
except EvidencePreparationError:
|
||||
raise
|
||||
except Exception as error:
|
||||
raise EvidencePreparationError("restructuring_failed", source_file) from error
|
||||
|
||||
reserved_ids = set(documents_by_id)
|
||||
for source_file, request in requests:
|
||||
source_text = source_texts[source_file]
|
||||
source_hash = request.source_sha256
|
||||
previous = request.previous_units
|
||||
candidates = candidates_by_source[source_file]
|
||||
model_calls += 1
|
||||
selected_existing_ids: set[str] = set()
|
||||
generated_ids: list[str] = []
|
||||
|
||||
Reference in New Issue
Block a user