fix(evidence): stabilize real Pi authoring
This commit is contained in:
@@ -0,0 +1,14 @@
|
||||
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
||||
|
||||
export default function (pi: ExtensionAPI) {
|
||||
pi.on("before_provider_request", (event) => {
|
||||
if (typeof event.payload !== "object" || event.payload === null) {
|
||||
return undefined;
|
||||
}
|
||||
return {
|
||||
...event.payload,
|
||||
temperature: 0,
|
||||
response_format: { type: "json_object" },
|
||||
};
|
||||
});
|
||||
}
|
||||
@@ -1,7 +1,13 @@
|
||||
---
|
||||
name: tht-evidence-authoring
|
||||
description: Restructure exactly one normalized Thoth Source Evidence request into strict, typed Evidence candidate JSON for deterministic host-side review.
|
||||
---
|
||||
|
||||
# Evidence authoring response contract
|
||||
|
||||
You receive exactly one normalized Source Evidence request. Return one JSON object with
|
||||
only a `candidates` array, without Markdown fences or explanatory text.
|
||||
only a `candidates` array, without Markdown fences, comments, or explanatory text. The
|
||||
host strictly rejects unknown or missing fields.
|
||||
|
||||
Use only facts present in `normalized_text`. Never merge, cite, or infer facts from
|
||||
another source. Reuse an `existing_id` only when it was supplied in `previous_units`;
|
||||
@@ -9,7 +15,78 @@ otherwise omit it. Preserve prior reviewed wording when it is still supported. W
|
||||
prior unit is no longer supported, omit its `existing_id` and add a
|
||||
`source_no_longer_supports_unit` review item to the related candidate when applicable.
|
||||
|
||||
Every candidate must match schema version 1, use one of the eight declared Evidence
|
||||
kinds, include one to five short exact excerpts copied from `normalized_text`, and add
|
||||
review items for ambiguities. Do not assign a new canonical ID; the host does that
|
||||
deterministically. Do not use tools or alter any repository state.
|
||||
Each candidate has exactly this shape:
|
||||
|
||||
```json
|
||||
{
|
||||
"schema_version": 1,
|
||||
"existing_id": "evidence:optional-existing-id",
|
||||
"title": "Short human title",
|
||||
"kind": "glossary",
|
||||
"purposes": ["disambiguation"],
|
||||
"applies_to": {
|
||||
"concepts": ["concept"],
|
||||
"tables": ["schema.table"],
|
||||
"columns": ["schema.table.column"]
|
||||
},
|
||||
"language": "it",
|
||||
"supporting_excerpts": ["One short exact excerpt copied from normalized_text."],
|
||||
"review_items": [
|
||||
{"code": "specific_ambiguity", "message": "What a human must decide.", "field": "payload"}
|
||||
],
|
||||
"payload": {"definition": "Typed payload described below.", "synonyms": [], "variants": []}
|
||||
}
|
||||
```
|
||||
|
||||
Omit `existing_id` for new candidates. `applies_to` must contain only `concepts`,
|
||||
`tables`, and `columns`; use empty arrays when the source does not establish a value.
|
||||
Every table identifier must be `schema.table`, and every column identifier must be
|
||||
`schema.table.column`. Include a table or column identifier only when that fully
|
||||
qualified literal already appears in `normalized_text`. Never qualify an unqualified
|
||||
name yourself. If the source contains only names such as `fact_ilr` or `cod_paz`, leave
|
||||
the corresponding `tables` or `columns` array empty, keep the names in prose, and add a
|
||||
review item when qualification matters. `review_items` may be empty. A review item
|
||||
contains only `code`, `message`, and optional `field`.
|
||||
|
||||
Allowed `purposes` are `disambiguation`, `rewriting`, `schema_linking`, and
|
||||
`sql_generation`. Allowed `kind` values and their exact `payload` shapes are:
|
||||
|
||||
- `glossary`: `{"definition": string, "synonyms": [string], "variants": [string]}`
|
||||
- `domain`: `{"rule": string}`
|
||||
- `enum`: `{"column": "schema.table.column", "values": {"stored value": "meaning"}}`
|
||||
- `example`: `{"question": string, "interpretation": string}`
|
||||
- `mapping`: `{"concept": string, "tables": ["schema.table"], "columns": ["schema.table.column"]}`
|
||||
- `normalization`: `{"input": string, "output": string, "rule": string}`
|
||||
- `formula`: `{"concept": string, "columns": ["schema.table.column"], "sql": "one PostgreSQL expression"}`
|
||||
- `reference`: `{"url": "https://...", "label": string, "description": string}`
|
||||
|
||||
Return exactly one candidate: the source's primary independent, reviewable Evidence
|
||||
Unit. Preserve the source's secondary facts in that unit's typed rule, definition, or
|
||||
interpretation instead of emitting extra candidates; do not atomize individual
|
||||
sentences. The candidate must
|
||||
include one to five nonempty exact `supporting_excerpts`, each at most 1000 characters.
|
||||
Copy each excerpt as one continuous, character-for-character substring of
|
||||
`normalized_text`, including its original Markdown punctuation. Prefer copying one
|
||||
complete source line. Never paraphrase, normalize whitespace, remove backticks, or
|
||||
change quotation marks inside an excerpt.
|
||||
|
||||
Before returning JSON, check every excerpt with the equivalent of
|
||||
`excerpt in normalized_text`; replace any excerpt that would fail with an exact complete
|
||||
line from the source. Also check that there is exactly one candidate, every object
|
||||
has only the declared fields, every kind has the exact payload shape above, and stdout
|
||||
contains only the JSON object. Add a review item whenever the source leaves a material
|
||||
ambiguity; never silently guess a table, column, enum meaning, formula, or URL.
|
||||
|
||||
Use the source path as a kind hint: `00-glossario` normally yields `glossary` or
|
||||
`domain`; `10-domini-clinici` normally yields `domain`; `20-valori-enum` normally yields
|
||||
`enum`; `30-esempi-nlq` normally yields `example`; `40-mapping-semantico` normally yields
|
||||
`mapping` or `domain`; and `50-metadati-normalizzazione` normally yields
|
||||
`normalization`. Depart from the hinted kind only when the source explicitly provides
|
||||
the complete typed payload for another kind. Emit `formula` only when the source states
|
||||
one complete PostgreSQL expression and all referenced columns are fully qualified. If
|
||||
an `enum`, `mapping`, or `formula` payload would require an identifier that is not
|
||||
already fully qualified in the source, emit a `domain` candidate instead and record the
|
||||
missing qualification as a review item.
|
||||
|
||||
Do not assign a new canonical ID; the host does that deterministically. Do not use tools
|
||||
or alter any repository state.
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import hashlib
|
||||
import threading
|
||||
import unicodedata
|
||||
|
||||
import pytest
|
||||
@@ -348,6 +349,40 @@ def test_prepare_changed_source_uses_one_model_call_and_applies_a_valid_batch(tm
|
||||
assert validate_workspace_evidence(tmp_path).publishable is True
|
||||
|
||||
|
||||
def test_prepare_can_issue_independent_source_calls_concurrently(tmp_path):
|
||||
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||
source_root = tmp_path / "evidence" / "source" / "domain"
|
||||
changed_text = source_text + "\nRegola revisionata.\n"
|
||||
(source_root / "patient.md").write_text(changed_text, encoding="utf-8")
|
||||
(source_root / "second.md").write_text(changed_text, encoding="utf-8")
|
||||
barrier = threading.Barrier(2)
|
||||
|
||||
class ConcurrentRestructurer:
|
||||
def __init__(self):
|
||||
self.requests = []
|
||||
|
||||
def restructure(self, request):
|
||||
self.requests.append(request)
|
||||
barrier.wait(timeout=2)
|
||||
return (_candidate(title=f"Unit {request.source_file}"),)
|
||||
|
||||
restructurer = ConcurrentRestructurer()
|
||||
|
||||
report = prepare_workspace_evidence(
|
||||
tmp_path,
|
||||
restructurer=restructurer,
|
||||
git_status=lambda _: (),
|
||||
max_workers=2,
|
||||
)
|
||||
|
||||
assert report.model_calls == 2
|
||||
assert sorted(request.source_file for request in restructurer.requests) == [
|
||||
"source/domain/patient.md",
|
||||
"source/domain/second.md",
|
||||
]
|
||||
|
||||
|
||||
def test_prepare_unchanged_source_skips_model_and_does_not_write(tmp_path):
|
||||
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||
|
||||
@@ -14,6 +14,32 @@ def test_evidence_authoring_commands_are_distinct_from_runtime_preprocessing():
|
||||
assert "validate" in result.output
|
||||
|
||||
|
||||
def test_evidence_prepare_failure_identifies_the_source_file(monkeypatch, tmp_path):
|
||||
from tht.cli import evidence_cmd
|
||||
from tht.evidence import EvidencePreparationError
|
||||
|
||||
monkeypatch.setattr(evidence_cmd, "_canonical_worktree", lambda root: root)
|
||||
monkeypatch.setattr(
|
||||
evidence_cmd,
|
||||
"prepare_workspace_evidence",
|
||||
lambda *args, **kwargs: (_ for _ in ()).throw(
|
||||
EvidencePreparationError("pi_restructure_invalid", "source/domain/patient.md")
|
||||
),
|
||||
)
|
||||
|
||||
result = CliRunner().invoke(app, ["evidence", "prepare", str(tmp_path), "--json"])
|
||||
|
||||
assert result.exit_code == 1
|
||||
assert result.stderr == ""
|
||||
assert json.loads(result.stdout) == {
|
||||
"code": "pi_restructure_invalid",
|
||||
"operation": "evidence_prepare",
|
||||
"schemaVersion": 1,
|
||||
"sourceFile": "source/domain/patient.md",
|
||||
"status": "failed",
|
||||
}
|
||||
|
||||
|
||||
def test_evidence_resolve_json_is_pristine(monkeypatch, tmp_path):
|
||||
from tht.cli import evidence_cmd
|
||||
from tht.evidence.authoring import EvidenceResolutionReport
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
@@ -35,6 +36,9 @@ def _candidate():
|
||||
def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeypatch):
|
||||
from tht.evidence import authoring
|
||||
|
||||
monkeypatch.setenv("PI_PROVIDER", "zai")
|
||||
monkeypatch.setenv("PI_MODEL", "glm-5.2")
|
||||
monkeypatch.setenv("PI_THINKING", "medium")
|
||||
calls = []
|
||||
|
||||
def run(argv, **kwargs):
|
||||
@@ -43,7 +47,9 @@ def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeyp
|
||||
return SimpleNamespace(returncode=0)
|
||||
|
||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md", timeout_seconds=12)
|
||||
skill_path = tmp_path / "skill.md"
|
||||
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||
restructurer = PiEvidenceRestructurer("pi-test", skill_path, timeout_seconds=12)
|
||||
|
||||
candidates = restructurer.restructure(_request())
|
||||
|
||||
@@ -52,7 +58,22 @@ def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeyp
|
||||
assert argv[:8] == [
|
||||
"pi-test", "--mode", "text", "--print", "--no-session", "--no-tools", "--no-extensions", "--no-context-files",
|
||||
]
|
||||
assert "--skill" in argv
|
||||
assert "--no-skills" in argv
|
||||
assert argv[argv.index("--extension") + 1].endswith(
|
||||
"/extensions/tht-evidence-json-mode.ts"
|
||||
)
|
||||
assert "--system-prompt" in argv
|
||||
assert argv[argv.index("--system-prompt") + 1] == "Evidence-only system prompt"
|
||||
assert "--skill" not in argv
|
||||
assert "--append-system-prompt" not in argv
|
||||
assert "/skill:tht-evidence-authoring" not in argv
|
||||
assert argv[argv.index("--provider") + 1] == "zai"
|
||||
assert argv[argv.index("--model") + 1] == "glm-5.2"
|
||||
assert argv[argv.index("--thinking") + 1] == "medium"
|
||||
assert any(
|
||||
"must not use kind enum or formula" in argument
|
||||
for argument in argv
|
||||
)
|
||||
assert any(argument.startswith("@") for argument in argv)
|
||||
assert kwargs["timeout"] == 12
|
||||
assert kwargs["shell"] is False
|
||||
@@ -71,10 +92,144 @@ def test_pi_restructurer_returns_a_bounded_error_without_model_output(tmp_path,
|
||||
return result
|
||||
|
||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md")
|
||||
skill_path = tmp_path / "skill.md"
|
||||
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||
restructurer = PiEvidenceRestructurer("pi-test", skill_path)
|
||||
|
||||
with pytest.raises(EvidencePreparationError) as error:
|
||||
restructurer.restructure(_request())
|
||||
|
||||
assert error.value.code in {"pi_restructure_failed", "pi_restructure_invalid"}
|
||||
assert "secret" not in str(error.value)
|
||||
|
||||
|
||||
def test_evidence_authoring_skill_is_loadable_and_declares_the_wire_schema():
|
||||
skill = (
|
||||
Path(__file__).parents[1] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
||||
).read_text(encoding="utf-8")
|
||||
|
||||
assert skill.startswith("---\nname: tht-evidence-authoring\ndescription:")
|
||||
for required_field in (
|
||||
"schema_version",
|
||||
"existing_id",
|
||||
"supporting_excerpts",
|
||||
"review_items",
|
||||
"payload",
|
||||
):
|
||||
assert required_field in skill
|
||||
|
||||
|
||||
def test_evidence_authoring_json_mode_extension_only_rewrites_the_provider_payload():
|
||||
extension = (
|
||||
Path(__file__).parents[1] / ".pi" / "extensions" / "tht-evidence-json-mode.ts"
|
||||
).read_text(encoding="utf-8")
|
||||
|
||||
assert 'pi.on("before_provider_request"' in extension
|
||||
assert 'response_format: { type: "json_object" }' in extension
|
||||
assert "temperature: 0" in extension
|
||||
assert "registerTool" not in extension
|
||||
|
||||
|
||||
@pytest.mark.parametrize("response", [_candidate(), [_candidate()]])
|
||||
def test_pi_restructurer_normalizes_bounded_candidate_envelopes(
|
||||
tmp_path, monkeypatch, response,
|
||||
):
|
||||
from tht.evidence import authoring
|
||||
|
||||
def run(argv, **kwargs):
|
||||
kwargs["stdout"].write(json.dumps(response))
|
||||
return SimpleNamespace(returncode=0)
|
||||
|
||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||
skill_path = tmp_path / "skill.md"
|
||||
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||
|
||||
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(_request())
|
||||
|
||||
assert [candidate.title for candidate in candidates] == ["Fascia pediatrica"]
|
||||
|
||||
|
||||
def test_pi_restructurer_restores_one_unique_markdown_source_line(tmp_path, monkeypatch):
|
||||
from tht.evidence import authoring
|
||||
|
||||
candidate = _candidate() | {
|
||||
"supporting_excerpts": ["Pazienti sotto i 18 anni sono pediatrici."],
|
||||
}
|
||||
|
||||
def run(argv, **kwargs):
|
||||
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
|
||||
return SimpleNamespace(returncode=0)
|
||||
|
||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||
skill_path = tmp_path / "skill.md"
|
||||
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||
request = _request().model_copy(update={
|
||||
"normalized_text": "- **Pazienti** sotto i 18 anni sono pediatrici.\n",
|
||||
})
|
||||
|
||||
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
|
||||
|
||||
assert candidates[0].supporting_excerpts == (
|
||||
"- **Pazienti** sotto i 18 anni sono pediatrici.",
|
||||
)
|
||||
|
||||
|
||||
def test_pi_restructurer_flags_one_unique_fuzzy_source_line_for_review(tmp_path, monkeypatch):
|
||||
from tht.evidence import authoring
|
||||
|
||||
candidate = _candidate() | {
|
||||
"supporting_excerpts": ["Pazienti sotto 18 anni sono pediatrici."],
|
||||
}
|
||||
|
||||
def run(argv, **kwargs):
|
||||
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
|
||||
return SimpleNamespace(returncode=0)
|
||||
|
||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||
skill_path = tmp_path / "skill.md"
|
||||
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||
request = _request().model_copy(update={
|
||||
"normalized_text": "Pazienti sotto i 18 anni sono pediatrici.\nAdulti sopra i 65 anni.\n",
|
||||
})
|
||||
|
||||
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
|
||||
|
||||
assert candidates[0].supporting_excerpts == (
|
||||
"Pazienti sotto i 18 anni sono pediatrici.",
|
||||
)
|
||||
assert [item.code for item in candidates[0].review_items] == [
|
||||
"supporting_excerpt_reconciled",
|
||||
]
|
||||
|
||||
|
||||
def test_pi_restructurer_restores_one_contiguous_multiline_source_excerpt(
|
||||
tmp_path, monkeypatch,
|
||||
):
|
||||
from tht.evidence import authoring
|
||||
|
||||
candidate = _candidate() | {
|
||||
"supporting_excerpts": [
|
||||
(
|
||||
"pivot/denormalization (solo per le FACT): legge le righe correlate "
|
||||
"nella tabella source_table, estrae source_column usando le join_keys."
|
||||
)
|
||||
],
|
||||
}
|
||||
|
||||
def run(argv, **kwargs):
|
||||
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
|
||||
return SimpleNamespace(returncode=0)
|
||||
|
||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||
skill_path = tmp_path / "skill.md"
|
||||
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||
exact = (
|
||||
"- **pivot/denormalization** (solo per le FACT):\n"
|
||||
" - legge le righe correlate nella tabella `source_table`,\n"
|
||||
" - estrae `source_column` usando le `join_keys`."
|
||||
)
|
||||
request = _request().model_copy(update={"normalized_text": exact + "\n"})
|
||||
|
||||
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
|
||||
|
||||
assert candidates[0].supporting_excerpts == (exact,)
|
||||
|
||||
@@ -117,9 +117,26 @@ def prepare_cmd(
|
||||
skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
||||
restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path)
|
||||
try:
|
||||
report = prepare_workspace_evidence(root, restructurer=restructurer, upgrade=upgrade)
|
||||
try:
|
||||
max_workers = int(os.environ.get("THT_EVIDENCE_AUTHORING_WORKERS", "1"))
|
||||
except ValueError as error:
|
||||
raise EvidencePreparationError("authoring_workers_invalid") from error
|
||||
report = prepare_workspace_evidence(
|
||||
root,
|
||||
restructurer=restructurer,
|
||||
upgrade=upgrade,
|
||||
max_workers=max_workers,
|
||||
)
|
||||
except EvidencePreparationError as error:
|
||||
_emit({"schemaVersion": 1, "operation": "evidence_prepare", "status": "failed", "code": error.code}, json_output)
|
||||
payload = {
|
||||
"schemaVersion": 1,
|
||||
"operation": "evidence_prepare",
|
||||
"status": "failed",
|
||||
"code": error.code,
|
||||
}
|
||||
if error.source_file is not None:
|
||||
payload["sourceFile"] = error.source_file
|
||||
_emit(payload, json_output)
|
||||
raise typer.Exit(code=1) from error
|
||||
_emit(_preparation_payload(report), json_output)
|
||||
if report.findings:
|
||||
|
||||
@@ -11,7 +11,9 @@ import subprocess
|
||||
import tempfile
|
||||
import unicodedata
|
||||
from collections.abc import Callable
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from dataclasses import dataclass
|
||||
from difflib import SequenceMatcher
|
||||
from pathlib import Path
|
||||
from typing import Literal, Protocol
|
||||
|
||||
@@ -103,6 +105,90 @@ class EvidenceRestructurer(Protocol):
|
||||
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ...
|
||||
|
||||
|
||||
def _excerpt_signature(value: str) -> str:
|
||||
value = "\n".join(
|
||||
re.sub(r"^\s*(?:(?:[-+*>]|#+)\s+)", "", line)
|
||||
for line in value.splitlines()
|
||||
)
|
||||
value = re.sub(r"[*_`]", "", value)
|
||||
return " ".join(value.split())
|
||||
|
||||
|
||||
def _request_specific_constraints(source_text: str) -> str:
|
||||
qualified_column = re.search(
|
||||
r"(?<![A-Za-z0-9_])[A-Za-z_][A-Za-z0-9_]*"
|
||||
r"\.[A-Za-z_][A-Za-z0-9_]*\.[A-Za-z_][A-Za-z0-9_]*"
|
||||
r"(?![A-Za-z0-9_])",
|
||||
source_text,
|
||||
)
|
||||
if qualified_column is None:
|
||||
return (
|
||||
"This request contains no literal schema.table.column identifier, so you "
|
||||
"must not use kind enum or formula. Use domain instead and add a review "
|
||||
"item when a missing qualification prevents the more specific kind."
|
||||
)
|
||||
return "Apply the Evidence authoring response contract exactly."
|
||||
|
||||
|
||||
def _restore_exact_source_excerpts(
|
||||
candidate: RestructureCandidate,
|
||||
source_text: str,
|
||||
) -> RestructureCandidate:
|
||||
source_lines = source_text.splitlines()
|
||||
source_spans: list[str] = []
|
||||
for start, first_line in enumerate(source_lines):
|
||||
if not first_line:
|
||||
continue
|
||||
for width in range(1, 6):
|
||||
selected = source_lines[start:start + width]
|
||||
if len(selected) != width or not selected[-1]:
|
||||
continue
|
||||
span = "\n".join(selected)
|
||||
if len(span) <= 1000:
|
||||
source_spans.append(span)
|
||||
restored: list[str] = []
|
||||
reconciled = False
|
||||
for excerpt in candidate.supporting_excerpts:
|
||||
if excerpt in source_text:
|
||||
restored.append(excerpt)
|
||||
continue
|
||||
signature = _excerpt_signature(excerpt)
|
||||
matches = tuple(span for span in source_spans if _excerpt_signature(span) == signature)
|
||||
if len(matches) == 1:
|
||||
restored.append(matches[0])
|
||||
continue
|
||||
ranked = sorted(
|
||||
(
|
||||
(SequenceMatcher(None, signature, _excerpt_signature(line)).ratio(), line)
|
||||
for line in source_spans
|
||||
),
|
||||
reverse=True,
|
||||
)
|
||||
best_score = ranked[0][0] if ranked else 0.0
|
||||
runner_up_score = ranked[1][0] if len(ranked) > 1 else 0.0
|
||||
if best_score >= 0.88 and best_score - runner_up_score >= 0.08:
|
||||
restored.append(ranked[0][1])
|
||||
reconciled = True
|
||||
else:
|
||||
restored.append(excerpt)
|
||||
review_items = candidate.review_items
|
||||
if reconciled and not any(
|
||||
item.code == "supporting_excerpt_reconciled" for item in review_items
|
||||
):
|
||||
review_items = (*review_items, ReviewItem(
|
||||
code="supporting_excerpt_reconciled",
|
||||
message=(
|
||||
"A model excerpt was reconciled to a unique exact source line; "
|
||||
"confirm that the restored quotation supports this unit."
|
||||
),
|
||||
field="supporting_excerpts",
|
||||
))
|
||||
return candidate.model_copy(update={
|
||||
"supporting_excerpts": tuple(restored),
|
||||
"review_items": review_items,
|
||||
})
|
||||
|
||||
|
||||
class PiEvidenceRestructurer:
|
||||
"""Invoke Pi once, without tools or session state, for one changed source."""
|
||||
|
||||
@@ -111,13 +197,20 @@ class PiEvidenceRestructurer:
|
||||
pi_executable: str,
|
||||
skill_path: Path,
|
||||
*,
|
||||
timeout_seconds: int = 120,
|
||||
timeout_seconds: int = 300,
|
||||
) -> None:
|
||||
self._pi_executable = pi_executable
|
||||
self._skill_path = skill_path
|
||||
self._json_mode_extension = (
|
||||
skill_path.parents[2] / "extensions" / "tht-evidence-json-mode.ts"
|
||||
)
|
||||
self._timeout_seconds = timeout_seconds
|
||||
|
||||
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]:
|
||||
try:
|
||||
system_prompt = self._skill_path.read_text(encoding="utf-8")
|
||||
except (OSError, UnicodeError) as error:
|
||||
raise EvidencePreparationError("pi_skill_invalid", request.source_file) from error
|
||||
with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary:
|
||||
request_path = Path(temporary) / "request.json"
|
||||
request_path.write_text(
|
||||
@@ -132,10 +225,25 @@ class PiEvidenceRestructurer:
|
||||
"--no-tools",
|
||||
"--no-extensions",
|
||||
"--no-context-files",
|
||||
"--skill", str(self._skill_path),
|
||||
f"@{request_path}",
|
||||
"Return only the JSON object required by the Evidence authoring skill.",
|
||||
"--no-skills",
|
||||
"--extension", str(self._json_mode_extension),
|
||||
]
|
||||
for option, environment_name in (
|
||||
("--provider", "PI_PROVIDER"),
|
||||
("--model", "PI_MODEL"),
|
||||
("--thinking", "PI_THINKING"),
|
||||
):
|
||||
value = os.environ.get(environment_name)
|
||||
if value:
|
||||
argv.extend((option, value))
|
||||
argv.extend((
|
||||
"--system-prompt", system_prompt,
|
||||
f"@{request_path}",
|
||||
(
|
||||
"Return only the JSON object required by the Evidence authoring skill. "
|
||||
+ _request_specific_constraints(request.normalized_text)
|
||||
),
|
||||
))
|
||||
try:
|
||||
with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response:
|
||||
result = subprocess.run(
|
||||
@@ -156,9 +264,21 @@ class PiEvidenceRestructurer:
|
||||
raise EvidencePreparationError("pi_restructure_failed", request.source_file)
|
||||
try:
|
||||
raw = json.loads(stdout)
|
||||
if not isinstance(raw, dict) or set(raw) != {"candidates"} or not isinstance(raw["candidates"], list):
|
||||
if isinstance(raw, dict) and set(raw) == {"candidates"}:
|
||||
candidates = raw["candidates"]
|
||||
elif isinstance(raw, list):
|
||||
candidates = raw
|
||||
elif isinstance(raw, dict):
|
||||
candidates = [raw]
|
||||
else:
|
||||
raise ValueError("response shape")
|
||||
return tuple(RestructureCandidate.model_validate(candidate) for candidate in raw["candidates"])
|
||||
if not isinstance(candidates, list):
|
||||
raise TypeError("response shape")
|
||||
parsed = tuple(RestructureCandidate.model_validate(candidate) for candidate in candidates)
|
||||
return tuple(
|
||||
_restore_exact_source_excerpts(candidate, request.normalized_text)
|
||||
for candidate in parsed
|
||||
)
|
||||
except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error:
|
||||
raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error
|
||||
|
||||
@@ -418,6 +538,7 @@ def prepare_workspace_evidence(
|
||||
restructurer: EvidenceRestructurer,
|
||||
git_status: Callable[[Path], tuple[str, ...]] | None = None,
|
||||
upgrade: bool = False,
|
||||
max_workers: int = 1,
|
||||
) -> EvidencePreparationReport:
|
||||
"""Prepare all changed Source Evidence without publishing or committing it.
|
||||
|
||||
@@ -425,6 +546,8 @@ def prepare_workspace_evidence(
|
||||
one call through ``restructurer``; all model results validate before the staged
|
||||
authoring tree replaces the current one.
|
||||
"""
|
||||
if not 1 <= max_workers <= 8:
|
||||
raise EvidencePreparationError("authoring_workers_invalid")
|
||||
workspace_root = workspace_root.resolve()
|
||||
evidence_root = workspace_root / "evidence"
|
||||
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
||||
@@ -454,7 +577,7 @@ def prepare_workspace_evidence(
|
||||
)
|
||||
_preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned)
|
||||
|
||||
reserved_ids = set(documents_by_id)
|
||||
requests: list[tuple[str, RestructureRequest]] = []
|
||||
for source_file in sorted(source_texts):
|
||||
source_text = source_texts[source_file]
|
||||
source_hash = _source_hash(source_text)
|
||||
@@ -476,12 +599,28 @@ def prepare_workspace_evidence(
|
||||
normalized_text=source_text,
|
||||
previous_units=previous,
|
||||
)
|
||||
try:
|
||||
candidates = restructurer.restructure(request)
|
||||
except EvidencePreparationError:
|
||||
raise
|
||||
except Exception as error:
|
||||
raise EvidencePreparationError("restructuring_failed", source_file) from error
|
||||
requests.append((source_file, request))
|
||||
|
||||
candidates_by_source: dict[str, tuple[RestructureCandidate, ...]] = {}
|
||||
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
||||
futures = {
|
||||
source_file: executor.submit(restructurer.restructure, request)
|
||||
for source_file, request in requests
|
||||
}
|
||||
for source_file, _request in requests:
|
||||
try:
|
||||
candidates_by_source[source_file] = futures[source_file].result()
|
||||
except EvidencePreparationError:
|
||||
raise
|
||||
except Exception as error:
|
||||
raise EvidencePreparationError("restructuring_failed", source_file) from error
|
||||
|
||||
reserved_ids = set(documents_by_id)
|
||||
for source_file, request in requests:
|
||||
source_text = source_texts[source_file]
|
||||
source_hash = request.source_sha256
|
||||
previous = request.previous_units
|
||||
candidates = candidates_by_source[source_file]
|
||||
model_calls += 1
|
||||
selected_existing_ids: set[str] = set()
|
||||
generated_ids: list[str] = []
|
||||
|
||||
Reference in New Issue
Block a user