fix(evidence): stabilize real Pi authoring
This commit is contained in:
@@ -0,0 +1,14 @@
|
|||||||
|
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
||||||
|
|
||||||
|
export default function (pi: ExtensionAPI) {
|
||||||
|
pi.on("before_provider_request", (event) => {
|
||||||
|
if (typeof event.payload !== "object" || event.payload === null) {
|
||||||
|
return undefined;
|
||||||
|
}
|
||||||
|
return {
|
||||||
|
...event.payload,
|
||||||
|
temperature: 0,
|
||||||
|
response_format: { type: "json_object" },
|
||||||
|
};
|
||||||
|
});
|
||||||
|
}
|
||||||
@@ -1,7 +1,13 @@
|
|||||||
|
---
|
||||||
|
name: tht-evidence-authoring
|
||||||
|
description: Restructure exactly one normalized Thoth Source Evidence request into strict, typed Evidence candidate JSON for deterministic host-side review.
|
||||||
|
---
|
||||||
|
|
||||||
# Evidence authoring response contract
|
# Evidence authoring response contract
|
||||||
|
|
||||||
You receive exactly one normalized Source Evidence request. Return one JSON object with
|
You receive exactly one normalized Source Evidence request. Return one JSON object with
|
||||||
only a `candidates` array, without Markdown fences or explanatory text.
|
only a `candidates` array, without Markdown fences, comments, or explanatory text. The
|
||||||
|
host strictly rejects unknown or missing fields.
|
||||||
|
|
||||||
Use only facts present in `normalized_text`. Never merge, cite, or infer facts from
|
Use only facts present in `normalized_text`. Never merge, cite, or infer facts from
|
||||||
another source. Reuse an `existing_id` only when it was supplied in `previous_units`;
|
another source. Reuse an `existing_id` only when it was supplied in `previous_units`;
|
||||||
@@ -9,7 +15,78 @@ otherwise omit it. Preserve prior reviewed wording when it is still supported. W
|
|||||||
prior unit is no longer supported, omit its `existing_id` and add a
|
prior unit is no longer supported, omit its `existing_id` and add a
|
||||||
`source_no_longer_supports_unit` review item to the related candidate when applicable.
|
`source_no_longer_supports_unit` review item to the related candidate when applicable.
|
||||||
|
|
||||||
Every candidate must match schema version 1, use one of the eight declared Evidence
|
Each candidate has exactly this shape:
|
||||||
kinds, include one to five short exact excerpts copied from `normalized_text`, and add
|
|
||||||
review items for ambiguities. Do not assign a new canonical ID; the host does that
|
```json
|
||||||
deterministically. Do not use tools or alter any repository state.
|
{
|
||||||
|
"schema_version": 1,
|
||||||
|
"existing_id": "evidence:optional-existing-id",
|
||||||
|
"title": "Short human title",
|
||||||
|
"kind": "glossary",
|
||||||
|
"purposes": ["disambiguation"],
|
||||||
|
"applies_to": {
|
||||||
|
"concepts": ["concept"],
|
||||||
|
"tables": ["schema.table"],
|
||||||
|
"columns": ["schema.table.column"]
|
||||||
|
},
|
||||||
|
"language": "it",
|
||||||
|
"supporting_excerpts": ["One short exact excerpt copied from normalized_text."],
|
||||||
|
"review_items": [
|
||||||
|
{"code": "specific_ambiguity", "message": "What a human must decide.", "field": "payload"}
|
||||||
|
],
|
||||||
|
"payload": {"definition": "Typed payload described below.", "synonyms": [], "variants": []}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Omit `existing_id` for new candidates. `applies_to` must contain only `concepts`,
|
||||||
|
`tables`, and `columns`; use empty arrays when the source does not establish a value.
|
||||||
|
Every table identifier must be `schema.table`, and every column identifier must be
|
||||||
|
`schema.table.column`. Include a table or column identifier only when that fully
|
||||||
|
qualified literal already appears in `normalized_text`. Never qualify an unqualified
|
||||||
|
name yourself. If the source contains only names such as `fact_ilr` or `cod_paz`, leave
|
||||||
|
the corresponding `tables` or `columns` array empty, keep the names in prose, and add a
|
||||||
|
review item when qualification matters. `review_items` may be empty. A review item
|
||||||
|
contains only `code`, `message`, and optional `field`.
|
||||||
|
|
||||||
|
Allowed `purposes` are `disambiguation`, `rewriting`, `schema_linking`, and
|
||||||
|
`sql_generation`. Allowed `kind` values and their exact `payload` shapes are:
|
||||||
|
|
||||||
|
- `glossary`: `{"definition": string, "synonyms": [string], "variants": [string]}`
|
||||||
|
- `domain`: `{"rule": string}`
|
||||||
|
- `enum`: `{"column": "schema.table.column", "values": {"stored value": "meaning"}}`
|
||||||
|
- `example`: `{"question": string, "interpretation": string}`
|
||||||
|
- `mapping`: `{"concept": string, "tables": ["schema.table"], "columns": ["schema.table.column"]}`
|
||||||
|
- `normalization`: `{"input": string, "output": string, "rule": string}`
|
||||||
|
- `formula`: `{"concept": string, "columns": ["schema.table.column"], "sql": "one PostgreSQL expression"}`
|
||||||
|
- `reference`: `{"url": "https://...", "label": string, "description": string}`
|
||||||
|
|
||||||
|
Return exactly one candidate: the source's primary independent, reviewable Evidence
|
||||||
|
Unit. Preserve the source's secondary facts in that unit's typed rule, definition, or
|
||||||
|
interpretation instead of emitting extra candidates; do not atomize individual
|
||||||
|
sentences. The candidate must
|
||||||
|
include one to five nonempty exact `supporting_excerpts`, each at most 1000 characters.
|
||||||
|
Copy each excerpt as one continuous, character-for-character substring of
|
||||||
|
`normalized_text`, including its original Markdown punctuation. Prefer copying one
|
||||||
|
complete source line. Never paraphrase, normalize whitespace, remove backticks, or
|
||||||
|
change quotation marks inside an excerpt.
|
||||||
|
|
||||||
|
Before returning JSON, check every excerpt with the equivalent of
|
||||||
|
`excerpt in normalized_text`; replace any excerpt that would fail with an exact complete
|
||||||
|
line from the source. Also check that there is exactly one candidate, every object
|
||||||
|
has only the declared fields, every kind has the exact payload shape above, and stdout
|
||||||
|
contains only the JSON object. Add a review item whenever the source leaves a material
|
||||||
|
ambiguity; never silently guess a table, column, enum meaning, formula, or URL.
|
||||||
|
|
||||||
|
Use the source path as a kind hint: `00-glossario` normally yields `glossary` or
|
||||||
|
`domain`; `10-domini-clinici` normally yields `domain`; `20-valori-enum` normally yields
|
||||||
|
`enum`; `30-esempi-nlq` normally yields `example`; `40-mapping-semantico` normally yields
|
||||||
|
`mapping` or `domain`; and `50-metadati-normalizzazione` normally yields
|
||||||
|
`normalization`. Depart from the hinted kind only when the source explicitly provides
|
||||||
|
the complete typed payload for another kind. Emit `formula` only when the source states
|
||||||
|
one complete PostgreSQL expression and all referenced columns are fully qualified. If
|
||||||
|
an `enum`, `mapping`, or `formula` payload would require an identifier that is not
|
||||||
|
already fully qualified in the source, emit a `domain` candidate instead and record the
|
||||||
|
missing qualification as a review item.
|
||||||
|
|
||||||
|
Do not assign a new canonical ID; the host does that deterministically. Do not use tools
|
||||||
|
or alter any repository state.
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import hashlib
|
import hashlib
|
||||||
|
import threading
|
||||||
import unicodedata
|
import unicodedata
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
@@ -348,6 +349,40 @@ def test_prepare_changed_source_uses_one_model_call_and_applies_a_valid_batch(tm
|
|||||||
assert validate_workspace_evidence(tmp_path).publishable is True
|
assert validate_workspace_evidence(tmp_path).publishable is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_prepare_can_issue_independent_source_calls_concurrently(tmp_path):
|
||||||
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
source_root = tmp_path / "evidence" / "source" / "domain"
|
||||||
|
changed_text = source_text + "\nRegola revisionata.\n"
|
||||||
|
(source_root / "patient.md").write_text(changed_text, encoding="utf-8")
|
||||||
|
(source_root / "second.md").write_text(changed_text, encoding="utf-8")
|
||||||
|
barrier = threading.Barrier(2)
|
||||||
|
|
||||||
|
class ConcurrentRestructurer:
|
||||||
|
def __init__(self):
|
||||||
|
self.requests = []
|
||||||
|
|
||||||
|
def restructure(self, request):
|
||||||
|
self.requests.append(request)
|
||||||
|
barrier.wait(timeout=2)
|
||||||
|
return (_candidate(title=f"Unit {request.source_file}"),)
|
||||||
|
|
||||||
|
restructurer = ConcurrentRestructurer()
|
||||||
|
|
||||||
|
report = prepare_workspace_evidence(
|
||||||
|
tmp_path,
|
||||||
|
restructurer=restructurer,
|
||||||
|
git_status=lambda _: (),
|
||||||
|
max_workers=2,
|
||||||
|
)
|
||||||
|
|
||||||
|
assert report.model_calls == 2
|
||||||
|
assert sorted(request.source_file for request in restructurer.requests) == [
|
||||||
|
"source/domain/patient.md",
|
||||||
|
"source/domain/second.md",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
def test_prepare_unchanged_source_skips_model_and_does_not_write(tmp_path):
|
def test_prepare_unchanged_source_skips_model_and_does_not_write(tmp_path):
|
||||||
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||||
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
_write_workspace(tmp_path, _evidence(source_text), source_text)
|
||||||
|
|||||||
@@ -14,6 +14,32 @@ def test_evidence_authoring_commands_are_distinct_from_runtime_preprocessing():
|
|||||||
assert "validate" in result.output
|
assert "validate" in result.output
|
||||||
|
|
||||||
|
|
||||||
|
def test_evidence_prepare_failure_identifies_the_source_file(monkeypatch, tmp_path):
|
||||||
|
from tht.cli import evidence_cmd
|
||||||
|
from tht.evidence import EvidencePreparationError
|
||||||
|
|
||||||
|
monkeypatch.setattr(evidence_cmd, "_canonical_worktree", lambda root: root)
|
||||||
|
monkeypatch.setattr(
|
||||||
|
evidence_cmd,
|
||||||
|
"prepare_workspace_evidence",
|
||||||
|
lambda *args, **kwargs: (_ for _ in ()).throw(
|
||||||
|
EvidencePreparationError("pi_restructure_invalid", "source/domain/patient.md")
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
result = CliRunner().invoke(app, ["evidence", "prepare", str(tmp_path), "--json"])
|
||||||
|
|
||||||
|
assert result.exit_code == 1
|
||||||
|
assert result.stderr == ""
|
||||||
|
assert json.loads(result.stdout) == {
|
||||||
|
"code": "pi_restructure_invalid",
|
||||||
|
"operation": "evidence_prepare",
|
||||||
|
"schemaVersion": 1,
|
||||||
|
"sourceFile": "source/domain/patient.md",
|
||||||
|
"status": "failed",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def test_evidence_resolve_json_is_pristine(monkeypatch, tmp_path):
|
def test_evidence_resolve_json_is_pristine(monkeypatch, tmp_path):
|
||||||
from tht.cli import evidence_cmd
|
from tht.cli import evidence_cmd
|
||||||
from tht.evidence.authoring import EvidenceResolutionReport
|
from tht.evidence.authoring import EvidenceResolutionReport
|
||||||
|
|||||||
@@ -1,4 +1,5 @@
|
|||||||
import json
|
import json
|
||||||
|
from pathlib import Path
|
||||||
from types import SimpleNamespace
|
from types import SimpleNamespace
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
@@ -35,6 +36,9 @@ def _candidate():
|
|||||||
def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeypatch):
|
def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeypatch):
|
||||||
from tht.evidence import authoring
|
from tht.evidence import authoring
|
||||||
|
|
||||||
|
monkeypatch.setenv("PI_PROVIDER", "zai")
|
||||||
|
monkeypatch.setenv("PI_MODEL", "glm-5.2")
|
||||||
|
monkeypatch.setenv("PI_THINKING", "medium")
|
||||||
calls = []
|
calls = []
|
||||||
|
|
||||||
def run(argv, **kwargs):
|
def run(argv, **kwargs):
|
||||||
@@ -43,7 +47,9 @@ def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeyp
|
|||||||
return SimpleNamespace(returncode=0)
|
return SimpleNamespace(returncode=0)
|
||||||
|
|
||||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||||
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md", timeout_seconds=12)
|
skill_path = tmp_path / "skill.md"
|
||||||
|
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||||
|
restructurer = PiEvidenceRestructurer("pi-test", skill_path, timeout_seconds=12)
|
||||||
|
|
||||||
candidates = restructurer.restructure(_request())
|
candidates = restructurer.restructure(_request())
|
||||||
|
|
||||||
@@ -52,7 +58,22 @@ def test_pi_restructurer_uses_an_ephemeral_no_tools_invocation(tmp_path, monkeyp
|
|||||||
assert argv[:8] == [
|
assert argv[:8] == [
|
||||||
"pi-test", "--mode", "text", "--print", "--no-session", "--no-tools", "--no-extensions", "--no-context-files",
|
"pi-test", "--mode", "text", "--print", "--no-session", "--no-tools", "--no-extensions", "--no-context-files",
|
||||||
]
|
]
|
||||||
assert "--skill" in argv
|
assert "--no-skills" in argv
|
||||||
|
assert argv[argv.index("--extension") + 1].endswith(
|
||||||
|
"/extensions/tht-evidence-json-mode.ts"
|
||||||
|
)
|
||||||
|
assert "--system-prompt" in argv
|
||||||
|
assert argv[argv.index("--system-prompt") + 1] == "Evidence-only system prompt"
|
||||||
|
assert "--skill" not in argv
|
||||||
|
assert "--append-system-prompt" not in argv
|
||||||
|
assert "/skill:tht-evidence-authoring" not in argv
|
||||||
|
assert argv[argv.index("--provider") + 1] == "zai"
|
||||||
|
assert argv[argv.index("--model") + 1] == "glm-5.2"
|
||||||
|
assert argv[argv.index("--thinking") + 1] == "medium"
|
||||||
|
assert any(
|
||||||
|
"must not use kind enum or formula" in argument
|
||||||
|
for argument in argv
|
||||||
|
)
|
||||||
assert any(argument.startswith("@") for argument in argv)
|
assert any(argument.startswith("@") for argument in argv)
|
||||||
assert kwargs["timeout"] == 12
|
assert kwargs["timeout"] == 12
|
||||||
assert kwargs["shell"] is False
|
assert kwargs["shell"] is False
|
||||||
@@ -71,10 +92,144 @@ def test_pi_restructurer_returns_a_bounded_error_without_model_output(tmp_path,
|
|||||||
return result
|
return result
|
||||||
|
|
||||||
monkeypatch.setattr(authoring.subprocess, "run", run)
|
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||||
restructurer = PiEvidenceRestructurer("pi-test", tmp_path / "skill.md")
|
skill_path = tmp_path / "skill.md"
|
||||||
|
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||||
|
restructurer = PiEvidenceRestructurer("pi-test", skill_path)
|
||||||
|
|
||||||
with pytest.raises(EvidencePreparationError) as error:
|
with pytest.raises(EvidencePreparationError) as error:
|
||||||
restructurer.restructure(_request())
|
restructurer.restructure(_request())
|
||||||
|
|
||||||
assert error.value.code in {"pi_restructure_failed", "pi_restructure_invalid"}
|
assert error.value.code in {"pi_restructure_failed", "pi_restructure_invalid"}
|
||||||
assert "secret" not in str(error.value)
|
assert "secret" not in str(error.value)
|
||||||
|
|
||||||
|
|
||||||
|
def test_evidence_authoring_skill_is_loadable_and_declares_the_wire_schema():
|
||||||
|
skill = (
|
||||||
|
Path(__file__).parents[1] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
||||||
|
).read_text(encoding="utf-8")
|
||||||
|
|
||||||
|
assert skill.startswith("---\nname: tht-evidence-authoring\ndescription:")
|
||||||
|
for required_field in (
|
||||||
|
"schema_version",
|
||||||
|
"existing_id",
|
||||||
|
"supporting_excerpts",
|
||||||
|
"review_items",
|
||||||
|
"payload",
|
||||||
|
):
|
||||||
|
assert required_field in skill
|
||||||
|
|
||||||
|
|
||||||
|
def test_evidence_authoring_json_mode_extension_only_rewrites_the_provider_payload():
|
||||||
|
extension = (
|
||||||
|
Path(__file__).parents[1] / ".pi" / "extensions" / "tht-evidence-json-mode.ts"
|
||||||
|
).read_text(encoding="utf-8")
|
||||||
|
|
||||||
|
assert 'pi.on("before_provider_request"' in extension
|
||||||
|
assert 'response_format: { type: "json_object" }' in extension
|
||||||
|
assert "temperature: 0" in extension
|
||||||
|
assert "registerTool" not in extension
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("response", [_candidate(), [_candidate()]])
|
||||||
|
def test_pi_restructurer_normalizes_bounded_candidate_envelopes(
|
||||||
|
tmp_path, monkeypatch, response,
|
||||||
|
):
|
||||||
|
from tht.evidence import authoring
|
||||||
|
|
||||||
|
def run(argv, **kwargs):
|
||||||
|
kwargs["stdout"].write(json.dumps(response))
|
||||||
|
return SimpleNamespace(returncode=0)
|
||||||
|
|
||||||
|
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||||
|
skill_path = tmp_path / "skill.md"
|
||||||
|
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||||
|
|
||||||
|
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(_request())
|
||||||
|
|
||||||
|
assert [candidate.title for candidate in candidates] == ["Fascia pediatrica"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_pi_restructurer_restores_one_unique_markdown_source_line(tmp_path, monkeypatch):
|
||||||
|
from tht.evidence import authoring
|
||||||
|
|
||||||
|
candidate = _candidate() | {
|
||||||
|
"supporting_excerpts": ["Pazienti sotto i 18 anni sono pediatrici."],
|
||||||
|
}
|
||||||
|
|
||||||
|
def run(argv, **kwargs):
|
||||||
|
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
|
||||||
|
return SimpleNamespace(returncode=0)
|
||||||
|
|
||||||
|
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||||
|
skill_path = tmp_path / "skill.md"
|
||||||
|
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||||
|
request = _request().model_copy(update={
|
||||||
|
"normalized_text": "- **Pazienti** sotto i 18 anni sono pediatrici.\n",
|
||||||
|
})
|
||||||
|
|
||||||
|
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
|
||||||
|
|
||||||
|
assert candidates[0].supporting_excerpts == (
|
||||||
|
"- **Pazienti** sotto i 18 anni sono pediatrici.",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_pi_restructurer_flags_one_unique_fuzzy_source_line_for_review(tmp_path, monkeypatch):
|
||||||
|
from tht.evidence import authoring
|
||||||
|
|
||||||
|
candidate = _candidate() | {
|
||||||
|
"supporting_excerpts": ["Pazienti sotto 18 anni sono pediatrici."],
|
||||||
|
}
|
||||||
|
|
||||||
|
def run(argv, **kwargs):
|
||||||
|
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
|
||||||
|
return SimpleNamespace(returncode=0)
|
||||||
|
|
||||||
|
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||||
|
skill_path = tmp_path / "skill.md"
|
||||||
|
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||||
|
request = _request().model_copy(update={
|
||||||
|
"normalized_text": "Pazienti sotto i 18 anni sono pediatrici.\nAdulti sopra i 65 anni.\n",
|
||||||
|
})
|
||||||
|
|
||||||
|
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
|
||||||
|
|
||||||
|
assert candidates[0].supporting_excerpts == (
|
||||||
|
"Pazienti sotto i 18 anni sono pediatrici.",
|
||||||
|
)
|
||||||
|
assert [item.code for item in candidates[0].review_items] == [
|
||||||
|
"supporting_excerpt_reconciled",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def test_pi_restructurer_restores_one_contiguous_multiline_source_excerpt(
|
||||||
|
tmp_path, monkeypatch,
|
||||||
|
):
|
||||||
|
from tht.evidence import authoring
|
||||||
|
|
||||||
|
candidate = _candidate() | {
|
||||||
|
"supporting_excerpts": [
|
||||||
|
(
|
||||||
|
"pivot/denormalization (solo per le FACT): legge le righe correlate "
|
||||||
|
"nella tabella source_table, estrae source_column usando le join_keys."
|
||||||
|
)
|
||||||
|
],
|
||||||
|
}
|
||||||
|
|
||||||
|
def run(argv, **kwargs):
|
||||||
|
kwargs["stdout"].write(json.dumps({"candidates": [candidate]}))
|
||||||
|
return SimpleNamespace(returncode=0)
|
||||||
|
|
||||||
|
monkeypatch.setattr(authoring.subprocess, "run", run)
|
||||||
|
skill_path = tmp_path / "skill.md"
|
||||||
|
skill_path.write_text("Evidence-only system prompt", encoding="utf-8")
|
||||||
|
exact = (
|
||||||
|
"- **pivot/denormalization** (solo per le FACT):\n"
|
||||||
|
" - legge le righe correlate nella tabella `source_table`,\n"
|
||||||
|
" - estrae `source_column` usando le `join_keys`."
|
||||||
|
)
|
||||||
|
request = _request().model_copy(update={"normalized_text": exact + "\n"})
|
||||||
|
|
||||||
|
candidates = PiEvidenceRestructurer("pi-test", skill_path).restructure(request)
|
||||||
|
|
||||||
|
assert candidates[0].supporting_excerpts == (exact,)
|
||||||
|
|||||||
@@ -117,9 +117,26 @@ def prepare_cmd(
|
|||||||
skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
skill_path = Path(__file__).resolve().parents[2] / ".pi" / "skills" / "tht-evidence-authoring" / "SKILL.md"
|
||||||
restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path)
|
restructurer = PiEvidenceRestructurer(os.environ.get("THT_PI_EXECUTABLE", "pi"), skill_path)
|
||||||
try:
|
try:
|
||||||
report = prepare_workspace_evidence(root, restructurer=restructurer, upgrade=upgrade)
|
try:
|
||||||
|
max_workers = int(os.environ.get("THT_EVIDENCE_AUTHORING_WORKERS", "1"))
|
||||||
|
except ValueError as error:
|
||||||
|
raise EvidencePreparationError("authoring_workers_invalid") from error
|
||||||
|
report = prepare_workspace_evidence(
|
||||||
|
root,
|
||||||
|
restructurer=restructurer,
|
||||||
|
upgrade=upgrade,
|
||||||
|
max_workers=max_workers,
|
||||||
|
)
|
||||||
except EvidencePreparationError as error:
|
except EvidencePreparationError as error:
|
||||||
_emit({"schemaVersion": 1, "operation": "evidence_prepare", "status": "failed", "code": error.code}, json_output)
|
payload = {
|
||||||
|
"schemaVersion": 1,
|
||||||
|
"operation": "evidence_prepare",
|
||||||
|
"status": "failed",
|
||||||
|
"code": error.code,
|
||||||
|
}
|
||||||
|
if error.source_file is not None:
|
||||||
|
payload["sourceFile"] = error.source_file
|
||||||
|
_emit(payload, json_output)
|
||||||
raise typer.Exit(code=1) from error
|
raise typer.Exit(code=1) from error
|
||||||
_emit(_preparation_payload(report), json_output)
|
_emit(_preparation_payload(report), json_output)
|
||||||
if report.findings:
|
if report.findings:
|
||||||
|
|||||||
@@ -11,7 +11,9 @@ import subprocess
|
|||||||
import tempfile
|
import tempfile
|
||||||
import unicodedata
|
import unicodedata
|
||||||
from collections.abc import Callable
|
from collections.abc import Callable
|
||||||
|
from concurrent.futures import ThreadPoolExecutor
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
|
from difflib import SequenceMatcher
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Literal, Protocol
|
from typing import Literal, Protocol
|
||||||
|
|
||||||
@@ -103,6 +105,90 @@ class EvidenceRestructurer(Protocol):
|
|||||||
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ...
|
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]: ...
|
||||||
|
|
||||||
|
|
||||||
|
def _excerpt_signature(value: str) -> str:
|
||||||
|
value = "\n".join(
|
||||||
|
re.sub(r"^\s*(?:(?:[-+*>]|#+)\s+)", "", line)
|
||||||
|
for line in value.splitlines()
|
||||||
|
)
|
||||||
|
value = re.sub(r"[*_`]", "", value)
|
||||||
|
return " ".join(value.split())
|
||||||
|
|
||||||
|
|
||||||
|
def _request_specific_constraints(source_text: str) -> str:
|
||||||
|
qualified_column = re.search(
|
||||||
|
r"(?<![A-Za-z0-9_])[A-Za-z_][A-Za-z0-9_]*"
|
||||||
|
r"\.[A-Za-z_][A-Za-z0-9_]*\.[A-Za-z_][A-Za-z0-9_]*"
|
||||||
|
r"(?![A-Za-z0-9_])",
|
||||||
|
source_text,
|
||||||
|
)
|
||||||
|
if qualified_column is None:
|
||||||
|
return (
|
||||||
|
"This request contains no literal schema.table.column identifier, so you "
|
||||||
|
"must not use kind enum or formula. Use domain instead and add a review "
|
||||||
|
"item when a missing qualification prevents the more specific kind."
|
||||||
|
)
|
||||||
|
return "Apply the Evidence authoring response contract exactly."
|
||||||
|
|
||||||
|
|
||||||
|
def _restore_exact_source_excerpts(
|
||||||
|
candidate: RestructureCandidate,
|
||||||
|
source_text: str,
|
||||||
|
) -> RestructureCandidate:
|
||||||
|
source_lines = source_text.splitlines()
|
||||||
|
source_spans: list[str] = []
|
||||||
|
for start, first_line in enumerate(source_lines):
|
||||||
|
if not first_line:
|
||||||
|
continue
|
||||||
|
for width in range(1, 6):
|
||||||
|
selected = source_lines[start:start + width]
|
||||||
|
if len(selected) != width or not selected[-1]:
|
||||||
|
continue
|
||||||
|
span = "\n".join(selected)
|
||||||
|
if len(span) <= 1000:
|
||||||
|
source_spans.append(span)
|
||||||
|
restored: list[str] = []
|
||||||
|
reconciled = False
|
||||||
|
for excerpt in candidate.supporting_excerpts:
|
||||||
|
if excerpt in source_text:
|
||||||
|
restored.append(excerpt)
|
||||||
|
continue
|
||||||
|
signature = _excerpt_signature(excerpt)
|
||||||
|
matches = tuple(span for span in source_spans if _excerpt_signature(span) == signature)
|
||||||
|
if len(matches) == 1:
|
||||||
|
restored.append(matches[0])
|
||||||
|
continue
|
||||||
|
ranked = sorted(
|
||||||
|
(
|
||||||
|
(SequenceMatcher(None, signature, _excerpt_signature(line)).ratio(), line)
|
||||||
|
for line in source_spans
|
||||||
|
),
|
||||||
|
reverse=True,
|
||||||
|
)
|
||||||
|
best_score = ranked[0][0] if ranked else 0.0
|
||||||
|
runner_up_score = ranked[1][0] if len(ranked) > 1 else 0.0
|
||||||
|
if best_score >= 0.88 and best_score - runner_up_score >= 0.08:
|
||||||
|
restored.append(ranked[0][1])
|
||||||
|
reconciled = True
|
||||||
|
else:
|
||||||
|
restored.append(excerpt)
|
||||||
|
review_items = candidate.review_items
|
||||||
|
if reconciled and not any(
|
||||||
|
item.code == "supporting_excerpt_reconciled" for item in review_items
|
||||||
|
):
|
||||||
|
review_items = (*review_items, ReviewItem(
|
||||||
|
code="supporting_excerpt_reconciled",
|
||||||
|
message=(
|
||||||
|
"A model excerpt was reconciled to a unique exact source line; "
|
||||||
|
"confirm that the restored quotation supports this unit."
|
||||||
|
),
|
||||||
|
field="supporting_excerpts",
|
||||||
|
))
|
||||||
|
return candidate.model_copy(update={
|
||||||
|
"supporting_excerpts": tuple(restored),
|
||||||
|
"review_items": review_items,
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
class PiEvidenceRestructurer:
|
class PiEvidenceRestructurer:
|
||||||
"""Invoke Pi once, without tools or session state, for one changed source."""
|
"""Invoke Pi once, without tools or session state, for one changed source."""
|
||||||
|
|
||||||
@@ -111,13 +197,20 @@ class PiEvidenceRestructurer:
|
|||||||
pi_executable: str,
|
pi_executable: str,
|
||||||
skill_path: Path,
|
skill_path: Path,
|
||||||
*,
|
*,
|
||||||
timeout_seconds: int = 120,
|
timeout_seconds: int = 300,
|
||||||
) -> None:
|
) -> None:
|
||||||
self._pi_executable = pi_executable
|
self._pi_executable = pi_executable
|
||||||
self._skill_path = skill_path
|
self._skill_path = skill_path
|
||||||
|
self._json_mode_extension = (
|
||||||
|
skill_path.parents[2] / "extensions" / "tht-evidence-json-mode.ts"
|
||||||
|
)
|
||||||
self._timeout_seconds = timeout_seconds
|
self._timeout_seconds = timeout_seconds
|
||||||
|
|
||||||
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]:
|
def restructure(self, request: RestructureRequest) -> tuple[RestructureCandidate, ...]:
|
||||||
|
try:
|
||||||
|
system_prompt = self._skill_path.read_text(encoding="utf-8")
|
||||||
|
except (OSError, UnicodeError) as error:
|
||||||
|
raise EvidencePreparationError("pi_skill_invalid", request.source_file) from error
|
||||||
with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary:
|
with tempfile.TemporaryDirectory(prefix="tht-evidence-request-") as temporary:
|
||||||
request_path = Path(temporary) / "request.json"
|
request_path = Path(temporary) / "request.json"
|
||||||
request_path.write_text(
|
request_path.write_text(
|
||||||
@@ -132,10 +225,25 @@ class PiEvidenceRestructurer:
|
|||||||
"--no-tools",
|
"--no-tools",
|
||||||
"--no-extensions",
|
"--no-extensions",
|
||||||
"--no-context-files",
|
"--no-context-files",
|
||||||
"--skill", str(self._skill_path),
|
"--no-skills",
|
||||||
f"@{request_path}",
|
"--extension", str(self._json_mode_extension),
|
||||||
"Return only the JSON object required by the Evidence authoring skill.",
|
|
||||||
]
|
]
|
||||||
|
for option, environment_name in (
|
||||||
|
("--provider", "PI_PROVIDER"),
|
||||||
|
("--model", "PI_MODEL"),
|
||||||
|
("--thinking", "PI_THINKING"),
|
||||||
|
):
|
||||||
|
value = os.environ.get(environment_name)
|
||||||
|
if value:
|
||||||
|
argv.extend((option, value))
|
||||||
|
argv.extend((
|
||||||
|
"--system-prompt", system_prompt,
|
||||||
|
f"@{request_path}",
|
||||||
|
(
|
||||||
|
"Return only the JSON object required by the Evidence authoring skill. "
|
||||||
|
+ _request_specific_constraints(request.normalized_text)
|
||||||
|
),
|
||||||
|
))
|
||||||
try:
|
try:
|
||||||
with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response:
|
with (Path(temporary) / "response.json").open("w+", encoding="utf-8") as response:
|
||||||
result = subprocess.run(
|
result = subprocess.run(
|
||||||
@@ -156,9 +264,21 @@ class PiEvidenceRestructurer:
|
|||||||
raise EvidencePreparationError("pi_restructure_failed", request.source_file)
|
raise EvidencePreparationError("pi_restructure_failed", request.source_file)
|
||||||
try:
|
try:
|
||||||
raw = json.loads(stdout)
|
raw = json.loads(stdout)
|
||||||
if not isinstance(raw, dict) or set(raw) != {"candidates"} or not isinstance(raw["candidates"], list):
|
if isinstance(raw, dict) and set(raw) == {"candidates"}:
|
||||||
|
candidates = raw["candidates"]
|
||||||
|
elif isinstance(raw, list):
|
||||||
|
candidates = raw
|
||||||
|
elif isinstance(raw, dict):
|
||||||
|
candidates = [raw]
|
||||||
|
else:
|
||||||
raise ValueError("response shape")
|
raise ValueError("response shape")
|
||||||
return tuple(RestructureCandidate.model_validate(candidate) for candidate in raw["candidates"])
|
if not isinstance(candidates, list):
|
||||||
|
raise TypeError("response shape")
|
||||||
|
parsed = tuple(RestructureCandidate.model_validate(candidate) for candidate in candidates)
|
||||||
|
return tuple(
|
||||||
|
_restore_exact_source_excerpts(candidate, request.normalized_text)
|
||||||
|
for candidate in parsed
|
||||||
|
)
|
||||||
except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error:
|
except (TypeError, ValueError, ValidationError, json.JSONDecodeError) as error:
|
||||||
raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error
|
raise EvidencePreparationError("pi_restructure_invalid", request.source_file) from error
|
||||||
|
|
||||||
@@ -418,6 +538,7 @@ def prepare_workspace_evidence(
|
|||||||
restructurer: EvidenceRestructurer,
|
restructurer: EvidenceRestructurer,
|
||||||
git_status: Callable[[Path], tuple[str, ...]] | None = None,
|
git_status: Callable[[Path], tuple[str, ...]] | None = None,
|
||||||
upgrade: bool = False,
|
upgrade: bool = False,
|
||||||
|
max_workers: int = 1,
|
||||||
) -> EvidencePreparationReport:
|
) -> EvidencePreparationReport:
|
||||||
"""Prepare all changed Source Evidence without publishing or committing it.
|
"""Prepare all changed Source Evidence without publishing or committing it.
|
||||||
|
|
||||||
@@ -425,6 +546,8 @@ def prepare_workspace_evidence(
|
|||||||
one call through ``restructurer``; all model results validate before the staged
|
one call through ``restructurer``; all model results validate before the staged
|
||||||
authoring tree replaces the current one.
|
authoring tree replaces the current one.
|
||||||
"""
|
"""
|
||||||
|
if not 1 <= max_workers <= 8:
|
||||||
|
raise EvidencePreparationError("authoring_workers_invalid")
|
||||||
workspace_root = workspace_root.resolve()
|
workspace_root = workspace_root.resolve()
|
||||||
evidence_root = workspace_root / "evidence"
|
evidence_root = workspace_root / "evidence"
|
||||||
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
_reject_dirty_authoring_state(workspace_root, git_status or _git_status)
|
||||||
@@ -454,7 +577,7 @@ def prepare_workspace_evidence(
|
|||||||
)
|
)
|
||||||
_preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned)
|
_preserve_removed_sources(source_texts, manifest, documents_by_id, source_units, orphaned)
|
||||||
|
|
||||||
reserved_ids = set(documents_by_id)
|
requests: list[tuple[str, RestructureRequest]] = []
|
||||||
for source_file in sorted(source_texts):
|
for source_file in sorted(source_texts):
|
||||||
source_text = source_texts[source_file]
|
source_text = source_texts[source_file]
|
||||||
source_hash = _source_hash(source_text)
|
source_hash = _source_hash(source_text)
|
||||||
@@ -476,12 +599,28 @@ def prepare_workspace_evidence(
|
|||||||
normalized_text=source_text,
|
normalized_text=source_text,
|
||||||
previous_units=previous,
|
previous_units=previous,
|
||||||
)
|
)
|
||||||
try:
|
requests.append((source_file, request))
|
||||||
candidates = restructurer.restructure(request)
|
|
||||||
except EvidencePreparationError:
|
candidates_by_source: dict[str, tuple[RestructureCandidate, ...]] = {}
|
||||||
raise
|
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
||||||
except Exception as error:
|
futures = {
|
||||||
raise EvidencePreparationError("restructuring_failed", source_file) from error
|
source_file: executor.submit(restructurer.restructure, request)
|
||||||
|
for source_file, request in requests
|
||||||
|
}
|
||||||
|
for source_file, _request in requests:
|
||||||
|
try:
|
||||||
|
candidates_by_source[source_file] = futures[source_file].result()
|
||||||
|
except EvidencePreparationError:
|
||||||
|
raise
|
||||||
|
except Exception as error:
|
||||||
|
raise EvidencePreparationError("restructuring_failed", source_file) from error
|
||||||
|
|
||||||
|
reserved_ids = set(documents_by_id)
|
||||||
|
for source_file, request in requests:
|
||||||
|
source_text = source_texts[source_file]
|
||||||
|
source_hash = request.source_sha256
|
||||||
|
previous = request.previous_units
|
||||||
|
candidates = candidates_by_source[source_file]
|
||||||
model_calls += 1
|
model_calls += 1
|
||||||
selected_existing_ids: set[str] = set()
|
selected_existing_ids: set[str] = set()
|
||||||
generated_ids: list[str] = []
|
generated_ids: list[str] = []
|
||||||
|
|||||||
Reference in New Issue
Block a user