feat(evidence): structure v3 domain rules for review

This commit is contained in:
Codex
2026-08-26 12:38:30 +02:00
parent 9d4f994d3e
commit 9898726069
9 changed files with 254 additions and 24 deletions
+2
View File
@@ -307,6 +307,8 @@ machine contract. Technical metadata belongs in progressive disclosure, not abov
- **Do** respect `prefers-reduced-motion` while preserving immediate non-kinetic feedback.
- **Do** use English for interface chrome and the workspace language for persisted document content.
- **Do** render curated metadata and scope as Markdown prose or lists, never as a frontmatter table.
- **Do** break long curated rules into paragraphs, labelled subsections, and lists at existing
punctuation boundaries while preserving the exact canonical text for machines.
### Don't:
+7 -5
View File
@@ -25,11 +25,13 @@ review gates and keeps the live transcript in memory. See
The evidence restructuring and PSD migration completed real acceptance on 2026-08-25.
- The curated PSD revision contains 35 approved Evidence units and 60 review items.
- The PSD authoring repository published all 35 units using Curated unit schema v2. A schema v3
table-free presentation is now available in the authoring flow: hidden canonical metadata,
wrapping Markdown scope lists, list-based enum values, and collapsed technical provenance.
`tht evidence migrate <workspace-root>` performs the deterministic v1/v2-to-v3 rewrite without
model calls. The 35-unit PSD v3 migration is currently local and pending commit/publication.
- The PSD authoring repository publishes all 35 units using Curated unit schema v3. Its table-free
presentation uses hidden canonical metadata, wrapping Markdown scope lists, list-based enum
values, and collapsed technical provenance. Long domain rules now have a deterministic
human-readable presentation while retaining their exact canonical text for vector ingestion.
`tht evidence migrate <workspace-root>` performs the deterministic v1/v2 upgrade and older-v3
presentation rewrite without model calls. The structured-rule PSD rewrite is currently local and
pending commit/publication.
- The accepted snapshot is
`psd-clinical-675990d90eae51da6f2bd51b1ae2609f245772ef-snapshot`.
- The active generation is `gen:f968b3bd7a553dbfef3cf47093698f2bc7f95f11`.
+7 -2
View File
@@ -59,8 +59,13 @@ the complete review surface as deterministic Markdown. It uses headings, paragra
lists, fenced SQL, blockquotes, and a collapsed technical-details block. It never emits YAML
frontmatter or Markdown tables. Invisible `tht:` comments delimit typed fields. Parsers must reject
missing, duplicate, unknown, desynchronized, or unstructured body content; they must never silently
ignore it. Newly prepared units use v3. `tht evidence migrate <workspace-root>` upgrades existing
v1 and v2 units locally without a model call, commit, publication, or semantic change.
ignore it. Domain rules also retain their exact canonical text in an invisible `tht:raw-rule`
comment while presenting long prose as paragraphs, labelled subsections, and semicolon-derived
lists. Runtime chunking reads the parsed canonical rule, not this review-only presentation.
Newly prepared units use v3. `tht evidence migrate <workspace-root>` upgrades v1 and v2 units and
canonicalizes an older v3 presentation locally without a model call, commit, publication, or
semantic change.
### Example: filesystem
+7 -1
View File
@@ -52,7 +52,13 @@ Canonical Curated Evidence v3 hides canonical machine metadata in an HTML commen
whole review surface as real Markdown. GitHub therefore shows no frontmatter table. The body layout
is deterministic for each Evidence kind: prose uses sections and paragraphs, scopes and enum values
use wrapping lists, formulas use fenced SQL, supporting excerpts use blockquotes, and unresolved
review items use dedicated blocks.
review items use dedicated blocks. Long domain rules are split into readable paragraphs, labelled
subsections, and lists at existing semicolon boundaries. Their exact original text remains canonical
in an invisible marker, so the formatting cannot change their meaning or bytes.
Preprocessing parses the unit first and builds semantic chunks from the typed payload. The vector
store therefore receives the original rule text and not headings, list markers, or invisible
presentation metadata.
```markdown
<!-- tht:metadata:<canonical metadata> -->
+44
View File
@@ -165,6 +165,50 @@ def test_pipeline_embeds_validated_curated_evidence_as_semantic_fragments(tmp_pa
assert vectors.records[0].record.metadata["provenance"]["source_file"] == "source/paziente.md"
def test_pipeline_strips_domain_rule_presentation_before_embedding(tmp_path):
rule = (
"Il dominio Ablazione descrive la procedura; indicazioni principali: fibrillazione "
"atriale; appartiene all'universo procedurale. Fact centrale: clinical.fact_ablazione."
)
evidence = CuratedEvidence.model_validate(
{
"schema_version": 3,
"id": "evidence:dominio-ablazione",
"title": "Dominio Ablazione",
"kind": "domain",
"purposes": ["disambiguation"],
"language": "it",
"provenance": {
"source_file": "source/ablazione.md",
"source_sha256": "sha256:" + "a" * 64,
"supporting_excerpts": ["Il dominio Ablazione descrive la procedura."],
},
"payload": {"rule": rule},
}
)
source_item = SourceObject(
source_id="fs:curated-domain",
uri="file:///safe/curated/domain/dominio-ablazione.md",
fingerprint="sha256:" + "b" * 64,
metadata={"relative_path": "curated/domain/dominio-ablazione.md"},
)
embedder = Embedder()
result = pipeline(
tmp_path,
Source([(source_item, dump_curated_markdown(evidence))]),
embedder=embedder,
vectors=Vectors(),
policy=ChunkPolicy(version="chunk-v1", max_chars=4000),
).run()
embedded = embedder.calls[0]
assert result.status == "succeeded"
assert f"Regola: {rule}" in embedded
assert "tht:raw-rule" not in embedded
assert "### Fact centrale:" not in embedded
def test_pipeline_exposes_atomic_content_review_code_when_candidate_is_blocked(tmp_path):
evidence = CuratedEvidence.model_validate(
{
+38 -3
View File
@@ -367,9 +367,9 @@ def test_migrate_workspace_evidence_rewrites_v1_units_as_v3_without_a_model_call
assert report.unchanged == ()
assert migrated.schema_version == 3
assert migrated.payload.rule == "La fascia pediatrica comprende i minori."
assert "## Regola\n\nLa fascia pediatrica comprende i minori." in curated_path.read_text(
encoding="utf-8",
)
text = curated_path.read_text(encoding="utf-8")
assert "## Regola\n\n<!-- tht:raw-rule:" in text
assert "La fascia pediatrica comprende i minori." in text
assert report.findings == ()
@@ -393,6 +393,41 @@ def test_migrate_workspace_evidence_rewrites_v2_units_as_table_free_v3(tmp_path)
assert not any(line.startswith("|") for line in text.splitlines())
def test_migrate_workspace_evidence_rewrites_legacy_v3_rule_presentation(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
evidence = _evidence(source_text).model_copy(update={"schema_version": 3})
_write_workspace(tmp_path, evidence, source_text)
curated_path = tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md"
legacy_text = curated_path.read_text(encoding="utf-8")
legacy_text = legacy_text.replace(
"## Regola\n\n<!-- tht:raw-rule:",
"## Regola\n\n<!-- tht:legacy-raw-rule:",
1,
)
# Recreate the exact pre-structured v3 body from the canonical value.
start = legacy_text.index("<!-- tht:field:rule -->")
end = legacy_text.index("<!-- /tht:field:rule -->", start)
legacy_rule = (
"<!-- tht:field:rule -->\n"
"## Regola\n\n"
f"{evidence.payload.rule}\n"
)
curated_path.write_text(
legacy_text[:start] + legacy_rule + legacy_text[end:],
encoding="utf-8",
)
report = migrate_workspace_evidence(tmp_path, git_status=lambda _: ())
migrated_text = curated_path.read_text(encoding="utf-8")
assert report.migrated == ("evidence:fascia-pediatrica",)
assert report.unchanged == ()
assert "## Regola\n\n<!-- tht:raw-rule:" in migrated_text
assert load_curated_tree(tmp_path / "evidence" / "curated")[0].payload.rule == (
evidence.payload.rule
)
def test_migrate_workspace_evidence_rejects_dirty_curated_files_in_a_nested_workspace(tmp_path):
subprocess.run(["git", "init", "--quiet", str(tmp_path)], check=True)
workspace_root = tmp_path / "psd-clinical"
+39
View File
@@ -262,6 +262,34 @@ def test_v3_curated_markdown_replaces_frontmatter_tables_with_readable_sections(
assert parsed == evidence
def test_v3_domain_rule_is_structured_without_changing_its_canonical_text(tmp_path):
rule = (
"Il dominio Ablazione descrive la procedura; indicazioni principali: fibrillazione "
"atriale, flutter; appartiene all'universo procedurale. Fact centrale: "
"`clinical.fact_ablazione`, collegata a `clinical.dim_patient`. Granularità: una riga "
"per procedura. Domande tipiche: numero di procedure per anno; pazienti distinti per "
"anno; distribuzione per indicazione."
)
base = _domain_evidence_v3()
evidence = base.model_copy(
update={"payload": base.payload.model_copy(update={"rule": rule})},
)
text = dump_curated_markdown(evidence)
parsed = parse_curated_markdown(
text,
path=tmp_path / "curated" / "domain" / "dominio-ablazione.md",
)
assert "## Regola\n\n<!-- tht:raw-rule:" in text
assert "- Il dominio Ablazione descrive la procedura;" in text
assert "- **indicazioni principali:** fibrillazione atriale, flutter;" in text
assert "### Fact centrale:" in text
assert "### Granularità:" in text
assert "### Domande tipiche:" in text
assert parsed.payload.rule == rule
def test_v3_curated_markdown_renders_enum_values_as_a_list_instead_of_a_table(tmp_path):
evidence = CuratedEvidence.model_validate({
**COMMON,
@@ -294,6 +322,17 @@ def test_v3_curated_markdown_rejects_visible_metadata_that_drifted_from_canonica
parse_curated_markdown(text)
def test_v3_curated_markdown_rejects_rule_presentation_that_drifted_from_raw_text():
text = dump_curated_markdown(_domain_evidence_v3()).replace(
"La fact centrale è",
"La tabella centrale è",
1,
)
with pytest.raises(ValueError, match="not canonical"):
parse_curated_markdown(text)
def test_v2_curated_markdown_renders_domain_content_in_the_markdown_body(tmp_path):
evidence = _domain_evidence_v2()
path = tmp_path / "curated" / "domain" / "dominio-ablazione.md"
+19 -10
View File
@@ -707,22 +707,31 @@ def migrate_workspace_evidence(
documents_by_id = {document.id: document for document in documents}
if len(documents_by_id) != len(documents):
raise EvidencePreparationError("duplicate_evidence_id")
migrated = tuple(sorted(
document.id for document in documents if document.schema_version in {1, 2}
))
unchanged = tuple(sorted(
document.id for document in documents if document.schema_version == 3
))
upgraded = {
evidence_id: document.model_copy(update={"schema_version": 3})
for evidence_id, document in documents_by_id.items()
}
migrated_ids: list[str] = []
unchanged_ids: list[str] = []
curated_root = evidence_root / "curated"
for evidence_id, document in upgraded.items():
path = curated_root / document.kind / f"{evidence_id.removeprefix('evidence:')}.md"
try:
presentation_is_current = path.read_text(encoding="utf-8") == dump_curated_markdown(
document,
)
except (OSError, UnicodeDecodeError):
presentation_is_current = False
target = unchanged_ids if presentation_is_current else migrated_ids
target.append(evidence_id)
migrated = tuple(sorted(migrated_ids))
unchanged = tuple(sorted(unchanged_ids))
if not migrated:
return EvidenceMigrationReport(
migrated=(),
unchanged=unchanged,
findings=validate_workspace_evidence(workspace_root).findings,
)
upgraded = {
evidence_id: document.model_copy(update={"schema_version": 3})
for evidence_id, document in documents_by_id.items()
}
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
return EvidenceMigrationReport(
migrated=migrated,
+91 -3
View File
@@ -319,6 +319,10 @@ _V2_EMPTY_LIST = "<!-- tht:empty-list -->"
_V2_REVIEW_SEPARATOR = "<!-- tht:review-separator -->"
_V2_REVIEW_FIELD = "<!-- tht:review-field -->"
_V3_METADATA = re.compile(r"\A<!-- tht:metadata:([A-Za-z0-9+/=]+) -->\n")
_V3_RAW_RULE = re.compile(
r"\A<!-- tht:raw-rule:([A-Za-z0-9+/=]+) -->\n\n(.+)\Z",
re.DOTALL,
)
_V3_KIND_LABELS = {
"en": {
"glossary": "Glossary",
@@ -507,6 +511,72 @@ def _parse_v3_values(value: str, labels: dict[str, str]) -> dict[str, str]:
return {key: "\n".join(lines) for key, lines in parsed.items()}
def _leading_rule_label(value: str) -> tuple[str, str] | None:
match = re.match(r"\A([^:;\n]{2,80}:)\s+(.+)\Z", value, re.DOTALL)
if match is None:
return None
label = match.group(1)
if len(label.removesuffix(":").split()) > 10:
return None
if label.lower() in {"http:", "https:"}:
return None
return label, match.group(2)
def _render_v3_rule_detail(value: str) -> str:
clauses = re.split(r"(?<=;)\s+", value)
if len(clauses) < 3:
return value
rendered: list[str] = []
for clause in clauses:
labelled = _leading_rule_label(clause)
if labelled is None:
rendered.append(f"- {clause}")
else:
label, detail = labelled
rendered.append(f"- **{label}** {detail}")
return "\n".join(rendered)
def _render_v3_rule_presentation(value: str) -> str:
if re.search(r"(?m)^\s*(?:[-*+] |#{1,6} |```|>)", value):
return value
rendered: list[str] = []
for paragraph in re.split(r"\n\s*\n", value):
sentences = re.split(
r"(?<=[.!?])\s+(?=[A-ZÀ-ÖØ-Þ0-9(`])",
paragraph,
)
for sentence in sentences:
labelled = _leading_rule_label(sentence)
if labelled is None:
rendered.append(_render_v3_rule_detail(sentence))
continue
label, detail = labelled
rendered.append(f"### {label}\n\n{_render_v3_rule_detail(detail)}")
return "\n\n".join(rendered)
def _render_v3_rule(value: str) -> str:
if "<!-- tht:raw-rule:" in value:
raise ValueError("curated evidence rule contains a reserved marker")
encoded = base64.b64encode(value.encode("utf-8")).decode("ascii")
return (
f"<!-- tht:raw-rule:{encoded} -->\n\n"
f"{_render_v3_rule_presentation(value)}"
)
def _parse_v3_rule(value: str) -> str:
match = _V3_RAW_RULE.fullmatch(value)
if match is None:
return value
try:
return base64.b64decode(match.group(1), validate=True).decode("utf-8")
except (binascii.Error, UnicodeDecodeError, ValueError) as error:
raise ValueError("curated evidence v3 rule metadata is malformed") from error
def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
payload = value.payload
if value.kind == "glossary":
@@ -568,6 +638,12 @@ def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[s
def _render_v3_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
if value.kind == "domain":
return [_render_v2_field(
"rule",
labels["rule"],
_render_v3_rule(value.payload.rule),
)]
if value.kind != "enum":
return _render_v2_payload(value, labels)
payload = value.payload
@@ -705,14 +781,22 @@ def _render_v3_metadata(value: CuratedEvidence) -> str:
return f"<!-- tht:metadata:{encoded} -->"
def _render_v3_body(value: CuratedEvidence) -> str:
def _render_v3_body(value: CuratedEvidence, *, structured_domain_rule: bool = True) -> str:
if "\n" in value.title:
raise ValueError("curated evidence title must be single-line in v3")
labels = _v2_labels(value.language)
fields = [
_render_v3_overview(value, labels),
_render_v3_scope(value, labels),
*_render_v3_payload(value, labels),
*(
_render_v3_payload(value, labels)
if structured_domain_rule
else (
[_render_v2_field("rule", labels["rule"], value.payload.rule)]
if value.kind == "domain"
else _render_v3_payload(value, labels)
)
),
_render_v2_field(
"supporting_excerpts",
labels["supporting_excerpts"],
@@ -827,6 +911,8 @@ def _parse_v2_payload(
def _parse_v3_payload(
kind: str, fields: dict[str, str], labels: dict[str, str],
) -> tuple[dict, set[str]]:
if kind == "domain":
return {"rule": _parse_v3_rule(fields.get("rule", ""))}, {"rule"}
if kind != "enum":
return _parse_v2_payload(kind, fields, labels)
return {
@@ -985,7 +1071,9 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
data["payload"] = data.pop(kind, None)
evidence = CuratedEvidence.model_validate(data)
if evidence.schema_version == 3 and dump_curated_markdown(evidence) != text:
raise ValueError("curated evidence v3 presentation is not canonical")
legacy = f"{_render_v3_metadata(evidence)}\n{_render_v3_body(evidence, structured_domain_rule=False)}"
if legacy != text:
raise ValueError("curated evidence v3 presentation is not canonical")
if path is not None:
_validate_kind_directory(path, evidence.kind)
return evidence