feat(evidence): structure v3 domain rules for review

This commit is contained in:
Codex
2026-08-26 12:38:30 +02:00
parent 9d4f994d3e
commit 9898726069
9 changed files with 254 additions and 24 deletions
+44
View File
@@ -165,6 +165,50 @@ def test_pipeline_embeds_validated_curated_evidence_as_semantic_fragments(tmp_pa
assert vectors.records[0].record.metadata["provenance"]["source_file"] == "source/paziente.md"
def test_pipeline_strips_domain_rule_presentation_before_embedding(tmp_path):
rule = (
"Il dominio Ablazione descrive la procedura; indicazioni principali: fibrillazione "
"atriale; appartiene all'universo procedurale. Fact centrale: clinical.fact_ablazione."
)
evidence = CuratedEvidence.model_validate(
{
"schema_version": 3,
"id": "evidence:dominio-ablazione",
"title": "Dominio Ablazione",
"kind": "domain",
"purposes": ["disambiguation"],
"language": "it",
"provenance": {
"source_file": "source/ablazione.md",
"source_sha256": "sha256:" + "a" * 64,
"supporting_excerpts": ["Il dominio Ablazione descrive la procedura."],
},
"payload": {"rule": rule},
}
)
source_item = SourceObject(
source_id="fs:curated-domain",
uri="file:///safe/curated/domain/dominio-ablazione.md",
fingerprint="sha256:" + "b" * 64,
metadata={"relative_path": "curated/domain/dominio-ablazione.md"},
)
embedder = Embedder()
result = pipeline(
tmp_path,
Source([(source_item, dump_curated_markdown(evidence))]),
embedder=embedder,
vectors=Vectors(),
policy=ChunkPolicy(version="chunk-v1", max_chars=4000),
).run()
embedded = embedder.calls[0]
assert result.status == "succeeded"
assert f"Regola: {rule}" in embedded
assert "tht:raw-rule" not in embedded
assert "### Fact centrale:" not in embedded
def test_pipeline_exposes_atomic_content_review_code_when_candidate_is_blocked(tmp_path):
evidence = CuratedEvidence.model_validate(
{
+38 -3
View File
@@ -367,9 +367,9 @@ def test_migrate_workspace_evidence_rewrites_v1_units_as_v3_without_a_model_call
assert report.unchanged == ()
assert migrated.schema_version == 3
assert migrated.payload.rule == "La fascia pediatrica comprende i minori."
assert "## Regola\n\nLa fascia pediatrica comprende i minori." in curated_path.read_text(
encoding="utf-8",
)
text = curated_path.read_text(encoding="utf-8")
assert "## Regola\n\n<!-- tht:raw-rule:" in text
assert "La fascia pediatrica comprende i minori." in text
assert report.findings == ()
@@ -393,6 +393,41 @@ def test_migrate_workspace_evidence_rewrites_v2_units_as_table_free_v3(tmp_path)
assert not any(line.startswith("|") for line in text.splitlines())
def test_migrate_workspace_evidence_rewrites_legacy_v3_rule_presentation(tmp_path):
source_text = "I pazienti sotto i 18 anni sono pediatrici."
evidence = _evidence(source_text).model_copy(update={"schema_version": 3})
_write_workspace(tmp_path, evidence, source_text)
curated_path = tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md"
legacy_text = curated_path.read_text(encoding="utf-8")
legacy_text = legacy_text.replace(
"## Regola\n\n<!-- tht:raw-rule:",
"## Regola\n\n<!-- tht:legacy-raw-rule:",
1,
)
# Recreate the exact pre-structured v3 body from the canonical value.
start = legacy_text.index("<!-- tht:field:rule -->")
end = legacy_text.index("<!-- /tht:field:rule -->", start)
legacy_rule = (
"<!-- tht:field:rule -->\n"
"## Regola\n\n"
f"{evidence.payload.rule}\n"
)
curated_path.write_text(
legacy_text[:start] + legacy_rule + legacy_text[end:],
encoding="utf-8",
)
report = migrate_workspace_evidence(tmp_path, git_status=lambda _: ())
migrated_text = curated_path.read_text(encoding="utf-8")
assert report.migrated == ("evidence:fascia-pediatrica",)
assert report.unchanged == ()
assert "## Regola\n\n<!-- tht:raw-rule:" in migrated_text
assert load_curated_tree(tmp_path / "evidence" / "curated")[0].payload.rule == (
evidence.payload.rule
)
def test_migrate_workspace_evidence_rejects_dirty_curated_files_in_a_nested_workspace(tmp_path):
subprocess.run(["git", "init", "--quiet", str(tmp_path)], check=True)
workspace_root = tmp_path / "psd-clinical"
+39
View File
@@ -262,6 +262,34 @@ def test_v3_curated_markdown_replaces_frontmatter_tables_with_readable_sections(
assert parsed == evidence
def test_v3_domain_rule_is_structured_without_changing_its_canonical_text(tmp_path):
rule = (
"Il dominio Ablazione descrive la procedura; indicazioni principali: fibrillazione "
"atriale, flutter; appartiene all'universo procedurale. Fact centrale: "
"`clinical.fact_ablazione`, collegata a `clinical.dim_patient`. Granularità: una riga "
"per procedura. Domande tipiche: numero di procedure per anno; pazienti distinti per "
"anno; distribuzione per indicazione."
)
base = _domain_evidence_v3()
evidence = base.model_copy(
update={"payload": base.payload.model_copy(update={"rule": rule})},
)
text = dump_curated_markdown(evidence)
parsed = parse_curated_markdown(
text,
path=tmp_path / "curated" / "domain" / "dominio-ablazione.md",
)
assert "## Regola\n\n<!-- tht:raw-rule:" in text
assert "- Il dominio Ablazione descrive la procedura;" in text
assert "- **indicazioni principali:** fibrillazione atriale, flutter;" in text
assert "### Fact centrale:" in text
assert "### Granularità:" in text
assert "### Domande tipiche:" in text
assert parsed.payload.rule == rule
def test_v3_curated_markdown_renders_enum_values_as_a_list_instead_of_a_table(tmp_path):
evidence = CuratedEvidence.model_validate({
**COMMON,
@@ -294,6 +322,17 @@ def test_v3_curated_markdown_rejects_visible_metadata_that_drifted_from_canonica
parse_curated_markdown(text)
def test_v3_curated_markdown_rejects_rule_presentation_that_drifted_from_raw_text():
text = dump_curated_markdown(_domain_evidence_v3()).replace(
"La fact centrale è",
"La tabella centrale è",
1,
)
with pytest.raises(ValueError, match="not canonical"):
parse_curated_markdown(text)
def test_v2_curated_markdown_renders_domain_content_in_the_markdown_body(tmp_path):
evidence = _domain_evidence_v2()
path = tmp_path / "curated" / "domain" / "dominio-ablazione.md"