feat(evidence): structure v3 domain rules for review
This commit is contained in:
@@ -307,6 +307,8 @@ machine contract. Technical metadata belongs in progressive disclosure, not abov
|
||||
- **Do** respect `prefers-reduced-motion` while preserving immediate non-kinetic feedback.
|
||||
- **Do** use English for interface chrome and the workspace language for persisted document content.
|
||||
- **Do** render curated metadata and scope as Markdown prose or lists, never as a frontmatter table.
|
||||
- **Do** break long curated rules into paragraphs, labelled subsections, and lists at existing
|
||||
punctuation boundaries while preserving the exact canonical text for machines.
|
||||
|
||||
### Don't:
|
||||
|
||||
|
||||
+7
-5
@@ -25,11 +25,13 @@ review gates and keeps the live transcript in memory. See
|
||||
The evidence restructuring and PSD migration completed real acceptance on 2026-08-25.
|
||||
|
||||
- The curated PSD revision contains 35 approved Evidence units and 60 review items.
|
||||
- The PSD authoring repository published all 35 units using Curated unit schema v2. A schema v3
|
||||
table-free presentation is now available in the authoring flow: hidden canonical metadata,
|
||||
wrapping Markdown scope lists, list-based enum values, and collapsed technical provenance.
|
||||
`tht evidence migrate <workspace-root>` performs the deterministic v1/v2-to-v3 rewrite without
|
||||
model calls. The 35-unit PSD v3 migration is currently local and pending commit/publication.
|
||||
- The PSD authoring repository publishes all 35 units using Curated unit schema v3. Its table-free
|
||||
presentation uses hidden canonical metadata, wrapping Markdown scope lists, list-based enum
|
||||
values, and collapsed technical provenance. Long domain rules now have a deterministic
|
||||
human-readable presentation while retaining their exact canonical text for vector ingestion.
|
||||
`tht evidence migrate <workspace-root>` performs the deterministic v1/v2 upgrade and older-v3
|
||||
presentation rewrite without model calls. The structured-rule PSD rewrite is currently local and
|
||||
pending commit/publication.
|
||||
- The accepted snapshot is
|
||||
`psd-clinical-675990d90eae51da6f2bd51b1ae2609f245772ef-snapshot`.
|
||||
- The active generation is `gen:f968b3bd7a553dbfef3cf47093698f2bc7f95f11`.
|
||||
|
||||
@@ -59,8 +59,13 @@ the complete review surface as deterministic Markdown. It uses headings, paragra
|
||||
lists, fenced SQL, blockquotes, and a collapsed technical-details block. It never emits YAML
|
||||
frontmatter or Markdown tables. Invisible `tht:` comments delimit typed fields. Parsers must reject
|
||||
missing, duplicate, unknown, desynchronized, or unstructured body content; they must never silently
|
||||
ignore it. Newly prepared units use v3. `tht evidence migrate <workspace-root>` upgrades existing
|
||||
v1 and v2 units locally without a model call, commit, publication, or semantic change.
|
||||
ignore it. Domain rules also retain their exact canonical text in an invisible `tht:raw-rule`
|
||||
comment while presenting long prose as paragraphs, labelled subsections, and semicolon-derived
|
||||
lists. Runtime chunking reads the parsed canonical rule, not this review-only presentation.
|
||||
|
||||
Newly prepared units use v3. `tht evidence migrate <workspace-root>` upgrades v1 and v2 units and
|
||||
canonicalizes an older v3 presentation locally without a model call, commit, publication, or
|
||||
semantic change.
|
||||
|
||||
### Example: filesystem
|
||||
|
||||
|
||||
+7
-1
@@ -52,7 +52,13 @@ Canonical Curated Evidence v3 hides canonical machine metadata in an HTML commen
|
||||
whole review surface as real Markdown. GitHub therefore shows no frontmatter table. The body layout
|
||||
is deterministic for each Evidence kind: prose uses sections and paragraphs, scopes and enum values
|
||||
use wrapping lists, formulas use fenced SQL, supporting excerpts use blockquotes, and unresolved
|
||||
review items use dedicated blocks.
|
||||
review items use dedicated blocks. Long domain rules are split into readable paragraphs, labelled
|
||||
subsections, and lists at existing semicolon boundaries. Their exact original text remains canonical
|
||||
in an invisible marker, so the formatting cannot change their meaning or bytes.
|
||||
|
||||
Preprocessing parses the unit first and builds semantic chunks from the typed payload. The vector
|
||||
store therefore receives the original rule text and not headings, list markers, or invisible
|
||||
presentation metadata.
|
||||
|
||||
```markdown
|
||||
<!-- tht:metadata:<canonical metadata> -->
|
||||
|
||||
@@ -165,6 +165,50 @@ def test_pipeline_embeds_validated_curated_evidence_as_semantic_fragments(tmp_pa
|
||||
assert vectors.records[0].record.metadata["provenance"]["source_file"] == "source/paziente.md"
|
||||
|
||||
|
||||
def test_pipeline_strips_domain_rule_presentation_before_embedding(tmp_path):
|
||||
rule = (
|
||||
"Il dominio Ablazione descrive la procedura; indicazioni principali: fibrillazione "
|
||||
"atriale; appartiene all'universo procedurale. Fact centrale: clinical.fact_ablazione."
|
||||
)
|
||||
evidence = CuratedEvidence.model_validate(
|
||||
{
|
||||
"schema_version": 3,
|
||||
"id": "evidence:dominio-ablazione",
|
||||
"title": "Dominio Ablazione",
|
||||
"kind": "domain",
|
||||
"purposes": ["disambiguation"],
|
||||
"language": "it",
|
||||
"provenance": {
|
||||
"source_file": "source/ablazione.md",
|
||||
"source_sha256": "sha256:" + "a" * 64,
|
||||
"supporting_excerpts": ["Il dominio Ablazione descrive la procedura."],
|
||||
},
|
||||
"payload": {"rule": rule},
|
||||
}
|
||||
)
|
||||
source_item = SourceObject(
|
||||
source_id="fs:curated-domain",
|
||||
uri="file:///safe/curated/domain/dominio-ablazione.md",
|
||||
fingerprint="sha256:" + "b" * 64,
|
||||
metadata={"relative_path": "curated/domain/dominio-ablazione.md"},
|
||||
)
|
||||
embedder = Embedder()
|
||||
|
||||
result = pipeline(
|
||||
tmp_path,
|
||||
Source([(source_item, dump_curated_markdown(evidence))]),
|
||||
embedder=embedder,
|
||||
vectors=Vectors(),
|
||||
policy=ChunkPolicy(version="chunk-v1", max_chars=4000),
|
||||
).run()
|
||||
|
||||
embedded = embedder.calls[0]
|
||||
assert result.status == "succeeded"
|
||||
assert f"Regola: {rule}" in embedded
|
||||
assert "tht:raw-rule" not in embedded
|
||||
assert "### Fact centrale:" not in embedded
|
||||
|
||||
|
||||
def test_pipeline_exposes_atomic_content_review_code_when_candidate_is_blocked(tmp_path):
|
||||
evidence = CuratedEvidence.model_validate(
|
||||
{
|
||||
|
||||
@@ -367,9 +367,9 @@ def test_migrate_workspace_evidence_rewrites_v1_units_as_v3_without_a_model_call
|
||||
assert report.unchanged == ()
|
||||
assert migrated.schema_version == 3
|
||||
assert migrated.payload.rule == "La fascia pediatrica comprende i minori."
|
||||
assert "## Regola\n\nLa fascia pediatrica comprende i minori." in curated_path.read_text(
|
||||
encoding="utf-8",
|
||||
)
|
||||
text = curated_path.read_text(encoding="utf-8")
|
||||
assert "## Regola\n\n<!-- tht:raw-rule:" in text
|
||||
assert "La fascia pediatrica comprende i minori." in text
|
||||
assert report.findings == ()
|
||||
|
||||
|
||||
@@ -393,6 +393,41 @@ def test_migrate_workspace_evidence_rewrites_v2_units_as_table_free_v3(tmp_path)
|
||||
assert not any(line.startswith("|") for line in text.splitlines())
|
||||
|
||||
|
||||
def test_migrate_workspace_evidence_rewrites_legacy_v3_rule_presentation(tmp_path):
|
||||
source_text = "I pazienti sotto i 18 anni sono pediatrici."
|
||||
evidence = _evidence(source_text).model_copy(update={"schema_version": 3})
|
||||
_write_workspace(tmp_path, evidence, source_text)
|
||||
curated_path = tmp_path / "evidence" / "curated" / "domain" / "fascia-pediatrica.md"
|
||||
legacy_text = curated_path.read_text(encoding="utf-8")
|
||||
legacy_text = legacy_text.replace(
|
||||
"## Regola\n\n<!-- tht:raw-rule:",
|
||||
"## Regola\n\n<!-- tht:legacy-raw-rule:",
|
||||
1,
|
||||
)
|
||||
# Recreate the exact pre-structured v3 body from the canonical value.
|
||||
start = legacy_text.index("<!-- tht:field:rule -->")
|
||||
end = legacy_text.index("<!-- /tht:field:rule -->", start)
|
||||
legacy_rule = (
|
||||
"<!-- tht:field:rule -->\n"
|
||||
"## Regola\n\n"
|
||||
f"{evidence.payload.rule}\n"
|
||||
)
|
||||
curated_path.write_text(
|
||||
legacy_text[:start] + legacy_rule + legacy_text[end:],
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
report = migrate_workspace_evidence(tmp_path, git_status=lambda _: ())
|
||||
|
||||
migrated_text = curated_path.read_text(encoding="utf-8")
|
||||
assert report.migrated == ("evidence:fascia-pediatrica",)
|
||||
assert report.unchanged == ()
|
||||
assert "## Regola\n\n<!-- tht:raw-rule:" in migrated_text
|
||||
assert load_curated_tree(tmp_path / "evidence" / "curated")[0].payload.rule == (
|
||||
evidence.payload.rule
|
||||
)
|
||||
|
||||
|
||||
def test_migrate_workspace_evidence_rejects_dirty_curated_files_in_a_nested_workspace(tmp_path):
|
||||
subprocess.run(["git", "init", "--quiet", str(tmp_path)], check=True)
|
||||
workspace_root = tmp_path / "psd-clinical"
|
||||
|
||||
@@ -262,6 +262,34 @@ def test_v3_curated_markdown_replaces_frontmatter_tables_with_readable_sections(
|
||||
assert parsed == evidence
|
||||
|
||||
|
||||
def test_v3_domain_rule_is_structured_without_changing_its_canonical_text(tmp_path):
|
||||
rule = (
|
||||
"Il dominio Ablazione descrive la procedura; indicazioni principali: fibrillazione "
|
||||
"atriale, flutter; appartiene all'universo procedurale. Fact centrale: "
|
||||
"`clinical.fact_ablazione`, collegata a `clinical.dim_patient`. Granularità: una riga "
|
||||
"per procedura. Domande tipiche: numero di procedure per anno; pazienti distinti per "
|
||||
"anno; distribuzione per indicazione."
|
||||
)
|
||||
base = _domain_evidence_v3()
|
||||
evidence = base.model_copy(
|
||||
update={"payload": base.payload.model_copy(update={"rule": rule})},
|
||||
)
|
||||
|
||||
text = dump_curated_markdown(evidence)
|
||||
parsed = parse_curated_markdown(
|
||||
text,
|
||||
path=tmp_path / "curated" / "domain" / "dominio-ablazione.md",
|
||||
)
|
||||
|
||||
assert "## Regola\n\n<!-- tht:raw-rule:" in text
|
||||
assert "- Il dominio Ablazione descrive la procedura;" in text
|
||||
assert "- **indicazioni principali:** fibrillazione atriale, flutter;" in text
|
||||
assert "### Fact centrale:" in text
|
||||
assert "### Granularità:" in text
|
||||
assert "### Domande tipiche:" in text
|
||||
assert parsed.payload.rule == rule
|
||||
|
||||
|
||||
def test_v3_curated_markdown_renders_enum_values_as_a_list_instead_of_a_table(tmp_path):
|
||||
evidence = CuratedEvidence.model_validate({
|
||||
**COMMON,
|
||||
@@ -294,6 +322,17 @@ def test_v3_curated_markdown_rejects_visible_metadata_that_drifted_from_canonica
|
||||
parse_curated_markdown(text)
|
||||
|
||||
|
||||
def test_v3_curated_markdown_rejects_rule_presentation_that_drifted_from_raw_text():
|
||||
text = dump_curated_markdown(_domain_evidence_v3()).replace(
|
||||
"La fact centrale è",
|
||||
"La tabella centrale è",
|
||||
1,
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="not canonical"):
|
||||
parse_curated_markdown(text)
|
||||
|
||||
|
||||
def test_v2_curated_markdown_renders_domain_content_in_the_markdown_body(tmp_path):
|
||||
evidence = _domain_evidence_v2()
|
||||
path = tmp_path / "curated" / "domain" / "dominio-ablazione.md"
|
||||
|
||||
@@ -707,22 +707,31 @@ def migrate_workspace_evidence(
|
||||
documents_by_id = {document.id: document for document in documents}
|
||||
if len(documents_by_id) != len(documents):
|
||||
raise EvidencePreparationError("duplicate_evidence_id")
|
||||
migrated = tuple(sorted(
|
||||
document.id for document in documents if document.schema_version in {1, 2}
|
||||
))
|
||||
unchanged = tuple(sorted(
|
||||
document.id for document in documents if document.schema_version == 3
|
||||
))
|
||||
upgraded = {
|
||||
evidence_id: document.model_copy(update={"schema_version": 3})
|
||||
for evidence_id, document in documents_by_id.items()
|
||||
}
|
||||
migrated_ids: list[str] = []
|
||||
unchanged_ids: list[str] = []
|
||||
curated_root = evidence_root / "curated"
|
||||
for evidence_id, document in upgraded.items():
|
||||
path = curated_root / document.kind / f"{evidence_id.removeprefix('evidence:')}.md"
|
||||
try:
|
||||
presentation_is_current = path.read_text(encoding="utf-8") == dump_curated_markdown(
|
||||
document,
|
||||
)
|
||||
except (OSError, UnicodeDecodeError):
|
||||
presentation_is_current = False
|
||||
target = unchanged_ids if presentation_is_current else migrated_ids
|
||||
target.append(evidence_id)
|
||||
migrated = tuple(sorted(migrated_ids))
|
||||
unchanged = tuple(sorted(unchanged_ids))
|
||||
if not migrated:
|
||||
return EvidenceMigrationReport(
|
||||
migrated=(),
|
||||
unchanged=unchanged,
|
||||
findings=validate_workspace_evidence(workspace_root).findings,
|
||||
)
|
||||
upgraded = {
|
||||
evidence_id: document.model_copy(update={"schema_version": 3})
|
||||
for evidence_id, document in documents_by_id.items()
|
||||
}
|
||||
findings = _stage_and_apply_authoring_tree(workspace_root, upgraded, manifest)
|
||||
return EvidenceMigrationReport(
|
||||
migrated=migrated,
|
||||
|
||||
@@ -319,6 +319,10 @@ _V2_EMPTY_LIST = "<!-- tht:empty-list -->"
|
||||
_V2_REVIEW_SEPARATOR = "<!-- tht:review-separator -->"
|
||||
_V2_REVIEW_FIELD = "<!-- tht:review-field -->"
|
||||
_V3_METADATA = re.compile(r"\A<!-- tht:metadata:([A-Za-z0-9+/=]+) -->\n")
|
||||
_V3_RAW_RULE = re.compile(
|
||||
r"\A<!-- tht:raw-rule:([A-Za-z0-9+/=]+) -->\n\n(.+)\Z",
|
||||
re.DOTALL,
|
||||
)
|
||||
_V3_KIND_LABELS = {
|
||||
"en": {
|
||||
"glossary": "Glossary",
|
||||
@@ -507,6 +511,72 @@ def _parse_v3_values(value: str, labels: dict[str, str]) -> dict[str, str]:
|
||||
return {key: "\n".join(lines) for key, lines in parsed.items()}
|
||||
|
||||
|
||||
def _leading_rule_label(value: str) -> tuple[str, str] | None:
|
||||
match = re.match(r"\A([^:;\n]{2,80}:)\s+(.+)\Z", value, re.DOTALL)
|
||||
if match is None:
|
||||
return None
|
||||
label = match.group(1)
|
||||
if len(label.removesuffix(":").split()) > 10:
|
||||
return None
|
||||
if label.lower() in {"http:", "https:"}:
|
||||
return None
|
||||
return label, match.group(2)
|
||||
|
||||
|
||||
def _render_v3_rule_detail(value: str) -> str:
|
||||
clauses = re.split(r"(?<=;)\s+", value)
|
||||
if len(clauses) < 3:
|
||||
return value
|
||||
rendered: list[str] = []
|
||||
for clause in clauses:
|
||||
labelled = _leading_rule_label(clause)
|
||||
if labelled is None:
|
||||
rendered.append(f"- {clause}")
|
||||
else:
|
||||
label, detail = labelled
|
||||
rendered.append(f"- **{label}** {detail}")
|
||||
return "\n".join(rendered)
|
||||
|
||||
|
||||
def _render_v3_rule_presentation(value: str) -> str:
|
||||
if re.search(r"(?m)^\s*(?:[-*+] |#{1,6} |```|>)", value):
|
||||
return value
|
||||
rendered: list[str] = []
|
||||
for paragraph in re.split(r"\n\s*\n", value):
|
||||
sentences = re.split(
|
||||
r"(?<=[.!?])\s+(?=[A-ZÀ-ÖØ-Þ0-9(`])",
|
||||
paragraph,
|
||||
)
|
||||
for sentence in sentences:
|
||||
labelled = _leading_rule_label(sentence)
|
||||
if labelled is None:
|
||||
rendered.append(_render_v3_rule_detail(sentence))
|
||||
continue
|
||||
label, detail = labelled
|
||||
rendered.append(f"### {label}\n\n{_render_v3_rule_detail(detail)}")
|
||||
return "\n\n".join(rendered)
|
||||
|
||||
|
||||
def _render_v3_rule(value: str) -> str:
|
||||
if "<!-- tht:raw-rule:" in value:
|
||||
raise ValueError("curated evidence rule contains a reserved marker")
|
||||
encoded = base64.b64encode(value.encode("utf-8")).decode("ascii")
|
||||
return (
|
||||
f"<!-- tht:raw-rule:{encoded} -->\n\n"
|
||||
f"{_render_v3_rule_presentation(value)}"
|
||||
)
|
||||
|
||||
|
||||
def _parse_v3_rule(value: str) -> str:
|
||||
match = _V3_RAW_RULE.fullmatch(value)
|
||||
if match is None:
|
||||
return value
|
||||
try:
|
||||
return base64.b64decode(match.group(1), validate=True).decode("utf-8")
|
||||
except (binascii.Error, UnicodeDecodeError, ValueError) as error:
|
||||
raise ValueError("curated evidence v3 rule metadata is malformed") from error
|
||||
|
||||
|
||||
def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
|
||||
payload = value.payload
|
||||
if value.kind == "glossary":
|
||||
@@ -568,6 +638,12 @@ def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[s
|
||||
|
||||
|
||||
def _render_v3_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]:
|
||||
if value.kind == "domain":
|
||||
return [_render_v2_field(
|
||||
"rule",
|
||||
labels["rule"],
|
||||
_render_v3_rule(value.payload.rule),
|
||||
)]
|
||||
if value.kind != "enum":
|
||||
return _render_v2_payload(value, labels)
|
||||
payload = value.payload
|
||||
@@ -705,14 +781,22 @@ def _render_v3_metadata(value: CuratedEvidence) -> str:
|
||||
return f"<!-- tht:metadata:{encoded} -->"
|
||||
|
||||
|
||||
def _render_v3_body(value: CuratedEvidence) -> str:
|
||||
def _render_v3_body(value: CuratedEvidence, *, structured_domain_rule: bool = True) -> str:
|
||||
if "\n" in value.title:
|
||||
raise ValueError("curated evidence title must be single-line in v3")
|
||||
labels = _v2_labels(value.language)
|
||||
fields = [
|
||||
_render_v3_overview(value, labels),
|
||||
_render_v3_scope(value, labels),
|
||||
*_render_v3_payload(value, labels),
|
||||
*(
|
||||
_render_v3_payload(value, labels)
|
||||
if structured_domain_rule
|
||||
else (
|
||||
[_render_v2_field("rule", labels["rule"], value.payload.rule)]
|
||||
if value.kind == "domain"
|
||||
else _render_v3_payload(value, labels)
|
||||
)
|
||||
),
|
||||
_render_v2_field(
|
||||
"supporting_excerpts",
|
||||
labels["supporting_excerpts"],
|
||||
@@ -827,6 +911,8 @@ def _parse_v2_payload(
|
||||
def _parse_v3_payload(
|
||||
kind: str, fields: dict[str, str], labels: dict[str, str],
|
||||
) -> tuple[dict, set[str]]:
|
||||
if kind == "domain":
|
||||
return {"rule": _parse_v3_rule(fields.get("rule", ""))}, {"rule"}
|
||||
if kind != "enum":
|
||||
return _parse_v2_payload(kind, fields, labels)
|
||||
return {
|
||||
@@ -985,7 +1071,9 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi
|
||||
data["payload"] = data.pop(kind, None)
|
||||
evidence = CuratedEvidence.model_validate(data)
|
||||
if evidence.schema_version == 3 and dump_curated_markdown(evidence) != text:
|
||||
raise ValueError("curated evidence v3 presentation is not canonical")
|
||||
legacy = f"{_render_v3_metadata(evidence)}\n{_render_v3_body(evidence, structured_domain_rule=False)}"
|
||||
if legacy != text:
|
||||
raise ValueError("curated evidence v3 presentation is not canonical")
|
||||
if path is not None:
|
||||
_validate_kind_directory(path, evidence.kind)
|
||||
return evidence
|
||||
|
||||
Reference in New Issue
Block a user