From 9898726069de718a7c500fc5908344db171de14b Mon Sep 17 00:00:00 2001 From: Codex Date: Wed, 26 Aug 2026 12:38:30 +0200 Subject: [PATCH] feat(evidence): structure v3 domain rules for review --- DESIGN.md | 2 + PROJECT_STATE.md | 12 +-- docs/contracts/workspace-evidence-v3.md | 9 ++- docs/evidence.md | 8 +- harness/tests/test_corpus_pipeline.py | 44 +++++++++++ harness/tests/test_evidence_authoring.py | 41 ++++++++++- harness/tests/test_evidence_canonical.py | 39 ++++++++++ harness/tht/evidence/authoring.py | 29 +++++--- harness/tht/evidence/canonical.py | 94 +++++++++++++++++++++++- 9 files changed, 254 insertions(+), 24 deletions(-) diff --git a/DESIGN.md b/DESIGN.md index e01c5d78..e466ce2c 100644 --- a/DESIGN.md +++ b/DESIGN.md @@ -307,6 +307,8 @@ machine contract. Technical metadata belongs in progressive disclosure, not abov - **Do** respect `prefers-reduced-motion` while preserving immediate non-kinetic feedback. - **Do** use English for interface chrome and the workspace language for persisted document content. - **Do** render curated metadata and scope as Markdown prose or lists, never as a frontmatter table. +- **Do** break long curated rules into paragraphs, labelled subsections, and lists at existing + punctuation boundaries while preserving the exact canonical text for machines. ### Don't: diff --git a/PROJECT_STATE.md b/PROJECT_STATE.md index b456bf94..4b128b23 100644 --- a/PROJECT_STATE.md +++ b/PROJECT_STATE.md @@ -25,11 +25,13 @@ review gates and keeps the live transcript in memory. See The evidence restructuring and PSD migration completed real acceptance on 2026-08-25. - The curated PSD revision contains 35 approved Evidence units and 60 review items. -- The PSD authoring repository published all 35 units using Curated unit schema v2. A schema v3 - table-free presentation is now available in the authoring flow: hidden canonical metadata, - wrapping Markdown scope lists, list-based enum values, and collapsed technical provenance. - `tht evidence migrate ` performs the deterministic v1/v2-to-v3 rewrite without - model calls. The 35-unit PSD v3 migration is currently local and pending commit/publication. +- The PSD authoring repository publishes all 35 units using Curated unit schema v3. Its table-free + presentation uses hidden canonical metadata, wrapping Markdown scope lists, list-based enum + values, and collapsed technical provenance. Long domain rules now have a deterministic + human-readable presentation while retaining their exact canonical text for vector ingestion. + `tht evidence migrate ` performs the deterministic v1/v2 upgrade and older-v3 + presentation rewrite without model calls. The structured-rule PSD rewrite is currently local and + pending commit/publication. - The accepted snapshot is `psd-clinical-675990d90eae51da6f2bd51b1ae2609f245772ef-snapshot`. - The active generation is `gen:f968b3bd7a553dbfef3cf47093698f2bc7f95f11`. diff --git a/docs/contracts/workspace-evidence-v3.md b/docs/contracts/workspace-evidence-v3.md index 734a5385..0e20e9cb 100644 --- a/docs/contracts/workspace-evidence-v3.md +++ b/docs/contracts/workspace-evidence-v3.md @@ -59,8 +59,13 @@ the complete review surface as deterministic Markdown. It uses headings, paragra lists, fenced SQL, blockquotes, and a collapsed technical-details block. It never emits YAML frontmatter or Markdown tables. Invisible `tht:` comments delimit typed fields. Parsers must reject missing, duplicate, unknown, desynchronized, or unstructured body content; they must never silently -ignore it. Newly prepared units use v3. `tht evidence migrate ` upgrades existing -v1 and v2 units locally without a model call, commit, publication, or semantic change. +ignore it. Domain rules also retain their exact canonical text in an invisible `tht:raw-rule` +comment while presenting long prose as paragraphs, labelled subsections, and semicolon-derived +lists. Runtime chunking reads the parsed canonical rule, not this review-only presentation. + +Newly prepared units use v3. `tht evidence migrate ` upgrades v1 and v2 units and +canonicalizes an older v3 presentation locally without a model call, commit, publication, or +semantic change. ### Example: filesystem diff --git a/docs/evidence.md b/docs/evidence.md index e4c27062..f5c9a82f 100644 --- a/docs/evidence.md +++ b/docs/evidence.md @@ -52,7 +52,13 @@ Canonical Curated Evidence v3 hides canonical machine metadata in an HTML commen whole review surface as real Markdown. GitHub therefore shows no frontmatter table. The body layout is deterministic for each Evidence kind: prose uses sections and paragraphs, scopes and enum values use wrapping lists, formulas use fenced SQL, supporting excerpts use blockquotes, and unresolved -review items use dedicated blocks. +review items use dedicated blocks. Long domain rules are split into readable paragraphs, labelled +subsections, and lists at existing semicolon boundaries. Their exact original text remains canonical +in an invisible marker, so the formatting cannot change their meaning or bytes. + +Preprocessing parses the unit first and builds semantic chunks from the typed payload. The vector +store therefore receives the original rule text and not headings, list markers, or invisible +presentation metadata. ```markdown diff --git a/harness/tests/test_corpus_pipeline.py b/harness/tests/test_corpus_pipeline.py index d8b464ee..a0323a6f 100644 --- a/harness/tests/test_corpus_pipeline.py +++ b/harness/tests/test_corpus_pipeline.py @@ -165,6 +165,50 @@ def test_pipeline_embeds_validated_curated_evidence_as_semantic_fragments(tmp_pa assert vectors.records[0].record.metadata["provenance"]["source_file"] == "source/paziente.md" +def test_pipeline_strips_domain_rule_presentation_before_embedding(tmp_path): + rule = ( + "Il dominio Ablazione descrive la procedura; indicazioni principali: fibrillazione " + "atriale; appartiene all'universo procedurale. Fact centrale: clinical.fact_ablazione." + ) + evidence = CuratedEvidence.model_validate( + { + "schema_version": 3, + "id": "evidence:dominio-ablazione", + "title": "Dominio Ablazione", + "kind": "domain", + "purposes": ["disambiguation"], + "language": "it", + "provenance": { + "source_file": "source/ablazione.md", + "source_sha256": "sha256:" + "a" * 64, + "supporting_excerpts": ["Il dominio Ablazione descrive la procedura."], + }, + "payload": {"rule": rule}, + } + ) + source_item = SourceObject( + source_id="fs:curated-domain", + uri="file:///safe/curated/domain/dominio-ablazione.md", + fingerprint="sha256:" + "b" * 64, + metadata={"relative_path": "curated/domain/dominio-ablazione.md"}, + ) + embedder = Embedder() + + result = pipeline( + tmp_path, + Source([(source_item, dump_curated_markdown(evidence))]), + embedder=embedder, + vectors=Vectors(), + policy=ChunkPolicy(version="chunk-v1", max_chars=4000), + ).run() + + embedded = embedder.calls[0] + assert result.status == "succeeded" + assert f"Regola: {rule}" in embedded + assert "tht:raw-rule" not in embedded + assert "### Fact centrale:" not in embedded + + def test_pipeline_exposes_atomic_content_review_code_when_candidate_is_blocked(tmp_path): evidence = CuratedEvidence.model_validate( { diff --git a/harness/tests/test_evidence_authoring.py b/harness/tests/test_evidence_authoring.py index c7cc8e33..cbfd23f9 100644 --- a/harness/tests/test_evidence_authoring.py +++ b/harness/tests/test_evidence_authoring.py @@ -367,9 +367,9 @@ def test_migrate_workspace_evidence_rewrites_v1_units_as_v3_without_a_model_call assert report.unchanged == () assert migrated.schema_version == 3 assert migrated.payload.rule == "La fascia pediatrica comprende i minori." - assert "## Regola\n\nLa fascia pediatrica comprende i minori." in curated_path.read_text( - encoding="utf-8", - ) + text = curated_path.read_text(encoding="utf-8") + assert "## Regola\n\n") + end = legacy_text.index("", start) + legacy_rule = ( + "\n" + "## Regola\n\n" + f"{evidence.payload.rule}\n" + ) + curated_path.write_text( + legacy_text[:start] + legacy_rule + legacy_text[end:], + encoding="utf-8", + ) + + report = migrate_workspace_evidence(tmp_path, git_status=lambda _: ()) + + migrated_text = curated_path.read_text(encoding="utf-8") + assert report.migrated == ("evidence:fascia-pediatrica",) + assert report.unchanged == () + assert "## Regola\n\n" _V2_REVIEW_SEPARATOR = "" _V2_REVIEW_FIELD = "" _V3_METADATA = re.compile(r"\A\n") +_V3_RAW_RULE = re.compile( + r"\A\n\n(.+)\Z", + re.DOTALL, +) _V3_KIND_LABELS = { "en": { "glossary": "Glossary", @@ -507,6 +511,72 @@ def _parse_v3_values(value: str, labels: dict[str, str]) -> dict[str, str]: return {key: "\n".join(lines) for key, lines in parsed.items()} +def _leading_rule_label(value: str) -> tuple[str, str] | None: + match = re.match(r"\A([^:;\n]{2,80}:)\s+(.+)\Z", value, re.DOTALL) + if match is None: + return None + label = match.group(1) + if len(label.removesuffix(":").split()) > 10: + return None + if label.lower() in {"http:", "https:"}: + return None + return label, match.group(2) + + +def _render_v3_rule_detail(value: str) -> str: + clauses = re.split(r"(?<=;)\s+", value) + if len(clauses) < 3: + return value + rendered: list[str] = [] + for clause in clauses: + labelled = _leading_rule_label(clause) + if labelled is None: + rendered.append(f"- {clause}") + else: + label, detail = labelled + rendered.append(f"- **{label}** {detail}") + return "\n".join(rendered) + + +def _render_v3_rule_presentation(value: str) -> str: + if re.search(r"(?m)^\s*(?:[-*+] |#{1,6} |```|>)", value): + return value + rendered: list[str] = [] + for paragraph in re.split(r"\n\s*\n", value): + sentences = re.split( + r"(?<=[.!?])\s+(?=[A-ZÀ-ÖØ-Þ0-9(`])", + paragraph, + ) + for sentence in sentences: + labelled = _leading_rule_label(sentence) + if labelled is None: + rendered.append(_render_v3_rule_detail(sentence)) + continue + label, detail = labelled + rendered.append(f"### {label}\n\n{_render_v3_rule_detail(detail)}") + return "\n\n".join(rendered) + + +def _render_v3_rule(value: str) -> str: + if "\n\n" + f"{_render_v3_rule_presentation(value)}" + ) + + +def _parse_v3_rule(value: str) -> str: + match = _V3_RAW_RULE.fullmatch(value) + if match is None: + return value + try: + return base64.b64decode(match.group(1), validate=True).decode("utf-8") + except (binascii.Error, UnicodeDecodeError, ValueError) as error: + raise ValueError("curated evidence v3 rule metadata is malformed") from error + + def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]: payload = value.payload if value.kind == "glossary": @@ -568,6 +638,12 @@ def _render_v2_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[s def _render_v3_payload(value: CuratedEvidence, labels: dict[str, str]) -> list[str]: + if value.kind == "domain": + return [_render_v2_field( + "rule", + labels["rule"], + _render_v3_rule(value.payload.rule), + )] if value.kind != "enum": return _render_v2_payload(value, labels) payload = value.payload @@ -705,14 +781,22 @@ def _render_v3_metadata(value: CuratedEvidence) -> str: return f"" -def _render_v3_body(value: CuratedEvidence) -> str: +def _render_v3_body(value: CuratedEvidence, *, structured_domain_rule: bool = True) -> str: if "\n" in value.title: raise ValueError("curated evidence title must be single-line in v3") labels = _v2_labels(value.language) fields = [ _render_v3_overview(value, labels), _render_v3_scope(value, labels), - *_render_v3_payload(value, labels), + *( + _render_v3_payload(value, labels) + if structured_domain_rule + else ( + [_render_v2_field("rule", labels["rule"], value.payload.rule)] + if value.kind == "domain" + else _render_v3_payload(value, labels) + ) + ), _render_v2_field( "supporting_excerpts", labels["supporting_excerpts"], @@ -827,6 +911,8 @@ def _parse_v2_payload( def _parse_v3_payload( kind: str, fields: dict[str, str], labels: dict[str, str], ) -> tuple[dict, set[str]]: + if kind == "domain": + return {"rule": _parse_v3_rule(fields.get("rule", ""))}, {"rule"} if kind != "enum": return _parse_v2_payload(kind, fields, labels) return { @@ -985,7 +1071,9 @@ def parse_curated_markdown(text: str, *, path: Path | None = None) -> CuratedEvi data["payload"] = data.pop(kind, None) evidence = CuratedEvidence.model_validate(data) if evidence.schema_version == 3 and dump_curated_markdown(evidence) != text: - raise ValueError("curated evidence v3 presentation is not canonical") + legacy = f"{_render_v3_metadata(evidence)}\n{_render_v3_body(evidence, structured_domain_rule=False)}" + if legacy != text: + raise ValueError("curated evidence v3 presentation is not canonical") if path is not None: _validate_kind_directory(path, evidence.kind) return evidence