"""Curated unit v4: visible Markdown fields are the sole human-content authority.""" from __future__ import annotations import json import re import yaml from .canonical import ( _PAYLOAD_TYPE_BY_KIND, CuratedEvidence, ManualEvidenceProvenance, _v2_labels, ) _LIST_FIELDS = {"synonyms", "variants", "columns", "tables"} def parse_metadata(text: str) -> dict: class UniqueKeysLoader(yaml.SafeLoader): pass def mapping(loader, node): pairs = loader.construct_pairs(node, deep=True) result = {} for key, value in pairs: if key in result: raise ValueError(f"Duplicate metadata key: {key}") result[key] = value return result UniqueKeysLoader.add_constructor(yaml.resolver.BaseResolver.DEFAULT_MAPPING_TAG, mapping) try: return yaml.load(text, Loader=UniqueKeysLoader) except (yaml.YAMLError, TypeError) as error: raise ValueError("Evidence metadata is malformed") from error def _sections(body: str, headings: dict[str, str]) -> dict[str, str]: """Recognize structural H2s outside code fences; all other Markdown is content.""" sections: dict[str, list[str]] = {} field = None fence = None for line in body.splitlines(): match = re.match(r"^\s{0,3}(`{3,}|~{3,})", line) if match: marker = match[1] if fence is None: fence = marker elif marker[0] == fence[0] and len(marker) >= len(fence): fence = None heading = headings.get(line[3:]) if line.startswith("## ") and fence is None else None if heading: if heading in sections: raise ValueError(f"Duplicate section: {line[3:]}") field = heading sections[field] = [] elif field is not None: sections[field].append(line) elif line.strip(): raise ValueError("Content must follow a documented section heading") if fence: raise ValueError("Unclosed Markdown code fence") return {key: "\n".join(lines).strip() for key, lines in sections.items()} def _list(text: str) -> list[str]: if not text: return [] values = [] for line in text.splitlines(): if not line.startswith("- ") or not line[2:].strip(): raise ValueError("List entries must use '- value', one per line") value = line[2:] # Quoted strings preserve multiline and unusual values during migration. values.append(json.loads(value) if value.startswith('"') else value) return values def _values(text: str) -> dict[str, str]: values = {} key = None lines = [] for line in text.splitlines(): if line.startswith("### "): if key is not None: values[key] = "\n".join(lines).strip() label = line[4:] key = json.loads(label) if label.startswith('"') else label if key in values: raise ValueError("Duplicate enum value") lines = [] elif key is None: if line.strip(): raise ValueError("Enum values require '### value' headings") else: lines.append(line) if key is not None: values[key] = "\n".join(lines).strip() return values def parse_document(metadata: dict, body: str) -> dict: data = dict(metadata) if {"title", "payload", *(_PAYLOAD_TYPE_BY_KIND)}.intersection(data): raise ValueError("Title and payload must be edited only in the Markdown body") lines = body.strip().splitlines() if not lines or not lines[0].startswith("# ") or not lines[0][2:].strip(): raise ValueError("A title starting with '# ' is required") data["title"] = lines[0][2:].strip() kind = data.get("kind") if kind not in _PAYLOAD_TYPE_BY_KIND: raise ValueError("Unknown Evidence kind") labels = _v2_labels(str(data.get("language", ""))) fields = _PAYLOAD_TYPE_BY_KIND[kind].model_fields sections = _sections("\n".join(lines[1:]), {labels[key]: key for key in fields}) required = {name for name, field in fields.items() if field.is_required()} if not required <= sections.keys(): raise ValueError( "Missing sections: " + ", ".join(labels[k] for k in sorted(required - sections.keys())) ) payload = {} for key, content in sections.items(): if key in _LIST_FIELDS: payload[key] = _list(content) elif key == "values": payload[key] = _values(content) elif key == "sql" and content.startswith("```sql\n") and content.endswith("\n```"): payload[key] = content[7:-4] else: payload[key] = content if fields[key].is_required() and not payload[key] and key != "values": raise ValueError(f"Section {labels[key]} must not be empty") data["payload"] = payload data.setdefault( "provenance", ManualEvidenceProvenance(declared_by="local curator").model_dump() ) return data def render_document(value: CuratedEvidence) -> str: metadata = value.model_dump(mode="json", exclude={"title", "payload"}) labels = _v2_labels(value.language) parts = [ f"---\n{yaml.safe_dump(metadata, allow_unicode=True, sort_keys=False)}---\n\n# {value.title}" ] for key, content in value.payload.model_dump(mode="json").items(): if key in _LIST_FIELDS: rendered = "\n".join( "- " + ( json.dumps(v, ensure_ascii=False) if "\n" in v or v.startswith('"') or v != v.strip() else v ) for v in content ) elif key == "values": rendered = "\n\n".join( f"### {json.dumps(k, ensure_ascii=False)}\n\n{v}" for k, v in content.items() ) elif key == "sql": rendered = f"```sql\n{content}\n```" else: rendered = content parts.append(f"## {labels[key]}\n\n{rendered}") rendered = "\n\n".join(parts) + "\n" # Migration must fail explicitly rather than silently changing unrepresentable content. restored = CuratedEvidence.model_validate( parse_document(metadata, rendered.split("---\n", 2)[2]) ) if restored != value: raise ValueError( f"{value.id}: content cannot be represented losslessly in editable Markdown" ) return rendered