#!/usr/bin/env python3 """Machine-checkable policy for current workspace-descriptor documentation.""" from __future__ import annotations import argparse import html import re import sys from collections.abc import Sequence from dataclasses import dataclass from pathlib import Path WORKSPACE_REGION = "workspace-descriptor-contract" NON_WORKSPACE_REGION = "non-workspace-migration" HISTORICAL_ARCHIVE_H1 = "# Historical archive" CANONICAL_V4_SENTENCE = "Schema v4 is the only accepted workspace descriptor." CANONICAL_REJECTION_SENTENCE = ( "Schema v1, v2, and v3 workspace descriptors are rejected before activation." ) _MARKER_LINE = re.compile( r"" ) _MARKER_LITERAL = re.compile( r"(?:workspace-descriptor-contract|non-workspace-migration):(start|end)" ) _MIGRATION_REQUIRED = re.compile(r"migration_required", re.IGNORECASE) _DELETED_TOOL = re.compile( r"migrate\s*-\s*legacy|" r"legacy[-\s]+(?:(?:workspace[-\s]+)?descriptor[-\s]+)?(?:transformer|migrator)|" r"(?:workspace[-\s]+)?descriptor[-\s]+legacy[-\s]+(?:transformer|migrator)|" r"migrat(?:e|ing)[-\s]+legacy[-\s]+descriptors?", re.IGNORECASE, ) _REMAINING_LEGACY_CLAIM = re.compile( r"\b(?:schema(?:[- ]v?|\s+version\s*)[123]|" r"schema_version\s*[:=]\s*[123]|version\s*[123]|v[123]|" r"legacy(?:[-\s]+workspace)?[-\s]+descriptors?)\b", re.IGNORECASE, ) _CODE_LITERAL_CHARS = "\\`*_~[]()<>!&" _CODE_PROTECT = str.maketrans( {character: chr(0xE000 + index) for index, character in enumerate(_CODE_LITERAL_CHARS)} ) _CODE_RESTORE = str.maketrans( {chr(0xE000 + index): character for index, character in enumerate(_CODE_LITERAL_CHARS)} ) _OUTSIDE_LEGACY_REFERENCE = re.compile( r"\b(?:schema(?:[- ]v?|\s+version\s*)[123]|" r"schema_version\s*[:=]\s*[123]|" r"(?:v[123](?:\s*(?:/|and|or|,)\s*v[123])*)\s+(?:workspace\s+)?descriptors?|" r"legacy(?:[-\s]+workspace)?[-\s]+descriptors?)\b", re.IGNORECASE, ) class ContractError(ValueError): """Raised when current documentation violates the descriptor-region policy.""" @dataclass(frozen=True) class ScannedLine: start: int end: int text: str active: bool @dataclass(frozen=True) class Heading: start: int level: int text: str style: str raw: str @dataclass(frozen=True) class MarkdownScan: lines: tuple[ScannedLine, ...] headings: tuple[Heading, ...] prose_text: str operator_text: str @dataclass(frozen=True) class Region: name: str start: int end: int def contains(self, offset: int) -> bool: return self.start <= offset < self.end def _opening_fence(line: str) -> tuple[str, int] | None: match = re.match(r"^ {0,3}(`{3,}|~{3,})(.*)$", line) if not match: return None run, rest = match.groups() if run[0] == "`" and "`" in rest: return None return run[0], len(run) def _closes_fence(line: str, fence: tuple[str, int]) -> bool: char, minimum = fence return re.fullmatch(rf" {{0,3}}{re.escape(char)}{{{minimum},}}[ \t]*", line) is not None def _atx_heading(line: ScannedLine) -> Heading | None: if not line.active: return None match = re.fullmatch(r" {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*", line.text) if not match: return None hashes, content = match.groups() content = content or "" content = re.sub(r"[ \t]+#+[ \t]*$", "", content).strip() return Heading(line.start, len(hashes), content, "atx", line.text) def _mask_html_comments(line: str, depth: int) -> tuple[str, int]: """Mask HTML comment ranges without changing raw character offsets.""" masked = list(line) position = 0 while position < len(line): if depth and line.startswith("--!>", position): raise ContractError("alternate HTML comment closer is not allowed") if line.startswith("", position): masked[position : position + 3] = " " * 3 depth -= 1 position += 3 else: if depth: masked[position] = " " position += 1 return "".join(masked), depth def scan_active_markdown(text: str, label: str = "document") -> MarkdownScan: """Scan structural Markdown and retain the whole visible active document.""" lines: list[ScannedLine] = [] prose_chunks: list[str] = [] operator_chunks: list[str] = [] fence: tuple[str, int] | None = None html_comment_depth = 0 offset = 0 for raw_line in text.splitlines(keepends=True): line = raw_line.rstrip("\r\n") ending = raw_line[len(line) :] start = offset offset += len(raw_line) if fence is not None: lines.append(ScannedLine(start, offset, line, False)) prose_chunks.append(" " * len(line) + ending) if _closes_fence(line, fence): operator_chunks.append(" " * len(line) + ending) fence = None else: operator_chunks.append(line.translate(_CODE_PROTECT) + ending) continue if not html_comment_depth and re.match(r"^(?: {4}|\t)", line): lines.append(ScannedLine(start, offset, line, False)) prose_chunks.append(" " * len(line) + ending) operator_chunks.append(line.translate(_CODE_PROTECT) + ending) continue if not html_comment_depth: opened_fence = _opening_fence(line) if opened_fence is not None: lines.append(ScannedLine(start, offset, line, False)) prose_chunks.append(" " * len(line) + ending) operator_chunks.append(" " * len(line) + ending) fence = opened_fence continue depth_before = html_comment_depth visible, html_comment_depth = _mask_html_comments(line, html_comment_depth) exact_marker = depth_before == 0 and _MARKER_LINE.fullmatch(line) is not None misplaced_marker = depth_before == 0 and _MARKER_LITERAL.search(line) is not None active = exact_marker or misplaced_marker or bool(visible.strip()) lines.append(ScannedLine(start, offset, line, active)) prose_chunks.append(visible + ending) operator_chunks.append(visible + ending) if html_comment_depth: raise ContractError(f"{label}: unclosed HTML comment") headings: list[Heading] = [] for index, line in enumerate(lines): atx = _atx_heading(line) if atx: headings.append(atx) continue if not line.active or not line.text.strip() or _MARKER_LINE.fullmatch(line.text): continue if index + 1 >= len(lines) or not lines[index + 1].active: continue underline = re.fullmatch(r" {0,3}(=+|-+)[ \t]*", lines[index + 1].text) if underline: headings.append( Heading( line.start, 1 if underline.group(1)[0] == "=" else 2, line.text.strip(), "setext", line.text, ) ) headings.sort(key=lambda heading: heading.start) prose_text = "".join(prose_chunks) operator_text = "".join(operator_chunks) if len(prose_text) != len(text) or len(operator_text) != len(text): raise ContractError(f"{label}: Markdown representations changed raw offsets") return MarkdownScan(tuple(lines), tuple(headings), prose_text, operator_text) def _regions(text: str, label: str, scan: MarkdownScan) -> list[Region]: events: list[tuple[ScannedLine, str, str]] = [] for line in scan.lines: if not line.active: continue marker = _MARKER_LINE.fullmatch(line.text) if marker: events.append((line, marker.group(1), marker.group(2))) elif _MARKER_LITERAL.search(line.text): raise ContractError( f"{label}: misplaced documentation region marker: {line.text.strip()}" ) regions: list[Region] = [] stack: tuple[str, int] | None = None for line, name, action in events: if action == "start": if stack is not None: raise ContractError(f"{label}: documentation regions may not nest or overlap") stack = (name, line.end) continue if stack is None: raise ContractError(f"{label}: unmatched {name}:end marker") open_name, content_start = stack if open_name != name: raise ContractError( f"{label}: marker {name}:end closes active {open_name}:start region" ) regions.append(Region(name, content_start, line.start)) stack = None if stack is not None: raise ContractError(f"{label}: unmatched {stack[0]}:start marker") workspace_regions = [region for region in regions if region.name == WORKSPACE_REGION] if len(workspace_regions) != 1: raise ContractError( f"{label}: expected exactly one {WORKSPACE_REGION}:start/end region, " f"found {len(workspace_regions)}" ) return regions def _render_markdown(markdown: str, *, retain_code_text: bool = True) -> str: """Normalize one offset-preserving Markdown representation to rendered text.""" rendered = markdown rendered = re.sub(r"(?m)^ {0,3}\[[^]\n]+\]:[^\n]*(?:\n|$)", "", rendered) # Retain multiline code-span text while discarding a matching delimiter run. def code_text(match: re.Match[str]) -> str: content = re.sub(r"[\r\n]+", " ", match.group(2)) if content.startswith(" ") and content.endswith(" ") and content.strip(): content = content[1:-1] return content rendered = re.sub( r"(?]*?)?\s*/?>", "", rendered, flags=re.DOTALL, ) rendered = re.sub( r"""\\([!"#$%&'()*+,\-./:;<=>?@\[\]\\^_`{|}~])""", r"\1", rendered, ) rendered = rendered.replace("*", "").replace("~", "") rendered = re.sub(r"(? str: return _render_markdown(scan.operator_text) def _render_prose_markdown(markdown: str) -> str: return _render_markdown(markdown, retain_code_text=False) def _exact_sentence_count(normalized: str, sentence: str) -> int: pattern = re.compile( rf"(?:^|(?<=[.!?]) )(?={re.escape(sentence)}(?:$| ))" ) return len(pattern.findall(normalized)) def _mask_region_ranges(text: str, regions: list[Region]) -> str: masked = list(text) for region in regions: for offset in range(region.start, region.end): if masked[offset] not in "\r\n": masked[offset] = " " return "".join(masked) def check_document_text(text: str, label: str = "document") -> None: """Validate one current (non-historical) document.""" scan = scan_active_markdown(text, label) regions = _regions(text, label, scan) workspace = next(region for region in regions if region.name == WORKSPACE_REGION) workspace_prose = _render_prose_markdown( scan.prose_text[workspace.start : workspace.end] ) if _exact_sentence_count(workspace_prose, CANONICAL_V4_SENTENCE) != 1: raise ContractError(f"{label}: workspace contract lacks the canonical v4-only sentence") if _exact_sentence_count(workspace_prose, CANONICAL_REJECTION_SENTENCE) != 1: raise ContractError( f"{label}: workspace contract lacks the canonical v1/v2/v3 rejection sentence" ) workspace_visible = _render_markdown(scan.operator_text[workspace.start : workspace.end]) workspace_remainder = workspace_visible.replace(CANONICAL_REJECTION_SENTENCE, " ", 1) forbidden_claim = _MIGRATION_REQUIRED.search( workspace_remainder ) or _REMAINING_LEGACY_CLAIM.search(workspace_remainder) if forbidden_claim: raise ContractError( f"{label}: workspace contract contains an extra legacy descriptor claim: " f"{forbidden_claim.group(0)}" ) rendered_operator_text = _render_active_markdown(scan) deleted = _DELETED_TOOL.search(rendered_operator_text) if deleted: raise ContractError( f"{label}: current text contains deleted tool instruction: {deleted.group(0)}" ) non_workspace = [region for region in regions if region.name == NON_WORKSPACE_REGION] outside_allowed = _mask_region_ranges( scan.operator_text, [workspace, *non_workspace], ) rendered_outside = _render_markdown(outside_allowed) outside_claim = _MIGRATION_REQUIRED.search( rendered_outside ) or _OUTSIDE_LEGACY_REFERENCE.search(rendered_outside) if outside_claim: raise ContractError( f"{label}: current legacy schema or migration reference outside a " f"{NON_WORKSPACE_REGION}:start/end region: {outside_claim.group(0)}" ) def check_project_state_text(text: str, label: str = "PROJECT_STATE.md") -> None: """Validate current PROJECT_STATE text and its terminal historical archive hierarchy.""" scan = scan_active_markdown(text, label) archive_headings = [ heading for heading in scan.headings if heading.level == 1 and heading.style == "atx" and heading.raw == HISTORICAL_ARCHIVE_H1 ] if len(archive_headings) != 1: raise ContractError( f"{label}: expected exactly one active terminal '{HISTORICAL_ARCHIVE_H1}' boundary" ) archive_heading = archive_headings[0] archive_line_index = next( index for index, line in enumerate(scan.lines) if line.start == archive_heading.start ) if archive_line_index + 1 >= len(scan.lines) or not ( scan.lines[archive_line_index + 1].active and scan.lines[archive_line_index + 1].text == "## Historical snapshots and archived reference notes" ): raise ContractError( f"{label}: the preserved Historical snapshots H2 must immediately follow " f"'{HISTORICAL_ARCHIVE_H1}'" ) later_headings = [heading for heading in scan.headings if heading.start > archive_heading.start] if not later_headings or not ( later_headings[0].level == 2 and later_headings[0].style == "atx" and later_headings[0].raw == "## Historical snapshots and archived reference notes" ): raise ContractError( f"{label}: Historical archive must be followed by the existing Historical snapshots H2" ) if any(heading.level == 1 for heading in later_headings): raise ContractError(f"{label}: Historical archive must be the terminal H1 hierarchy") check_document_text(text[:archive_heading.start], f"{label} current section") def check_document_path(path: Path) -> None: check_document_text(path.read_text(), str(path)) def check_project_state_path(path: Path) -> None: check_project_state_text(path.read_text(), str(path)) def _parser() -> argparse.ArgumentParser: parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--document", action="append", default=[], type=Path) parser.add_argument("--project-state", action="append", default=[], type=Path) return parser def main(argv: Sequence[str] | None = None) -> int: args = _parser().parse_args(argv) if not args.document and not args.project_state: raise SystemExit("at least one --document or --project-state path is required") try: for path in args.document: check_document_path(path) for path in args.project_state: check_project_state_path(path) except (ContractError, OSError) as error: print(error, file=sys.stderr) return 1 return 0 if __name__ == "__main__": raise SystemExit(main())