Files
ThothII/scripts/workspace_descriptor_doc_contract.py

475 lines
16 KiB
Python
Executable File

#!/usr/bin/env python3
"""Machine-checkable policy for current workspace-descriptor documentation."""
from __future__ import annotations
import argparse
import html
import re
import sys
from collections.abc import Sequence
from dataclasses import dataclass
from pathlib import Path
WORKSPACE_REGION = "workspace-descriptor-contract"
NON_WORKSPACE_REGION = "non-workspace-migration"
HISTORICAL_ARCHIVE_H1 = "# Historical archive"
CANONICAL_V4_SENTENCE = "Schema v4 is the only accepted workspace descriptor."
CANONICAL_REJECTION_SENTENCE = (
"Schema v1, v2, and v3 workspace descriptors are rejected before activation."
)
_MARKER_LINE = re.compile(
r"<!-- (workspace-descriptor-contract|non-workspace-migration):(start|end) -->"
)
_MARKER_LITERAL = re.compile(
r"(?:workspace-descriptor-contract|non-workspace-migration):(start|end)"
)
_MIGRATION_REQUIRED = re.compile(r"migration_required", re.IGNORECASE)
_DELETED_TOOL = re.compile(
r"migrate\s*-\s*legacy|"
r"legacy[-\s]+(?:(?:workspace[-\s]+)?descriptor[-\s]+)?(?:transformer|migrator)|"
r"(?:workspace[-\s]+)?descriptor[-\s]+legacy[-\s]+(?:transformer|migrator)|"
r"migrat(?:e|ing)[-\s]+legacy[-\s]+descriptors?",
re.IGNORECASE,
)
_REMAINING_LEGACY_CLAIM = re.compile(
r"\b(?:schema(?:[- ]v?|\s+version\s*)[123]|"
r"schema_version\s*[:=]\s*[123]|version\s*[123]|v[123]|"
r"legacy(?:[-\s]+workspace)?[-\s]+descriptors?)\b",
re.IGNORECASE,
)
_CODE_LITERAL_CHARS = "\\`*_~[]()<>!&"
_CODE_PROTECT = str.maketrans(
{character: chr(0xE000 + index) for index, character in enumerate(_CODE_LITERAL_CHARS)}
)
_CODE_RESTORE = str.maketrans(
{chr(0xE000 + index): character for index, character in enumerate(_CODE_LITERAL_CHARS)}
)
_OUTSIDE_LEGACY_REFERENCE = re.compile(
r"\b(?:schema(?:[- ]v?|\s+version\s*)[123]|"
r"schema_version\s*[:=]\s*[123]|"
r"(?:v[123](?:\s*(?:/|and|or|,)\s*v[123])*)\s+(?:workspace\s+)?descriptors?|"
r"legacy(?:[-\s]+workspace)?[-\s]+descriptors?)\b",
re.IGNORECASE,
)
class ContractError(ValueError):
"""Raised when current documentation violates the descriptor-region policy."""
@dataclass(frozen=True)
class ScannedLine:
start: int
end: int
text: str
active: bool
@dataclass(frozen=True)
class Heading:
start: int
level: int
text: str
style: str
raw: str
@dataclass(frozen=True)
class MarkdownScan:
lines: tuple[ScannedLine, ...]
headings: tuple[Heading, ...]
prose_text: str
operator_text: str
@dataclass(frozen=True)
class Region:
name: str
start: int
end: int
def contains(self, offset: int) -> bool:
return self.start <= offset < self.end
def _opening_fence(line: str) -> tuple[str, int] | None:
match = re.match(r"^ {0,3}(`{3,}|~{3,})(.*)$", line)
if not match:
return None
run, rest = match.groups()
if run[0] == "`" and "`" in rest:
return None
return run[0], len(run)
def _closes_fence(line: str, fence: tuple[str, int]) -> bool:
char, minimum = fence
return re.fullmatch(rf" {{0,3}}{re.escape(char)}{{{minimum},}}[ \t]*", line) is not None
def _atx_heading(line: ScannedLine) -> Heading | None:
if not line.active:
return None
match = re.fullmatch(r" {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*", line.text)
if not match:
return None
hashes, content = match.groups()
content = content or ""
content = re.sub(r"[ \t]+#+[ \t]*$", "", content).strip()
return Heading(line.start, len(hashes), content, "atx", line.text)
def _mask_html_comments(line: str, depth: int) -> tuple[str, int]:
"""Mask HTML comment ranges without changing raw character offsets."""
masked = list(line)
position = 0
while position < len(line):
if depth and line.startswith("--!>", position):
raise ContractError("alternate HTML comment closer is not allowed")
if line.startswith("<!--", position):
if depth:
raise ContractError("nested HTML comments are not allowed")
depth = 1
masked[position : position + 4] = " " * 4
position += 4
elif depth and line.startswith("-->", position):
masked[position : position + 3] = " " * 3
depth -= 1
position += 3
else:
if depth:
masked[position] = " "
position += 1
return "".join(masked), depth
def scan_active_markdown(text: str, label: str = "document") -> MarkdownScan:
"""Scan structural Markdown and retain the whole visible active document."""
lines: list[ScannedLine] = []
prose_chunks: list[str] = []
operator_chunks: list[str] = []
fence: tuple[str, int] | None = None
html_comment_depth = 0
offset = 0
for raw_line in text.splitlines(keepends=True):
line = raw_line.rstrip("\r\n")
ending = raw_line[len(line) :]
start = offset
offset += len(raw_line)
if fence is not None:
lines.append(ScannedLine(start, offset, line, False))
prose_chunks.append(" " * len(line) + ending)
if _closes_fence(line, fence):
operator_chunks.append(" " * len(line) + ending)
fence = None
else:
operator_chunks.append(line.translate(_CODE_PROTECT) + ending)
continue
if not html_comment_depth and re.match(r"^(?: {4}|\t)", line):
lines.append(ScannedLine(start, offset, line, False))
prose_chunks.append(" " * len(line) + ending)
operator_chunks.append(line.translate(_CODE_PROTECT) + ending)
continue
if not html_comment_depth:
opened_fence = _opening_fence(line)
if opened_fence is not None:
lines.append(ScannedLine(start, offset, line, False))
prose_chunks.append(" " * len(line) + ending)
operator_chunks.append(" " * len(line) + ending)
fence = opened_fence
continue
depth_before = html_comment_depth
visible, html_comment_depth = _mask_html_comments(line, html_comment_depth)
exact_marker = depth_before == 0 and _MARKER_LINE.fullmatch(line) is not None
misplaced_marker = depth_before == 0 and _MARKER_LITERAL.search(line) is not None
active = exact_marker or misplaced_marker or bool(visible.strip())
lines.append(ScannedLine(start, offset, line, active))
prose_chunks.append(visible + ending)
operator_chunks.append(visible + ending)
if html_comment_depth:
raise ContractError(f"{label}: unclosed HTML comment")
headings: list[Heading] = []
for index, line in enumerate(lines):
atx = _atx_heading(line)
if atx:
headings.append(atx)
continue
if not line.active or not line.text.strip() or _MARKER_LINE.fullmatch(line.text):
continue
if index + 1 >= len(lines) or not lines[index + 1].active:
continue
underline = re.fullmatch(r" {0,3}(=+|-+)[ \t]*", lines[index + 1].text)
if underline:
headings.append(
Heading(
line.start,
1 if underline.group(1)[0] == "=" else 2,
line.text.strip(),
"setext",
line.text,
)
)
headings.sort(key=lambda heading: heading.start)
prose_text = "".join(prose_chunks)
operator_text = "".join(operator_chunks)
if len(prose_text) != len(text) or len(operator_text) != len(text):
raise ContractError(f"{label}: Markdown representations changed raw offsets")
return MarkdownScan(tuple(lines), tuple(headings), prose_text, operator_text)
def _regions(text: str, label: str, scan: MarkdownScan) -> list[Region]:
events: list[tuple[ScannedLine, str, str]] = []
for line in scan.lines:
if not line.active:
continue
marker = _MARKER_LINE.fullmatch(line.text)
if marker:
events.append((line, marker.group(1), marker.group(2)))
elif _MARKER_LITERAL.search(line.text):
raise ContractError(
f"{label}: misplaced documentation region marker: {line.text.strip()}"
)
regions: list[Region] = []
stack: tuple[str, int] | None = None
for line, name, action in events:
if action == "start":
if stack is not None:
raise ContractError(f"{label}: documentation regions may not nest or overlap")
stack = (name, line.end)
continue
if stack is None:
raise ContractError(f"{label}: unmatched {name}:end marker")
open_name, content_start = stack
if open_name != name:
raise ContractError(
f"{label}: marker {name}:end closes active {open_name}:start region"
)
regions.append(Region(name, content_start, line.start))
stack = None
if stack is not None:
raise ContractError(f"{label}: unmatched {stack[0]}:start marker")
workspace_regions = [region for region in regions if region.name == WORKSPACE_REGION]
if len(workspace_regions) != 1:
raise ContractError(
f"{label}: expected exactly one {WORKSPACE_REGION}:start/end region, "
f"found {len(workspace_regions)}"
)
return regions
def _render_markdown(markdown: str, *, retain_code_text: bool = True) -> str:
"""Normalize one offset-preserving Markdown representation to rendered text."""
rendered = markdown
rendered = re.sub(r"(?m)^ {0,3}\[[^]\n]+\]:[^\n]*(?:\n|$)", "", rendered)
# Retain multiline code-span text while discarding a matching delimiter run.
def code_text(match: re.Match[str]) -> str:
content = re.sub(r"[\r\n]+", " ", match.group(2))
if content.startswith(" ") and content.endswith(" ") and content.strip():
content = content[1:-1]
return content
rendered = re.sub(
r"(?<!`)(`+)(?!`)(.*?)(?<!`)\1(?!`)",
code_text if retain_code_text else " ",
rendered,
flags=re.DOTALL,
)
rendered = re.sub(
r"!?\[([^]]*?)\]\([^)]*\)",
r"\1",
rendered,
flags=re.DOTALL,
)
rendered = re.sub(
r"!?\[([^]]*?)\]\s*\[[^]]*?\]",
r"\1",
rendered,
flags=re.DOTALL,
)
rendered = re.sub(r"\[([^]\n]+)\]", r"\1", rendered)
rendered = re.sub(
r"</?[A-Za-z][A-Za-z0-9-]*(?:\s[^<>]*?)?\s*/?>",
"",
rendered,
flags=re.DOTALL,
)
rendered = re.sub(
r"""\\([!"#$%&'()*+,\-./:;<=>?@\[\]\\^_`{|}~])""",
r"\1",
rendered,
)
rendered = rendered.replace("*", "").replace("~", "")
rendered = re.sub(r"(?<!\w)_{1,3}|_{1,3}(?!\w)", "", rendered)
normalized = re.sub(r"\s+", " ", html.unescape(rendered)).strip()
return normalized.translate(_CODE_RESTORE)
def _render_active_markdown(scan: MarkdownScan) -> str:
return _render_markdown(scan.operator_text)
def _render_prose_markdown(markdown: str) -> str:
return _render_markdown(markdown, retain_code_text=False)
def _exact_sentence_count(normalized: str, sentence: str) -> int:
pattern = re.compile(
rf"(?:^|(?<=[.!?]) )(?={re.escape(sentence)}(?:$| ))"
)
return len(pattern.findall(normalized))
def _mask_region_ranges(text: str, regions: list[Region]) -> str:
masked = list(text)
for region in regions:
for offset in range(region.start, region.end):
if masked[offset] not in "\r\n":
masked[offset] = " "
return "".join(masked)
def check_document_text(text: str, label: str = "document") -> None:
"""Validate one current (non-historical) document."""
scan = scan_active_markdown(text, label)
regions = _regions(text, label, scan)
workspace = next(region for region in regions if region.name == WORKSPACE_REGION)
workspace_prose = _render_prose_markdown(
scan.prose_text[workspace.start : workspace.end]
)
if _exact_sentence_count(workspace_prose, CANONICAL_V4_SENTENCE) != 1:
raise ContractError(f"{label}: workspace contract lacks the canonical v4-only sentence")
if _exact_sentence_count(workspace_prose, CANONICAL_REJECTION_SENTENCE) != 1:
raise ContractError(
f"{label}: workspace contract lacks the canonical v1/v2/v3 rejection sentence"
)
workspace_visible = _render_markdown(scan.operator_text[workspace.start : workspace.end])
workspace_remainder = workspace_visible.replace(CANONICAL_REJECTION_SENTENCE, " ", 1)
forbidden_claim = _MIGRATION_REQUIRED.search(
workspace_remainder
) or _REMAINING_LEGACY_CLAIM.search(workspace_remainder)
if forbidden_claim:
raise ContractError(
f"{label}: workspace contract contains an extra legacy descriptor claim: "
f"{forbidden_claim.group(0)}"
)
rendered_operator_text = _render_active_markdown(scan)
deleted = _DELETED_TOOL.search(rendered_operator_text)
if deleted:
raise ContractError(
f"{label}: current text contains deleted tool instruction: {deleted.group(0)}"
)
non_workspace = [region for region in regions if region.name == NON_WORKSPACE_REGION]
outside_allowed = _mask_region_ranges(
scan.operator_text,
[workspace, *non_workspace],
)
rendered_outside = _render_markdown(outside_allowed)
outside_claim = _MIGRATION_REQUIRED.search(
rendered_outside
) or _OUTSIDE_LEGACY_REFERENCE.search(rendered_outside)
if outside_claim:
raise ContractError(
f"{label}: current legacy schema or migration reference outside a "
f"{NON_WORKSPACE_REGION}:start/end region: {outside_claim.group(0)}"
)
def check_project_state_text(text: str, label: str = "PROJECT_STATE.md") -> None:
"""Validate current PROJECT_STATE text and its terminal historical archive hierarchy."""
scan = scan_active_markdown(text, label)
archive_headings = [
heading
for heading in scan.headings
if heading.level == 1
and heading.style == "atx"
and heading.raw == HISTORICAL_ARCHIVE_H1
]
if len(archive_headings) != 1:
raise ContractError(
f"{label}: expected exactly one active terminal '{HISTORICAL_ARCHIVE_H1}' boundary"
)
archive_heading = archive_headings[0]
archive_line_index = next(
index for index, line in enumerate(scan.lines) if line.start == archive_heading.start
)
if archive_line_index + 1 >= len(scan.lines) or not (
scan.lines[archive_line_index + 1].active
and scan.lines[archive_line_index + 1].text
== "## Historical snapshots and archived reference notes"
):
raise ContractError(
f"{label}: the preserved Historical snapshots H2 must immediately follow "
f"'{HISTORICAL_ARCHIVE_H1}'"
)
later_headings = [heading for heading in scan.headings if heading.start > archive_heading.start]
if not later_headings or not (
later_headings[0].level == 2
and later_headings[0].style == "atx"
and later_headings[0].raw == "## Historical snapshots and archived reference notes"
):
raise ContractError(
f"{label}: Historical archive must be followed by the existing Historical snapshots H2"
)
if any(heading.level == 1 for heading in later_headings):
raise ContractError(f"{label}: Historical archive must be the terminal H1 hierarchy")
check_document_text(text[:archive_heading.start], f"{label} current section")
def check_document_path(path: Path) -> None:
check_document_text(path.read_text(), str(path))
def check_project_state_path(path: Path) -> None:
check_project_state_text(path.read_text(), str(path))
def _parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--document", action="append", default=[], type=Path)
parser.add_argument("--project-state", action="append", default=[], type=Path)
return parser
def main(argv: Sequence[str] | None = None) -> int:
args = _parser().parse_args(argv)
if not args.document and not args.project_state:
raise SystemExit("at least one --document or --project-state path is required")
try:
for path in args.document:
check_document_path(path)
for path in args.project_state:
check_project_state_path(path)
except (ContractError, OSError) as error:
print(error, file=sys.stderr)
return 1
return 0
if __name__ == "__main__":
raise SystemExit(main())