475 lines
16 KiB
Python
Executable File
475 lines
16 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Machine-checkable policy for current workspace-descriptor documentation."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import html
|
|
import re
|
|
import sys
|
|
from collections.abc import Sequence
|
|
from dataclasses import dataclass
|
|
from pathlib import Path
|
|
|
|
WORKSPACE_REGION = "workspace-descriptor-contract"
|
|
NON_WORKSPACE_REGION = "non-workspace-migration"
|
|
HISTORICAL_ARCHIVE_H1 = "# Historical archive"
|
|
CANONICAL_V4_SENTENCE = "Schema v4 is the only accepted workspace descriptor."
|
|
CANONICAL_REJECTION_SENTENCE = (
|
|
"Schema v1, v2, and v3 workspace descriptors are rejected before activation."
|
|
)
|
|
|
|
_MARKER_LINE = re.compile(
|
|
r"<!-- (workspace-descriptor-contract|non-workspace-migration):(start|end) -->"
|
|
)
|
|
_MARKER_LITERAL = re.compile(
|
|
r"(?:workspace-descriptor-contract|non-workspace-migration):(start|end)"
|
|
)
|
|
_MIGRATION_REQUIRED = re.compile(r"migration_required", re.IGNORECASE)
|
|
_DELETED_TOOL = re.compile(
|
|
r"migrate\s*-\s*legacy|"
|
|
r"legacy[-\s]+(?:(?:workspace[-\s]+)?descriptor[-\s]+)?(?:transformer|migrator)|"
|
|
r"(?:workspace[-\s]+)?descriptor[-\s]+legacy[-\s]+(?:transformer|migrator)|"
|
|
r"migrat(?:e|ing)[-\s]+legacy[-\s]+descriptors?",
|
|
re.IGNORECASE,
|
|
)
|
|
_REMAINING_LEGACY_CLAIM = re.compile(
|
|
r"\b(?:schema(?:[- ]v?|\s+version\s*)[123]|"
|
|
r"schema_version\s*[:=]\s*[123]|version\s*[123]|v[123]|"
|
|
r"legacy(?:[-\s]+workspace)?[-\s]+descriptors?)\b",
|
|
re.IGNORECASE,
|
|
)
|
|
_CODE_LITERAL_CHARS = "\\`*_~[]()<>!&"
|
|
_CODE_PROTECT = str.maketrans(
|
|
{character: chr(0xE000 + index) for index, character in enumerate(_CODE_LITERAL_CHARS)}
|
|
)
|
|
_CODE_RESTORE = str.maketrans(
|
|
{chr(0xE000 + index): character for index, character in enumerate(_CODE_LITERAL_CHARS)}
|
|
)
|
|
|
|
_OUTSIDE_LEGACY_REFERENCE = re.compile(
|
|
r"\b(?:schema(?:[- ]v?|\s+version\s*)[123]|"
|
|
r"schema_version\s*[:=]\s*[123]|"
|
|
r"(?:v[123](?:\s*(?:/|and|or|,)\s*v[123])*)\s+(?:workspace\s+)?descriptors?|"
|
|
r"legacy(?:[-\s]+workspace)?[-\s]+descriptors?)\b",
|
|
re.IGNORECASE,
|
|
)
|
|
|
|
|
|
class ContractError(ValueError):
|
|
"""Raised when current documentation violates the descriptor-region policy."""
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class ScannedLine:
|
|
start: int
|
|
end: int
|
|
text: str
|
|
active: bool
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Heading:
|
|
start: int
|
|
level: int
|
|
text: str
|
|
style: str
|
|
raw: str
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class MarkdownScan:
|
|
lines: tuple[ScannedLine, ...]
|
|
headings: tuple[Heading, ...]
|
|
prose_text: str
|
|
operator_text: str
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Region:
|
|
name: str
|
|
start: int
|
|
end: int
|
|
|
|
def contains(self, offset: int) -> bool:
|
|
return self.start <= offset < self.end
|
|
|
|
|
|
def _opening_fence(line: str) -> tuple[str, int] | None:
|
|
match = re.match(r"^ {0,3}(`{3,}|~{3,})(.*)$", line)
|
|
if not match:
|
|
return None
|
|
run, rest = match.groups()
|
|
if run[0] == "`" and "`" in rest:
|
|
return None
|
|
return run[0], len(run)
|
|
|
|
|
|
def _closes_fence(line: str, fence: tuple[str, int]) -> bool:
|
|
char, minimum = fence
|
|
return re.fullmatch(rf" {{0,3}}{re.escape(char)}{{{minimum},}}[ \t]*", line) is not None
|
|
|
|
|
|
def _atx_heading(line: ScannedLine) -> Heading | None:
|
|
if not line.active:
|
|
return None
|
|
match = re.fullmatch(r" {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*", line.text)
|
|
if not match:
|
|
return None
|
|
hashes, content = match.groups()
|
|
content = content or ""
|
|
content = re.sub(r"[ \t]+#+[ \t]*$", "", content).strip()
|
|
return Heading(line.start, len(hashes), content, "atx", line.text)
|
|
|
|
|
|
def _mask_html_comments(line: str, depth: int) -> tuple[str, int]:
|
|
"""Mask HTML comment ranges without changing raw character offsets."""
|
|
|
|
masked = list(line)
|
|
position = 0
|
|
while position < len(line):
|
|
if depth and line.startswith("--!>", position):
|
|
raise ContractError("alternate HTML comment closer is not allowed")
|
|
if line.startswith("<!--", position):
|
|
if depth:
|
|
raise ContractError("nested HTML comments are not allowed")
|
|
depth = 1
|
|
masked[position : position + 4] = " " * 4
|
|
position += 4
|
|
elif depth and line.startswith("-->", position):
|
|
masked[position : position + 3] = " " * 3
|
|
depth -= 1
|
|
position += 3
|
|
else:
|
|
if depth:
|
|
masked[position] = " "
|
|
position += 1
|
|
return "".join(masked), depth
|
|
|
|
|
|
def scan_active_markdown(text: str, label: str = "document") -> MarkdownScan:
|
|
"""Scan structural Markdown and retain the whole visible active document."""
|
|
|
|
lines: list[ScannedLine] = []
|
|
prose_chunks: list[str] = []
|
|
operator_chunks: list[str] = []
|
|
fence: tuple[str, int] | None = None
|
|
html_comment_depth = 0
|
|
offset = 0
|
|
|
|
for raw_line in text.splitlines(keepends=True):
|
|
line = raw_line.rstrip("\r\n")
|
|
ending = raw_line[len(line) :]
|
|
start = offset
|
|
offset += len(raw_line)
|
|
|
|
if fence is not None:
|
|
lines.append(ScannedLine(start, offset, line, False))
|
|
prose_chunks.append(" " * len(line) + ending)
|
|
if _closes_fence(line, fence):
|
|
operator_chunks.append(" " * len(line) + ending)
|
|
fence = None
|
|
else:
|
|
operator_chunks.append(line.translate(_CODE_PROTECT) + ending)
|
|
continue
|
|
|
|
if not html_comment_depth and re.match(r"^(?: {4}|\t)", line):
|
|
lines.append(ScannedLine(start, offset, line, False))
|
|
prose_chunks.append(" " * len(line) + ending)
|
|
operator_chunks.append(line.translate(_CODE_PROTECT) + ending)
|
|
continue
|
|
|
|
if not html_comment_depth:
|
|
opened_fence = _opening_fence(line)
|
|
if opened_fence is not None:
|
|
lines.append(ScannedLine(start, offset, line, False))
|
|
prose_chunks.append(" " * len(line) + ending)
|
|
operator_chunks.append(" " * len(line) + ending)
|
|
fence = opened_fence
|
|
continue
|
|
|
|
depth_before = html_comment_depth
|
|
visible, html_comment_depth = _mask_html_comments(line, html_comment_depth)
|
|
exact_marker = depth_before == 0 and _MARKER_LINE.fullmatch(line) is not None
|
|
misplaced_marker = depth_before == 0 and _MARKER_LITERAL.search(line) is not None
|
|
active = exact_marker or misplaced_marker or bool(visible.strip())
|
|
lines.append(ScannedLine(start, offset, line, active))
|
|
prose_chunks.append(visible + ending)
|
|
operator_chunks.append(visible + ending)
|
|
|
|
if html_comment_depth:
|
|
raise ContractError(f"{label}: unclosed HTML comment")
|
|
|
|
headings: list[Heading] = []
|
|
for index, line in enumerate(lines):
|
|
atx = _atx_heading(line)
|
|
if atx:
|
|
headings.append(atx)
|
|
continue
|
|
if not line.active or not line.text.strip() or _MARKER_LINE.fullmatch(line.text):
|
|
continue
|
|
if index + 1 >= len(lines) or not lines[index + 1].active:
|
|
continue
|
|
underline = re.fullmatch(r" {0,3}(=+|-+)[ \t]*", lines[index + 1].text)
|
|
if underline:
|
|
headings.append(
|
|
Heading(
|
|
line.start,
|
|
1 if underline.group(1)[0] == "=" else 2,
|
|
line.text.strip(),
|
|
"setext",
|
|
line.text,
|
|
)
|
|
)
|
|
|
|
headings.sort(key=lambda heading: heading.start)
|
|
prose_text = "".join(prose_chunks)
|
|
operator_text = "".join(operator_chunks)
|
|
if len(prose_text) != len(text) or len(operator_text) != len(text):
|
|
raise ContractError(f"{label}: Markdown representations changed raw offsets")
|
|
return MarkdownScan(tuple(lines), tuple(headings), prose_text, operator_text)
|
|
|
|
|
|
def _regions(text: str, label: str, scan: MarkdownScan) -> list[Region]:
|
|
events: list[tuple[ScannedLine, str, str]] = []
|
|
for line in scan.lines:
|
|
if not line.active:
|
|
continue
|
|
marker = _MARKER_LINE.fullmatch(line.text)
|
|
if marker:
|
|
events.append((line, marker.group(1), marker.group(2)))
|
|
elif _MARKER_LITERAL.search(line.text):
|
|
raise ContractError(
|
|
f"{label}: misplaced documentation region marker: {line.text.strip()}"
|
|
)
|
|
|
|
regions: list[Region] = []
|
|
stack: tuple[str, int] | None = None
|
|
for line, name, action in events:
|
|
if action == "start":
|
|
if stack is not None:
|
|
raise ContractError(f"{label}: documentation regions may not nest or overlap")
|
|
stack = (name, line.end)
|
|
continue
|
|
if stack is None:
|
|
raise ContractError(f"{label}: unmatched {name}:end marker")
|
|
open_name, content_start = stack
|
|
if open_name != name:
|
|
raise ContractError(
|
|
f"{label}: marker {name}:end closes active {open_name}:start region"
|
|
)
|
|
regions.append(Region(name, content_start, line.start))
|
|
stack = None
|
|
if stack is not None:
|
|
raise ContractError(f"{label}: unmatched {stack[0]}:start marker")
|
|
|
|
workspace_regions = [region for region in regions if region.name == WORKSPACE_REGION]
|
|
if len(workspace_regions) != 1:
|
|
raise ContractError(
|
|
f"{label}: expected exactly one {WORKSPACE_REGION}:start/end region, "
|
|
f"found {len(workspace_regions)}"
|
|
)
|
|
return regions
|
|
|
|
|
|
def _render_markdown(markdown: str, *, retain_code_text: bool = True) -> str:
|
|
"""Normalize one offset-preserving Markdown representation to rendered text."""
|
|
|
|
rendered = markdown
|
|
rendered = re.sub(r"(?m)^ {0,3}\[[^]\n]+\]:[^\n]*(?:\n|$)", "", rendered)
|
|
|
|
# Retain multiline code-span text while discarding a matching delimiter run.
|
|
def code_text(match: re.Match[str]) -> str:
|
|
content = re.sub(r"[\r\n]+", " ", match.group(2))
|
|
if content.startswith(" ") and content.endswith(" ") and content.strip():
|
|
content = content[1:-1]
|
|
return content
|
|
|
|
rendered = re.sub(
|
|
r"(?<!`)(`+)(?!`)(.*?)(?<!`)\1(?!`)",
|
|
code_text if retain_code_text else " ",
|
|
rendered,
|
|
flags=re.DOTALL,
|
|
)
|
|
rendered = re.sub(
|
|
r"!?\[([^]]*?)\]\([^)]*\)",
|
|
r"\1",
|
|
rendered,
|
|
flags=re.DOTALL,
|
|
)
|
|
rendered = re.sub(
|
|
r"!?\[([^]]*?)\]\s*\[[^]]*?\]",
|
|
r"\1",
|
|
rendered,
|
|
flags=re.DOTALL,
|
|
)
|
|
rendered = re.sub(r"\[([^]\n]+)\]", r"\1", rendered)
|
|
rendered = re.sub(
|
|
r"</?[A-Za-z][A-Za-z0-9-]*(?:\s[^<>]*?)?\s*/?>",
|
|
"",
|
|
rendered,
|
|
flags=re.DOTALL,
|
|
)
|
|
rendered = re.sub(
|
|
r"""\\([!"#$%&'()*+,\-./:;<=>?@\[\]\\^_`{|}~])""",
|
|
r"\1",
|
|
rendered,
|
|
)
|
|
rendered = rendered.replace("*", "").replace("~", "")
|
|
rendered = re.sub(r"(?<!\w)_{1,3}|_{1,3}(?!\w)", "", rendered)
|
|
normalized = re.sub(r"\s+", " ", html.unescape(rendered)).strip()
|
|
return normalized.translate(_CODE_RESTORE)
|
|
|
|
|
|
def _render_active_markdown(scan: MarkdownScan) -> str:
|
|
return _render_markdown(scan.operator_text)
|
|
|
|
|
|
def _render_prose_markdown(markdown: str) -> str:
|
|
return _render_markdown(markdown, retain_code_text=False)
|
|
|
|
|
|
def _exact_sentence_count(normalized: str, sentence: str) -> int:
|
|
pattern = re.compile(
|
|
rf"(?:^|(?<=[.!?]) )(?={re.escape(sentence)}(?:$| ))"
|
|
)
|
|
return len(pattern.findall(normalized))
|
|
|
|
|
|
def _mask_region_ranges(text: str, regions: list[Region]) -> str:
|
|
masked = list(text)
|
|
for region in regions:
|
|
for offset in range(region.start, region.end):
|
|
if masked[offset] not in "\r\n":
|
|
masked[offset] = " "
|
|
return "".join(masked)
|
|
|
|
|
|
def check_document_text(text: str, label: str = "document") -> None:
|
|
"""Validate one current (non-historical) document."""
|
|
|
|
scan = scan_active_markdown(text, label)
|
|
regions = _regions(text, label, scan)
|
|
workspace = next(region for region in regions if region.name == WORKSPACE_REGION)
|
|
|
|
workspace_prose = _render_prose_markdown(
|
|
scan.prose_text[workspace.start : workspace.end]
|
|
)
|
|
if _exact_sentence_count(workspace_prose, CANONICAL_V4_SENTENCE) != 1:
|
|
raise ContractError(f"{label}: workspace contract lacks the canonical v4-only sentence")
|
|
if _exact_sentence_count(workspace_prose, CANONICAL_REJECTION_SENTENCE) != 1:
|
|
raise ContractError(
|
|
f"{label}: workspace contract lacks the canonical v1/v2/v3 rejection sentence"
|
|
)
|
|
|
|
workspace_visible = _render_markdown(scan.operator_text[workspace.start : workspace.end])
|
|
workspace_remainder = workspace_visible.replace(CANONICAL_REJECTION_SENTENCE, " ", 1)
|
|
forbidden_claim = _MIGRATION_REQUIRED.search(
|
|
workspace_remainder
|
|
) or _REMAINING_LEGACY_CLAIM.search(workspace_remainder)
|
|
if forbidden_claim:
|
|
raise ContractError(
|
|
f"{label}: workspace contract contains an extra legacy descriptor claim: "
|
|
f"{forbidden_claim.group(0)}"
|
|
)
|
|
|
|
rendered_operator_text = _render_active_markdown(scan)
|
|
deleted = _DELETED_TOOL.search(rendered_operator_text)
|
|
if deleted:
|
|
raise ContractError(
|
|
f"{label}: current text contains deleted tool instruction: {deleted.group(0)}"
|
|
)
|
|
|
|
non_workspace = [region for region in regions if region.name == NON_WORKSPACE_REGION]
|
|
outside_allowed = _mask_region_ranges(
|
|
scan.operator_text,
|
|
[workspace, *non_workspace],
|
|
)
|
|
rendered_outside = _render_markdown(outside_allowed)
|
|
outside_claim = _MIGRATION_REQUIRED.search(
|
|
rendered_outside
|
|
) or _OUTSIDE_LEGACY_REFERENCE.search(rendered_outside)
|
|
if outside_claim:
|
|
raise ContractError(
|
|
f"{label}: current legacy schema or migration reference outside a "
|
|
f"{NON_WORKSPACE_REGION}:start/end region: {outside_claim.group(0)}"
|
|
)
|
|
|
|
|
|
def check_project_state_text(text: str, label: str = "PROJECT_STATE.md") -> None:
|
|
"""Validate current PROJECT_STATE text and its terminal historical archive hierarchy."""
|
|
|
|
scan = scan_active_markdown(text, label)
|
|
archive_headings = [
|
|
heading
|
|
for heading in scan.headings
|
|
if heading.level == 1
|
|
and heading.style == "atx"
|
|
and heading.raw == HISTORICAL_ARCHIVE_H1
|
|
]
|
|
if len(archive_headings) != 1:
|
|
raise ContractError(
|
|
f"{label}: expected exactly one active terminal '{HISTORICAL_ARCHIVE_H1}' boundary"
|
|
)
|
|
archive_heading = archive_headings[0]
|
|
archive_line_index = next(
|
|
index for index, line in enumerate(scan.lines) if line.start == archive_heading.start
|
|
)
|
|
if archive_line_index + 1 >= len(scan.lines) or not (
|
|
scan.lines[archive_line_index + 1].active
|
|
and scan.lines[archive_line_index + 1].text
|
|
== "## Historical snapshots and archived reference notes"
|
|
):
|
|
raise ContractError(
|
|
f"{label}: the preserved Historical snapshots H2 must immediately follow "
|
|
f"'{HISTORICAL_ARCHIVE_H1}'"
|
|
)
|
|
|
|
later_headings = [heading for heading in scan.headings if heading.start > archive_heading.start]
|
|
if not later_headings or not (
|
|
later_headings[0].level == 2
|
|
and later_headings[0].style == "atx"
|
|
and later_headings[0].raw == "## Historical snapshots and archived reference notes"
|
|
):
|
|
raise ContractError(
|
|
f"{label}: Historical archive must be followed by the existing Historical snapshots H2"
|
|
)
|
|
if any(heading.level == 1 for heading in later_headings):
|
|
raise ContractError(f"{label}: Historical archive must be the terminal H1 hierarchy")
|
|
|
|
check_document_text(text[:archive_heading.start], f"{label} current section")
|
|
|
|
|
|
def check_document_path(path: Path) -> None:
|
|
check_document_text(path.read_text(), str(path))
|
|
|
|
|
|
def check_project_state_path(path: Path) -> None:
|
|
check_project_state_text(path.read_text(), str(path))
|
|
|
|
|
|
def _parser() -> argparse.ArgumentParser:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("--document", action="append", default=[], type=Path)
|
|
parser.add_argument("--project-state", action="append", default=[], type=Path)
|
|
return parser
|
|
|
|
|
|
def main(argv: Sequence[str] | None = None) -> int:
|
|
args = _parser().parse_args(argv)
|
|
if not args.document and not args.project_state:
|
|
raise SystemExit("at least one --document or --project-state path is required")
|
|
try:
|
|
for path in args.document:
|
|
check_document_path(path)
|
|
for path in args.project_state:
|
|
check_project_state_path(path)
|
|
except (ContractError, OSError) as error:
|
|
print(error, file=sys.stderr)
|
|
return 1
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|