docs: make schema v3 the only workspace contract
This commit is contained in:
Executable
+474
@@ -0,0 +1,474 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Machine-checkable policy for current workspace-descriptor documentation."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import html
|
||||
import re
|
||||
import sys
|
||||
from collections.abc import Sequence
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
WORKSPACE_REGION = "workspace-descriptor-contract"
|
||||
NON_WORKSPACE_REGION = "non-workspace-migration"
|
||||
HISTORICAL_ARCHIVE_H1 = "# Historical archive"
|
||||
CANONICAL_V3_SENTENCE = "Schema v3 is the only accepted workspace descriptor."
|
||||
CANONICAL_REJECTION_SENTENCE = (
|
||||
"Schema v1 and v2 workspace descriptors are rejected before activation."
|
||||
)
|
||||
|
||||
_MARKER_LINE = re.compile(
|
||||
r"<!-- (workspace-descriptor-contract|non-workspace-migration):(start|end) -->"
|
||||
)
|
||||
_MARKER_LITERAL = re.compile(
|
||||
r"(?:workspace-descriptor-contract|non-workspace-migration):(start|end)"
|
||||
)
|
||||
_MIGRATION_REQUIRED = re.compile(r"migration_required", re.IGNORECASE)
|
||||
_DELETED_TOOL = re.compile(
|
||||
r"migrate\s*-\s*legacy|"
|
||||
r"legacy[-\s]+(?:(?:workspace[-\s]+)?descriptor[-\s]+)?(?:transformer|migrator)|"
|
||||
r"(?:workspace[-\s]+)?descriptor[-\s]+legacy[-\s]+(?:transformer|migrator)|"
|
||||
r"migrat(?:e|ing)[-\s]+legacy[-\s]+descriptors?",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_REMAINING_LEGACY_CLAIM = re.compile(
|
||||
r"\b(?:schema(?:[- ]v?|\s+version\s*)[12]|"
|
||||
r"schema_version\s*[:=]\s*[12]|version\s*[12]|v[12]|"
|
||||
r"legacy(?:[-\s]+workspace)?[-\s]+descriptors?)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_CODE_LITERAL_CHARS = "\\`*_~[]()<>!&"
|
||||
_CODE_PROTECT = str.maketrans(
|
||||
{character: chr(0xE000 + index) for index, character in enumerate(_CODE_LITERAL_CHARS)}
|
||||
)
|
||||
_CODE_RESTORE = str.maketrans(
|
||||
{chr(0xE000 + index): character for index, character in enumerate(_CODE_LITERAL_CHARS)}
|
||||
)
|
||||
|
||||
_OUTSIDE_LEGACY_REFERENCE = re.compile(
|
||||
r"\b(?:schema(?:[- ]v?|\s+version\s*)[12]|"
|
||||
r"schema_version\s*[:=]\s*[12]|"
|
||||
r"(?:v[12](?:\s*(?:/|and|or)\s*v[12])?)\s+(?:workspace\s+)?descriptors?|"
|
||||
r"legacy(?:[-\s]+workspace)?[-\s]+descriptors?)\b",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
class ContractError(ValueError):
|
||||
"""Raised when current documentation violates the descriptor-region policy."""
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ScannedLine:
|
||||
start: int
|
||||
end: int
|
||||
text: str
|
||||
active: bool
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Heading:
|
||||
start: int
|
||||
level: int
|
||||
text: str
|
||||
style: str
|
||||
raw: str
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MarkdownScan:
|
||||
lines: tuple[ScannedLine, ...]
|
||||
headings: tuple[Heading, ...]
|
||||
prose_text: str
|
||||
operator_text: str
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Region:
|
||||
name: str
|
||||
start: int
|
||||
end: int
|
||||
|
||||
def contains(self, offset: int) -> bool:
|
||||
return self.start <= offset < self.end
|
||||
|
||||
|
||||
def _opening_fence(line: str) -> tuple[str, int] | None:
|
||||
match = re.match(r"^ {0,3}(`{3,}|~{3,})(.*)$", line)
|
||||
if not match:
|
||||
return None
|
||||
run, rest = match.groups()
|
||||
if run[0] == "`" and "`" in rest:
|
||||
return None
|
||||
return run[0], len(run)
|
||||
|
||||
|
||||
def _closes_fence(line: str, fence: tuple[str, int]) -> bool:
|
||||
char, minimum = fence
|
||||
return re.fullmatch(rf" {{0,3}}{re.escape(char)}{{{minimum},}}[ \t]*", line) is not None
|
||||
|
||||
|
||||
def _atx_heading(line: ScannedLine) -> Heading | None:
|
||||
if not line.active:
|
||||
return None
|
||||
match = re.fullmatch(r" {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*", line.text)
|
||||
if not match:
|
||||
return None
|
||||
hashes, content = match.groups()
|
||||
content = content or ""
|
||||
content = re.sub(r"[ \t]+#+[ \t]*$", "", content).strip()
|
||||
return Heading(line.start, len(hashes), content, "atx", line.text)
|
||||
|
||||
|
||||
def _mask_html_comments(line: str, depth: int) -> tuple[str, int]:
|
||||
"""Mask HTML comment ranges without changing raw character offsets."""
|
||||
|
||||
masked = list(line)
|
||||
position = 0
|
||||
while position < len(line):
|
||||
if depth and line.startswith("--!>", position):
|
||||
raise ContractError("alternate HTML comment closer is not allowed")
|
||||
if line.startswith("<!--", position):
|
||||
if depth:
|
||||
raise ContractError("nested HTML comments are not allowed")
|
||||
depth = 1
|
||||
masked[position : position + 4] = " " * 4
|
||||
position += 4
|
||||
elif depth and line.startswith("-->", position):
|
||||
masked[position : position + 3] = " " * 3
|
||||
depth -= 1
|
||||
position += 3
|
||||
else:
|
||||
if depth:
|
||||
masked[position] = " "
|
||||
position += 1
|
||||
return "".join(masked), depth
|
||||
|
||||
|
||||
def scan_active_markdown(text: str, label: str = "document") -> MarkdownScan:
|
||||
"""Scan structural Markdown and retain the whole visible active document."""
|
||||
|
||||
lines: list[ScannedLine] = []
|
||||
prose_chunks: list[str] = []
|
||||
operator_chunks: list[str] = []
|
||||
fence: tuple[str, int] | None = None
|
||||
html_comment_depth = 0
|
||||
offset = 0
|
||||
|
||||
for raw_line in text.splitlines(keepends=True):
|
||||
line = raw_line.rstrip("\r\n")
|
||||
ending = raw_line[len(line) :]
|
||||
start = offset
|
||||
offset += len(raw_line)
|
||||
|
||||
if fence is not None:
|
||||
lines.append(ScannedLine(start, offset, line, False))
|
||||
prose_chunks.append(" " * len(line) + ending)
|
||||
if _closes_fence(line, fence):
|
||||
operator_chunks.append(" " * len(line) + ending)
|
||||
fence = None
|
||||
else:
|
||||
operator_chunks.append(line.translate(_CODE_PROTECT) + ending)
|
||||
continue
|
||||
|
||||
if not html_comment_depth and re.match(r"^(?: {4}|\t)", line):
|
||||
lines.append(ScannedLine(start, offset, line, False))
|
||||
prose_chunks.append(" " * len(line) + ending)
|
||||
operator_chunks.append(line.translate(_CODE_PROTECT) + ending)
|
||||
continue
|
||||
|
||||
if not html_comment_depth:
|
||||
opened_fence = _opening_fence(line)
|
||||
if opened_fence is not None:
|
||||
lines.append(ScannedLine(start, offset, line, False))
|
||||
prose_chunks.append(" " * len(line) + ending)
|
||||
operator_chunks.append(" " * len(line) + ending)
|
||||
fence = opened_fence
|
||||
continue
|
||||
|
||||
depth_before = html_comment_depth
|
||||
visible, html_comment_depth = _mask_html_comments(line, html_comment_depth)
|
||||
exact_marker = depth_before == 0 and _MARKER_LINE.fullmatch(line) is not None
|
||||
misplaced_marker = depth_before == 0 and _MARKER_LITERAL.search(line) is not None
|
||||
active = exact_marker or misplaced_marker or bool(visible.strip())
|
||||
lines.append(ScannedLine(start, offset, line, active))
|
||||
prose_chunks.append(visible + ending)
|
||||
operator_chunks.append(visible + ending)
|
||||
|
||||
if html_comment_depth:
|
||||
raise ContractError(f"{label}: unclosed HTML comment")
|
||||
|
||||
headings: list[Heading] = []
|
||||
for index, line in enumerate(lines):
|
||||
atx = _atx_heading(line)
|
||||
if atx:
|
||||
headings.append(atx)
|
||||
continue
|
||||
if not line.active or not line.text.strip() or _MARKER_LINE.fullmatch(line.text):
|
||||
continue
|
||||
if index + 1 >= len(lines) or not lines[index + 1].active:
|
||||
continue
|
||||
underline = re.fullmatch(r" {0,3}(=+|-+)[ \t]*", lines[index + 1].text)
|
||||
if underline:
|
||||
headings.append(
|
||||
Heading(
|
||||
line.start,
|
||||
1 if underline.group(1)[0] == "=" else 2,
|
||||
line.text.strip(),
|
||||
"setext",
|
||||
line.text,
|
||||
)
|
||||
)
|
||||
|
||||
headings.sort(key=lambda heading: heading.start)
|
||||
prose_text = "".join(prose_chunks)
|
||||
operator_text = "".join(operator_chunks)
|
||||
if len(prose_text) != len(text) or len(operator_text) != len(text):
|
||||
raise ContractError(f"{label}: Markdown representations changed raw offsets")
|
||||
return MarkdownScan(tuple(lines), tuple(headings), prose_text, operator_text)
|
||||
|
||||
|
||||
def _regions(text: str, label: str, scan: MarkdownScan) -> list[Region]:
|
||||
events: list[tuple[ScannedLine, str, str]] = []
|
||||
for line in scan.lines:
|
||||
if not line.active:
|
||||
continue
|
||||
marker = _MARKER_LINE.fullmatch(line.text)
|
||||
if marker:
|
||||
events.append((line, marker.group(1), marker.group(2)))
|
||||
elif _MARKER_LITERAL.search(line.text):
|
||||
raise ContractError(
|
||||
f"{label}: misplaced documentation region marker: {line.text.strip()}"
|
||||
)
|
||||
|
||||
regions: list[Region] = []
|
||||
stack: tuple[str, int] | None = None
|
||||
for line, name, action in events:
|
||||
if action == "start":
|
||||
if stack is not None:
|
||||
raise ContractError(f"{label}: documentation regions may not nest or overlap")
|
||||
stack = (name, line.end)
|
||||
continue
|
||||
if stack is None:
|
||||
raise ContractError(f"{label}: unmatched {name}:end marker")
|
||||
open_name, content_start = stack
|
||||
if open_name != name:
|
||||
raise ContractError(
|
||||
f"{label}: marker {name}:end closes active {open_name}:start region"
|
||||
)
|
||||
regions.append(Region(name, content_start, line.start))
|
||||
stack = None
|
||||
if stack is not None:
|
||||
raise ContractError(f"{label}: unmatched {stack[0]}:start marker")
|
||||
|
||||
workspace_regions = [region for region in regions if region.name == WORKSPACE_REGION]
|
||||
if len(workspace_regions) != 1:
|
||||
raise ContractError(
|
||||
f"{label}: expected exactly one {WORKSPACE_REGION}:start/end region, "
|
||||
f"found {len(workspace_regions)}"
|
||||
)
|
||||
return regions
|
||||
|
||||
|
||||
def _render_markdown(markdown: str, *, retain_code_text: bool = True) -> str:
|
||||
"""Normalize one offset-preserving Markdown representation to rendered text."""
|
||||
|
||||
rendered = markdown
|
||||
rendered = re.sub(r"(?m)^ {0,3}\[[^]\n]+\]:[^\n]*(?:\n|$)", "", rendered)
|
||||
|
||||
# Retain multiline code-span text while discarding a matching delimiter run.
|
||||
def code_text(match: re.Match[str]) -> str:
|
||||
content = re.sub(r"[\r\n]+", " ", match.group(2))
|
||||
if content.startswith(" ") and content.endswith(" ") and content.strip():
|
||||
content = content[1:-1]
|
||||
return content
|
||||
|
||||
rendered = re.sub(
|
||||
r"(?<!`)(`+)(?!`)(.*?)(?<!`)\1(?!`)",
|
||||
code_text if retain_code_text else " ",
|
||||
rendered,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
rendered = re.sub(
|
||||
r"!?\[([^]]*?)\]\([^)]*\)",
|
||||
r"\1",
|
||||
rendered,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
rendered = re.sub(
|
||||
r"!?\[([^]]*?)\]\s*\[[^]]*?\]",
|
||||
r"\1",
|
||||
rendered,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
rendered = re.sub(r"\[([^]\n]+)\]", r"\1", rendered)
|
||||
rendered = re.sub(
|
||||
r"</?[A-Za-z][A-Za-z0-9-]*(?:\s[^<>]*?)?\s*/?>",
|
||||
"",
|
||||
rendered,
|
||||
flags=re.DOTALL,
|
||||
)
|
||||
rendered = re.sub(
|
||||
r"""\\([!"#$%&'()*+,\-./:;<=>?@\[\]\\^_`{|}~])""",
|
||||
r"\1",
|
||||
rendered,
|
||||
)
|
||||
rendered = rendered.replace("*", "").replace("~", "")
|
||||
rendered = re.sub(r"(?<!\w)_{1,3}|_{1,3}(?!\w)", "", rendered)
|
||||
normalized = re.sub(r"\s+", " ", html.unescape(rendered)).strip()
|
||||
return normalized.translate(_CODE_RESTORE)
|
||||
|
||||
|
||||
def _render_active_markdown(scan: MarkdownScan) -> str:
|
||||
return _render_markdown(scan.operator_text)
|
||||
|
||||
|
||||
def _render_prose_markdown(markdown: str) -> str:
|
||||
return _render_markdown(markdown, retain_code_text=False)
|
||||
|
||||
|
||||
def _exact_sentence_count(normalized: str, sentence: str) -> int:
|
||||
pattern = re.compile(
|
||||
rf"(?:^|(?<=[.!?]) )(?={re.escape(sentence)}(?:$| ))"
|
||||
)
|
||||
return len(pattern.findall(normalized))
|
||||
|
||||
|
||||
def _mask_region_ranges(text: str, regions: list[Region]) -> str:
|
||||
masked = list(text)
|
||||
for region in regions:
|
||||
for offset in range(region.start, region.end):
|
||||
if masked[offset] not in "\r\n":
|
||||
masked[offset] = " "
|
||||
return "".join(masked)
|
||||
|
||||
|
||||
def check_document_text(text: str, label: str = "document") -> None:
|
||||
"""Validate one current (non-historical) document."""
|
||||
|
||||
scan = scan_active_markdown(text, label)
|
||||
regions = _regions(text, label, scan)
|
||||
workspace = next(region for region in regions if region.name == WORKSPACE_REGION)
|
||||
|
||||
workspace_prose = _render_prose_markdown(
|
||||
scan.prose_text[workspace.start : workspace.end]
|
||||
)
|
||||
if _exact_sentence_count(workspace_prose, CANONICAL_V3_SENTENCE) != 1:
|
||||
raise ContractError(f"{label}: workspace contract lacks the canonical v3-only sentence")
|
||||
if _exact_sentence_count(workspace_prose, CANONICAL_REJECTION_SENTENCE) != 1:
|
||||
raise ContractError(
|
||||
f"{label}: workspace contract lacks the canonical v1/v2 rejection sentence"
|
||||
)
|
||||
|
||||
workspace_visible = _render_markdown(scan.operator_text[workspace.start : workspace.end])
|
||||
workspace_remainder = workspace_visible.replace(CANONICAL_REJECTION_SENTENCE, " ", 1)
|
||||
forbidden_claim = _MIGRATION_REQUIRED.search(
|
||||
workspace_remainder
|
||||
) or _REMAINING_LEGACY_CLAIM.search(workspace_remainder)
|
||||
if forbidden_claim:
|
||||
raise ContractError(
|
||||
f"{label}: workspace contract contains an extra legacy descriptor claim: "
|
||||
f"{forbidden_claim.group(0)}"
|
||||
)
|
||||
|
||||
rendered_operator_text = _render_active_markdown(scan)
|
||||
deleted = _DELETED_TOOL.search(rendered_operator_text)
|
||||
if deleted:
|
||||
raise ContractError(
|
||||
f"{label}: current text contains deleted tool instruction: {deleted.group(0)}"
|
||||
)
|
||||
|
||||
non_workspace = [region for region in regions if region.name == NON_WORKSPACE_REGION]
|
||||
outside_allowed = _mask_region_ranges(
|
||||
scan.operator_text,
|
||||
[workspace, *non_workspace],
|
||||
)
|
||||
rendered_outside = _render_markdown(outside_allowed)
|
||||
outside_claim = _MIGRATION_REQUIRED.search(
|
||||
rendered_outside
|
||||
) or _OUTSIDE_LEGACY_REFERENCE.search(rendered_outside)
|
||||
if outside_claim:
|
||||
raise ContractError(
|
||||
f"{label}: current legacy schema or migration reference outside a "
|
||||
f"{NON_WORKSPACE_REGION}:start/end region: {outside_claim.group(0)}"
|
||||
)
|
||||
|
||||
|
||||
def check_project_state_text(text: str, label: str = "PROJECT_STATE.md") -> None:
|
||||
"""Validate current PROJECT_STATE text and its terminal historical archive hierarchy."""
|
||||
|
||||
scan = scan_active_markdown(text, label)
|
||||
archive_headings = [
|
||||
heading
|
||||
for heading in scan.headings
|
||||
if heading.level == 1
|
||||
and heading.style == "atx"
|
||||
and heading.raw == HISTORICAL_ARCHIVE_H1
|
||||
]
|
||||
if len(archive_headings) != 1:
|
||||
raise ContractError(
|
||||
f"{label}: expected exactly one active terminal '{HISTORICAL_ARCHIVE_H1}' boundary"
|
||||
)
|
||||
archive_heading = archive_headings[0]
|
||||
archive_line_index = next(
|
||||
index for index, line in enumerate(scan.lines) if line.start == archive_heading.start
|
||||
)
|
||||
if archive_line_index + 1 >= len(scan.lines) or not (
|
||||
scan.lines[archive_line_index + 1].active
|
||||
and scan.lines[archive_line_index + 1].text
|
||||
== "## Historical snapshots and archived reference notes"
|
||||
):
|
||||
raise ContractError(
|
||||
f"{label}: the preserved Historical snapshots H2 must immediately follow "
|
||||
f"'{HISTORICAL_ARCHIVE_H1}'"
|
||||
)
|
||||
|
||||
later_headings = [heading for heading in scan.headings if heading.start > archive_heading.start]
|
||||
if not later_headings or not (
|
||||
later_headings[0].level == 2
|
||||
and later_headings[0].style == "atx"
|
||||
and later_headings[0].raw == "## Historical snapshots and archived reference notes"
|
||||
):
|
||||
raise ContractError(
|
||||
f"{label}: Historical archive must be followed by the existing Historical snapshots H2"
|
||||
)
|
||||
if any(heading.level == 1 for heading in later_headings):
|
||||
raise ContractError(f"{label}: Historical archive must be the terminal H1 hierarchy")
|
||||
|
||||
check_document_text(text[:archive_heading.start], f"{label} current section")
|
||||
|
||||
|
||||
def check_document_path(path: Path) -> None:
|
||||
check_document_text(path.read_text(), str(path))
|
||||
|
||||
|
||||
def check_project_state_path(path: Path) -> None:
|
||||
check_project_state_text(path.read_text(), str(path))
|
||||
|
||||
|
||||
def _parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--document", action="append", default=[], type=Path)
|
||||
parser.add_argument("--project-state", action="append", default=[], type=Path)
|
||||
return parser
|
||||
|
||||
|
||||
def main(argv: Sequence[str] | None = None) -> int:
|
||||
args = _parser().parse_args(argv)
|
||||
if not args.document and not args.project_state:
|
||||
raise SystemExit("at least one --document or --project-state path is required")
|
||||
try:
|
||||
for path in args.document:
|
||||
check_document_path(path)
|
||||
for path in args.project_state:
|
||||
check_project_state_path(path)
|
||||
except (ContractError, OSError) as error:
|
||||
print(error, file=sys.stderr)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user