docs: make schema v3 the only workspace contract

This commit is contained in:
2026-08-11 02:53:41 +02:00
parent edef085fea
commit 5310c6555b
9 changed files with 1180 additions and 123 deletions
+18 -64
View File
@@ -137,57 +137,32 @@ sed '/^case "\$mode" in/,$d' "$root/scripts/verify-workspace-install-docs.sh" >"
# shellcheck source=/dev/null
source "$verifier_functions"
project_state_fixture="$negative_root/project-state.md"
python3 - "$root/PROJECT_STATE.md" "$project_state_fixture" <<'PY'
python3 "$root/scripts/test_workspace_descriptor_doc_contract.py"
project_topology_fixture="$negative_root/project-topology-contradiction.md"
python3 - "$root/PROJECT_STATE.md" "$project_topology_fixture" <<'PY'
import pathlib, sys
source = pathlib.Path(sys.argv[1]).read_text()
target = pathlib.Path(sys.argv[2])
marker = source.index("## Historical snapshots")
contradiction = """
## Contradictory release note — LIVE 2026-08-08
- Schema-v2 descriptors are operational again.
- The supported Compose stack is exactly `frontend` plus `core`.
- DWH, vector DB, embedding, LLM, and reverse-proxy services are external configurable endpoints.
"""
target.write_text(source[:marker] + contradiction + source[marker:])
marker = source.index("# Historical archive")
contradiction = (
"The supported Compose stack is exactly `frontend` plus `core`.\n"
"DWH, vector DB, embedding, LLM, and reverse-proxy services are external endpoints.\n\n"
)
pathlib.Path(sys.argv[2]).write_text(source[:marker] + contradiction + source[marker:])
PY
project_state_output="$negative_root/project-state-output"
project_topology_output="$negative_root/project-topology-output"
set +e
verify_project_state_current_contract "$project_state_fixture" contradictory-project-state >"$project_state_output" 2>&1
project_state_status=$?
verify_project_state_current_contract "$project_topology_fixture" contradictory-project-topology \
>"$project_topology_output" 2>&1
project_topology_status=$?
set -e
if [[ $project_state_status -eq 0 ]] || ! grep -Fq "contradictory active text" "$project_state_output"; then
echo "contradictory current-state fixture was not rejected correctly" >&2
cat "$project_state_output" >&2
if [[ $project_topology_status -eq 0 ]] || \
! grep -Fq "contradictory active text" "$project_topology_output"; then
echo "contradictory current PROJECT_STATE topology was not rejected" >&2
cat "$project_topology_output" >&2
exit 1
fi
for level in 1 2 3 4 5 6; do
project_state_live_heading="$negative_root/project-state-live-heading-h$level.md"
python3 - "$root/PROJECT_STATE.md" "$project_state_live_heading" "$level" <<'PY'
import pathlib, sys
source = pathlib.Path(sys.argv[1]).read_text()
target = pathlib.Path(sys.argv[2])
level = int(sys.argv[3])
marker = source.index("## Historical snapshots")
historical = source[marker:]
replacement = "#" * level + " Session summary redesign — LIVE 2026-07-23"
historical = historical.replace("### Historical snapshot — Session summary redesign (2026-07-23)", replacement, 1)
target.write_text(source[:marker] + historical)
PY
set +e
verify_project_state_current_contract "$project_state_live_heading" "historical-live-heading-h$level" >"$project_state_output" 2>&1
project_state_status=$?
set -e
if [[ $project_state_status -eq 0 ]] || ! grep -Fq "active/live heading markers" "$project_state_output"; then
echo "historical LIVE-heading fixture was not rejected correctly for heading level $level" >&2
cat "$project_state_output" >&2
exit 1
fi
done
workspace_fixture="$negative_root/workspace-invalid.yaml"
python3 - "$root/deploy/workspaces/example.yaml" "$workspace_fixture" <<'PY'
import pathlib, sys, yaml
@@ -206,27 +181,6 @@ if [[ $workspace_status -eq 0 ]] || ! grep -Fq "embedding dimensions must be 102
exit 1
fi
project_state_positive="$negative_root/project-state-positive.md"
cat >"$project_state_positive" <<'EOF'
# ThothII — Project State
> Starting-point snapshot.
## Internal Qdrant + Ollama semantic infrastructure — LIVE 2026-08-08
- Schema-v3 descriptors are operational and v1/v2 remain `migration_required`.
- One workspace owns one Qdrant collection.
- Only DWH and LLM remain external runtime application endpoints.
- The internal stack includes `qdrant`, `embedding`, and `embedding-model-init`.
## Historical snapshots — superseded context
### Historical snapshot — previous deployment
- Older notes intentionally live only here.
EOF
verify_project_state_current_contract "$project_state_positive" positive-project-state >/dev/null
local_manual_paraphrase="$negative_root/local-manual-paraphrase.md"
cp "$root/docs/install/local-workspace-registry.md" "$local_manual_paraphrase"
python3 - "$local_manual_paraphrase" <<'PY'
+602
View File
@@ -0,0 +1,602 @@
#!/usr/bin/env python3
"""Unit and policy-matrix coverage for workspace descriptor documentation regions."""
import unittest
from dataclasses import dataclass
try:
from workspace_descriptor_doc_contract import (
ContractError,
_render_active_markdown,
check_document_text,
check_project_state_text,
scan_active_markdown,
)
except ModuleNotFoundError: # Support `python -m unittest scripts.test_...` from repo root.
from scripts.workspace_descriptor_doc_contract import (
ContractError,
_render_active_markdown,
check_document_text,
check_project_state_text,
scan_active_markdown,
)
WORKSPACE_START = "<!-- workspace-descriptor-contract:start -->"
WORKSPACE_END = "<!-- workspace-descriptor-contract:end -->"
NON_WORKSPACE_START = "<!-- non-workspace-migration:start -->"
NON_WORKSPACE_END = "<!-- non-workspace-migration:end -->"
@dataclass(frozen=True)
class PolicyCase:
name: str
expected: str
content: str
CASES = [
PolicyCase('m_schema_v1', 'R', 'Schema v1 descriptors return `migration_required`.'),
PolicyCase('m_schema_dash_v2', 'R', 'Schema-v2 descriptors report `migration_required`.'),
PolicyCase('m_schema_num1', 'R', 'Schema 1 descriptors remain `migration_required`.'),
PolicyCase('m_schema_word1', 'R', 'Schema version 1 descriptors remain `migration_required`.'),
PolicyCase('m_schema_under_colon', 'R', 'A descriptor with schema_version: 1 returns `migration_required`.'),
PolicyCase('m_schema_under_eq', 'R', 'A descriptor with schema_version=2 returns `migration_required`.'),
PolicyCase('m_schema_v1_and_v2', 'R', 'Schema v1 and v2 descriptors remain `migration_required`.'),
PolicyCase('m_bare_and', 'R', 'v1 and v2 descriptors remain `migration_required`.'),
PolicyCase('m_bare_or', 'R', 'v1 or v2 descriptors remain `migration_required`.'),
PolicyCase('m_bare_slash', 'R', 'v1/v2 descriptors remain `migration_required`.'),
PolicyCase('a_they', 'R', 'Schema v1 descriptors are rejected. They return `migration_required`.'),
PolicyCase('a_these_desc', 'R', 'Schema v1 descriptors are rejected. These descriptors return `migration_required`.'),
PolicyCase('a_those_desc', 'R', 'Schema v1 descriptors are rejected. Those descriptors return `migration_required`.'),
PolicyCase('a_such_desc', 'R', 'Schema v1 descriptors are rejected. Such descriptors return `migration_required`.'),
PolicyCase('a_their', 'R', 'Schema v1 descriptors are rejected. Their validation returns `migration_required`.'),
PolicyCase('a_the_desc', 'R', 'Schema v1 descriptors are rejected. The descriptors return `migration_required`.'),
PolicyCase('a_the_candidate', 'R', 'Schema v1 candidates are rejected. The candidate returns `migration_required`.'),
PolicyCase('a_still', 'R', 'Schema v1 descriptors are rejected, but still return `migration_required`.'),
PolicyCase('a_it', 'R', 'Schema v1 descriptor is rejected. It still returns `migration_required`.'),
PolicyCase('a_this_desc', 'R', 'Schema v1 descriptor is rejected. This descriptor still returns `migration_required`.'),
PolicyCase('a_that_desc', 'R', 'Schema v1 descriptor is rejected. That descriptor still returns `migration_required`.'),
PolicyCase('a_such_candidate', 'R', 'Schema v1 candidate is rejected. Such a candidate still returns `migration_required`.'),
PolicyCase('d_workspace', 'R', 'Workspace descriptors remain `migration_required`.'),
PolicyCase('d_possessive', 'R', "Workspace descriptors' status is `migration_required`."),
PolicyCase('d_validation', 'R', 'Workspace descriptor validation returns `migration_required`.'),
PolicyCase('d_legacy', 'R', 'Legacy descriptors remain `migration_required`.'),
PolicyCase('d_bare_descriptor', 'R', 'The descriptor remains `migration_required`.'),
PolicyCase('r_workspace', 'R', '`migration_required` is returned by workspace descriptors.'),
PolicyCase('r_workspace_status', 'R', '`migration_required` is the status for workspace descriptors.'),
PolicyCase('r_legacy', 'R', '`migration_required` applies to legacy descriptors.'),
PolicyCase('r_bare_descriptor', 'R', '`migration_required` is the status for the descriptor.'),
PolicyCase('u_standalone', 'A', 'The unrelated session database may report `migration_required`.'),
PolicyCase('u_after_period', 'A', 'Schema v1 descriptors are rejected. The unrelated session database may report `migration_required`.'),
PolicyCase('u_while', 'A', 'Schema v1 descriptors are rejected while the unrelated session database may report `migration_required`.'),
PolicyCase('u_and_the', 'A', 'Schema v1 descriptors are rejected and the unrelated session database may report `migration_required`.'),
PolicyCase('u_and_a', 'A', 'Schema v1 descriptors are rejected and a session database upgrade may report `migration_required`.'),
PolicyCase('u_and_this', 'A', 'Schema v1 descriptors are rejected and this unrelated session database may report `migration_required`.'),
PolicyCase('u_and_these', 'A', 'Schema v1 descriptors are rejected and these unrelated session database upgrades may report `migration_required`.'),
PolicyCase('u_and_our', 'A', 'Schema v1 descriptors are rejected and our unrelated session database may report `migration_required`.'),
PolicyCase('u_and_bare_subject', 'A', 'Schema v1 descriptors are rejected and session database upgrades may report `migration_required`.'),
PolicyCase('u_prev_then_these', 'A', 'Schema v1 descriptors are rejected. These unrelated session database upgrades report `migration_required`.'),
PolicyCase('u_api_versions', 'A', 'The unrelated session API v1 and v2 may report `migration_required`.'),
PolicyCase('u_db_schema_version', 'A', 'The unrelated session database schema version 1 may report `migration_required`.'),
PolicyCase('u_reverse_negative', 'A', '`migration_required` is not returned by workspace descriptors; it belongs to the session database.'),
PolicyCase('s_accept_although', 'A', 'Although schema v1 descriptors are rejected, the unrelated session database may report `migration_required`.'),
PolicyCase('s_accept_because', 'A', 'Schema v1 descriptors are rejected because the unrelated session database may report `migration_required`.'),
PolicyCase('s_accept_mentions_that', 'A', 'Schema v1 descriptor documentation mentions that the unrelated session database may report `migration_required`.'),
PolicyCase('s_accept_unrelated_descriptor', 'A', 'The unrelated session database descriptor reports `migration_required`.'),
PolicyCase('s_accept_unrelated_candidate', 'A', 'An unrelated candidate remains `migration_required`.'),
PolicyCase('s_reject_workspace_anaphor_they', 'R', 'Workspace descriptors are rejected. They return `migration_required`.'),
PolicyCase('s_reject_workspace_anaphor_it', 'R', 'A workspace descriptor is rejected. It returns `migration_required`.'),
PolicyCase('s_reject_descriptor_anaphor', 'R', 'The descriptor is rejected. It returns `migration_required`.'),
]
def workspace_region(extra=""):
body = (
"Schema v3 is the only accepted workspace descriptor. "
"Schema v1 and v2 workspace descriptors are rejected before activation."
)
if extra:
body += "\n" + extra
return f"{WORKSPACE_START}\n{body}\n{WORKSPACE_END}"
def non_workspace_region(content):
return f"{NON_WORKSPACE_START}\n{content}\n{NON_WORKSPACE_END}"
def generic_fixture(*, workspace_extra="", current_extra=""):
return f"# Fixture\n\n{workspace_region(workspace_extra)}\n\n{current_extra}\n"
def project_fixture(*, workspace_extra="", current_extra="", historical_extra=""):
return (
"# Project state\n\n"
+ workspace_region(workspace_extra)
+ "\n\n"
+ current_extra
+ "\n\n# Historical archive\n"
+ "## Historical snapshots and archived reference notes\n\n"
+ historical_extra
+ "\n"
)
class PolicyMatrixTests(unittest.TestCase):
def test_52_case_matrix_across_both_entry_points(self):
self.assertEqual(len(CASES), 52)
for case in CASES:
if case.expected == "R":
generic = generic_fixture(workspace_extra=case.content)
project = project_fixture(workspace_extra=case.content)
else:
allowed = non_workspace_region(case.content)
generic = generic_fixture(current_extra=allowed)
project = project_fixture(current_extra=allowed)
for path, checker, text in (
("generic", check_document_text, generic),
("project", check_project_state_text, project),
):
with self.subTest(case=case.name, path=path, expected=case.expected):
if case.expected == "R":
with self.assertRaises(ContractError):
checker(text, f"{path}-{case.name}")
else:
checker(text, f"{path}-{case.name}")
class RegionStructureTests(unittest.TestCase):
def assert_invalid_both(self, generic, project=None):
with self.assertRaises(ContractError):
check_document_text(generic, "generic-invalid")
with self.assertRaises(ContractError):
check_project_state_text(project or project_fixture(current_extra=generic), "project-invalid")
def test_unmarked_unrelated_migration_required_is_rejected(self):
text = "The unrelated session database may report `migration_required`."
self.assert_invalid_both(generic_fixture(current_extra=text), project_fixture(current_extra=text))
def test_marked_unrelated_migration_required_is_accepted(self):
text = non_workspace_region("The session database may report `migration_required`.")
check_document_text(generic_fixture(current_extra=text), "generic-allowed")
check_project_state_text(project_fixture(current_extra=text), "project-allowed")
def test_missing_reversed_duplicate_and_nested_markers_are_rejected(self):
malformed = (
WORKSPACE_START,
WORKSPACE_END + "\n" + WORKSPACE_START,
workspace_region() + "\n" + workspace_region(),
WORKSPACE_START + "\n" + NON_WORKSPACE_START + "\n" + WORKSPACE_END + "\n" + NON_WORKSPACE_END,
NON_WORKSPACE_START + "\n" + NON_WORKSPACE_START + "\n" + NON_WORKSPACE_END + "\n" + NON_WORKSPACE_END,
NON_WORKSPACE_START + "\n" + workspace_region() + "\n" + NON_WORKSPACE_END,
workspace_region() + "\n" + NON_WORKSPACE_END,
)
for index, text in enumerate(malformed):
with self.subTest(index=index):
self.assert_invalid_both(text)
def test_migration_required_inside_workspace_block_is_rejected(self):
text = generic_fixture(workspace_extra="Descriptors return `migration_required`.")
self.assert_invalid_both(text, project_fixture(workspace_extra="Descriptors return `migration_required`."))
def test_legacy_support_inside_workspace_block_is_rejected(self):
text = generic_fixture(workspace_extra="Schema v1 descriptors are operational and readable.")
self.assert_invalid_both(
text,
project_fixture(workspace_extra="Schema v1 descriptors are operational and readable."),
)
def test_migrate_legacy_is_forbidden_even_in_non_workspace_region(self):
allowed = non_workspace_region("Run `migrate-legacy` for the session database.")
self.assert_invalid_both(generic_fixture(current_extra=allowed), project_fixture(current_extra=allowed))
def test_project_requires_terminal_historical_h1(self):
valid = project_fixture(historical_extra="Schema v2 returned `migration_required` historically.")
check_project_state_text(valid, "project-valid-history")
for name, text in (
("missing", valid.replace("# Historical archive\n", "")),
(
"blank-physical-line",
valid.replace(
"# Historical archive\n## Historical",
"# Historical archive\n\n## Historical",
),
),
(
"prose-physical-line",
valid.replace(
"# Historical archive\n## Historical",
"# Historical archive\nArchived notes follow.\n## Historical",
),
),
(
"comment-physical-line",
valid.replace(
"# Historical archive\n## Historical",
"# Historical archive\n<!-- archived -->\n## Historical",
),
),
(
"intervening-h2",
valid.replace(
"# Historical archive\n## Historical",
"# Historical archive\n## Other archive\n## Historical",
),
),
(
"intervening-setext-h2",
valid.replace(
"# Historical archive\n## Historical",
"# Historical archive\nOther archive\n-------------\n## Historical",
),
),
(
"setext-replacement-h2",
valid.replace(
"## Historical snapshots",
"Historical snapshots\n--------------------",
),
),
("later-h1", valid + "\n# Returned live section\n"),
("duplicate-h1", valid + "\n# Historical archive\n"),
):
with self.subTest(name=name), self.assertRaises(ContractError):
check_project_state_text(text, f"project-{name}")
def test_markers_inside_fences_comments_or_code_do_not_count(self):
fake = workspace_region()
for name, wrapped in (
("backtick-fence", f"```markdown\n{fake}\n```"),
("tilde-fence", f"~~~~\n{fake}\n~~~~"),
("outer-comment", f"<!--\n{fake}\n-->"),
("indented-code", "\n".join(" " + line for line in fake.splitlines())),
):
generic = f"# Fixture\n\n{wrapped}\n"
project = (
f"# Project\n\n{wrapped}\n\n# Historical archive\n"
"## Historical snapshots\n"
)
with self.subTest(name=name):
self.assert_invalid_both(generic, project)
def test_extra_inactive_marker_literals_do_not_duplicate_active_region(self):
fake_region = f"{WORKSPACE_START}\ninactive example\n{WORKSPACE_END}"
extras = (
f"```\n{fake_region}\n```",
f" ~~~\n{fake_region}\n ~~~~",
"\n".join(" " + line for line in fake_region.splitlines()),
)
for index, extra in enumerate(extras):
with self.subTest(index=index):
check_document_text(generic_fixture(current_extra=extra), f"generic-extra-{index}")
check_project_state_text(
project_fixture(current_extra=extra), f"project-extra-{index}"
)
nested_comment = f"<!--\n{fake_region}\n-->"
self.assert_invalid_both(
generic_fixture(current_extra=nested_comment),
project_fixture(current_extra=nested_comment),
)
def test_inline_and_partially_indented_markers_are_misplaced(self):
for name, marker in (
("inline", "text <!-- workspace-descriptor-contract:start -->"),
("one-space", " <!-- workspace-descriptor-contract:start -->"),
("inline-allow", "text <!-- non-workspace-migration:start -->"),
("one-space-allow", " <!-- non-workspace-migration:start -->"),
):
generic = generic_fixture(current_extra=marker)
project = project_fixture(current_extra=marker)
with self.subTest(name=name):
self.assert_invalid_both(generic, project)
def test_inactive_allow_markers_do_not_authorize_but_extra_literals_are_ignored(self):
allow = non_workspace_region("migration_required")
fake_token_region = non_workspace_region("migration_required")
fake_empty_region = non_workspace_region("inactive example")
wrappers = (
(f"```\n{fake_token_region}\n```", f"```\n{fake_empty_region}\n```", True),
(f"<!--\n{fake_token_region}\n-->", f"<!--\n{fake_empty_region}\n-->", False),
(
"\n".join(" " + line for line in fake_token_region.splitlines()),
"\n".join(" " + line for line in fake_empty_region.splitlines()),
True,
),
)
for index, (fake_with_token, fake_without_token, extra_allowed) in enumerate(wrappers):
with self.subTest(index=index, mode="sole"):
self.assert_invalid_both(
generic_fixture(current_extra=fake_with_token + "\nmigration_required"),
project_fixture(current_extra=fake_with_token + "\nmigration_required"),
)
extra_generic = generic_fixture(current_extra=fake_without_token + "\n" + allow)
extra_project = project_fixture(current_extra=fake_without_token + "\n" + allow)
with self.subTest(index=index, mode="extra"):
if extra_allowed:
check_document_text(extra_generic, f"generic-extra-allow-{index}")
check_project_state_text(extra_project, f"project-extra-allow-{index}")
else:
self.assert_invalid_both(extra_generic, extra_project)
def test_fence_closer_must_match_character_and_minimum_length(self):
fake = workspace_region()
for name, fenced in (
("short-backtick-close", f"````\n```\n{fake}"),
("wrong-character-close", f"~~~~\n```\n{fake}\n~~~~"),
):
project = (
f"# Project\n\n{fenced}\n\n# Historical archive\n"
"## Historical snapshots\n"
)
with self.subTest(name=name):
self.assert_invalid_both(f"# Fixture\n\n{fenced}\n", project)
safe_fake = f"{WORKSPACE_START}\ninactive example\n{WORKSPACE_END}"
valid_extra = f"````\n{safe_fake}\n`````"
check_document_text(generic_fixture(current_extra=valid_extra), "valid-long-close")
def test_noncanonical_contract_sentences_are_rejected(self):
variants = (
workspace_region().replace(
"Schema v3 is the only accepted workspace descriptor.",
"Only schema v3 workspace descriptors are accepted.",
),
workspace_region().replace(
"Schema v3 is the only accepted workspace descriptor.",
"Not Schema v3 is the only accepted workspace descriptor.",
),
workspace_region().replace(
"Schema v1 and v2 workspace descriptors are rejected before activation.",
"Not Schema v1 and v2 workspace descriptors are rejected before activation.",
),
)
for index, noncanonical in enumerate(variants):
with self.subTest(index=index):
self.assert_invalid_both(
generic_fixture().replace(workspace_region(), noncanonical),
project_fixture().replace(workspace_region(), noncanonical),
)
def test_remaining_legacy_claim_and_deleted_transformer_are_rejected(self):
for claim in (
"The system supports schema v1 descriptors.",
"A v2 descriptor remains readable.",
"Legacy-descriptor activation is operational.",
):
with self.subTest(claim=claim):
self.assert_invalid_both(
generic_fixture(workspace_extra=claim),
project_fixture(workspace_extra=claim),
)
for deleted in (
"Run the deleted legacy descriptor transformer.",
"Invoke the descriptor legacy migrator.",
"Use the legacy transformer.",
):
with self.subTest(deleted=deleted):
self.assert_invalid_both(
generic_fixture(current_extra=deleted),
project_fixture(current_extra=deleted),
)
def test_deleted_tools_are_rejected_after_active_markdown_rendering(self):
examples = (
"migrate-`**legacy**`",
"migrate-[legacy](https://example.invalid/tool)",
"migrate-[legacy][tool]\n\n[tool]: https://example.invalid/tool",
"legacy descriptor **transformer**",
"migrate&#45;legacy",
"migrate-<span>legacy</span>",
r"migrate\-legacy",
"legacy descriptor trans<!-- hidden -->former",
"Run migrate-[legacy\n](https://example.invalid/tool)",
"Run migrate-[legacy\n][tool]\n\n[tool]: https://example.invalid/tool",
"migrate-``\nlegacy\n``",
"<!-- hidden --> migrate-legacy",
"<!-- hidden --> legacy descriptor transformer",
"migrate-`\nlegacy\n`",
"<!-- hidden\ncomment --> migrate-legacy",
"migrate-legacy <!-- hidden\ncomment --> safe suffix",
)
for index, example in enumerate(examples):
with self.subTest(index=index, example=example):
self.assert_invalid_both(
generic_fixture(current_extra=example),
project_fixture(current_extra=example),
)
def test_deleted_tool_examples_in_inactive_markdown_are_allowed(self):
inactive_examples = (
"<!--\nmigrate-legacy\nlegacy descriptor transformer\n-->",
"<!-- migrate-legacy and legacy descriptor transformer -->",
"[safe label](https://example.invalid/migrate-legacy)",
"[safe label][tool]\n\n[tool]: https://example.invalid/migrate-legacy",
'<span data-example="migrate-legacy">safe text</span>',
"safe prefix <!-- migrate-legacy --> safe suffix",
"safe prefix <!--\nmigrate-legacy\nlegacy descriptor transformer\n--> safe suffix",
)
for index, example in enumerate(inactive_examples):
with self.subTest(index=index):
check_document_text(
generic_fixture(current_extra=example), f"generic-inactive-tool-{index}"
)
check_project_state_text(
project_fixture(current_extra=example), f"project-inactive-tool-{index}"
)
def test_deleted_tools_in_operator_visible_code_blocks_are_rejected(self):
examples = (
"```text\nmigrate-legacy\n```",
"~~~~\nlegacy descriptor transformer\n~~~~",
" migrate-legacy",
"\tlegacy descriptor transformer",
"```text\n[safe](https://example.invalid/migrate-legacy)\n```",
" <!-- migrate-legacy -->",
"```html\n<!-- migrate-legacy -->\n```",
"```html\n<!-- outer <!-- migrate-legacy --> -->\n```",
)
for index, example in enumerate(examples):
with self.subTest(index=index):
self.assert_invalid_both(
generic_fixture(current_extra=example),
project_fixture(current_extra=example),
)
def test_hidden_canonical_sentences_cannot_satisfy_workspace_contract(self):
canonical = workspace_region()[len(WORKSPACE_START) + 1 : -len(WORKSPACE_END) - 1]
hidden_bodies = (
f"<!--\n{canonical}\n-->",
f"```text\n{canonical}\n```",
"\n".join(" " + line for line in canonical.splitlines()),
f"`{canonical}`",
)
for index, hidden in enumerate(hidden_bodies):
hidden_region = f"{WORKSPACE_START}\n{hidden}\n{WORKSPACE_END}"
with self.subTest(index=index):
self.assert_invalid_both(
generic_fixture().replace(workspace_region(), hidden_region),
project_fixture().replace(workspace_region(), hidden_region),
)
def test_current_legacy_schema_references_require_non_workspace_region(self):
claim = "Schema v1 workspace descriptors remain operational and readable."
visible_claims = (
claim,
f"```text\n{claim}\n```",
" " + claim,
)
for index, visible_claim in enumerate(visible_claims):
with self.subTest(index=index):
self.assert_invalid_both(
generic_fixture(current_extra=visible_claim),
project_fixture(current_extra=visible_claim),
)
hidden_claim = f"<!-- {claim} -->"
check_document_text(
generic_fixture(current_extra=hidden_claim), "generic-hidden-legacy"
)
check_project_state_text(
project_fixture(current_extra=hidden_claim), "project-hidden-legacy"
)
allowed = non_workspace_region(claim)
check_document_text(generic_fixture(current_extra=allowed), "generic-allowed-legacy")
check_project_state_text(
project_fixture(current_extra=allowed), "project-allowed-legacy"
)
def test_html_comments_fail_closed_when_unclosed_or_nested(self):
malformed = (
"<!-- unclosed comment",
"safe prefix <!-- unclosed comment",
"<!-- outer\n<!-- inner -->\n-->\nmigrate-legacy",
(
"<!-- outer\n<!-- inner -->\n-->\n"
"Schema v1 workspace descriptors remain operational and readable."
),
"<!-- outer\n<!-- inner -->\n-->\nsafe trailing text",
)
for index, text in enumerate(malformed):
with self.subTest(index=index):
self.assert_invalid_both(
generic_fixture(current_extra=text),
project_fixture(current_extra=text),
)
visible_suffixes = (
"<!-- hidden --> migrate-legacy",
"<!-- hidden\n--> legacy descriptor transformer",
)
for index, text in enumerate(visible_suffixes):
with self.subTest(index=index, kind="visible-suffix"):
self.assert_invalid_both(
generic_fixture(current_extra=text),
project_fixture(current_extra=text),
)
pure_comments = (
"<!-- migrate-legacy -->",
"<!--\nlegacy descriptor transformer\n-->",
)
for index, text in enumerate(pure_comments):
with self.subTest(index=index, kind="pure-comment"):
check_document_text(
generic_fixture(current_extra=text), f"generic-pure-comment-{index}"
)
check_project_state_text(
project_fixture(current_extra=text), f"project-pure-comment-{index}"
)
def test_alternate_html_comment_closer_fails_closed_outside_code(self):
malformed = (
"<!-- hidden --!> migrate-legacy",
(
"<!-- hidden --!> "
"Schema v1 workspace descriptors remain operational and readable."
),
"<!-- hidden\n--!> migrate-legacy",
)
for index, text in enumerate(malformed):
generic = generic_fixture(current_extra=text)
project = project_fixture(current_extra=text)
with (
self.subTest(index=index, entry_point="generic"),
self.assertRaisesRegex(ContractError, "alternate HTML comment closer"),
):
check_document_text(generic, f"generic-alternate-closer-{index}")
with (
self.subTest(index=index, entry_point="project"),
self.assertRaisesRegex(ContractError, "alternate HTML comment closer"),
):
check_project_state_text(project, f"project-alternate-closer-{index}")
safe_code = (
"```html\n<!-- hidden --!> safe fenced example\n```\n"
" <!-- hidden --!> safe indented example"
)
check_document_text(generic_fixture(current_extra=safe_code), "generic-safe-code-closer")
check_project_state_text(
project_fixture(current_extra=safe_code), "project-safe-code-closer"
)
visible_code = (
"```html\n<!-- hidden --!> migrate-legacy\n```\n"
" <!-- hidden --!> Schema v1 workspace descriptors remain operational."
)
self.assert_invalid_both(
generic_fixture(current_extra=visible_code),
project_fixture(current_extra=visible_code),
)
def test_visible_comment_prefix_and_suffix_are_preserved_without_structural_reparse(self):
text = (
"safe prefix <!-- hidden --> safe suffix\n"
"before <!--\nhidden\n--> after\n"
"<!-- hidden --> # Reconstructed heading must not count\n"
)
scan = scan_active_markdown(text, "visible-comment-test")
self.assertEqual(
_render_active_markdown(scan),
"safe prefix safe suffix before after # Reconstructed heading must not count",
)
self.assertFalse(any(heading.text.startswith("Reconstructed") for heading in scan.headings))
def test_setext_h1_after_archive_is_rejected(self):
project = project_fixture(historical_extra="Returned live section\n=====================")
with self.assertRaises(ContractError):
check_project_state_text(project, "project-setext-live-tail")
def test_heading_literals_in_adversarial_fences_and_comments_are_inactive(self):
historical = (
"````\nFake live H1\n============\n```\n`````\n"
"<!--\nComment live H1\n===============\n-->"
)
check_project_state_text(
project_fixture(historical_extra=historical), "project-inactive-headings"
)
def test_historical_markers_cannot_satisfy_current_contract(self):
current_without = "# Project state\n\nNo current workspace contract.\n"
historical = "# Historical archive\n## Historical snapshots\n\n" + workspace_region()
with self.assertRaises(ContractError):
check_project_state_text(current_without + historical, "project-historical-marker")
if __name__ == "__main__":
unittest.main(verbosity=2)
+18 -12
View File
@@ -105,6 +105,14 @@ for token in tokens:
PY
}
verify_workspace_descriptor_doc_contract() {
local source="$1" label="$2"
if ! python3 "$root/scripts/workspace_descriptor_doc_contract.py" --document "$source"; then
echo "$label violates the workspace descriptor documentation contract" >&2
return 1
fi
}
verify_markdown_table_relationships() {
local source="$1" label="$2" heading="$3" spec_json="$4"
python3 - "$source" "$label" "$heading" "$spec_json" <<'PY'
@@ -588,21 +596,20 @@ verify_vector_helper_interfaces() {
verify_project_state_current_contract() {
local source="${1:-$root/PROJECT_STATE.md}"
local label="${2:-PROJECT_STATE.md}"
if ! python3 "$root/scripts/workspace_descriptor_doc_contract.py" --project-state "$source"; then
echo "$label violates the workspace descriptor documentation contract" >&2
return 1
fi
python3 - "$source" "$label" <<'PY'
import pathlib, re, sys
text = pathlib.Path(sys.argv[1]).read_text()
label = sys.argv[2]
marker = re.search(r"^## Historical snapshots\b", text, re.MULTILINE)
marker = re.search(r"^# Historical archive$", text, re.MULTILINE)
if not marker:
raise SystemExit(f"{label}: missing Historical snapshots boundary")
raise SystemExit(f"{label}: missing Historical archive boundary")
current = text[:marker.start()]
historical = text[marker.start():]
if not re.search(r"Internal Qdrant \+ Ollama semantic infrastructure", current, re.MULTILINE):
raise SystemExit(f"{label}: current section missing internal semantic snapshot heading")
if not re.search(r"Schema-v3 descriptors are operational", current, re.MULTILINE):
raise SystemExit(f"{label}: current section must say schema-v3 is operational")
if "migration_required" not in current:
raise SystemExit(f"{label}: current section must mention migration_required")
if not re.search(r"\b(one|single)\b.*\bworkspace\b.*\b(one|single)\b.*\bQdrant\b.*\bcollection\b", current, re.IGNORECASE | re.DOTALL):
raise SystemExit(f"{label}: current section must describe one-workspace/one-collection ownership")
if not re.search(r"\bDWH\b", current) or not re.search(r"\bLLM\b", current):
@@ -612,7 +619,6 @@ if not re.search(r"\bexternal\b", current, re.IGNORECASE):
if "embedding-model-init" not in current:
raise SystemExit(f"{label}: current section missing embedding-model-init")
forbidden = [
r"Schema-v2 descriptors are operational",
r"supported Compose stack is exactly `frontend` plus `core`",
r"DWH, vector DB, embedding, LLM",
r"vector DB, embedding, and LLM remain external",
@@ -620,8 +626,6 @@ forbidden = [
for pattern in forbidden:
if re.search(pattern, current, re.MULTILINE):
raise SystemExit(f"{label}: current section still contains contradictory active text: {pattern}")
if re.search(r"^#{1,6}[^\n]*\b(LIVE|live|current state|current-state|Current state|Current-state)\b", historical, re.MULTILINE):
raise SystemExit(f"{label}: historical section still contains active/live heading markers")
PY
}
@@ -640,6 +644,10 @@ verify_internal_semantic_infrastructure_docs() {
verify_workspace_descriptor_semantic_contract "$root/deploy/workspaces/psd.yaml.example" "psd workspace example" || return 1
verify_vector_helper_interfaces || return 1
verify_project_state_current_contract "$root/PROJECT_STATE.md" "PROJECT_STATE.md" || return 1
verify_workspace_descriptor_doc_contract "$readme" "README" || return 1
verify_workspace_descriptor_doc_contract "$local_manual" "local workspace manual" || return 1
verify_workspace_descriptor_doc_contract "$server_manual" "server workspace manual" || return 1
verify_workspace_descriptor_doc_contract "$diagnostics" "workspace diagnostic protocol" || return 1
local ownership_spec semantic_index_spec compact_spec
ownership_spec='{"rows":[
@@ -670,13 +678,11 @@ verify_internal_semantic_infrastructure_docs() {
require_pattern "$agents" "AGENTS.md" 'DWH and LLM remain external configuration endpoints' || return 1
for manual in "$local_manual" "$server_manual"; do
require_pattern "$manual" "$(basename "$manual")" 'qwen3-embedding:0\.6b' || return 1
require_pattern "$manual" "$(basename "$manual")" 'migration_required' || return 1
done
require_pattern "$local_manual" "local workspace manual" 'CPU-first' || return 1
require_pattern "$local_manual" "local workspace manual" 'THOTH_ENABLE_EMBEDDING_GPU=1' || return 1
require_pattern "$server_manual" "server workspace manual" 'Qdrant backup/restore' || return 1
require_pattern "$compact_manual" "four-context install note" '1024 dimensioni' || return 1
require_pattern "$diagnostics" "workspace diagnostic protocol" 'schema version 3' || return 1
require_pattern "$diagnostics" "workspace diagnostic protocol" 'semantic_index_incompatible' || return 1
require_absent "$diagnostics" "workspace diagnostic protocol" \
'engine: pgvector' \
+474
View File
@@ -0,0 +1,474 @@
#!/usr/bin/env python3
"""Machine-checkable policy for current workspace-descriptor documentation."""
from __future__ import annotations
import argparse
import html
import re
import sys
from collections.abc import Sequence
from dataclasses import dataclass
from pathlib import Path
WORKSPACE_REGION = "workspace-descriptor-contract"
NON_WORKSPACE_REGION = "non-workspace-migration"
HISTORICAL_ARCHIVE_H1 = "# Historical archive"
CANONICAL_V3_SENTENCE = "Schema v3 is the only accepted workspace descriptor."
CANONICAL_REJECTION_SENTENCE = (
"Schema v1 and v2 workspace descriptors are rejected before activation."
)
_MARKER_LINE = re.compile(
r"<!-- (workspace-descriptor-contract|non-workspace-migration):(start|end) -->"
)
_MARKER_LITERAL = re.compile(
r"(?:workspace-descriptor-contract|non-workspace-migration):(start|end)"
)
_MIGRATION_REQUIRED = re.compile(r"migration_required", re.IGNORECASE)
_DELETED_TOOL = re.compile(
r"migrate\s*-\s*legacy|"
r"legacy[-\s]+(?:(?:workspace[-\s]+)?descriptor[-\s]+)?(?:transformer|migrator)|"
r"(?:workspace[-\s]+)?descriptor[-\s]+legacy[-\s]+(?:transformer|migrator)|"
r"migrat(?:e|ing)[-\s]+legacy[-\s]+descriptors?",
re.IGNORECASE,
)
_REMAINING_LEGACY_CLAIM = re.compile(
r"\b(?:schema(?:[- ]v?|\s+version\s*)[12]|"
r"schema_version\s*[:=]\s*[12]|version\s*[12]|v[12]|"
r"legacy(?:[-\s]+workspace)?[-\s]+descriptors?)\b",
re.IGNORECASE,
)
_CODE_LITERAL_CHARS = "\\`*_~[]()<>!&"
_CODE_PROTECT = str.maketrans(
{character: chr(0xE000 + index) for index, character in enumerate(_CODE_LITERAL_CHARS)}
)
_CODE_RESTORE = str.maketrans(
{chr(0xE000 + index): character for index, character in enumerate(_CODE_LITERAL_CHARS)}
)
_OUTSIDE_LEGACY_REFERENCE = re.compile(
r"\b(?:schema(?:[- ]v?|\s+version\s*)[12]|"
r"schema_version\s*[:=]\s*[12]|"
r"(?:v[12](?:\s*(?:/|and|or)\s*v[12])?)\s+(?:workspace\s+)?descriptors?|"
r"legacy(?:[-\s]+workspace)?[-\s]+descriptors?)\b",
re.IGNORECASE,
)
class ContractError(ValueError):
"""Raised when current documentation violates the descriptor-region policy."""
@dataclass(frozen=True)
class ScannedLine:
start: int
end: int
text: str
active: bool
@dataclass(frozen=True)
class Heading:
start: int
level: int
text: str
style: str
raw: str
@dataclass(frozen=True)
class MarkdownScan:
lines: tuple[ScannedLine, ...]
headings: tuple[Heading, ...]
prose_text: str
operator_text: str
@dataclass(frozen=True)
class Region:
name: str
start: int
end: int
def contains(self, offset: int) -> bool:
return self.start <= offset < self.end
def _opening_fence(line: str) -> tuple[str, int] | None:
match = re.match(r"^ {0,3}(`{3,}|~{3,})(.*)$", line)
if not match:
return None
run, rest = match.groups()
if run[0] == "`" and "`" in rest:
return None
return run[0], len(run)
def _closes_fence(line: str, fence: tuple[str, int]) -> bool:
char, minimum = fence
return re.fullmatch(rf" {{0,3}}{re.escape(char)}{{{minimum},}}[ \t]*", line) is not None
def _atx_heading(line: ScannedLine) -> Heading | None:
if not line.active:
return None
match = re.fullmatch(r" {0,3}(#{1,6})(?:[ \t]+(.*?))?[ \t]*", line.text)
if not match:
return None
hashes, content = match.groups()
content = content or ""
content = re.sub(r"[ \t]+#+[ \t]*$", "", content).strip()
return Heading(line.start, len(hashes), content, "atx", line.text)
def _mask_html_comments(line: str, depth: int) -> tuple[str, int]:
"""Mask HTML comment ranges without changing raw character offsets."""
masked = list(line)
position = 0
while position < len(line):
if depth and line.startswith("--!>", position):
raise ContractError("alternate HTML comment closer is not allowed")
if line.startswith("<!--", position):
if depth:
raise ContractError("nested HTML comments are not allowed")
depth = 1
masked[position : position + 4] = " " * 4
position += 4
elif depth and line.startswith("-->", position):
masked[position : position + 3] = " " * 3
depth -= 1
position += 3
else:
if depth:
masked[position] = " "
position += 1
return "".join(masked), depth
def scan_active_markdown(text: str, label: str = "document") -> MarkdownScan:
"""Scan structural Markdown and retain the whole visible active document."""
lines: list[ScannedLine] = []
prose_chunks: list[str] = []
operator_chunks: list[str] = []
fence: tuple[str, int] | None = None
html_comment_depth = 0
offset = 0
for raw_line in text.splitlines(keepends=True):
line = raw_line.rstrip("\r\n")
ending = raw_line[len(line) :]
start = offset
offset += len(raw_line)
if fence is not None:
lines.append(ScannedLine(start, offset, line, False))
prose_chunks.append(" " * len(line) + ending)
if _closes_fence(line, fence):
operator_chunks.append(" " * len(line) + ending)
fence = None
else:
operator_chunks.append(line.translate(_CODE_PROTECT) + ending)
continue
if not html_comment_depth and re.match(r"^(?: {4}|\t)", line):
lines.append(ScannedLine(start, offset, line, False))
prose_chunks.append(" " * len(line) + ending)
operator_chunks.append(line.translate(_CODE_PROTECT) + ending)
continue
if not html_comment_depth:
opened_fence = _opening_fence(line)
if opened_fence is not None:
lines.append(ScannedLine(start, offset, line, False))
prose_chunks.append(" " * len(line) + ending)
operator_chunks.append(" " * len(line) + ending)
fence = opened_fence
continue
depth_before = html_comment_depth
visible, html_comment_depth = _mask_html_comments(line, html_comment_depth)
exact_marker = depth_before == 0 and _MARKER_LINE.fullmatch(line) is not None
misplaced_marker = depth_before == 0 and _MARKER_LITERAL.search(line) is not None
active = exact_marker or misplaced_marker or bool(visible.strip())
lines.append(ScannedLine(start, offset, line, active))
prose_chunks.append(visible + ending)
operator_chunks.append(visible + ending)
if html_comment_depth:
raise ContractError(f"{label}: unclosed HTML comment")
headings: list[Heading] = []
for index, line in enumerate(lines):
atx = _atx_heading(line)
if atx:
headings.append(atx)
continue
if not line.active or not line.text.strip() or _MARKER_LINE.fullmatch(line.text):
continue
if index + 1 >= len(lines) or not lines[index + 1].active:
continue
underline = re.fullmatch(r" {0,3}(=+|-+)[ \t]*", lines[index + 1].text)
if underline:
headings.append(
Heading(
line.start,
1 if underline.group(1)[0] == "=" else 2,
line.text.strip(),
"setext",
line.text,
)
)
headings.sort(key=lambda heading: heading.start)
prose_text = "".join(prose_chunks)
operator_text = "".join(operator_chunks)
if len(prose_text) != len(text) or len(operator_text) != len(text):
raise ContractError(f"{label}: Markdown representations changed raw offsets")
return MarkdownScan(tuple(lines), tuple(headings), prose_text, operator_text)
def _regions(text: str, label: str, scan: MarkdownScan) -> list[Region]:
events: list[tuple[ScannedLine, str, str]] = []
for line in scan.lines:
if not line.active:
continue
marker = _MARKER_LINE.fullmatch(line.text)
if marker:
events.append((line, marker.group(1), marker.group(2)))
elif _MARKER_LITERAL.search(line.text):
raise ContractError(
f"{label}: misplaced documentation region marker: {line.text.strip()}"
)
regions: list[Region] = []
stack: tuple[str, int] | None = None
for line, name, action in events:
if action == "start":
if stack is not None:
raise ContractError(f"{label}: documentation regions may not nest or overlap")
stack = (name, line.end)
continue
if stack is None:
raise ContractError(f"{label}: unmatched {name}:end marker")
open_name, content_start = stack
if open_name != name:
raise ContractError(
f"{label}: marker {name}:end closes active {open_name}:start region"
)
regions.append(Region(name, content_start, line.start))
stack = None
if stack is not None:
raise ContractError(f"{label}: unmatched {stack[0]}:start marker")
workspace_regions = [region for region in regions if region.name == WORKSPACE_REGION]
if len(workspace_regions) != 1:
raise ContractError(
f"{label}: expected exactly one {WORKSPACE_REGION}:start/end region, "
f"found {len(workspace_regions)}"
)
return regions
def _render_markdown(markdown: str, *, retain_code_text: bool = True) -> str:
"""Normalize one offset-preserving Markdown representation to rendered text."""
rendered = markdown
rendered = re.sub(r"(?m)^ {0,3}\[[^]\n]+\]:[^\n]*(?:\n|$)", "", rendered)
# Retain multiline code-span text while discarding a matching delimiter run.
def code_text(match: re.Match[str]) -> str:
content = re.sub(r"[\r\n]+", " ", match.group(2))
if content.startswith(" ") and content.endswith(" ") and content.strip():
content = content[1:-1]
return content
rendered = re.sub(
r"(?<!`)(`+)(?!`)(.*?)(?<!`)\1(?!`)",
code_text if retain_code_text else " ",
rendered,
flags=re.DOTALL,
)
rendered = re.sub(
r"!?\[([^]]*?)\]\([^)]*\)",
r"\1",
rendered,
flags=re.DOTALL,
)
rendered = re.sub(
r"!?\[([^]]*?)\]\s*\[[^]]*?\]",
r"\1",
rendered,
flags=re.DOTALL,
)
rendered = re.sub(r"\[([^]\n]+)\]", r"\1", rendered)
rendered = re.sub(
r"</?[A-Za-z][A-Za-z0-9-]*(?:\s[^<>]*?)?\s*/?>",
"",
rendered,
flags=re.DOTALL,
)
rendered = re.sub(
r"""\\([!"#$%&'()*+,\-./:;<=>?@\[\]\\^_`{|}~])""",
r"\1",
rendered,
)
rendered = rendered.replace("*", "").replace("~", "")
rendered = re.sub(r"(?<!\w)_{1,3}|_{1,3}(?!\w)", "", rendered)
normalized = re.sub(r"\s+", " ", html.unescape(rendered)).strip()
return normalized.translate(_CODE_RESTORE)
def _render_active_markdown(scan: MarkdownScan) -> str:
return _render_markdown(scan.operator_text)
def _render_prose_markdown(markdown: str) -> str:
return _render_markdown(markdown, retain_code_text=False)
def _exact_sentence_count(normalized: str, sentence: str) -> int:
pattern = re.compile(
rf"(?:^|(?<=[.!?]) )(?={re.escape(sentence)}(?:$| ))"
)
return len(pattern.findall(normalized))
def _mask_region_ranges(text: str, regions: list[Region]) -> str:
masked = list(text)
for region in regions:
for offset in range(region.start, region.end):
if masked[offset] not in "\r\n":
masked[offset] = " "
return "".join(masked)
def check_document_text(text: str, label: str = "document") -> None:
"""Validate one current (non-historical) document."""
scan = scan_active_markdown(text, label)
regions = _regions(text, label, scan)
workspace = next(region for region in regions if region.name == WORKSPACE_REGION)
workspace_prose = _render_prose_markdown(
scan.prose_text[workspace.start : workspace.end]
)
if _exact_sentence_count(workspace_prose, CANONICAL_V3_SENTENCE) != 1:
raise ContractError(f"{label}: workspace contract lacks the canonical v3-only sentence")
if _exact_sentence_count(workspace_prose, CANONICAL_REJECTION_SENTENCE) != 1:
raise ContractError(
f"{label}: workspace contract lacks the canonical v1/v2 rejection sentence"
)
workspace_visible = _render_markdown(scan.operator_text[workspace.start : workspace.end])
workspace_remainder = workspace_visible.replace(CANONICAL_REJECTION_SENTENCE, " ", 1)
forbidden_claim = _MIGRATION_REQUIRED.search(
workspace_remainder
) or _REMAINING_LEGACY_CLAIM.search(workspace_remainder)
if forbidden_claim:
raise ContractError(
f"{label}: workspace contract contains an extra legacy descriptor claim: "
f"{forbidden_claim.group(0)}"
)
rendered_operator_text = _render_active_markdown(scan)
deleted = _DELETED_TOOL.search(rendered_operator_text)
if deleted:
raise ContractError(
f"{label}: current text contains deleted tool instruction: {deleted.group(0)}"
)
non_workspace = [region for region in regions if region.name == NON_WORKSPACE_REGION]
outside_allowed = _mask_region_ranges(
scan.operator_text,
[workspace, *non_workspace],
)
rendered_outside = _render_markdown(outside_allowed)
outside_claim = _MIGRATION_REQUIRED.search(
rendered_outside
) or _OUTSIDE_LEGACY_REFERENCE.search(rendered_outside)
if outside_claim:
raise ContractError(
f"{label}: current legacy schema or migration reference outside a "
f"{NON_WORKSPACE_REGION}:start/end region: {outside_claim.group(0)}"
)
def check_project_state_text(text: str, label: str = "PROJECT_STATE.md") -> None:
"""Validate current PROJECT_STATE text and its terminal historical archive hierarchy."""
scan = scan_active_markdown(text, label)
archive_headings = [
heading
for heading in scan.headings
if heading.level == 1
and heading.style == "atx"
and heading.raw == HISTORICAL_ARCHIVE_H1
]
if len(archive_headings) != 1:
raise ContractError(
f"{label}: expected exactly one active terminal '{HISTORICAL_ARCHIVE_H1}' boundary"
)
archive_heading = archive_headings[0]
archive_line_index = next(
index for index, line in enumerate(scan.lines) if line.start == archive_heading.start
)
if archive_line_index + 1 >= len(scan.lines) or not (
scan.lines[archive_line_index + 1].active
and scan.lines[archive_line_index + 1].text
== "## Historical snapshots and archived reference notes"
):
raise ContractError(
f"{label}: the preserved Historical snapshots H2 must immediately follow "
f"'{HISTORICAL_ARCHIVE_H1}'"
)
later_headings = [heading for heading in scan.headings if heading.start > archive_heading.start]
if not later_headings or not (
later_headings[0].level == 2
and later_headings[0].style == "atx"
and later_headings[0].raw == "## Historical snapshots and archived reference notes"
):
raise ContractError(
f"{label}: Historical archive must be followed by the existing Historical snapshots H2"
)
if any(heading.level == 1 for heading in later_headings):
raise ContractError(f"{label}: Historical archive must be the terminal H1 hierarchy")
check_document_text(text[:archive_heading.start], f"{label} current section")
def check_document_path(path: Path) -> None:
check_document_text(path.read_text(), str(path))
def check_project_state_path(path: Path) -> None:
check_project_state_text(path.read_text(), str(path))
def _parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--document", action="append", default=[], type=Path)
parser.add_argument("--project-state", action="append", default=[], type=Path)
return parser
def main(argv: Sequence[str] | None = None) -> int:
args = _parser().parse_args(argv)
if not args.document and not args.project_state:
raise SystemExit("at least one --document or --project-state path is required")
try:
for path in args.document:
check_document_path(path)
for path in args.project_state:
check_project_state_path(path)
except (ContractError, OSError) as error:
print(error, file=sys.stderr)
return 1
return 0
if __name__ == "__main__":
raise SystemExit(main())