From 9023f28cc0315404e48ecf8d449b2cf9a5feeb51 Mon Sep 17 00:00:00 2001 From: Andraxion Date: Wed, 29 Jul 2026 15:21:29 -0400 Subject: [PATCH] Add strict documentation graph gate --- Makefile | 8 +- tests/test_documentation_check.py | 176 ++++++++ tools/check_documentation.py | 698 ++++++++++++++++++++++++++++++ tools/docs_pending_m4_pages.txt | 13 + 4 files changed, 893 insertions(+), 2 deletions(-) create mode 100644 tests/test_documentation_check.py create mode 100644 tools/check_documentation.py create mode 100644 tools/docs_pending_m4_pages.txt diff --git a/Makefile b/Makefile index 5aa7783..203ae12 100644 --- a/Makefile +++ b/Makefile @@ -5,7 +5,7 @@ NPM := npm PYTHONPYCACHEPREFIX := /tmp/docforge-quality-pycache PYTEST_BASETEMP := /tmp/docforge-quality-pytest -.PHONY: accessibility adoption-m4 benchmark benchmark-m1 benchmark-m1-smoke benchmark-m2 benchmark-m2-smoke benchmark-m3 benchmark-m3-full benchmark-m3-smoke benchmark-m4 benchmark-m4-full benchmark-m4-smoke benchmark-smoke build command-reference-check compile contract dependencies format-check gate lint lock test type +.PHONY: accessibility adoption-m4 benchmark benchmark-m1 benchmark-m1-smoke benchmark-m2 benchmark-m2-smoke benchmark-m3 benchmark-m3-full benchmark-m3-smoke benchmark-m4 benchmark-m4-full benchmark-m4-smoke benchmark-smoke build command-reference-check compile contract dependencies docs-check format-check gate lint lock test type accessibility: $(NPM) run test:accessibility @@ -66,6 +66,10 @@ command-reference-check: $(PYTHON) tools/generate_command_reference.py \ --output docs/COMMAND_REFERENCE.md --check > /dev/null +docs-check: command-reference-check + $(PYTHON) tools/check_documentation.py \ + --pending-inventory tools/docs_pending_m4_pages.txt + benchmark-smoke: $(PYTHON) tools/milestone0_baseline.py --nodes 25 --samples 1 --cold-samples 1 \ --output /tmp/docforge-milestone0-smoke.json > /dev/null @@ -105,4 +109,4 @@ benchmark-m4: benchmark-m4-full: benchmark-m4 -gate: format-check lint type compile contract test accessibility lock dependencies build command-reference-check benchmark-smoke benchmark-m1-smoke benchmark-m2-smoke benchmark-m3-smoke benchmark-m4-smoke +gate: format-check lint type compile contract test accessibility lock dependencies build docs-check benchmark-smoke benchmark-m1-smoke benchmark-m2-smoke benchmark-m3-smoke benchmark-m4-smoke diff --git a/tests/test_documentation_check.py b/tests/test_documentation_check.py new file mode 100644 index 0000000..de2d851 --- /dev/null +++ b/tests/test_documentation_check.py @@ -0,0 +1,176 @@ +from __future__ import annotations + +import json +import tempfile +import unittest +from pathlib import Path + +from tools.check_documentation import ( + DocumentationPolicy, + check_documentation, +) + + +class DocumentationCheckTests(unittest.TestCase): + def repository(self, parent: Path) -> Path: + root = parent / "repository" + (root / "docs").mkdir(parents=True) + (root / "schemas").mkdir() + (root / "schemas" / "reference-adapter.schema.json").write_text( + json.dumps( + { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "additionalProperties": False, + "required": [ + "schema_version", + "project_id", + "title", + "language", + "source_roots", + ], + "properties": { + "schema_version": {"const": 1}, + "project_id": {"type": "string"}, + "title": {"type": "string"}, + "language": {"enum": ["python", "javascript", "typescript", "cpp"]}, + "source_roots": { + "type": "array", + "minItems": 1, + "items": {"type": "string"}, + }, + "compilation_database": {"type": "string"}, + }, + } + ), + encoding="utf-8", + ) + return root + + def test_linked_pages_h1_anchors_and_reference_example_pass(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = self.repository(Path(directory)) + (root / "README.md").write_text( + "# Project\n\n[Guide](docs/GUIDE.md#reference-adapter-configuration)\n", + encoding="utf-8", + ) + (root / "docs" / "GUIDE.md").write_text( + "\n".join( + ( + "# Guide", + "", + "## Reference adapter configuration", + "", + "`.docforge/reference-adapter.toml`:", + "", + "```toml", + "schema_version = 1", + 'project_id = "fixture"', + 'title = "Fixture"', + 'language = "python"', + 'source_roots = ["src"]', + "```", + "", + ) + ), + encoding="utf-8", + ) + + report = check_documentation( + root, + policy=DocumentationPolicy( + required_pages=(Path("README.md"), Path("docs/GUIDE.md")), + pending_inventory=None, + ), + ) + + self.assertTrue(report.ok, [item.render() for item in report.diagnostics]) + self.assertEqual(2, report.markdown_pages) + self.assertEqual(1, report.reference_adapter_examples) + + def test_missing_anchor_orphan_and_duplicate_h1_fail_deterministically(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = self.repository(Path(directory)) + (root / "README.md").write_text( + "# Project\n\n[Broken](docs/GUIDE.md#missing)\n", + encoding="utf-8", + ) + (root / "docs" / "GUIDE.md").write_text( + "# Guide\n\n# Duplicate\n", + encoding="utf-8", + ) + (root / "docs" / "ORPHAN.md").write_text("# Orphan\n", encoding="utf-8") + + report = check_documentation( + root, + policy=DocumentationPolicy( + required_pages=(Path("README.md"), Path("docs/GUIDE.md")), + pending_inventory=None, + ), + ) + + self.assertFalse(report.ok) + codes = [item.code for item in report.diagnostics] + self.assertEqual(["DOC016", "DOC017", "DOC023"], sorted(codes)) + + def test_pending_inventory_allows_absence_but_rejects_existing_page(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = self.repository(Path(directory)) + (root / "README.md").write_text("# Project\n", encoding="utf-8") + pending_path = root / "pending.txt" + pending_path.write_text("docs/FUTURE.md\n", encoding="utf-8") + policy = DocumentationPolicy( + required_pages=(Path("README.md"), Path("docs/FUTURE.md")), + pending_inventory=Path("pending.txt"), + ) + + missing = check_documentation(root, policy=policy) + self.assertTrue(missing.ok) + self.assertEqual(1, missing.pending_pages) + + (root / "docs" / "FUTURE.md").write_text("# Future\n", encoding="utf-8") + stale = check_documentation(root, policy=policy) + self.assertEqual(["DOC014", "DOC023"], [item.code for item in stale.diagnostics]) + + def test_invalid_reference_example_reports_schema_path(self) -> None: + with tempfile.TemporaryDirectory() as directory: + root = self.repository(Path(directory)) + (root / "README.md").write_text( + "# Project\n\n[Guide](docs/GUIDE.md)\n", + encoding="utf-8", + ) + (root / "docs" / "GUIDE.md").write_text( + "\n".join( + ( + "# Guide", + "", + "Reference adapter configuration:", + "", + "```toml", + 'language = "python"', + 'source_roots = ["src"]', + "```", + "", + ) + ), + encoding="utf-8", + ) + + report = check_documentation( + root, + policy=DocumentationPolicy( + required_pages=(Path("README.md"), Path("docs/GUIDE.md")), + pending_inventory=None, + ), + ) + + schema_errors = [item for item in report.diagnostics if item.code == "DOC026"] + self.assertEqual(3, len(schema_errors)) + messages = " ".join(item.message for item in schema_errors) + self.assertIn("schema_version", messages) + self.assertIn("project_id", messages) + self.assertIn("title", messages) + + +if __name__ == "__main__": + unittest.main() diff --git a/tools/check_documentation.py b/tools/check_documentation.py new file mode 100644 index 0000000..df32b56 --- /dev/null +++ b/tools/check_documentation.py @@ -0,0 +1,698 @@ +"""Validate the maintained DocForge Markdown documentation graph.""" + +from __future__ import annotations + +import argparse +import html +import json +import re +import stat +import sys +import tomllib +import unicodedata +from collections import deque +from collections.abc import Iterable, Sequence +from dataclasses import dataclass +from pathlib import Path +from typing import Protocol, cast +from urllib.parse import unquote, urlsplit + +from jsonschema import Draft202012Validator +from jsonschema.exceptions import ValidationError +from markdown_it import MarkdownIt +from markdown_it.token import Token + +REPOSITORY_ROOT = Path(__file__).resolve().parents[1] +DEFAULT_PENDING_INVENTORY = Path("tools/docs_pending_m4_pages.txt") +COMMAND_REFERENCE = Path("docs/COMMAND_REFERENCE.md") +COMMAND_REFERENCE_H1 = "DocForge command reference" +COMMAND_REFERENCE_NOTICE = ( + "> Generated from the live CLI parser and MCP registrations. Do not edit this file by hand." +) +REFERENCE_ADAPTER_SCHEMA = Path("schemas/reference-adapter.schema.json") +MAX_MARKDOWN_PAGES = 256 +MAX_PAGE_BYTES = 2_000_000 +MAX_TOTAL_BYTES = 20_000_000 +MAX_DIAGNOSTICS = 100 + +REQUIRED_MILESTONE_4_PAGES = ( + Path("README.md"), + Path("docs/ADAPTER_AUTHORING_GUIDE.md"), + Path("docs/AGENT_INTEGRATION.md"), + Path("docs/APPLICATION_DECISION.md"), + Path("docs/COMMAND_REFERENCE.md"), + Path("docs/COMPATIBILITY.md"), + Path("docs/CONTRACT.md"), + Path("docs/CORE_CONCEPTS_AND_AUTHORITY.md"), + Path("docs/INCREMENTAL_INDEXING.md"), + Path("docs/LEGACY_AND_NO_AST.md"), + Path("docs/MCP_CONTRACT.md"), + Path("docs/MIGRATING_FROM_V1.md"), + Path("docs/MILESTONE_4_BASELINE.md"), + Path("docs/MILESTONE_4_CLOSEOUT.md"), + Path("docs/NEW_PROJECT_QUICKSTART.md"), + Path("docs/POLICY_PRECEDENCE.md"), + Path("docs/PROJECT_DESCRIPTOR.md"), + Path("docs/PROJECT_ONBOARDING.md"), + Path("docs/RECOVERY_AND_PERFORMANCE.md"), + Path("docs/REFERENCE_ADAPTERS.md"), + Path("docs/RENDERING_AND_VISUALIZATION.md"), + Path("docs/SECURITY.md"), + Path("docs/USER_MANUAL.md"), + Path("docs/VIEWER_MANAGER.md"), +) + +_HISTORICAL_RECORD = re.compile(r"^docs/MILESTONE_[0-3]_(?:BASELINE|CLOSEOUT)\.md$") +_HTML_ID = re.compile( + r"""\bid\s*=\s*(?:"([^"]+)"|'([^']+)')""", + re.IGNORECASE, +) +_REFERENCE_ADAPTER_KEY = re.compile(r"(?m)^(?:language|source_roots|compilation_database)\s*=") +_URI_SCHEME = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:") + + +@dataclass(frozen=True, order=True) +class Diagnostic: + """One deterministic documentation validation failure.""" + + path: str + line: int + code: str + message: str + + def render(self) -> str: + """Render one bounded compiler-style diagnostic.""" + + return f"{self.path}:{self.line}: {self.code}: {self.message}" + + +@dataclass(frozen=True) +class DocumentationPolicy: + """Repository-relative pages and temporary inventory accepted by one run.""" + + required_pages: tuple[Path, ...] = REQUIRED_MILESTONE_4_PAGES + pending_inventory: Path | None = DEFAULT_PENDING_INVENTORY + + +@dataclass(frozen=True) +class DocumentationReport: + """Bounded deterministic result from one complete documentation check.""" + + diagnostics: tuple[Diagnostic, ...] + omitted_diagnostics: int + markdown_pages: int + pending_pages: int + reference_adapter_examples: int + + @property + def ok(self) -> bool: + """Return whether every documentation invariant passed.""" + + return not self.diagnostics and self.omitted_diagnostics == 0 + + +@dataclass(frozen=True) +class _Page: + relative: Path + source: str + tokens: tuple[Token, ...] + anchors: frozenset[str] + + +class _Diagnostics: + def __init__(self) -> None: + self._items: list[Diagnostic] = [] + self._omitted = 0 + + def add(self, path: Path | str, line: int, code: str, message: str) -> None: + relative = path.as_posix() if isinstance(path, Path) else path + item = Diagnostic(relative, max(1, line), code, message) + if len(self._items) < MAX_DIAGNOSTICS: + self._items.append(item) + else: + self._omitted += 1 + + def result(self) -> tuple[tuple[Diagnostic, ...], int]: + return tuple(sorted(self._items)), self._omitted + + +class _Validator(Protocol): + def iter_errors(self, instance: object) -> Iterable[ValidationError]: ... + + +def check_documentation( + repository_root: Path, + *, + policy: DocumentationPolicy | None = None, +) -> DocumentationReport: + """Check one repository's maintained Markdown pages without modifying it.""" + + root = repository_root.resolve(strict=True) + if not root.is_dir(): + raise ValueError(f"repository root is not a directory: {root}") + effective_policy = policy or DocumentationPolicy() + + diagnostics = _Diagnostics() + parser = MarkdownIt("commonmark", {"html": True, "typographer": False}) + page_paths = _discover_pages(root, diagnostics) + pages = _load_pages(root, page_paths, parser, diagnostics) + pending = _load_pending_inventory(root, effective_policy, diagnostics) + _check_required_inventory( + root, + effective_policy.required_pages, + pending, + diagnostics, + ) + _check_h1s(pages, diagnostics) + link_graph = _check_links(root, pages, parser, diagnostics) + _check_reachability(pages, link_graph, diagnostics) + reference_examples = _check_reference_adapter_examples(root, pages, diagnostics) + _check_command_reference(pages, diagnostics) + items, omitted = diagnostics.result() + return DocumentationReport( + diagnostics=items, + omitted_diagnostics=omitted, + markdown_pages=len(pages), + pending_pages=len(pending), + reference_adapter_examples=reference_examples, + ) + + +def _discover_pages(root: Path, diagnostics: _Diagnostics) -> tuple[Path, ...]: + candidates = [Path("README.md")] + docs = root / "docs" + if not docs.is_dir(): + diagnostics.add("docs", 1, "DOC001", "maintained documentation directory is missing") + return tuple(candidates) + + candidates.extend( + path.relative_to(root) + for path in docs.rglob("*.md") + if not path.is_symlink() and path.is_file() + ) + unique = tuple(sorted(set(candidates), key=Path.as_posix)) + if len(unique) > MAX_MARKDOWN_PAGES: + diagnostics.add( + "docs", + 1, + "DOC002", + f"found {len(unique)} Markdown pages; limit is {MAX_MARKDOWN_PAGES}", + ) + return unique[:MAX_MARKDOWN_PAGES] + return unique + + +def _load_pages( + root: Path, + paths: tuple[Path, ...], + parser: MarkdownIt, + diagnostics: _Diagnostics, +) -> dict[Path, _Page]: + pages: dict[Path, _Page] = {} + total_bytes = 0 + for relative in paths: + absolute = root / relative + try: + metadata = absolute.lstat() + except FileNotFoundError: + diagnostics.add(relative, 1, "DOC003", "maintained Markdown page is missing") + continue + if not stat.S_ISREG(metadata.st_mode): + diagnostics.add(relative, 1, "DOC004", "maintained Markdown page is not a regular file") + continue + if metadata.st_size > MAX_PAGE_BYTES: + diagnostics.add( + relative, + 1, + "DOC005", + f"page is {metadata.st_size} bytes; limit is {MAX_PAGE_BYTES}", + ) + continue + total_bytes += metadata.st_size + if total_bytes > MAX_TOTAL_BYTES: + diagnostics.add( + relative, + 1, + "DOC006", + f"total Markdown input exceeds {MAX_TOTAL_BYTES} bytes", + ) + break + try: + source = absolute.read_text(encoding="utf-8") + except UnicodeDecodeError: + diagnostics.add(relative, 1, "DOC007", "page is not valid UTF-8") + continue + tokens = tuple(parser.parse(source)) + pages[relative] = _Page( + relative=relative, + source=source, + tokens=tokens, + anchors=_anchors(tokens), + ) + return pages + + +def _load_pending_inventory( + root: Path, + policy: DocumentationPolicy, + diagnostics: _Diagnostics, +) -> frozenset[Path]: + inventory = policy.pending_inventory + if inventory is None: + return frozenset() + if inventory.is_absolute(): + diagnostics.add( + inventory.as_posix(), + 1, + "DOC008", + "pending inventory path must be repository-relative", + ) + return frozenset() + absolute = root / inventory + try: + lines = absolute.read_text(encoding="utf-8").splitlines() + except FileNotFoundError: + diagnostics.add(inventory, 1, "DOC009", "pending inventory file is missing") + return frozenset() + except UnicodeDecodeError: + diagnostics.add(inventory, 1, "DOC010", "pending inventory is not valid UTF-8") + return frozenset() + + required = set(policy.required_pages) + pending: set[Path] = set() + for line_number, raw in enumerate(lines, start=1): + value = raw.strip() + if not value or value.startswith("#"): + continue + path = Path(value) + if path.is_absolute() or ".." in path.parts or path.as_posix() != value: + diagnostics.add( + inventory, + line_number, + "DOC011", + f"unsafe pending inventory path: {value!r}", + ) + continue + if path not in required: + diagnostics.add( + inventory, + line_number, + "DOC012", + f"pending page is not in the required inventory: {value}", + ) + continue + if path in pending: + diagnostics.add( + inventory, + line_number, + "DOC013", + f"duplicate pending page: {value}", + ) + continue + pending.add(path) + return frozenset(pending) + + +def _check_required_inventory( + root: Path, + required_pages: tuple[Path, ...], + pending: frozenset[Path], + diagnostics: _Diagnostics, +) -> None: + for relative in sorted(set(required_pages), key=Path.as_posix): + exists = (root / relative).is_file() + if relative in pending: + if exists: + diagnostics.add( + relative, + 1, + "DOC014", + "page exists but remains in the pending inventory", + ) + continue + if not exists: + diagnostics.add(relative, 1, "DOC015", "required Milestone 4 page is missing") + + +def _check_h1s(pages: dict[Path, _Page], diagnostics: _Diagnostics) -> None: + for relative, page in pages.items(): + h1_lines = [ + _token_line(token) + for token in page.tokens + if token.type == "heading_open" and token.tag == "h1" + ] + if len(h1_lines) != 1: + rendered = ", ".join(str(line) for line in h1_lines) or "none" + diagnostics.add( + relative, + h1_lines[1] if len(h1_lines) > 1 else 1, + "DOC016", + f"expected exactly one H1; found {len(h1_lines)} (lines: {rendered})", + ) + + +def _check_links( + root: Path, + pages: dict[Path, _Page], + parser: MarkdownIt, + diagnostics: _Diagnostics, +) -> dict[Path, frozenset[Path]]: + graph: dict[Path, frozenset[Path]] = {} + anchor_cache = {relative: page.anchors for relative, page in pages.items()} + for relative, page in pages.items(): + destinations: set[Path] = set() + for href, line in _page_links(page.tokens): + target = _resolve_local_link(root, relative, href, line, diagnostics) + if target is None: + continue + target_relative, fragment = target + if target_relative in pages: + destinations.add(target_relative) + if not fragment: + continue + anchors = anchor_cache.get(target_relative) + if anchors is None: + anchors = _load_link_target_anchors( + root, + target_relative, + parser, + diagnostics, + ) + if anchors is None: + continue + anchor_cache[target_relative] = anchors + if fragment not in anchors: + diagnostics.add( + relative, + line, + "DOC017", + f"missing anchor #{fragment} in {target_relative.as_posix()}", + ) + graph[relative] = frozenset(destinations) + return graph + + +def _page_links(tokens: tuple[Token, ...]) -> Iterable[tuple[str, int]]: + for token in tokens: + children = token.children or [] + for child in children: + attribute = "href" if child.type == "link_open" else "src" + if child.type not in {"link_open", "image"}: + continue + raw = child.attrs.get(attribute) + if isinstance(raw, str): + yield raw, _token_line(token) + + +def _resolve_local_link( + root: Path, + source: Path, + href: str, + line: int, + diagnostics: _Diagnostics, +) -> tuple[Path, str] | None: + if not href or href.startswith("//") or _URI_SCHEME.match(href): + return None + try: + parsed = urlsplit(href) + except ValueError as error: + diagnostics.add(source, line, "DOC018", f"invalid link {href!r}: {error}") + return None + if parsed.netloc: + return None + decoded_path = unquote(parsed.path) + fragment = unquote(parsed.fragment) + if "\x00" in decoded_path or decoded_path.startswith("/"): + diagnostics.add(source, line, "DOC019", f"local link escapes the repository: {href!r}") + return None + absolute = (root / source if not decoded_path else root / source.parent / decoded_path).resolve( + strict=False + ) + try: + target = absolute.relative_to(root) + except ValueError: + diagnostics.add(source, line, "DOC019", f"local link escapes the repository: {href!r}") + return None + if not absolute.is_file(): + diagnostics.add(source, line, "DOC020", f"missing local link target: {target.as_posix()}") + return None + return target, fragment + + +def _load_link_target_anchors( + root: Path, + relative: Path, + parser: MarkdownIt, + diagnostics: _Diagnostics, +) -> frozenset[str] | None: + if relative.suffix.lower() not in {".md", ".markdown"}: + diagnostics.add( + relative, + 1, + "DOC021", + "link uses an anchor on a non-Markdown target", + ) + return None + absolute = root / relative + try: + metadata = absolute.lstat() + if not stat.S_ISREG(metadata.st_mode) or metadata.st_size > MAX_PAGE_BYTES: + raise ValueError("target is not a bounded regular Markdown file") + source = absolute.read_text(encoding="utf-8") + except (OSError, UnicodeDecodeError, ValueError) as error: + diagnostics.add(relative, 1, "DOC022", f"cannot inspect linked Markdown anchors: {error}") + return None + return _anchors(tuple(parser.parse(source))) + + +def _anchors(tokens: tuple[Token, ...]) -> frozenset[str]: + anchors: set[str] = set() + slug_counts: dict[str, int] = {} + for index, token in enumerate(tokens): + if token.type in {"html_block", "html_inline"}: + anchors.update(_html_ids(token.content)) + if token.children: + for child in token.children: + if child.type == "html_inline": + anchors.update(_html_ids(child.content)) + if token.type != "heading_open" or index + 1 >= len(tokens): + continue + inline = tokens[index + 1] + base = _heading_slug(_inline_text(inline)) + duplicate = slug_counts.get(base, 0) + slug_counts[base] = duplicate + 1 + anchors.add(base if duplicate == 0 else f"{base}-{duplicate}") + return frozenset(anchors) + + +def _inline_text(token: Token) -> str: + if not token.children: + return token.content + pieces: list[str] = [] + for child in token.children: + if child.type in {"text", "code_inline"}: + pieces.append(child.content) + elif child.type in {"softbreak", "hardbreak"}: + pieces.append(" ") + elif child.type == "image": + pieces.append(child.content) + elif child.type == "html_inline": + pieces.append(re.sub(r"<[^>]*>", "", child.content)) + return "".join(pieces) + + +def _heading_slug(value: str) -> str: + characters: list[str] = [] + for character in html.unescape(value).casefold().strip(): + category = unicodedata.category(character) + if category[0] in {"L", "M", "N"} or character in {"-", "_"}: + characters.append(character) + elif character.isspace(): + characters.append("-") + return "".join(characters) + + +def _html_ids(value: str) -> set[str]: + return {first or second for first, second in _HTML_ID.findall(value)} + + +def _check_reachability( + pages: dict[Path, _Page], + graph: dict[Path, frozenset[Path]], + diagnostics: _Diagnostics, +) -> None: + root_page = Path("README.md") + if root_page not in pages: + return + reached = {root_page} + queue = deque([root_page]) + while queue: + current = queue.popleft() + for target in sorted(graph.get(current, frozenset()), key=Path.as_posix): + if target not in reached: + reached.add(target) + queue.append(target) + for relative in sorted(set(pages) - reached, key=Path.as_posix): + if _HISTORICAL_RECORD.fullmatch(relative.as_posix()): + continue + diagnostics.add( + relative, + 1, + "DOC023", + "maintained page is not reachable from README.md", + ) + + +def _check_reference_adapter_examples( + root: Path, + pages: dict[Path, _Page], + diagnostics: _Diagnostics, +) -> int: + schema_path = root / REFERENCE_ADAPTER_SCHEMA + try: + schema_document = cast( + dict[str, object], + json.loads(schema_path.read_text(encoding="utf-8")), + ) + Draft202012Validator.check_schema(schema_document) + except (OSError, UnicodeDecodeError, json.JSONDecodeError, TypeError) as error: + diagnostics.add( + REFERENCE_ADAPTER_SCHEMA, + 1, + "DOC024", + f"cannot load reference-adapter schema: {error}", + ) + return 0 + validator = cast(_Validator, Draft202012Validator(schema_document)) + count = 0 + for relative, page in pages.items(): + previous_inline: Token | None = None + for token in page.tokens: + if token.type == "inline": + previous_inline = token + continue + if token.type != "fence" or token.info.split(maxsplit=1)[:1] != ["toml"]: + continue + if not _is_reference_adapter_example(token, previous_inline): + continue + count += 1 + line = _token_line(token) + try: + document = tomllib.loads(token.content) + except tomllib.TOMLDecodeError as error: + diagnostics.add( + relative, + line, + "DOC025", + f"invalid reference-adapter TOML example: {error}", + ) + continue + errors = sorted( + validator.iter_errors(document), + key=lambda error: tuple(str(part) for part in error.absolute_path), + ) + for error in errors: + location = ".".join(str(part) for part in error.absolute_path) or "" + diagnostics.add( + relative, + line, + "DOC026", + f"reference-adapter example {location}: {error.message}", + ) + return count + + +def _is_reference_adapter_example(token: Token, previous_inline: Token | None) -> bool: + info = token.info.casefold() + if "reference-adapter" in info or "reference_adapter" in info: + return True + if previous_inline is not None and previous_inline.map is not None and token.map is not None: + adjacent = token.map[0] - previous_inline.map[1] <= 1 + context = previous_inline.content.casefold() + if adjacent and ( + "reference-adapter.toml" in context or "reference adapter configuration" in context + ): + return True + keys = set(_REFERENCE_ADAPTER_KEY.findall(token.content)) + return {"language", "source_roots"}.issubset(keys) + + +def _check_command_reference( + pages: dict[Path, _Page], + diagnostics: _Diagnostics, +) -> None: + page = pages.get(COMMAND_REFERENCE) + if page is None: + return + lines = page.source.splitlines() + if not lines or lines[0] != f"# {COMMAND_REFERENCE_H1}": + diagnostics.add( + COMMAND_REFERENCE, + 1, + "DOC027", + f"generated reference must begin with '# {COMMAND_REFERENCE_H1}'", + ) + if COMMAND_REFERENCE_NOTICE not in lines[:5]: + diagnostics.add( + COMMAND_REFERENCE, + 2, + "DOC028", + "generated reference notice is missing or changed", + ) + + +def _token_line(token: Token) -> int: + return token.map[0] + 1 if token.map else 1 + + +def _parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser( + description="Check deterministic DocForge documentation invariants." + ) + parser.add_argument("--repository-root", type=Path, default=REPOSITORY_ROOT) + parser.add_argument( + "--pending-inventory", + type=Path, + default=DEFAULT_PENDING_INVENTORY, + help="repository-relative list of required pages not authored yet", + ) + return parser + + +def main(arguments: Sequence[str] | None = None) -> int: + parsed = _parser().parse_args(arguments) + try: + report = check_documentation( + parsed.repository_root, + policy=DocumentationPolicy(pending_inventory=parsed.pending_inventory), + ) + except Exception as error: + print(f"documentation check failed: {error}", file=sys.stderr) + return 2 + if not report.ok: + for diagnostic in report.diagnostics: + print(diagnostic.render(), file=sys.stderr) + if report.omitted_diagnostics: + print( + f"... {report.omitted_diagnostics} additional diagnostics omitted " + f"(limit {MAX_DIAGNOSTICS})", + file=sys.stderr, + ) + return 1 + print( + json.dumps( + { + "markdown_pages": report.markdown_pages, + "pending_pages": report.pending_pages, + "reference_adapter_examples": report.reference_adapter_examples, + "status": "ok", + }, + sort_keys=True, + separators=(",", ":"), + ) + ) + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tools/docs_pending_m4_pages.txt b/tools/docs_pending_m4_pages.txt new file mode 100644 index 0000000..852d59b --- /dev/null +++ b/tools/docs_pending_m4_pages.txt @@ -0,0 +1,13 @@ +# Remove each entry in the same commit that authors the required page. +docs/AGENT_INTEGRATION.md +docs/CORE_CONCEPTS_AND_AUTHORITY.md +docs/LEGACY_AND_NO_AST.md +docs/MIGRATING_FROM_V1.md +docs/MILESTONE_4_BASELINE.md +docs/MILESTONE_4_CLOSEOUT.md +docs/POLICY_PRECEDENCE.md +docs/PROJECT_DESCRIPTOR.md +docs/RECOVERY_AND_PERFORMANCE.md +docs/REFERENCE_ADAPTERS.md +docs/RENDERING_AND_VISUALIZATION.md +docs/SECURITY.md