Add strict documentation graph gate
This commit is contained in:
parent
983a55f156
commit
9023f28cc0
4 changed files with 893 additions and 2 deletions
8
Makefile
8
Makefile
|
|
@ -5,7 +5,7 @@ NPM := npm
|
|||
PYTHONPYCACHEPREFIX := /tmp/docforge-quality-pycache
|
||||
PYTEST_BASETEMP := /tmp/docforge-quality-pytest
|
||||
|
||||
.PHONY: accessibility adoption-m4 benchmark benchmark-m1 benchmark-m1-smoke benchmark-m2 benchmark-m2-smoke benchmark-m3 benchmark-m3-full benchmark-m3-smoke benchmark-m4 benchmark-m4-full benchmark-m4-smoke benchmark-smoke build command-reference-check compile contract dependencies format-check gate lint lock test type
|
||||
.PHONY: accessibility adoption-m4 benchmark benchmark-m1 benchmark-m1-smoke benchmark-m2 benchmark-m2-smoke benchmark-m3 benchmark-m3-full benchmark-m3-smoke benchmark-m4 benchmark-m4-full benchmark-m4-smoke benchmark-smoke build command-reference-check compile contract dependencies docs-check format-check gate lint lock test type
|
||||
|
||||
accessibility:
|
||||
$(NPM) run test:accessibility
|
||||
|
|
@ -66,6 +66,10 @@ command-reference-check:
|
|||
$(PYTHON) tools/generate_command_reference.py \
|
||||
--output docs/COMMAND_REFERENCE.md --check > /dev/null
|
||||
|
||||
docs-check: command-reference-check
|
||||
$(PYTHON) tools/check_documentation.py \
|
||||
--pending-inventory tools/docs_pending_m4_pages.txt
|
||||
|
||||
benchmark-smoke:
|
||||
$(PYTHON) tools/milestone0_baseline.py --nodes 25 --samples 1 --cold-samples 1 \
|
||||
--output /tmp/docforge-milestone0-smoke.json > /dev/null
|
||||
|
|
@ -105,4 +109,4 @@ benchmark-m4:
|
|||
|
||||
benchmark-m4-full: benchmark-m4
|
||||
|
||||
gate: format-check lint type compile contract test accessibility lock dependencies build command-reference-check benchmark-smoke benchmark-m1-smoke benchmark-m2-smoke benchmark-m3-smoke benchmark-m4-smoke
|
||||
gate: format-check lint type compile contract test accessibility lock dependencies build docs-check benchmark-smoke benchmark-m1-smoke benchmark-m2-smoke benchmark-m3-smoke benchmark-m4-smoke
|
||||
|
|
|
|||
176
tests/test_documentation_check.py
Normal file
176
tests/test_documentation_check.py
Normal file
|
|
@ -0,0 +1,176 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
from tools.check_documentation import (
|
||||
DocumentationPolicy,
|
||||
check_documentation,
|
||||
)
|
||||
|
||||
|
||||
class DocumentationCheckTests(unittest.TestCase):
|
||||
def repository(self, parent: Path) -> Path:
|
||||
root = parent / "repository"
|
||||
(root / "docs").mkdir(parents=True)
|
||||
(root / "schemas").mkdir()
|
||||
(root / "schemas" / "reference-adapter.schema.json").write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
||||
"type": "object",
|
||||
"additionalProperties": False,
|
||||
"required": [
|
||||
"schema_version",
|
||||
"project_id",
|
||||
"title",
|
||||
"language",
|
||||
"source_roots",
|
||||
],
|
||||
"properties": {
|
||||
"schema_version": {"const": 1},
|
||||
"project_id": {"type": "string"},
|
||||
"title": {"type": "string"},
|
||||
"language": {"enum": ["python", "javascript", "typescript", "cpp"]},
|
||||
"source_roots": {
|
||||
"type": "array",
|
||||
"minItems": 1,
|
||||
"items": {"type": "string"},
|
||||
},
|
||||
"compilation_database": {"type": "string"},
|
||||
},
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
return root
|
||||
|
||||
def test_linked_pages_h1_anchors_and_reference_example_pass(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = self.repository(Path(directory))
|
||||
(root / "README.md").write_text(
|
||||
"# Project\n\n[Guide](docs/GUIDE.md#reference-adapter-configuration)\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(root / "docs" / "GUIDE.md").write_text(
|
||||
"\n".join(
|
||||
(
|
||||
"# Guide",
|
||||
"",
|
||||
"## Reference adapter configuration",
|
||||
"",
|
||||
"`.docforge/reference-adapter.toml`:",
|
||||
"",
|
||||
"```toml",
|
||||
"schema_version = 1",
|
||||
'project_id = "fixture"',
|
||||
'title = "Fixture"',
|
||||
'language = "python"',
|
||||
'source_roots = ["src"]',
|
||||
"```",
|
||||
"",
|
||||
)
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
report = check_documentation(
|
||||
root,
|
||||
policy=DocumentationPolicy(
|
||||
required_pages=(Path("README.md"), Path("docs/GUIDE.md")),
|
||||
pending_inventory=None,
|
||||
),
|
||||
)
|
||||
|
||||
self.assertTrue(report.ok, [item.render() for item in report.diagnostics])
|
||||
self.assertEqual(2, report.markdown_pages)
|
||||
self.assertEqual(1, report.reference_adapter_examples)
|
||||
|
||||
def test_missing_anchor_orphan_and_duplicate_h1_fail_deterministically(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = self.repository(Path(directory))
|
||||
(root / "README.md").write_text(
|
||||
"# Project\n\n[Broken](docs/GUIDE.md#missing)\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(root / "docs" / "GUIDE.md").write_text(
|
||||
"# Guide\n\n# Duplicate\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(root / "docs" / "ORPHAN.md").write_text("# Orphan\n", encoding="utf-8")
|
||||
|
||||
report = check_documentation(
|
||||
root,
|
||||
policy=DocumentationPolicy(
|
||||
required_pages=(Path("README.md"), Path("docs/GUIDE.md")),
|
||||
pending_inventory=None,
|
||||
),
|
||||
)
|
||||
|
||||
self.assertFalse(report.ok)
|
||||
codes = [item.code for item in report.diagnostics]
|
||||
self.assertEqual(["DOC016", "DOC017", "DOC023"], sorted(codes))
|
||||
|
||||
def test_pending_inventory_allows_absence_but_rejects_existing_page(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = self.repository(Path(directory))
|
||||
(root / "README.md").write_text("# Project\n", encoding="utf-8")
|
||||
pending_path = root / "pending.txt"
|
||||
pending_path.write_text("docs/FUTURE.md\n", encoding="utf-8")
|
||||
policy = DocumentationPolicy(
|
||||
required_pages=(Path("README.md"), Path("docs/FUTURE.md")),
|
||||
pending_inventory=Path("pending.txt"),
|
||||
)
|
||||
|
||||
missing = check_documentation(root, policy=policy)
|
||||
self.assertTrue(missing.ok)
|
||||
self.assertEqual(1, missing.pending_pages)
|
||||
|
||||
(root / "docs" / "FUTURE.md").write_text("# Future\n", encoding="utf-8")
|
||||
stale = check_documentation(root, policy=policy)
|
||||
self.assertEqual(["DOC014", "DOC023"], [item.code for item in stale.diagnostics])
|
||||
|
||||
def test_invalid_reference_example_reports_schema_path(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = self.repository(Path(directory))
|
||||
(root / "README.md").write_text(
|
||||
"# Project\n\n[Guide](docs/GUIDE.md)\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(root / "docs" / "GUIDE.md").write_text(
|
||||
"\n".join(
|
||||
(
|
||||
"# Guide",
|
||||
"",
|
||||
"Reference adapter configuration:",
|
||||
"",
|
||||
"```toml",
|
||||
'language = "python"',
|
||||
'source_roots = ["src"]',
|
||||
"```",
|
||||
"",
|
||||
)
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
report = check_documentation(
|
||||
root,
|
||||
policy=DocumentationPolicy(
|
||||
required_pages=(Path("README.md"), Path("docs/GUIDE.md")),
|
||||
pending_inventory=None,
|
||||
),
|
||||
)
|
||||
|
||||
schema_errors = [item for item in report.diagnostics if item.code == "DOC026"]
|
||||
self.assertEqual(3, len(schema_errors))
|
||||
messages = " ".join(item.message for item in schema_errors)
|
||||
self.assertIn("schema_version", messages)
|
||||
self.assertIn("project_id", messages)
|
||||
self.assertIn("title", messages)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
698
tools/check_documentation.py
Normal file
698
tools/check_documentation.py
Normal file
|
|
@ -0,0 +1,698 @@
|
|||
"""Validate the maintained DocForge Markdown documentation graph."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import html
|
||||
import json
|
||||
import re
|
||||
import stat
|
||||
import sys
|
||||
import tomllib
|
||||
import unicodedata
|
||||
from collections import deque
|
||||
from collections.abc import Iterable, Sequence
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Protocol, cast
|
||||
from urllib.parse import unquote, urlsplit
|
||||
|
||||
from jsonschema import Draft202012Validator
|
||||
from jsonschema.exceptions import ValidationError
|
||||
from markdown_it import MarkdownIt
|
||||
from markdown_it.token import Token
|
||||
|
||||
REPOSITORY_ROOT = Path(__file__).resolve().parents[1]
|
||||
DEFAULT_PENDING_INVENTORY = Path("tools/docs_pending_m4_pages.txt")
|
||||
COMMAND_REFERENCE = Path("docs/COMMAND_REFERENCE.md")
|
||||
COMMAND_REFERENCE_H1 = "DocForge command reference"
|
||||
COMMAND_REFERENCE_NOTICE = (
|
||||
"> Generated from the live CLI parser and MCP registrations. Do not edit this file by hand."
|
||||
)
|
||||
REFERENCE_ADAPTER_SCHEMA = Path("schemas/reference-adapter.schema.json")
|
||||
MAX_MARKDOWN_PAGES = 256
|
||||
MAX_PAGE_BYTES = 2_000_000
|
||||
MAX_TOTAL_BYTES = 20_000_000
|
||||
MAX_DIAGNOSTICS = 100
|
||||
|
||||
REQUIRED_MILESTONE_4_PAGES = (
|
||||
Path("README.md"),
|
||||
Path("docs/ADAPTER_AUTHORING_GUIDE.md"),
|
||||
Path("docs/AGENT_INTEGRATION.md"),
|
||||
Path("docs/APPLICATION_DECISION.md"),
|
||||
Path("docs/COMMAND_REFERENCE.md"),
|
||||
Path("docs/COMPATIBILITY.md"),
|
||||
Path("docs/CONTRACT.md"),
|
||||
Path("docs/CORE_CONCEPTS_AND_AUTHORITY.md"),
|
||||
Path("docs/INCREMENTAL_INDEXING.md"),
|
||||
Path("docs/LEGACY_AND_NO_AST.md"),
|
||||
Path("docs/MCP_CONTRACT.md"),
|
||||
Path("docs/MIGRATING_FROM_V1.md"),
|
||||
Path("docs/MILESTONE_4_BASELINE.md"),
|
||||
Path("docs/MILESTONE_4_CLOSEOUT.md"),
|
||||
Path("docs/NEW_PROJECT_QUICKSTART.md"),
|
||||
Path("docs/POLICY_PRECEDENCE.md"),
|
||||
Path("docs/PROJECT_DESCRIPTOR.md"),
|
||||
Path("docs/PROJECT_ONBOARDING.md"),
|
||||
Path("docs/RECOVERY_AND_PERFORMANCE.md"),
|
||||
Path("docs/REFERENCE_ADAPTERS.md"),
|
||||
Path("docs/RENDERING_AND_VISUALIZATION.md"),
|
||||
Path("docs/SECURITY.md"),
|
||||
Path("docs/USER_MANUAL.md"),
|
||||
Path("docs/VIEWER_MANAGER.md"),
|
||||
)
|
||||
|
||||
_HISTORICAL_RECORD = re.compile(r"^docs/MILESTONE_[0-3]_(?:BASELINE|CLOSEOUT)\.md$")
|
||||
_HTML_ID = re.compile(
|
||||
r"""\bid\s*=\s*(?:"([^"]+)"|'([^']+)')""",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_REFERENCE_ADAPTER_KEY = re.compile(r"(?m)^(?:language|source_roots|compilation_database)\s*=")
|
||||
_URI_SCHEME = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:")
|
||||
|
||||
|
||||
@dataclass(frozen=True, order=True)
|
||||
class Diagnostic:
|
||||
"""One deterministic documentation validation failure."""
|
||||
|
||||
path: str
|
||||
line: int
|
||||
code: str
|
||||
message: str
|
||||
|
||||
def render(self) -> str:
|
||||
"""Render one bounded compiler-style diagnostic."""
|
||||
|
||||
return f"{self.path}:{self.line}: {self.code}: {self.message}"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DocumentationPolicy:
|
||||
"""Repository-relative pages and temporary inventory accepted by one run."""
|
||||
|
||||
required_pages: tuple[Path, ...] = REQUIRED_MILESTONE_4_PAGES
|
||||
pending_inventory: Path | None = DEFAULT_PENDING_INVENTORY
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class DocumentationReport:
|
||||
"""Bounded deterministic result from one complete documentation check."""
|
||||
|
||||
diagnostics: tuple[Diagnostic, ...]
|
||||
omitted_diagnostics: int
|
||||
markdown_pages: int
|
||||
pending_pages: int
|
||||
reference_adapter_examples: int
|
||||
|
||||
@property
|
||||
def ok(self) -> bool:
|
||||
"""Return whether every documentation invariant passed."""
|
||||
|
||||
return not self.diagnostics and self.omitted_diagnostics == 0
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _Page:
|
||||
relative: Path
|
||||
source: str
|
||||
tokens: tuple[Token, ...]
|
||||
anchors: frozenset[str]
|
||||
|
||||
|
||||
class _Diagnostics:
|
||||
def __init__(self) -> None:
|
||||
self._items: list[Diagnostic] = []
|
||||
self._omitted = 0
|
||||
|
||||
def add(self, path: Path | str, line: int, code: str, message: str) -> None:
|
||||
relative = path.as_posix() if isinstance(path, Path) else path
|
||||
item = Diagnostic(relative, max(1, line), code, message)
|
||||
if len(self._items) < MAX_DIAGNOSTICS:
|
||||
self._items.append(item)
|
||||
else:
|
||||
self._omitted += 1
|
||||
|
||||
def result(self) -> tuple[tuple[Diagnostic, ...], int]:
|
||||
return tuple(sorted(self._items)), self._omitted
|
||||
|
||||
|
||||
class _Validator(Protocol):
|
||||
def iter_errors(self, instance: object) -> Iterable[ValidationError]: ...
|
||||
|
||||
|
||||
def check_documentation(
|
||||
repository_root: Path,
|
||||
*,
|
||||
policy: DocumentationPolicy | None = None,
|
||||
) -> DocumentationReport:
|
||||
"""Check one repository's maintained Markdown pages without modifying it."""
|
||||
|
||||
root = repository_root.resolve(strict=True)
|
||||
if not root.is_dir():
|
||||
raise ValueError(f"repository root is not a directory: {root}")
|
||||
effective_policy = policy or DocumentationPolicy()
|
||||
|
||||
diagnostics = _Diagnostics()
|
||||
parser = MarkdownIt("commonmark", {"html": True, "typographer": False})
|
||||
page_paths = _discover_pages(root, diagnostics)
|
||||
pages = _load_pages(root, page_paths, parser, diagnostics)
|
||||
pending = _load_pending_inventory(root, effective_policy, diagnostics)
|
||||
_check_required_inventory(
|
||||
root,
|
||||
effective_policy.required_pages,
|
||||
pending,
|
||||
diagnostics,
|
||||
)
|
||||
_check_h1s(pages, diagnostics)
|
||||
link_graph = _check_links(root, pages, parser, diagnostics)
|
||||
_check_reachability(pages, link_graph, diagnostics)
|
||||
reference_examples = _check_reference_adapter_examples(root, pages, diagnostics)
|
||||
_check_command_reference(pages, diagnostics)
|
||||
items, omitted = diagnostics.result()
|
||||
return DocumentationReport(
|
||||
diagnostics=items,
|
||||
omitted_diagnostics=omitted,
|
||||
markdown_pages=len(pages),
|
||||
pending_pages=len(pending),
|
||||
reference_adapter_examples=reference_examples,
|
||||
)
|
||||
|
||||
|
||||
def _discover_pages(root: Path, diagnostics: _Diagnostics) -> tuple[Path, ...]:
|
||||
candidates = [Path("README.md")]
|
||||
docs = root / "docs"
|
||||
if not docs.is_dir():
|
||||
diagnostics.add("docs", 1, "DOC001", "maintained documentation directory is missing")
|
||||
return tuple(candidates)
|
||||
|
||||
candidates.extend(
|
||||
path.relative_to(root)
|
||||
for path in docs.rglob("*.md")
|
||||
if not path.is_symlink() and path.is_file()
|
||||
)
|
||||
unique = tuple(sorted(set(candidates), key=Path.as_posix))
|
||||
if len(unique) > MAX_MARKDOWN_PAGES:
|
||||
diagnostics.add(
|
||||
"docs",
|
||||
1,
|
||||
"DOC002",
|
||||
f"found {len(unique)} Markdown pages; limit is {MAX_MARKDOWN_PAGES}",
|
||||
)
|
||||
return unique[:MAX_MARKDOWN_PAGES]
|
||||
return unique
|
||||
|
||||
|
||||
def _load_pages(
|
||||
root: Path,
|
||||
paths: tuple[Path, ...],
|
||||
parser: MarkdownIt,
|
||||
diagnostics: _Diagnostics,
|
||||
) -> dict[Path, _Page]:
|
||||
pages: dict[Path, _Page] = {}
|
||||
total_bytes = 0
|
||||
for relative in paths:
|
||||
absolute = root / relative
|
||||
try:
|
||||
metadata = absolute.lstat()
|
||||
except FileNotFoundError:
|
||||
diagnostics.add(relative, 1, "DOC003", "maintained Markdown page is missing")
|
||||
continue
|
||||
if not stat.S_ISREG(metadata.st_mode):
|
||||
diagnostics.add(relative, 1, "DOC004", "maintained Markdown page is not a regular file")
|
||||
continue
|
||||
if metadata.st_size > MAX_PAGE_BYTES:
|
||||
diagnostics.add(
|
||||
relative,
|
||||
1,
|
||||
"DOC005",
|
||||
f"page is {metadata.st_size} bytes; limit is {MAX_PAGE_BYTES}",
|
||||
)
|
||||
continue
|
||||
total_bytes += metadata.st_size
|
||||
if total_bytes > MAX_TOTAL_BYTES:
|
||||
diagnostics.add(
|
||||
relative,
|
||||
1,
|
||||
"DOC006",
|
||||
f"total Markdown input exceeds {MAX_TOTAL_BYTES} bytes",
|
||||
)
|
||||
break
|
||||
try:
|
||||
source = absolute.read_text(encoding="utf-8")
|
||||
except UnicodeDecodeError:
|
||||
diagnostics.add(relative, 1, "DOC007", "page is not valid UTF-8")
|
||||
continue
|
||||
tokens = tuple(parser.parse(source))
|
||||
pages[relative] = _Page(
|
||||
relative=relative,
|
||||
source=source,
|
||||
tokens=tokens,
|
||||
anchors=_anchors(tokens),
|
||||
)
|
||||
return pages
|
||||
|
||||
|
||||
def _load_pending_inventory(
|
||||
root: Path,
|
||||
policy: DocumentationPolicy,
|
||||
diagnostics: _Diagnostics,
|
||||
) -> frozenset[Path]:
|
||||
inventory = policy.pending_inventory
|
||||
if inventory is None:
|
||||
return frozenset()
|
||||
if inventory.is_absolute():
|
||||
diagnostics.add(
|
||||
inventory.as_posix(),
|
||||
1,
|
||||
"DOC008",
|
||||
"pending inventory path must be repository-relative",
|
||||
)
|
||||
return frozenset()
|
||||
absolute = root / inventory
|
||||
try:
|
||||
lines = absolute.read_text(encoding="utf-8").splitlines()
|
||||
except FileNotFoundError:
|
||||
diagnostics.add(inventory, 1, "DOC009", "pending inventory file is missing")
|
||||
return frozenset()
|
||||
except UnicodeDecodeError:
|
||||
diagnostics.add(inventory, 1, "DOC010", "pending inventory is not valid UTF-8")
|
||||
return frozenset()
|
||||
|
||||
required = set(policy.required_pages)
|
||||
pending: set[Path] = set()
|
||||
for line_number, raw in enumerate(lines, start=1):
|
||||
value = raw.strip()
|
||||
if not value or value.startswith("#"):
|
||||
continue
|
||||
path = Path(value)
|
||||
if path.is_absolute() or ".." in path.parts or path.as_posix() != value:
|
||||
diagnostics.add(
|
||||
inventory,
|
||||
line_number,
|
||||
"DOC011",
|
||||
f"unsafe pending inventory path: {value!r}",
|
||||
)
|
||||
continue
|
||||
if path not in required:
|
||||
diagnostics.add(
|
||||
inventory,
|
||||
line_number,
|
||||
"DOC012",
|
||||
f"pending page is not in the required inventory: {value}",
|
||||
)
|
||||
continue
|
||||
if path in pending:
|
||||
diagnostics.add(
|
||||
inventory,
|
||||
line_number,
|
||||
"DOC013",
|
||||
f"duplicate pending page: {value}",
|
||||
)
|
||||
continue
|
||||
pending.add(path)
|
||||
return frozenset(pending)
|
||||
|
||||
|
||||
def _check_required_inventory(
|
||||
root: Path,
|
||||
required_pages: tuple[Path, ...],
|
||||
pending: frozenset[Path],
|
||||
diagnostics: _Diagnostics,
|
||||
) -> None:
|
||||
for relative in sorted(set(required_pages), key=Path.as_posix):
|
||||
exists = (root / relative).is_file()
|
||||
if relative in pending:
|
||||
if exists:
|
||||
diagnostics.add(
|
||||
relative,
|
||||
1,
|
||||
"DOC014",
|
||||
"page exists but remains in the pending inventory",
|
||||
)
|
||||
continue
|
||||
if not exists:
|
||||
diagnostics.add(relative, 1, "DOC015", "required Milestone 4 page is missing")
|
||||
|
||||
|
||||
def _check_h1s(pages: dict[Path, _Page], diagnostics: _Diagnostics) -> None:
|
||||
for relative, page in pages.items():
|
||||
h1_lines = [
|
||||
_token_line(token)
|
||||
for token in page.tokens
|
||||
if token.type == "heading_open" and token.tag == "h1"
|
||||
]
|
||||
if len(h1_lines) != 1:
|
||||
rendered = ", ".join(str(line) for line in h1_lines) or "none"
|
||||
diagnostics.add(
|
||||
relative,
|
||||
h1_lines[1] if len(h1_lines) > 1 else 1,
|
||||
"DOC016",
|
||||
f"expected exactly one H1; found {len(h1_lines)} (lines: {rendered})",
|
||||
)
|
||||
|
||||
|
||||
def _check_links(
|
||||
root: Path,
|
||||
pages: dict[Path, _Page],
|
||||
parser: MarkdownIt,
|
||||
diagnostics: _Diagnostics,
|
||||
) -> dict[Path, frozenset[Path]]:
|
||||
graph: dict[Path, frozenset[Path]] = {}
|
||||
anchor_cache = {relative: page.anchors for relative, page in pages.items()}
|
||||
for relative, page in pages.items():
|
||||
destinations: set[Path] = set()
|
||||
for href, line in _page_links(page.tokens):
|
||||
target = _resolve_local_link(root, relative, href, line, diagnostics)
|
||||
if target is None:
|
||||
continue
|
||||
target_relative, fragment = target
|
||||
if target_relative in pages:
|
||||
destinations.add(target_relative)
|
||||
if not fragment:
|
||||
continue
|
||||
anchors = anchor_cache.get(target_relative)
|
||||
if anchors is None:
|
||||
anchors = _load_link_target_anchors(
|
||||
root,
|
||||
target_relative,
|
||||
parser,
|
||||
diagnostics,
|
||||
)
|
||||
if anchors is None:
|
||||
continue
|
||||
anchor_cache[target_relative] = anchors
|
||||
if fragment not in anchors:
|
||||
diagnostics.add(
|
||||
relative,
|
||||
line,
|
||||
"DOC017",
|
||||
f"missing anchor #{fragment} in {target_relative.as_posix()}",
|
||||
)
|
||||
graph[relative] = frozenset(destinations)
|
||||
return graph
|
||||
|
||||
|
||||
def _page_links(tokens: tuple[Token, ...]) -> Iterable[tuple[str, int]]:
|
||||
for token in tokens:
|
||||
children = token.children or []
|
||||
for child in children:
|
||||
attribute = "href" if child.type == "link_open" else "src"
|
||||
if child.type not in {"link_open", "image"}:
|
||||
continue
|
||||
raw = child.attrs.get(attribute)
|
||||
if isinstance(raw, str):
|
||||
yield raw, _token_line(token)
|
||||
|
||||
|
||||
def _resolve_local_link(
|
||||
root: Path,
|
||||
source: Path,
|
||||
href: str,
|
||||
line: int,
|
||||
diagnostics: _Diagnostics,
|
||||
) -> tuple[Path, str] | None:
|
||||
if not href or href.startswith("//") or _URI_SCHEME.match(href):
|
||||
return None
|
||||
try:
|
||||
parsed = urlsplit(href)
|
||||
except ValueError as error:
|
||||
diagnostics.add(source, line, "DOC018", f"invalid link {href!r}: {error}")
|
||||
return None
|
||||
if parsed.netloc:
|
||||
return None
|
||||
decoded_path = unquote(parsed.path)
|
||||
fragment = unquote(parsed.fragment)
|
||||
if "\x00" in decoded_path or decoded_path.startswith("/"):
|
||||
diagnostics.add(source, line, "DOC019", f"local link escapes the repository: {href!r}")
|
||||
return None
|
||||
absolute = (root / source if not decoded_path else root / source.parent / decoded_path).resolve(
|
||||
strict=False
|
||||
)
|
||||
try:
|
||||
target = absolute.relative_to(root)
|
||||
except ValueError:
|
||||
diagnostics.add(source, line, "DOC019", f"local link escapes the repository: {href!r}")
|
||||
return None
|
||||
if not absolute.is_file():
|
||||
diagnostics.add(source, line, "DOC020", f"missing local link target: {target.as_posix()}")
|
||||
return None
|
||||
return target, fragment
|
||||
|
||||
|
||||
def _load_link_target_anchors(
|
||||
root: Path,
|
||||
relative: Path,
|
||||
parser: MarkdownIt,
|
||||
diagnostics: _Diagnostics,
|
||||
) -> frozenset[str] | None:
|
||||
if relative.suffix.lower() not in {".md", ".markdown"}:
|
||||
diagnostics.add(
|
||||
relative,
|
||||
1,
|
||||
"DOC021",
|
||||
"link uses an anchor on a non-Markdown target",
|
||||
)
|
||||
return None
|
||||
absolute = root / relative
|
||||
try:
|
||||
metadata = absolute.lstat()
|
||||
if not stat.S_ISREG(metadata.st_mode) or metadata.st_size > MAX_PAGE_BYTES:
|
||||
raise ValueError("target is not a bounded regular Markdown file")
|
||||
source = absolute.read_text(encoding="utf-8")
|
||||
except (OSError, UnicodeDecodeError, ValueError) as error:
|
||||
diagnostics.add(relative, 1, "DOC022", f"cannot inspect linked Markdown anchors: {error}")
|
||||
return None
|
||||
return _anchors(tuple(parser.parse(source)))
|
||||
|
||||
|
||||
def _anchors(tokens: tuple[Token, ...]) -> frozenset[str]:
|
||||
anchors: set[str] = set()
|
||||
slug_counts: dict[str, int] = {}
|
||||
for index, token in enumerate(tokens):
|
||||
if token.type in {"html_block", "html_inline"}:
|
||||
anchors.update(_html_ids(token.content))
|
||||
if token.children:
|
||||
for child in token.children:
|
||||
if child.type == "html_inline":
|
||||
anchors.update(_html_ids(child.content))
|
||||
if token.type != "heading_open" or index + 1 >= len(tokens):
|
||||
continue
|
||||
inline = tokens[index + 1]
|
||||
base = _heading_slug(_inline_text(inline))
|
||||
duplicate = slug_counts.get(base, 0)
|
||||
slug_counts[base] = duplicate + 1
|
||||
anchors.add(base if duplicate == 0 else f"{base}-{duplicate}")
|
||||
return frozenset(anchors)
|
||||
|
||||
|
||||
def _inline_text(token: Token) -> str:
|
||||
if not token.children:
|
||||
return token.content
|
||||
pieces: list[str] = []
|
||||
for child in token.children:
|
||||
if child.type in {"text", "code_inline"}:
|
||||
pieces.append(child.content)
|
||||
elif child.type in {"softbreak", "hardbreak"}:
|
||||
pieces.append(" ")
|
||||
elif child.type == "image":
|
||||
pieces.append(child.content)
|
||||
elif child.type == "html_inline":
|
||||
pieces.append(re.sub(r"<[^>]*>", "", child.content))
|
||||
return "".join(pieces)
|
||||
|
||||
|
||||
def _heading_slug(value: str) -> str:
|
||||
characters: list[str] = []
|
||||
for character in html.unescape(value).casefold().strip():
|
||||
category = unicodedata.category(character)
|
||||
if category[0] in {"L", "M", "N"} or character in {"-", "_"}:
|
||||
characters.append(character)
|
||||
elif character.isspace():
|
||||
characters.append("-")
|
||||
return "".join(characters)
|
||||
|
||||
|
||||
def _html_ids(value: str) -> set[str]:
|
||||
return {first or second for first, second in _HTML_ID.findall(value)}
|
||||
|
||||
|
||||
def _check_reachability(
|
||||
pages: dict[Path, _Page],
|
||||
graph: dict[Path, frozenset[Path]],
|
||||
diagnostics: _Diagnostics,
|
||||
) -> None:
|
||||
root_page = Path("README.md")
|
||||
if root_page not in pages:
|
||||
return
|
||||
reached = {root_page}
|
||||
queue = deque([root_page])
|
||||
while queue:
|
||||
current = queue.popleft()
|
||||
for target in sorted(graph.get(current, frozenset()), key=Path.as_posix):
|
||||
if target not in reached:
|
||||
reached.add(target)
|
||||
queue.append(target)
|
||||
for relative in sorted(set(pages) - reached, key=Path.as_posix):
|
||||
if _HISTORICAL_RECORD.fullmatch(relative.as_posix()):
|
||||
continue
|
||||
diagnostics.add(
|
||||
relative,
|
||||
1,
|
||||
"DOC023",
|
||||
"maintained page is not reachable from README.md",
|
||||
)
|
||||
|
||||
|
||||
def _check_reference_adapter_examples(
|
||||
root: Path,
|
||||
pages: dict[Path, _Page],
|
||||
diagnostics: _Diagnostics,
|
||||
) -> int:
|
||||
schema_path = root / REFERENCE_ADAPTER_SCHEMA
|
||||
try:
|
||||
schema_document = cast(
|
||||
dict[str, object],
|
||||
json.loads(schema_path.read_text(encoding="utf-8")),
|
||||
)
|
||||
Draft202012Validator.check_schema(schema_document)
|
||||
except (OSError, UnicodeDecodeError, json.JSONDecodeError, TypeError) as error:
|
||||
diagnostics.add(
|
||||
REFERENCE_ADAPTER_SCHEMA,
|
||||
1,
|
||||
"DOC024",
|
||||
f"cannot load reference-adapter schema: {error}",
|
||||
)
|
||||
return 0
|
||||
validator = cast(_Validator, Draft202012Validator(schema_document))
|
||||
count = 0
|
||||
for relative, page in pages.items():
|
||||
previous_inline: Token | None = None
|
||||
for token in page.tokens:
|
||||
if token.type == "inline":
|
||||
previous_inline = token
|
||||
continue
|
||||
if token.type != "fence" or token.info.split(maxsplit=1)[:1] != ["toml"]:
|
||||
continue
|
||||
if not _is_reference_adapter_example(token, previous_inline):
|
||||
continue
|
||||
count += 1
|
||||
line = _token_line(token)
|
||||
try:
|
||||
document = tomllib.loads(token.content)
|
||||
except tomllib.TOMLDecodeError as error:
|
||||
diagnostics.add(
|
||||
relative,
|
||||
line,
|
||||
"DOC025",
|
||||
f"invalid reference-adapter TOML example: {error}",
|
||||
)
|
||||
continue
|
||||
errors = sorted(
|
||||
validator.iter_errors(document),
|
||||
key=lambda error: tuple(str(part) for part in error.absolute_path),
|
||||
)
|
||||
for error in errors:
|
||||
location = ".".join(str(part) for part in error.absolute_path) or "<root>"
|
||||
diagnostics.add(
|
||||
relative,
|
||||
line,
|
||||
"DOC026",
|
||||
f"reference-adapter example {location}: {error.message}",
|
||||
)
|
||||
return count
|
||||
|
||||
|
||||
def _is_reference_adapter_example(token: Token, previous_inline: Token | None) -> bool:
|
||||
info = token.info.casefold()
|
||||
if "reference-adapter" in info or "reference_adapter" in info:
|
||||
return True
|
||||
if previous_inline is not None and previous_inline.map is not None and token.map is not None:
|
||||
adjacent = token.map[0] - previous_inline.map[1] <= 1
|
||||
context = previous_inline.content.casefold()
|
||||
if adjacent and (
|
||||
"reference-adapter.toml" in context or "reference adapter configuration" in context
|
||||
):
|
||||
return True
|
||||
keys = set(_REFERENCE_ADAPTER_KEY.findall(token.content))
|
||||
return {"language", "source_roots"}.issubset(keys)
|
||||
|
||||
|
||||
def _check_command_reference(
|
||||
pages: dict[Path, _Page],
|
||||
diagnostics: _Diagnostics,
|
||||
) -> None:
|
||||
page = pages.get(COMMAND_REFERENCE)
|
||||
if page is None:
|
||||
return
|
||||
lines = page.source.splitlines()
|
||||
if not lines or lines[0] != f"# {COMMAND_REFERENCE_H1}":
|
||||
diagnostics.add(
|
||||
COMMAND_REFERENCE,
|
||||
1,
|
||||
"DOC027",
|
||||
f"generated reference must begin with '# {COMMAND_REFERENCE_H1}'",
|
||||
)
|
||||
if COMMAND_REFERENCE_NOTICE not in lines[:5]:
|
||||
diagnostics.add(
|
||||
COMMAND_REFERENCE,
|
||||
2,
|
||||
"DOC028",
|
||||
"generated reference notice is missing or changed",
|
||||
)
|
||||
|
||||
|
||||
def _token_line(token: Token) -> int:
|
||||
return token.map[0] + 1 if token.map else 1
|
||||
|
||||
|
||||
def _parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Check deterministic DocForge documentation invariants."
|
||||
)
|
||||
parser.add_argument("--repository-root", type=Path, default=REPOSITORY_ROOT)
|
||||
parser.add_argument(
|
||||
"--pending-inventory",
|
||||
type=Path,
|
||||
default=DEFAULT_PENDING_INVENTORY,
|
||||
help="repository-relative list of required pages not authored yet",
|
||||
)
|
||||
return parser
|
||||
|
||||
|
||||
def main(arguments: Sequence[str] | None = None) -> int:
|
||||
parsed = _parser().parse_args(arguments)
|
||||
try:
|
||||
report = check_documentation(
|
||||
parsed.repository_root,
|
||||
policy=DocumentationPolicy(pending_inventory=parsed.pending_inventory),
|
||||
)
|
||||
except Exception as error:
|
||||
print(f"documentation check failed: {error}", file=sys.stderr)
|
||||
return 2
|
||||
if not report.ok:
|
||||
for diagnostic in report.diagnostics:
|
||||
print(diagnostic.render(), file=sys.stderr)
|
||||
if report.omitted_diagnostics:
|
||||
print(
|
||||
f"... {report.omitted_diagnostics} additional diagnostics omitted "
|
||||
f"(limit {MAX_DIAGNOSTICS})",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return 1
|
||||
print(
|
||||
json.dumps(
|
||||
{
|
||||
"markdown_pages": report.markdown_pages,
|
||||
"pending_pages": report.pending_pages,
|
||||
"reference_adapter_examples": report.reference_adapter_examples,
|
||||
"status": "ok",
|
||||
},
|
||||
sort_keys=True,
|
||||
separators=(",", ":"),
|
||||
)
|
||||
)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
13
tools/docs_pending_m4_pages.txt
Normal file
13
tools/docs_pending_m4_pages.txt
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
# Remove each entry in the same commit that authors the required page.
|
||||
docs/AGENT_INTEGRATION.md
|
||||
docs/CORE_CONCEPTS_AND_AUTHORITY.md
|
||||
docs/LEGACY_AND_NO_AST.md
|
||||
docs/MIGRATING_FROM_V1.md
|
||||
docs/MILESTONE_4_BASELINE.md
|
||||
docs/MILESTONE_4_CLOSEOUT.md
|
||||
docs/POLICY_PRECEDENCE.md
|
||||
docs/PROJECT_DESCRIPTOR.md
|
||||
docs/RECOVERY_AND_PERFORMANCE.md
|
||||
docs/REFERENCE_ADAPTERS.md
|
||||
docs/RENDERING_AND_VISUALIZATION.md
|
||||
docs/SECURITY.md
|
||||
Loading…
Add table
Add a link
Reference in a new issue