1
0
Fork 0
Code Issues Pull requests Projects Releases 2 Packages Wiki Activity Actions Pages

Compare commits

..

86 commits
v1.0.0 ... main

Author SHA1 Message Date
3d5fa4e9e8 Release DocForge 2.0.0 2026-08-02 20:52:36 -04:00
7b21541ab3 Add explicit cross-identity proposal acceptance 2026-08-01 02:53:15 -04:00
377cca0531 docs: isolate current DocForge2 documentation 2026-07-31 10:07:30 -04:00
f9a05f868e Close Milestone 5 for release 2026-07-29 16:57:44 -04:00
49e1a87c13 Document the 1.4.0 release candidate 2026-07-29 16:50:39 -04:00
2b98059b44 Prove derived artifact recovery against exact oracles 2026-07-29 16:41:42 -04:00
97f3b6b1ae Verify legacy tag in fresh-clone gate 2026-07-29 16:39:21 -04:00
d2bb95fe61 Close canonical cleanup race windows 2026-07-29 16:29:42 -04:00
1a6f33e2de Add pinned real-package task evidence 2026-07-29 16:21:10 -04:00
f41a45213b Add full and fresh-clone release rehearsals 2026-07-29 16:10:37 -04:00
a901c9705b Harden canonical publication against races 2026-07-29 16:03:14 -04:00
8919e2af32 Add maintained Milestone 5 release gates 2026-07-29 16:01:24 -04:00
9071bcc0bc Add comparative Milestone 5 task evidence 2026-07-29 15:59:55 -04:00
44f20eac08 Keep release identity test valid after tagging 2026-07-29 15:57:04 -04:00
ea3f9c082b Prove exact migration from tagged v1 state 2026-07-29 15:54:02 -04:00
7b79d65dae Add reproducible release artifact proof 2026-07-29 15:49:46 -04:00
dc9500e50e Bind generated clients to product version 2026-07-29 15:47:08 -04:00
47ac329947 Centralize release version identity 2026-07-29 15:42:45 -04:00
c3552f2328 Activate Milestone 5 release contract 2026-07-29 15:38:49 -04:00
6d06195950 Close Milestone 4 with adapter adoption evidence 2026-07-29 15:34:25 -04:00
95271dcf2e Expand documentation gate coverage 2026-07-29 15:21:54 -04:00
9023f28cc0 Add strict documentation graph gate 2026-07-29 15:21:29 -04:00
983a55f156 Add fresh-wheel adoption proof 2026-07-29 15:11:56 -04:00
f0dea0cedc Document generated command surface 2026-07-29 15:10:47 -04:00
cd3c761988 Freeze adapter launcher public imports 2026-07-29 15:10:07 -04:00
663811b1d3 Format Milestone 4 hardening code 2026-07-29 15:05:11 -04:00
cca374bf99 Gate generated command reference 2026-07-29 15:04:12 -04:00
e525efd2fc Add Milestone 4 adapter benchmark gate 2026-07-29 15:03:31 -04:00
ecbf94d14a Prove adapter launcher modules are runnable 2026-07-29 15:01:11 -04:00
3bf3836ac3 Make command reference publication race-safe 2026-07-29 15:00:28 -04:00
efe4a443b7 Add strict reference project MCP binding 2026-07-29 14:54:10 -04:00
7bee0220c2 Add deterministic C++ reference adapter 2026-07-29 14:53:21 -04:00
f00aa65a73 Bound adapter assemblies and extraction caches 2026-07-29 14:52:38 -04:00
9cd7c4e424 Keep reference manifests parser-free 2026-07-29 14:52:05 -04:00
58c0196be5 Add JavaScript and TypeScript reference adapter 2026-07-29 14:39:53 -04:00
cb52bf8ae6 Add project-owned adapter client launchers 2026-07-29 14:33:42 -04:00
0d498fc385 Add deterministic command reference tool 2026-07-29 14:31:15 -04:00
8956c5c5f5 Add deterministic Python reference adapter 2026-07-29 14:30:05 -04:00
e6523c0c00 Add adapter SDK conformance proofs 2026-07-29 14:23:29 -04:00
85f6629cb1 Generate command references from live registrations 2026-07-29 14:20:43 -04:00
395d732348 Make language frontends optional 2026-07-29 14:18:31 -04:00
fee2bb0080 Activate Milestone 4 adapter SDK contract 2026-07-29 14:14:52 -04:00
d6d9f47672 Close Milestone 3 with measured projection evidence 2026-07-29 13:10:14 -04:00
f5dccb5e1c Gate production projection worker memory 2026-07-29 13:00:35 -04:00
d65b83a140 Harden detached worker module startup 2026-07-29 12:48:52 -04:00
5b43c44dd7 Make projection cycle planning scale-safe 2026-07-29 12:41:17 -04:00
f1fabaf0ca Complete independent projection runtime 2026-07-29 12:38:25 -04:00
1134c2d375 Add durable portable graph publication 2026-07-29 11:37:33 -04:00
96e3965855 Add versioned independent projection contracts 2026-07-29 10:50:05 -04:00
4c5773c865 Close Milestone 2 with measured evidence 2026-07-29 10:22:56 -04:00
fb0df5e4a1 Add deterministic client integration diagnostics 2026-07-29 10:15:27 -04:00
eb9355b003 Make task context page packing logarithmic 2026-07-29 08:51:22 -04:00
9a48233983 Add bounded generation transition receipts 2026-07-29 08:23:04 -04:00
4cc6277054 Add versioned task context capsules 2026-07-29 07:10:18 -04:00
34cd5f74c1 Add versioned effective policy 2026-07-29 06:26:40 -04:00
a9a75c5c27 Close Milestone 1 fast core 2026-07-29 06:11:22 -04:00
6253c45a5e Cache validated source generation receipts 2026-07-29 06:07:03 -04:00
529accf858 Bound paged retrieval responses 2026-07-29 06:02:07 -04:00
176b2d2784 Bind visualization status to snapshot freshness 2026-07-29 05:23:52 -04:00
24bd13f9d9 Add structured compiler diagnostics 2026-07-29 05:07:16 -04:00
0fe968c475 Make render status receipt based 2026-07-29 04:42:55 -04:00
21c4992f9c Guarantee bounded mutation receipts 2026-07-29 04:24:06 -04:00
4ae9b31db5 Refine bounded traversal contracts 2026-07-29 04:15:13 -04:00
c69cd16515 Bound indexed retrieval operations 2026-07-29 04:09:28 -04:00
ad4f52b239 Add persistent source generations 2026-07-29 04:00:23 -04:00
3bee200234 Make dependency validation linear 2026-07-29 03:45:09 -04:00
7e0eff347c Close Milestone 0 successor foundation 2026-07-29 03:29:22 -04:00
cb59a822a8 Record Milestone 0 performance baseline 2026-07-29 03:20:09 -04:00
fd4759096e Fix baseline proposal sampling 2026-07-29 03:16:18 -04:00
8ebb78a71d Establish Milestone 0 compatibility and quality gates 2026-07-29 03:12:30 -04:00
15a913003c Merge adapter lifecycle safeguards 2026-07-29 02:59:24 -04:00
6c05607b14 Preserve no-AST adapter bindings 2026-07-29 02:59:15 -04:00
1ef76f0271 Guard long-running adapter implementations 2026-07-28 19:44:25 -04:00
bb13258861 docs: adopt release-candidate closeout cadence 2026-07-28 18:23:11 -04:00
7bc2ac1e3f Document language adapter construction 2026-07-27 20:29:07 -04:00
cd54cae71d Add deterministic incremental adapter assembly 2026-07-27 16:01:40 -04:00
09c09300b1 Add language-neutral project onboarding 2026-07-27 15:50:33 -04:00
73165c9f51 Make project MCP workflows self-synchronizing 2026-07-26 09:32:25 -04:00
a30f021a52 Group source comments in logic views 2026-07-25 22:51:56 -04:00
6d659ba381 Ignore comments in Tree-sitter logic graphs 2026-07-25 22:46:01 -04:00
9161889492 Add multi-language logic exploration 2026-07-25 22:29:15 -04:00
9b4258c852 Add function-scoped Logic visualization 2026-07-25 21:08:43 -04:00
9fcafc290c Avoid rewriting warm extraction cache 2026-07-25 20:00:21 -04:00
4a8980110d Document Release 1 adapter compatibility 2026-07-25 19:21:23 -04:00
696b62f9f8 Add incremental adapter compiler boundary 2026-07-25 19:08:39 -04:00
82b3b90521 Document the DocForge 1.0 release 2026-07-25 18:33:42 -04:00
198 changed files with 64768 additions and 1702 deletions

View file

@ -1,14 +1,57 @@
# Active slice
```text
Slice: DFG-14 durable graph navigation (complete)
Goal: Make the generic graph browser durable across short MCP transactions and efficient for navigating dense project manuals.
In scope: A browser-renewed listener lease; bounded abandoned-viewer shutdown; explicit disconnected state; resizable side panels; a draggable and resizable unblurred modal; topology-derived primary, child, and edge/context navigation sections; hop-ring layout; role palettes; progressive distance shading; keyboard-operable panel resizing; deterministic interaction checks; and complete regression verification.
Out of scope: Graph mutation; source editing; persisted UI layout; project-specific relationship vocabulary; arbitrary templates; external hosting; canonical writes; unbounded listener lifetime; or non-loopback binding.
Done when: An open viewer survives MCP transport completion, closes after its browser lease disappears or its process is terminated, all requested panels can be resized, the modal can be moved and resized without backdrop blur, every neighborhood exposes generic role sections, hop distance is visually encoded up to fifty-percent darkening, and the complete DocForge gate passes.
Owners: DocForge owns viewer lease and generic presentation behavior. The configured project continues to own graph facts and relationship semantics. The process owner retains explicit termination authority.
Last completed milestone: 5 - stabilization and first DocForge2 release.
Release: 2.0.0.
Active milestone: none.
Maintenance: DocForge2 version identity and cross-identity proposal acceptance released.
Status: Complete.
```
**Next gate:** None planned. Measure actual graph-browser use before extending layout, export,
minimap, or remote-access policy. Canonical application remains permanently out of scope under
`docs/APPLICATION_DECISION.md`.
## Current state
DocForge2 `2.0.0` is the released product baseline. Current behavior is defined by the contracts
and guides under `docs/`. `SLICE_HISTORY.md` contains DocForge2 milestone summaries only.
The release adds an explicit accepted-writer allowlist for project-owned canonical appliers.
Same-identity application remains the default. Contributor processes receive no application tool,
and successful receipts record both the proposal creator and applier.
The live documentation set contains current product contracts, current operating guidance, and
current successor milestone evidence. Superseded product plans and policy notes remain available
only through repository history and are not part of normal documentation validation or retrieval.
## Maintenance proof - 2026-07-31
- Replaced the historical application-decision memo with the current canonical-application
contract.
- Removed the predecessor chronology from the live slice history and removed the transition-only
repository closeout page.
- Updated the README and documentation validator to reference only current product documentation.
- Formatting, Python lint, web lint, command-reference validation, and the 33-page documentation
graph passed.
- Focused adapter and changeset verification passed 45 tests plus 2 subtests.
## Maintenance proof - 2026-08-02
- Compared exact released commit `f9a05f868eec6c35e2b74a69467459e6ff2190bf` with exact development
commit `7b21541ab37c35dc9f3d47f631fef0c1884f9525` in isolated locked environments.
- The development repository gate passed formatting, Ruff, web lint, strict Pyright, compilation,
142 contract tests plus 272 subtests, 381 complete tests plus 422 subtests, three accessibility
flows, dependency, build, documentation, and five smoke benchmark gates.
- Compatibility passed 116 tests plus 263 subtests, concurrency and application passed 31 tests
plus 2 subtests, recovery passed 72 tests plus 62 subtests, offline fresh-wheel adoption passed,
reproducible artifact checks passed, and Git plus directory secret scans found no leaks.
- All six full maintained benchmark tracks passed. Across 66 multi-sample operations, the median
absolute development-versus-release difference was 0.71%; 65 remained within 5%, and the only
larger difference was 0.062 ms on a 1.156 ms viewer-overview operation.
- Reference-adapter graph and Logic evidence and representative-task semantic evidence matched the
release exactly. All six representative tasks retained exact answers.
- Anonymous exact-commit fresh clones of both revisions completed the full release gate offline and
remained clean. The development rehearsal completed in 283.490 seconds versus 282.091 seconds for
the release.
## Next gate
No implementation milestone is active. Any additional product behavior, version identity change,
release, or public integration requires its own bounded contract.

View file

@ -11,6 +11,18 @@
execute shell commands, mutate Git, deploy, or publish.
- Use deterministic ordering, hashes, JSON results, and structured errors.
- Fail closed on stale caches, invalid configuration, ambiguous IDs, and unauthorized families.
- Automatically repair only disposable derived state. Keep canonical sources and proposal
conflicts fail-closed.
- Prefer one synchronized bootstrap, one atomic proposal registration, one reviewed diff, and one
exact hash-bound application over caller-managed operation chaining.
- Read canonical documentation during intake, but keep it read-only while implementation and
focused testing are still changing the candidate.
- Freeze and validate one release candidate before registering documentation changes. After the
candidate is green, perform one atomic documentation closeout, run documentation-only
validation, and then publish the final revision.
- Allow at most one narrow evidence-only documentation correction after deployment. If validation
finds an implementation defect, abandon or rebase the proposal and return to implementation
instead of documenting a failed candidate.
- Keep dependencies small and pinned by compatible major version.
- Run strict `pyright`, `npm run lint:web`, formatting, Ruff, compilation, focused tests, and the
complete warning-strict test suite before closing a gate.

67
CHANGELOG.md Normal file
View file

@ -0,0 +1,67 @@
# Changelog
Notable changes in the DocForge product line are recorded here. Historical milestone evidence
remains in `docs/MILESTONE_*_BASELINE.md` and `docs/MILESTONE_*_CLOSEOUT.md`.
## Unreleased
## 2.0.0 - 2026-08-02
- Established `2.0.0` as the unambiguous package, Python, CLI, MCP, viewer-manager, and generated
client identity for DocForge2 without changing descriptor, result, adapter, or index schemas.
- Project-owned MCP servers can explicitly authorize a canonical applier to accept exact-hash
changesets from additional configured proposal writers. The default remains same-identity
application, contributor processes receive no application tool, and application receipts record
both the proposal creator and applier.
- Retained the full DocForge 1.4 compatibility, recovery, migration, benchmark, and reproducible
artifact gates.
## 1.4.0 - 2026-07-29
Version `1.4.0` is the first additive DocForge2 successor release. Annotated tag `v1.4.0`
identifies the exact synchronized documentation-bearing commit that passed the fresh-clone gate.
Added:
- One authoritative version shared by package metadata, Python, CLI, generic MCP, reference MCP,
viewer manager, generated client bindings, and release checks.
- Public adapter SDK and complete graph-plus-Logic conformance checks.
- Base Python and optional JavaScript, TypeScript, and C++ reference adapters.
- Task-shaped retrieval, generation diffs, generated client configuration, doctor checks, and
bounded response continuation.
- Independent manual, portable-graph, and live-viewer projections with immutable packages,
receipts, policy enforcement, and accessibility gates.
- Maintained compatibility, migration, concurrency, recovery, representative-task,
fresh-wheel, artifact-reproducibility, secret-scan, and benchmark gates.
- MIT licensing and Forgejo repository metadata.
Changed:
- Warm graph reads use generation-bound SQLite state without reparsing canonical project sources.
- Incremental adapters use bounded extraction caches while retaining a complete-build equivalence
oracle.
- Generated client configuration is bound to the exact DocForge version.
- Derived publication uses durable atomic replacement and fails closed on malformed, foreign,
stale, oversized, or incompatible state.
- Generic canonical application uses project-owned locking, compare-and-swap publication, and
exact-hash proposal checks while preserving raced source data.
Compatibility:
- The `docforge` distribution and Python package, `docforge` CLI, `docforge-mcp`, MCP tool names,
legacy one-method adapters, generic projects, and descriptor schema version 1 remain supported.
- Version `1.4.0` rebuilds disposable version-1 indexes as version 3 without rewriting canonical
source or active proposal bytes.
- The base wheel has no Tree-sitter dependency. Optional language extras remain explicit.
Release channels:
- The intended release channel is the public Forgejo repository and its release artifacts.
- PyPI publication is not planned because the `docforge` name is occupied by an unrelated
project.
## 1.0.0 - 2026-07-25
The first stable release established the project-scoped graph, generic Markdown/TOML adapter,
SQLite index, project-bound CLI and MCP server, reviewable exact-hash changesets, declared manual
rendering, visualization, and the original versioned compatibility surface.

959
DEVELOPMENT_NOTES.md Normal file
View file

@ -0,0 +1,959 @@
# DocForge2 development notes
This is the running implementation record for DocForge2. It records what is active, what was
measured, what changed, what failed, why architectural decisions were made, and which ideas were
deferred. Stable user and compatibility contracts still belong in dedicated documentation.
## Working rules
- Only one milestone is active at a time.
- `main` remains the last fully verified milestone.
- Active implementation occurs on `dev`.
- Every milestone begins from direct repository evidence and ends with focused tests, the complete
repository gate, updated measurements, documentation closeout, and a clean pushed state.
- WorldForge, ScrapeStation, legacy DocForge, and production MCP bindings remain out of scope.
- DocForge2 does not self-host during this program.
- Release tags and Forgejo releases require Rob's explicit approval.
Milestone 5 was explicitly authorized on 2026-07-29 through the end of the roadmap, including
evidence-based delegation and the milestone release actions frozen in `ACTIVE_SLICE.md`. That
authorization does not widen the project, production-binding, or external-publication boundaries
recorded there.
## Milestone 5 — complete: stabilization and first DocForge2 release
Milestone 5 starts from the clean, merged, and pushed Milestone 4 closeout
`6d06195950d33bcd2d712f8819bbfb3d6652ad03`. The candidate version is `1.4.0`, subject to the full
release contract rather than version work alone.
The release work is evidence-first. Maintained compatibility, migration, recovery, concurrency,
policy, projection, task-comparison, package, clone, browser, benchmark, and secret-scan gates must
pass before documentation closeout, tag creation, or publication. `main` remains the Milestone 4
baseline while implementation proceeds on `dev`.
### Release-candidate freeze — 2026-07-29
The executable implementation froze at
`d2bb95fe6190e659cf66ba57c78be53b63b53240`. It centralizes version `1.4.0`, binds generated
clients to that version, adds MIT license and package metadata, proves reproducible artifacts,
migrates a real archived `v1.0.0` project, maintains comparative real-task evidence, and hardens
derived and canonical publication against measured race and durability failures.
The clean executable release gate passed:
- Ruff formatting and lint, web lint, strict Pyright, compilation, lock, dependency, build,
generated-reference, and documentation checks.
- 142 contract tests plus 272 subtests and 371 complete tests plus 419 subtests.
- Three manual, portable-graph, and live-viewer accessibility flows.
- 116 compatibility tests plus 263 subtests.
- 29 concurrency tests plus 2 subtests.
- Offline fresh-wheel adoption, reproducible artifacts, version checks, and secret scans.
- Every maintained full Milestone 0 through Milestone 4 benchmark.
Commit `97f3b6b` made the fresh-clone rehearsal require and verify the exact annotated legacy tag.
Commit `2b98059b44f4d46b4d4cce776f163e893c647c76` added four maintained recovery proofs without
changing product code. The recovery aggregate now passes 72 tests plus 62 subtests. Those tests
record exact canonical bytes, canonical collection hash, complete snapshot hash, and index
graph-plus-Logic identity before corrupting and repairing the index attestation, manual render
receipt, generation-diff receipt, and portable-graph manifest through normal public work paths.
The real migration evidence uses tag object
`2d7d306a37da89f1c860c7f0be161c45386acf61`, peeled commit
`593c173b453236a6872d0a4e88e7a51a67a21cde`, and a real schema-1 index plus active proposal.
Canonical bytes, the graph snapshot, proposal hash, and proposal file bytes remain exact after the
schema-3 rebuild. The proof retains the inherited version-1 metadata/runtime mismatch: package
metadata is `1.0.0`, while the module and server report `0.15.0`.
The representative real-task track pins `markdown-it-py 4.2.0` at 66 Python files and 225,945
bytes. Both workflows answer all three reviewed questions exactly. Graph-assisted work inspects
349 versus 10,628 bytes, 419 versus 225,945 bytes, and 212 versus 225,945 bytes, with medians of
0.008 versus 1.341 ms, 0.009 versus 32.108 ms, and 0.013 versus 32.142 ms. Source-only final
responses are smaller, so the evidence claims reduced inspected source and faster maintained task
latency rather than a universal response-size advantage.
### Release closeout
Documentation candidate `49e1a87c138cdc63fb5abb85fc6eb2cf9f4a9d73` passed the complete
release gate with 378 tests plus 422 subtests, all three accessibility flows, every focused
Milestone 5 gate, fresh-wheel adoption, reproducible artifacts, both secret scans, and every full
Milestone 0 through Milestone 4 benchmark.
The final documentation-only descendant synchronizes `main` and `dev`, passes the anonymous exact-
commit clone rehearsal, and is identified by annotated tag `v1.4.0`. The public Forgejo release
attaches the reproducible wheel, source distribution, and machine-readable release-identity
evidence. PyPI remains excluded.
## Milestone 0 — complete
Milestone 0 established the public successor, preserved the complete lineage and v1 tag, integrated
the no-AST and adapter-lifecycle work, froze compatibility guarantees, added repository-native
quality and contract gates, and recorded cold/warm performance, memory, rendering, and response
sizes.
The central measurement was decisive: a 1,000-node warm exact lookup took about 286 ms while the
generation-pinned SQLite query path took about 0.41.4 ms. Repeated whole-source loading and
validation, not SQLite, is the first optimization target.
## Milestone 1 — complete: fast, observable core
### Outcome
Warm retrieval should disappear into normal tool overhead. Routine reads must not parse project
sources. Status must not render or rebuild hidden work. Results must remain bounded independently
of project size.
### Starting evidence
- Generic `Project.load()` walks, captures, parses, rereads, validates, hashes, and checks Git for
the complete source set.
- Exact retrieval validates twice around one bounded SQLite query.
- Context compilation performs three full project loads.
- Render status recompiles the complete manual.
- Incremental adapters already prove that manifest attestation can make no-change synchronization
and exact retrieval sub-millisecond on a tiny fixture.
- Pinned viewer queries prove the current SQLite schema can serve bounded reads quickly.
### Final outcome
Routine generic reads now parse zero canonical source files, use one generation-pinned SQLite
snapshot, and return bounded results. Status checks perform no hidden rendering or rebuilding.
Structured counters prove those invariants independently of machine timing. Complete loading and
deep validation remain recovery and equivalence oracles.
### Work log
#### Linear dependency validation
The inherited dependency-cycle preparation scanned every edge once for every node. The graph
validator now constructs dependency adjacency in one edge pass and sorts each adjacency list before
an iterative deterministic depth-first cycle check. The iterative stack also removes recursion
depth as a failure mode on large valid graphs. The same edge pass now rejects missing sources as
well as missing targets.
A 10,000-node regression test counts complete edge-collection iteration passes and caps them at
four. The focused correctness and bounded-pass tests pass, and the configured strict source type
gate is clean.
One validation command initially included `tests/test_core.py` in a direct Pyright invocation.
Repository Pyright intentionally covers `src` and `tools`, so that command reported existing
untyped test-result indexing rather than a source defect. Rerunning the repository-configured type
gate produced zero diagnostics.
#### Persistent generic source generations
Generic projects now persist a version-1 source-generation receipt only after complete source
loading and fully verified index publication. The receipt binds the explicit generic source
contract, project and root identity, adapter, revision, source hash, every canonical/authority/
descriptor regular-file identity, and every source-membership directory identity.
A normal warm check reads no canonical source bytes. It validates the known directories and files
directly using device, inode, mode, size, nanosecond modification time, and nanosecond change time.
Directory identities detect add, delete, and rename operations without an `rglob`. Any missing,
malformed, incompatible, foreign, or dirty receipt becomes a cache miss and falls back to the full
canonical load and row-verification oracle. Successful fallback verification repairs the disposable
receipt.
The final 1,000-node evidence run initially exposed a repeatable 52 ms exact-read maximum against
the 50 ms target. Profiling showed no source parsing or SQLite cost; each request rebuilt and
revalidated 1,000 `Path` objects from the unchanged JSON receipt twice. The project binding now
caches only the strictly validated receipt structure behind its device, inode, size, modification
time, and change-time signature. Canonical file and directory identities are still recaptured
before and after every query. Receipt replacement or mutation invalidates the cache and fails
closed. The same ordered exact-read profile fell from about 4052 ms to about 18 ms without
weakening stale-read refusal.
The receipt is deliberately generic-project behavior. Incremental adapter manifests retain
authority over generated or specialist source identities. A one-method legacy adapter continues to
work even when it cannot provide a cheap generation.
#### Request-scoped immutable reads
Index reads now use one read-only SQLite transaction pinned to one verified file signature and one
source identity. Existence checks and queries share that connection. Before returning, the request
rechecks the index signature and current cheap source generation. A concurrent source or index
change fails closed.
Context compilation hydrates nodes and edges from the pinned derived snapshot while retaining
profiles from the immutable descriptor. It no longer loads or parses canonical sources. The public
full `Project.load()` and deep `ProjectIndex.check()` behavior remains the recovery and equivalence
oracle.
Focused tests prove that fresh-process-style generic reads can run exact, search, filter,
backlinks, dependency, impact, context, and no-change synchronization operations while
`Project.load()` is forbidden. They also prove a final source-generation change is rejected before
return and missing/corrupt receipts fall back and repair.
The final clean 1,000-file run recorded:
| Operation | Milestone 0 median | Milestone 1 median | Milestone 1 p95 |
|---|---:|---:|---:|
| Warm no-change synchronize | 142.479 ms | 9.192 ms | 9.324 ms |
| Exact node | 286.306 ms | 17.887 ms | 18.577 ms |
| Search, limit 20 | 288.793 ms | 19.518 ms | 19.884 ms |
| Dependencies, depth 8 | 287.791 ms | 18.117 ms | 19.061 ms |
| Context, 32k, page 20 | 436.897 ms | 25.395 ms | 25.867 ms |
| Render status | 150.591 ms | 18.969 ms | 19.613 ms |
| Visualization status | — | 9.875 ms | 10.836 ms |
The recorded run came from clean commit `6253c45a5eca01efa8c73ea3dfe4d85c55878ada`.
Every measured operation passed its p95 threshold and work-counter contract.
#### Read-only audit reconciliation
The three Milestone 1 audits agreed on the main architecture:
- Keep complete loading and deep checking as independent truth oracles.
- Trust only versioned, identity-bound disposable generation receipts.
- Use one pinned read transaction and retain a final dirty check.
- Hydrate context from the current index.
- Replace full-edge traversal scans with bounded indexed frontier reads.
- Add compact success receipts before allowing large mutations to report post-write size errors.
- Replace hidden render-status rendering with a receipt comparison.
- Add algorithmic counters and parse-count gates alongside wall-clock thresholds.
One audit identified a correctness risk beyond latency: a large mutating MCP operation can commit
successfully and then be replaced by `result_too_large`. This must be fixed in Milestone 1 so
exactly-once operations never report a false failure after mutation.
#### Mutation success receipts
Proposal, preview, and canonical-application MCP mutations now declare an internal response policy.
Before runtime validation or mutation, the service proves that a minimum receipt containing the
actual input identity and fixed-length hash fields fits the configured output limit. If it cannot,
the operation returns a preflight size error with `mutation_committed = false` and does not call the
mutation.
Small results retain the existing full payload. Oversized successful results become a version-1
compact receipt that preserves exact changeset identity, hash, workflow scalars, and lifecycle
state while omitting full operations. Application receipts also preserve changed-source counts and
derived-refresh status/counts. If the compact form is still too large, the service returns the
minimum receipt proven by preflight. It never converts committed success into a post-write size
failure.
End-to-end MCP tests exercise two large hash-chained appends followed by canonical application.
Each response stays within 1,600 compact JSON characters, exposes the new exact hash, and reports
committed success. A separate 700-character preflight test proves the callback and changeset file
are never created. Changeset lifecycle receipts now obey the configured changeset byte limit on
both write and read.
#### Receipt-based render status
Successful declared renders now publish a bounded, atomic version-1 receipt below the disposable
cache. It binds project/root/adapter/source identity, the normalized view configuration, renderer
identity, template and output hashes, byte size, and safe regular-file identities. Generic renders
also publish the verified source generation used by cheap status.
Normal status compares only source-generation, descriptor/view, template-file, output-file, and
receipt identities. It does not call `Project.load()`, prepare the renderer, construct HTML, read
the full output, rebuild the index, or repair missing state. Missing and corrupt receipts are
`unverified`; source, template, or output changes are `stale`. An explicit `deep` option on the
Python, CLI, and MCP status surfaces preserves the old side-effect-free full-render equivalence
oracle.
Receipt failure after atomic output replacement is reported as degraded publication success, not a
false render failure. Canonical application converts the same condition into a degraded
derived-refresh report while retaining canonical success. Focused tests forbid source loading and
renderer preparation during warm status and cover output, template, missing-receipt, corrupt-
receipt, and post-publication receipt-failure behavior.
After race hardening, a 50-sample three-node receipt-status check measured a 3.202 ms median and
3.509 ms p95, compared with the 1.941 ms Milestone 0 three-node full-render status. The small
fixture does not show the scaling benefit; the 1,000-node Milestone 0 status baseline was
150.591 ms and will be rerun in the final Milestone 1 evidence pass.
#### Bounded indexed retrieval
Search, metadata filtering, backlinks, dependency traversal, and impact traversal now query one
extra row beyond the requested bound and report `limit` plus `truncated`. Backlinks, dependency,
and impact APIs accept the same additive `limit` option through Python, CLI, and MCP surfaces.
Omitted limits are capped by the project `max_results` policy.
Traversal no longer loads the complete edge table and repeatedly scans it. It performs
deterministically ordered frontier queries through the existing source primary key or target index.
Each request also has a deterministic edge-examination budget derived from its result limit. The
response includes `candidate_edges_consumed`, `candidate_edges_limit`, and `truncation_reason`
counters so algorithmic work can be asserted independently of machine timing. `truncated` is true
when either another unique result exists or the work budget prevents proving completeness.
The read-only query-plan audit found that source-ordered unfiltered incoming traversal required a
temporary SQLite sort with the version-2 `(target_id, relation, source_id)` index. Direct
`EXPLAIN QUERY PLAN` evidence showed `USE TEMP B-TREE FOR ORDER BY`. A measured additive
`(target_id, source_id, relation)` index removes that sort. The disposable index schema is now
version 3, so existing version-2 indexes rebuild without changing canonical source or proposals.
Frontier cursors are streamed and stop immediately on the first omitted unique result. A focused
core, CLI, MCP, Ruff, and Pyright gate passes for this work-in-progress slice.
#### Structured profiling and zero-work gates
DocForge now has an opt-in, request-local diagnostics collector backed by `ContextVar`. It emits
one bounded version-1 aggregate with a fixed operation name, outcome, total elapsed nanoseconds,
fixed stage timing keys, and fixed integer counters. It never records paths, node IDs, queries,
source text, or SQL. Disabled mode reads no clock and adds no response field, preserving the
existing CLI and MCP payloads.
The generic loader, adapter projection and extraction paths, source-generation checks, index
checks/synchronization/build/read transactions, render status/preparation/output hashing, MCP
runtime validation, and viewer-manager requests now expose direct proof counters. A warm
incremental adapter cache hit still counts the enclosing project load, so the counters cannot hide
full adapter assembly merely because extraction was reused.
MCP servers and the CLI accept the additive `--diagnostics` startup option. Diagnostics are
attached to structured successes and errors only when the complete MCP response still fits its
configured output budget; they are discarded before any primary result or compact mutation
receipt. Warm generic error decoration now reads the persisted source generation before falling
back to complete loading. Render- and visualization-status error paths explicitly disable both
recovery synchronization and complete identity loading.
Context isolation tests cover threads, concurrent async tasks, repeated stages, nested collectors,
exceptions, and disabled collection. Repository tests assert that warm success and error reads,
render status, and visualization status perform zero project loads, source parses, adapter
projection/extraction, index builds, render preparation, output construction, and output hashing.
The result JSON schema contains the same closed operation, stage, and counter sets as the
implementation.
The maintained `tools/milestone1_benchmark.py` harness adds hard counter and p95 latency gates to a
disposable generic project. The smoke target is part of `make gate`; the final 1,000-node evidence
is recorded in `benchmarks/milestone1-2026-07-29.json`. The historical Milestone 0 harness remains
behaviorally unchanged as comparison evidence; it only exposes shared fixture and measurement
helpers to the Milestone 1 harness.
#### Visualization snapshot freshness
Visualization workers now receive a version-1 snapshot specification containing the exact
validated index publication signature: device, inode, size, modification time, and change time.
Both the manager and worker reject a launch if that publication changes before startup. The
transmitted project root, root fingerprint, source identity, adapter, counts, limits, and confined
index path are strictly validated before the worker may serve source or graph data.
Worker health reports index freshness through stat-only comparison. It does not open SQLite and
does not renew the browser activity lease. The version-2 viewer-manager protocol validates the
complete worker identity and distinguishes an unreachable worker from a live stale worker. A stale
worker stays lifecycle `running` for accurate diagnosis, but the next visualize request stops it
and launches a newly validated snapshot instead of reusing it.
Client status separately compares the worker's pinned source identity with
`IncrementalStateProject.incremental_state()`. The composite snapshot is stale if either proof is
stale, current only when both proofs are current, and unknown otherwise. A stopped worker has
unknown snapshot identity. MCP preserves this state at the top-level `staleness` field and disables
recovery synchronization and full-load error decoration.
Tests cover signature mutation before worker startup, malformed identity, missing and symlinked
indexes, stat-only health, unchanged activity, current/unknown/stale source states, live stale
workers, non-reuse, and zero-load status. The Milestone 1 benchmark now measures current, stale,
not-running, and unavailable visualization status separately with the same zero-work and 50 ms p95
gates as other receipt status operations.
#### Bounded pagination and exact large-result review
The final Milestone 1 contract audit found that count limits and the global MCP output ceiling were
not sufficient. A 1,000-node context response already exceeded the normal 200,000-character tool
limit, and one allowed changeset operation can be larger than that limit. Returning
`result_too_large` kept transport bounded but stranded useful evidence.
Version-1 pagination now uses canonical, base64url cursors with a domain-separated SHA-256
corruption checksum. Cursors bind the project, adapter, canonical generation, semantic query,
collection hash, and position. They are deliberately unkeyed read tokens rather than authorization
credentials. Corrupt tokens fail as `invalid_cursor`; changed generations or collections fail as
`stale_cursor` with explicit pagination-restart remediation.
Context transport flattens the compiler's deterministic selected entries followed by all explicit
omissions, then partitions each page back into the existing arrays. Both item count and exact
compact-JSON response size constrain packing. An individually oversized entry becomes a bounded,
hash-identified omission and advances the cursor, avoiding an infinite retry while preserving the
fact that evidence was excluded.
Changeset list, inspection, validation, and diff reads preserve direct full-result defaults while
MCP uses bounded pages. Pages retain exact changeset identity and hash. Large operation pages
compact content-bearing fields into hashes and character counts. A single oversized structured
diff is serialized once as canonical ASCII JSON and returned through hash-bound chunks that
reconstruct the exact legacy `operations` and `changes` arrays. This solves transport growth
without lowering canonical changeset limits or adding cursor storage.
The benchmark now validates zero-work and operation-specific counters for every warmup and measured
sample, records bounded semantic response summaries, covers filter, backlinks, outgoing and
incoming traversal, and measures current/stale/missing/corrupt render receipts plus all
visualization lifecycle states. A maintained query-plan test prevents the incoming traversal
temporary sort from returning.
#### Milestone closeout
The complete repository gate passed with 138 tests and 77 subtests, zero Pyright diagnostics,
warning-strict execution, package builds, contract checks, and both benchmark smoke gates. The
clean 1,000-node benchmark passed all maintained thresholds and is interpreted in
`docs/MILESTONE_1_BASELINE.md`. Compatibility, measured decisions, limitations, and scope evidence
are frozen in `docs/MILESTONE_1_CLOSEOUT.md`.
Milestone 1 made no storage rewrite, self-hosting change, production integration change, tag, or
release.
### Initial design constraints
- Full rebuild remains the recovery and equivalence oracle.
- Canonical content remains authoritative.
- Existing one-method `load_projection()` adapters remain unchanged.
- No-AST adapters remain first-class.
- Indexes, source-generation receipts, and caches remain disposable.
- Cheap reads may trust only identity-bound, versioned, corruption-checked receipts.
- Any optimization must fail closed on source mutation and must preserve stale-read refusal.
### Future ideas and suggestions
These are notes, not commitments:
- A stable source-generation provider may deserve a public adapter capability only after both the
generic project and one incremental adapter prove the same boundary.
- Profiling receipts could eventually feed the human-facing project control panel, but Milestone 1
should expose structured data before adding UI.
- A durable telemetry exporter remains deliberately deferred. Request-local bounded aggregates are
enough to prove compiler work in Milestone 1 without adding persistence, cardinality, or privacy
risks.
- The stat identity is a cheap publication proof, not a cryptographic integrity scan. Full index
validation remains the launch and query oracle.
- Cursor authentication remains deliberately absent. If read cursors ever carry authority rather
than bounded positions, they will need a different versioned security contract and persisted key
lifecycle.
## Milestone 2 — complete: agent retrieval and MCP experience
### Audit reconciliation
Three independent read-only audits covered effective policy and bootstrap, task-shaped retrieval
and context capsules, and generation diffs plus client configuration and doctor checks.
They agreed on these boundaries:
- Keep the project descriptor at schema version 1. Process capability and client configuration are
machine-specific bindings, not canonical project content.
- Preserve the legacy adapter-policy payload, no-AST shorthand, tool names, default tool ordering,
one-method adapters, and custom context provider.
- Add one versioned effective-policy authority and derive bootstrap, contract, instructions, and
access reporting from it.
- Add one task-context operation with a closed task-kind vocabulary and one immutable,
generation-pinned retrieval plan. Do not create a tool for every task kind.
- Produce evidence gaps only from declared plan requirements and completed bounded checks. Never
infer missing facts from arbitrary project naming.
- Record only the latest bounded generation transition as disposable evidence. Do not add a
history database.
- Preview client configuration by default. Any write must be explicit, atomic, merge-preserving,
and backed by a verified client-format driver.
- Keep doctor strictly read-only. It must not bootstrap, synchronize, build, render, start a
viewer, or rewrite client configuration.
### Versioned effective policy and session contract
The binding now composes an immutable version-1 policy containing capability mode, adapter
evolution, AST and Logic behavior, synchronization and integrity levels, render and viewer
behavior, profiling, blocked tools, prohibitions, and explicit precedence. `--no-ast` is a
restrictive override. The exact legacy `adapter_policy` response remains a projection of the new
object.
Bootstrap reuses the identity already proven by synchronization and no longer reloads the complete
project. Its additive version-1 session contract reports binding, generation, effective policy,
actual registered surfaces and mutation access, render policies, first operation, filtered
workflow, and prohibitions. Read mode does not recommend proposals. Proposal mode recommends
registration and review only with writer access. Application is recommended only when the
exact-hash applier is enabled.
Existing factory defaults and tool order remain unchanged. Explicit application mode fails closed
without an applier. Operator mode is reserved and currently adds no tools.
### Versioned task retrieval and context capsules
The first Milestone 2 retrieval slice adds one `docforge_get_task_context` read tool rather than a
family of task-specific tools. Its closed task kinds are change, implementation, failure,
ownership, test, operation, and release. One immutable `RetrievalPlanV1` derives exact or lexical
focus, bounded bidirectional graph traversal, metadata hydration, required evidence categories,
and fixed work budgets from the project descriptor and effective policy.
The public executor re-derives every submitted plan before opening SQLite. It rejects modified
steps, task identity, requirements, category order, bounds, policy identity, or hashes as
`invalid_retrieval_plan`. Traversal binds the project relation set by canonical hash and queries
the already-validated edge table by endpoint, avoiding relation-sized SQL parameter lists.
Version-1 internal ceilings are 1,000 evidence items, 100,000 candidate edges, and 10,000 task
query characters.
The executor uses one immutable SQLite read generation. It rejects missing explicit focus, blocks
unresolved or tied lexical focus, stops at deterministic evidence and candidate-edge limits, and
checks source identity again when the transaction closes. `ContextCapsuleV1` binds the generation,
policy, request, plan, evidence collection, and complete capsule with canonical hashes.
Project relation names remain authoritative. The core recognizes only a versioned alias map for
structure, implementation, dependency, execution, data, evidence, and context. Unknown allowed
relations stay visible under their raw names as `unclassified`. Required evidence diagnostics
distinguish categories the project never declared, completed bounded checks with no selected
evidence, and incomplete checks caused by a result, work, token, or response limit.
Each evidence item carries a stable content hash, confined source identity, shortest selected graph
path, every additional qualifying relationship reason observed during traversal, and explicit
limitations where the current graph cannot prove evidence type, extractor identity, relationship
source provenance, or observation time. The planner contains no Logic operation, so no-AST
bindings can use task context without weakening their existing Logic prohibition.
Path direction is relative to the preceding traversal node. Additional relationship reasons use
the evidence node as their direction subject. Candidate-edge and unclassified-relation ceilings
produce explicit omissions and bounded summaries.
MCP pagination preserves the complete plan, collection, and capsule hashes while returning bounded
pages. Its cursor additionally binds the effective policy and task request. An individually
oversized item advances once as a hash-identified omission. A later generation or policy change
fails closed as `stale_cursor`.
The legacy profile-context contract remains intact. A custom context provider does not silently
gain core task planning. Version 1 defines no custom task-planner extension, so the additive tool
returns `task_context_unavailable` without synchronization or a complete projection load.
Two independent pre-commit audits reproduced and closed plan-forgery, relation-sized SQL,
SQLite-parameter portability, ambiguous relationship-direction, missing work-limit evidence,
schema/runtime drift, incomplete page hashing, and custom-provider hidden-load defects. Regression
coverage includes 33,005 valid relation names, fixed extreme project limits, tampered plans,
evidence-relative directions, edge and unclassified limits, schema-valid pages, changed cursor
semantics, oversized evidence advancement, no-AST retrieval, and legacy complete-projection
adapters.
The complete repository gate passes with 158 tests and 101 subtests, zero Pyright diagnostics,
warning-strict execution, package builds, public-contract validation, and the maintained Milestone
0 and Milestone 1 smoke benchmarks. Gitleaks 8.30.1 reports no secret findings in the working tree.
### Latest-generation diff receipt
Three read-only audits reconciled the index publication, public transport, compatibility, and
no-AST boundaries before implementation. The selected design stores one disposable
`generation-diff.json` receipt. It does not add a history database, arbitrary generation
selectors, source text, rendered content, or Logic details.
Before a build loads current source, it accepts an existing index only when its exact main-file
inode has a matching stable whole-file attestation and no WAL, journal, or shared-memory sidecar.
It then captures that predecessor through an immutable main-file transaction. The capture validates
the SQLite application and schema IDs, project/root/adapter binding, integrity, complete node and
edge rows, Logic aggregate identity, FTS count, metadata hashes and counts, and final file
signature. It never calls normal check or synchronization and never repairs predecessor evidence.
The final source revalidation now compares exact nodes and edges in addition to source hash,
revision, and Logic. A verified predecessor that maps the same source identity to different graph
content fails before publication as `generation_collision`. This closes a pre-existing adapter
determinism gap found during the generation-diff audit.
SQLite replacement is now the explicit derived mutation commit point. Whole-file attestation,
cheap source-generation, and generation-diff receipts publish independently afterward. Any
post-commit receipt failure returns `status = ok`, `index = published`, a bounded degraded
publication record, and receipt-stage names. It never rolls back the new index or reports a false failed
mutation. Attestation hashing checks the exact index signature before, during, and immediately
before receipt publication.
Version-1 diff semantics compare every core `Node` field by stable node ID and exact edge triples.
Node renames are removal plus addition. Edge changes are removal plus addition. Exact summary
counts and a full ordered item-hash collection cover every change. Retained details are
deterministically ordered and independently capped at 1,000 items and 1 MiB with explicit item- or
byte-limit evidence. A first build or untrusted predecessor is a baseline with no fabricated
all-added result. A same-generation reindex republishes the existing meaningful transition against
the new index file identity instead of erasing it with an empty diff.
The additive public surfaces are:
- CLI `generation-diff [--limit N] [--cursor OPAQUE]`.
- MCP `docforge_get_generation_diff(limit=None, cursor=None)`.
- Telemetry operations `cli.generation-diff` and `mcp.generation_diff`.
Public reads do not open SQLite, call `project.load()`, extract an adapter projection, parse source,
check, synchronize, build, or repair. They strictly validate the bounded receipt, compare stable
receipt and index file identities, require two matching cheap source-generation checks, and report
unknown for legacy adapters without that capability. Missing, corrupt, foreign, oversized,
symlinked, stale, or concurrently changed evidence remains a read-only status outcome.
Generation-diff pagination binds the complete stored receipt hash and effective policy. That hash
already covers project, generation, graph, collection, and committed-index identity. Page size may
change. A replaced receipt returns `stale_cursor`. One top-level pagination object owns the only
cursor. The nested version-1 page uses `receipt_header.stored_receipt_hash` so it never
misrepresents the complete receipt hash as the hash of a partial header. The summary distinguishes
additional retained pages from details permanently omitted by the fixed publication limits.
Adversarial coverage now includes strict runtime/schema rejection, predecessor attestation and
generation identity, live and synthetic SQLite sidecars, cache-root symlink substitution,
source/sidecar changes during diff preparation, independent receipt failures, and degraded
post-commit identity and durability failures. Focused verification passes the direct, CLI, MCP,
schema, pagination, legacy, incremental no-AST, and zero-work suites. The complete repository gate
passes with 176 tests and 113 subtests, zero Pyright diagnostics, package builds, web checks, and
the maintained Milestone 0 and Milestone 1 smoke benchmarks. Final independent re-audit is in
progress before this slice is committed.
The final dense benchmark uses a 1,000-node transition with 1,000 changed details. Its receipt is
775,663 bytes. Direct status is 37.659 ms median and 39.108 ms p95. A maximum-size MCP request
returns 307 items in 199,754 bytes at 54.037 ms median and 56.617 ms p95. Four pages reconstruct
all 1,000 retained details in 652,798 bytes at 201.55 ms median. Peak RSS is 79,096 KiB.
Every hidden-work counter remains zero; the read performs exactly two cheap source-generation
checks.
Measurement found and removed two avoidable costs before commit. Receipt loading had repeated the
complete 1,000-item validator solely to check project identity; it now validates once and compares
the three binding fields directly. Page fitting had encoded every growing prefix; it now uses an
exact logarithmic search and retains the hash-only oversized-item omission path. The maximum page
fell from 272.06 ms p95 to 56.617 ms p95, while full traversal fell from roughly 859 ms to
203.55 ms p95. Regression tests require one receipt validation and at most 15 response encodes for
1,000 page candidates. Final independent publication, contract, and performance audits approve
the slice for commit.
### Deterministic client configuration and read-only doctor
Client integration remains an explicit machine-local boundary rather than canonical project
content. `docforge configure {codex,claude,openclaw} --project PATH` previews a deterministic
version-1 fragment by default. An optional output path publishes only a standalone fragment into
an existing real directory. Publication is create-only, private-mode, no-follow, bounded, and
conflict-aware. Existing differing client configuration is never merged, replaced, or silently
overwritten.
Generated commands use the exact current virtual-environment Python executable with isolated
module startup. The binding records explicit read, proposal, or application mode, no-AST policy,
render policy, empty environment, and bounded timeouts. Proposal and application generation fail
closed unless the descriptor declares the required writer and matching applier identity. A generic
CLI cannot reconstruct project-owned adapter composition, so custom adapters return an explicit
unavailable result instead of generating a misleading command.
Codex and OpenClaw fragments include their verified timeout fields. Claude JSON fragment syntax is
supported, while its timeout representation remains an explicit warning. The configuration result
has a strict JSON schema and canonical plan hash. Diagnostics are additive and remain disabled by
default.
`docforge doctor --client CLIENT` performs bounded, non-mutating inspection only. It reads the
project descriptor and selected client file through stable, directory-bound, no-follow handles;
parses at most 1 MiB and 256 server entries; selects at most one exact project binding; validates
the closed server argument set; checks executable, capability, declared authority, no-AST,
timeouts, environment-key names, and tool-filter presence; and performs only a stat-level index
presence check. It never loads canonical sources, opens SQLite, starts MCP, executes the configured
command, synchronizes, builds, renders, starts a viewer, or writes client configuration.
Doctor reports healthy, degraded, or unhealthy with stable process exit codes 0, 1, and 2. Secret
environment values are parsed only to enforce bounded string limits and are never returned.
Unknown or unverified client tool filtering, Claude timeout representation, implicit legacy
capability mode, shadowed authority, and missing disposable indexes are warnings. Unsafe paths,
malformed matching entries, unexpected executables, wrong project roots, invalid authorities, and
missing configuration are failures.
The first benchmark smoke failed for the correct product reason: its disposable doctor fragment
used the shared path `/tmp/doctor-codex.toml`, where a previous run had left different content. The
harness now creates a project subdirectory inside one unique temporary root and places the client
fragment beside it. This preserves create-only conflict safety and makes every run disposable.
The final pre-commit 1,000-node audit sample passes every provisional Milestone 2 gate. Task
context reconstructs 1,000 candidates as 108 cited evidence records and 892 explicit bounded
omissions across 11 pages in 703.808 ms. Generation diff reconstructs 1,000 changed details across
10 pages in 427.450 ms. Maximum pages remain below the 200,000-byte MCP budget; generation diff
uses 199,566 bytes and proves that diagnostics are discarded before the primary result. Isolated
peak RSS is 86,168 KiB.
Configuration preview now includes a bounded real import probe of the exact isolated interpreter,
so its provisional single-sample latency is about 315 ms rather than the earlier sub-millisecond
derivation-only figure. Doctor remains below 1 ms on generated disposable configurations.
Every configuration and doctor hidden-work counter is zero.
The aggregate `make gate` includes the Milestone 2 smoke benchmark. The frozen candidate passes
205 tests and 120 schema subtests, strict warnings, Ruff, formatting, Pyright, web checks,
compilation, lock and dependency checks, package builds, and all three milestone smoke benchmarks.
Three independent final audits approve client publication and policy binding, doctor fail-closed
behavior, and benchmark/contract coverage. Clean-revision benchmark evidence is still required
before closeout.
### Milestone 2 closeout
Candidate commit `fb0df5e4a1c591c2a84788fd4814d98550f11863` passed the clean ten-sample
Milestone 2 benchmark. Task-context complete traversal measured 703.561 ms median and 721.847 ms
p95 across 11 bounded pages. It reconstructed the exact 1,000-candidate collection from 108 cited
evidence records, 891 original token-budget omissions, and one hash-attested response-limit
surrogate. Generation-diff complete traversal measured 418.607 ms median and 425.315 ms p95 across
10 pages.
Read and no-AST bootstrap remained below 10 ms p95. The maximum generation page used 199,566 bytes
of the 200,000-byte budget and correctly discarded diagnostics before primary evidence.
Configuration preview measured about 314 ms median and 365 ms p95 because it proves the real
isolated interpreter import on every invocation. Codex and OpenClaw doctor checks remained below
0.6 ms p95; Claude remained explicitly degraded because its timeout format is unverified.
Isolated-process peak RSS was 86,448 KiB against the 262,144 KiB gate.
All measured configuration and doctor counters were zero. Task-context pages performed one index
check and two cheap generation checks with no loads, parses, synchronization, builds, extraction,
rendering, or viewer work. Generation-diff pages performed two cheap generation checks and no
index check. The canonical machine-readable result is
`benchmarks/milestone2-2026-07-29.json`.
Milestone 2 is complete. Follow-up ideas stay explicitly later-scope: avoid recomputing the
task-shaped capsule for every continuation page, add authenticated continuation when the threat
model requires it, verify a native Claude timeout representation, and introduce adapter-owned
launcher metadata before generating configurations for custom adapters.
## Milestone 3 — complete: independent projections
Milestone 3 began only after `main` and `dev` were aligned at the verified Milestone 2 closeout.
Three read-only audits ran before source changes:
- Manual planning, immutable packages, renderer isolation, receipts, preview/application
integration, and full/incremental equivalence.
- Portable graph planning, static artifacts, the live viewer boundary, worker protocol, and static
plus interactive accessibility.
- Packaging, optional dependencies, public contracts, projection policies, performance,
incremental fragments, and maintained gates.
The active design constraints are unchanged: renderers consume one validated immutable generation;
manual and graph plans remain separate; the live viewer is not retrieval authority; core remains
usable without rendering; status performs no hidden rendering; full rendering remains the recovery
and equivalence oracle; no storage rewrite is assumed.
### Milestone 3 architecture decision
The three audits converged on one compatibility-first boundary:
- The existing `docforge.render_contract` names, `GenericHtmlRenderer.prepare()` signature,
`generic_html` renderer identity, and byte output remain the version-1 compatibility surface.
They become adapters over the new manual-planning path rather than being changed in place.
- New `ManualRenderPlanV1`, `GraphViewPlanV1`, `ProjectionPackageV1`, and
`ProjectionReceiptV1` contracts use strict canonical JSON, deterministic ordering, independent
item and byte bounds, exact generation and policy binding, and content-derived identities.
- Plans and packages contain selected graph facts and bounded content. They never contain a
project object, SQLite handle, absolute project or index path, arbitrary query, command, or
project-provided executable code.
- The planner owns graph selection and meaning. A renderer may transform only a validated package
into declared artifacts and cannot select nodes, invent relationships, crawl the project, choose
publication paths, or mutate canonical sources.
- Manual and portable graph renderers live behind independent import boundaries. Renderer
dependencies load lazily. Default installation behavior remains compatible during the initial
migration; optional dependency changes require their own verified packaging decision.
- Portable graph rendering is additive. It does not replace or silently change
`docforge_visualize`, `graph-browser@17`, the viewer-manager protocol, or the query-backed live
viewer.
- Effective policy version 1 remains frozen. Milestone 3 introduces a version-2 projection-policy
view for manual `auto|explicit|disabled`, portable graph `explicit|disabled`, and live viewer
`on-demand|disabled` enforcement, while retaining the version-1 projection for existing clients.
- Publication commits content-addressed artifacts first, renderer evidence second, and a bounded
generation/view manifest last. Status remains receipt-only. Failures after artifact replacement
report committed degraded success rather than an ordinary failed mutation.
- Full planning and rendering remain the recovery and equivalence oracle. Incremental fragments
are disposable, keyed from complete plan semantics, and may be reused only when byte-exact
artifact equivalence is proven.
- The live source endpoint must stop reading mutable canonical files behind a pinned graph
snapshot. Portable artifacts never inherit that path-bearing behavior.
The first implementation slice freezes existing golden output, adds the four versioned contracts
and validators, introduces pure manual and graph planners, and makes the legacy manual renderer a
compatibility wrapper. Publication hardening, detached rendering, incremental fragments, portable
graph publication, independent policy enforcement, accessibility, and maintained performance
gates follow on top of that frozen boundary.
### Milestone 3 contract slice
The first slice now implements:
- Strict Draft 2020-12 schemas and runtime canonical-hash validation for manual plans, graph plans,
projection packages, and projection receipts.
- A deterministic manual planner that owns page selection, navigation, cross-references,
backlinks, search documents, component assignments, orphan diagnostics, and cycle diagnostics.
- A deterministic graph planner with exact-root or metadata-only lexical scope, closed filters,
explicit node/edge/work bounds, deterministic omissions, path/source-body exclusion, and
no-AST Logic exclusion.
- A separate `docforge_renderers.manual` package. Its renderer accepts only a validated package and
has no project, SQLite, publication-path, or filesystem-write API.
- The frozen `GenericHtmlRenderer` compatibility shim over the new planner/package/renderer
pipeline. The alpha artifact remains exactly 2,043 bytes with output SHA-256
`81656bb89debc7ad1fbe8bc290e9a3ba90664442b17a6d57e908d30d20c47f77` and legacy render identity
`1c0a49c28ba3b0dabf94be36e75def197dee1be3cb73ac405b09875383c8dc5f`.
- Rejection of project-template scripts, inline event handlers, `javascript:` URLs, embedded
browsing contexts, and refresh redirects.
- Wheel inclusion for both typed packages and every published JSON schema. Importing `docforge`
no longer imports `markdown_it` or the manual renderer package.
- A live-viewer correction: source evidence now comes from the pinned index generation. The
viewer no longer reopens mutable canonical files behind an older graph snapshot.
The new repository-native contract target passed 91 tests and 120 subtests at the slice boundary.
The combined projection, rendering, and live-viewer focus passed with byte-exact compatibility and
no hidden source/path authority.
### Durable portable graph publication
The portable graph path now has its own declared `graph_render` views, pure plans, fixed
`portable_graph_html` renderer, content-addressed artifact store, renderer receipts, and one bounded
generation/view manifest as the publication commit. It supports Nodes, Flow, and Web without
including Logic. Static HTML contains the complete pre-rendered graph and treats JavaScript as
progressive enhancement.
Publication revalidates source, view, artifact, receipt, and output identities across replacement.
Status reads only bounded manifest and receipt evidence. It never plans or renders. Repair may
restore a declared output from its content-addressed artifact. A post-artifact failure that cannot
be rolled back returns explicit degraded committed evidence rather than reporting an ordinary
failed mutation.
### Detached workers and incremental fragments
Manual and portable graph packages execute through one fixed one-request child protocol. The
parent launches isolated Python from a trusted working directory with a sanitized environment,
spools stdout to disk, reads one bounded canonical response, and validates the complete artifact
and receipt identity. The worker accepts only the two built-in renderer identities. Requests are
bounded by the 24,000,000-byte package contract, actual artifact transfer by 20,000,000 bytes, and
execution by a 30-second timeout.
Manual fragment records are semantic, versioned, canonical, hash-bound, and stored below a
dedicated confined cache. The worker independently recomputes the expected page fragment before
using a record. Corrupt, forged, oversized, stale, or aggregate-oversized records fall back to the
full detached render. Cold fragment creation is compared byte-for-byte with that full oracle before
cache publication. The cache retains only the current inventory and is capped at 10,000 entries
and 64,000,000 bytes.
### Independent policies and accessibility
Projection policy version 2 independently composes manual `auto|explicit|disabled`, portable graph
`explicit|disabled`, and live viewer `on-demand|disabled`. CLI, MCP, generated client
configuration, doctor, render services, canonical application, onboarding, and viewer-manager
entry points enforce their relevant policy. Status remains available when an active operation is
disabled.
Generated client evidence binds the projection policy, its hash, projection availability, and the
current descriptor hash into the configuration hash. Validation cross-checks omitted default
selectors against the bound descriptor so coordinated policy and availability drift fails closed.
The version-1 effective-policy payload remains unchanged for existing clients.
Pinned Playwright 1.62.0 and axe-core 4.12.1 gates exercise the frozen manual, portable graph, and
live viewer with selected WCAG A/AA axe tags and keyboard interaction flows. Portable and live
graph presentation received only the minimal contrast and nested-role corrections needed by those
gates.
### Scale and runtime hardening
The first 1,000-node full benchmark exposed recursive strongly connected-component traversal in
manual planning. Cycle detection now uses an iterative two-pass traversal. A regression covers the
descriptor maximum of 10,000 nodes as both a deep acyclic chain and one strongly connected
component.
The isolated wheel proof also exposed a Python `runpy` warning when the worker module was imported
during package initialization before `-m` execution. A private fixed module entrypoint now owns
child startup. Malformed child input returns code 2 with empty stdout and stderr.
Configured render ceilings above 20,000,000 bytes remain accepted for compatibility, and small
actual artifacts render normally. The detached protocol still rejects an actual transfer beyond
its fixed 20,000,000-byte boundary.
### Milestone 3 closeout
Candidate `f5dccb5e1c312121f1af63780162f593d9363b98` passed the complete repository gate: formatting,
Python and web lint, strict types, compilation, 281 tests and 272 subtests, three accessibility
flows, lock and dependency checks, package builds, and all milestone smoke benchmarks. The
maintained projection contract subset passed 142 tests and 236 subtests.
The clean ten-sample 1,000-node benchmark passed every latency, memory, response-size, no-work, and
equivalence gate. Manual full rendering measured 810.490 ms p95, portable graph full rendering
323.690 ms p95, and receipt-only status 111.381 ms and 59.331 ms p95 respectively. Direct detached
worker peaks were 88,580,096 and 89,583,616 bytes. The separately gated production manual worker
peak was 104,771,584 bytes. Production cold, warm, forced-full, add, change, delete, and reorder
outputs were byte-identical.
Production warm fragment rendering measured 2,206.540 ms p95 versus 978.870 ms for forced full.
Milestone 3 therefore closes the fragment isolation, invalidation, equivalence, and recovery
contract without claiming a throughput win. Later optimization must begin from that evidence.
The exact method and measurements are recorded in `docs/MILESTONE_3_BASELINE.md` and
`benchmarks/milestone3-2026-07-29.json`. The candidate passed an isolated wheel CLI/MCP/worker
proof. Gitleaks 8.30.1 found no findings across the six Milestone 3 commits or candidate tree.
Milestone 3 is complete. No tag, release, production integration repointing, WorldForge change,
ScrapeStation change, storage rewrite, or self-hosting dependency was introduced. Milestone 4
remains directional and has not started.
## Milestone 4 — complete: adapter SDK and product documentation
Milestone 4 activated from the verified Milestone 3 closeout. Its scope was deliberately additive:
freeze an adapter-authoring boundary, prove narrow language references, make heavy frontends
optional, generate command and client integration evidence, and document adoption. WorldForge,
ScrapeStation, the legacy repository, production bindings, self-hosting, storage replacement, and
release publication remained excluded.
### Adapter SDK and conformance
`docforge.adapter_sdk` is now the public authoring facade for typed projections, manifests, source
contributions, complete assemblies, incremental loaders, project bindings, core graph models, and
conformance reports. Complete evidence includes primary graph and function Logic. An incremental
adapter that publishes Logic must provide an independent `load_complete_assembly()` oracle.
The conformance helper repeats complete loading for determinism, compares the complete assembly
with `load_projection()`, and proves exact complete/incremental graph-plus-Logic parity. It does not
pretend to replace separate confinement, restart, no-AST, corrupt-cache, or retrieval tests.
An adversarial review found that extraction caches and adapter assemblies lacked aggregate
boundaries. Version-1 cache reads and writes are now regular-file-only, identity-checked, capped at
64,000,000 bytes and 10,000 sources, and fail safely to a miss. Adapter projects enforce primary
node limits and deterministic edge and Logic multipliers before publication.
### Reference language evidence
The Python reference uses the standard-library AST and publishes files, modules, classes,
functions, arguments, local static imports, and function Logic. Its manifest uses tokenization
rather than AST parsing, so unchanged warm builds perform no syntax parse.
JavaScript and TypeScript use separate pinned optional Tree-sitter grammars. They publish files,
modules, classes, functions/methods, static project-relative imports/re-exports, and function
Logic. Their manifests use a closed comment/string-aware ESM scan and do not load Tree-sitter.
C++ uses explicit non-overlapping source roots and one confined `compile_commands.json` as
translation-unit inventory and fingerprint evidence. It validates entries but never executes a
command, compiler, response file, or project program. It publishes syntax and directly resolvable
project-local quoted includes. Its manifest currently parses for include discovery, so no
zero-warm-parser claim is made.
None of the references claim compiler-resolved calls, inheritance, types, symbol references,
compiler include semantics, macro semantics, runtime behavior, or semantic ownership.
The first packaging proof exposed that language parsers were inherited as mandatory dependencies.
The base wheel now depends only on Markdown and MCP packages. Python works in the base wheel;
JavaScript, TypeScript, C++, and all-language extras install their respective Tree-sitter
frontends. Missing extras return a closed error with the exact install target.
### Fixed reference MCP and launchers
A strict `.docforge/reference-adapter.toml` selects one fixed reference language, project
identity, and explicit source roots. C++ additionally requires the compilation database. The fixed
`python -m docforge.reference_mcp` binding constructs only repository-owned providers and exposes
the 21 read tools.
The initial launcher design generated `python -I -m` for arbitrary project module names without
proving that isolated Python could resolve them. The corrected `AdapterLauncherV1` accepts only an
installed project-confined top-level module or the exact trusted `docforge.reference_mcp`
exception. A fixed bounded `find_spec` probe runs under isolated Python without importing project
code. Tests then launch both a real project module and the real reference MCP process.
Generated Codex, Claude, and OpenClaw fragments bind the exact launcher, descriptor, source
availability, effective and projection policy, interpreter, canonical arguments, empty
environment, artifact bytes, and hashes. There is no arbitrary command, argument list, working
directory, environment, discovery, or callable selector.
### Generated reference and documentation gates
CLI tables are derived from the real argparse subcommands. MCP tables are derived from a real
application-enabled server's `list_tools()` registrations, including capability surface,
arguments, descriptions, and input-schema hashes. The generated artifact contains 28 CLI rows and
36 MCP rows.
An adversarial review found a target-change race between the generator's identity check and
replacement. Publication now locks cooperating generators, uses no-clobber publication for a
missing target, and uses Linux atomic exchange plus displaced-byte and file-identity verification
for an existing target. Raced data is restored or retained for recovery rather than discarded.
`docs-check` validates generated drift, local files and anchors, one H1 per page, README-rooted
reachability, required Milestone 4 inventory, the generated notice, and every documented
reference-adapter TOML example against the packaged schema.
### Candidate, adoption, and scale evidence
Frozen executable candidate `95271dcf2e48045b9d3aed9b9ea09c7fc155692c` passed formatting,
Python and web lint, strict Pyright, compilation, 142 contract tests and 268 subtests, 347 total
tests and 402 subtests, three accessibility flows, lock and dependency checks, package builds,
offline wheel adoption, and every maintained smoke benchmark.
The offline adoption proof installed the base wheel from the lock without network access. It
proved no Tree-sitter package or module was installed, built and checked a real Python reference
project, performed isolated MCP bootstrap/search/get-node over the exact 21 read tools, and proved
that base-only C++ fails with `optional_dependency_missing` and `docforge[cpp]` remediation.
The clean 334-source benchmark produced 1,002 primary nodes, 1,001 primary edges, 334 Logic
projections, 2,338 Logic nodes, and 2,338 Logic edges. Complete and incremental output matched
exactly. Warm build p95 was 1,265.387 ms with zero `ast.parse` and zero `extract_source` calls.
Complete equivalence, corrupt cache recovery, and corrupt index recovery were all below 1.9
seconds. The largest per-operation traced peak was 69,997,166 bytes and process high-water was
78,798,848 bytes. Exact hashes and method are recorded in
`benchmarks/milestone4-2026-07-29.json` and `docs/MILESTONE_4_BASELINE.md`.
Milestone 4 is complete. No tag or release was created. Milestone 5 remains unstarted until its own
active-slice contract is frozen.

21
LICENSE Normal file
View file

@ -0,0 +1,21 @@
MIT License
Copyright (c) 2026 Worldforge contributors
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.

197
Makefile Normal file
View file

@ -0,0 +1,197 @@
PYTHON := .venv/bin/python
PYRIGHT := pyright
UV := uv
NPM := npm
GITLEAKS := gitleaks
PYTHONPYCACHEPREFIX := /tmp/docforge-quality-pycache
PYTEST_BASETEMP := /tmp/docforge-quality-pytest
.PHONY: accessibility adoption-m4 benchmark benchmark-full benchmark-m1 benchmark-m1-smoke benchmark-m2 benchmark-m2-smoke benchmark-m3 benchmark-m3-full benchmark-m3-smoke benchmark-m4 benchmark-m4-full benchmark-m4-smoke benchmark-smoke build command-reference-check compatibility-m5 compile concurrency-m5 contract dependencies docs-check format-check fresh-clone-m5 gate lint lock migration-m5 recovery-m5 release-artifacts release-gate release-posttag release-pretag secret-scan task-evidence-m5 task-evidence-m5-smoke test type version-check
accessibility:
$(NPM) run test:accessibility
format-check:
$(PYTHON) -m ruff format --check src tests tools
lint:
$(PYTHON) -m ruff check src tests tools
$(NPM) run lint:web
type:
$(PYRIGHT) --pythonpath .venv/bin/python
compile:
PYTHONPYCACHEPREFIX=$(PYTHONPYCACHEPREFIX) $(PYTHON) -m compileall -q src tests tools
contract:
PYTHONPYCACHEPREFIX=$(PYTHONPYCACHEPREFIX) $(PYTHON) -m pytest -q \
-p no:cacheprovider --basetemp=$(PYTEST_BASETEMP) \
tests/test_public_contract.py \
tests/test_policy.py \
tests/test_projection_policy.py \
tests/test_projection_policy_integration.py \
tests/test_projection_worker.py \
tests/test_projection_fragments.py \
tests/test_retrieval.py \
tests/test_generation_diff.py \
tests/test_client_integration.py \
tests/test_projection_contract.py \
tests/test_projection_schemas.py \
tests/test_graph_projection.py \
tests/test_graph_rendering.py \
tests/test_graph_publication.py \
tests/test_observability.py::TelemetryContractTests::test_schema_fixed_names_match_the_implementation \
tests/test_adapter_contract.py::AdapterContractTests::test_no_ast_index_policy_rejects_logic_publication \
tests/test_adapter_contract.py::AdapterContractTests::test_no_ast_index_accepts_legacy_and_non_logic_incremental_adapters \
tests/test_adapter_contract.py::AdapterContractTests::test_no_ast_rejects_preexisting_logic_index_and_viewer_snapshot \
tests/test_mcp_server.py::DocForgeMcpTests::test_no_ast_binding_preserves_adapter_and_blocks_logic
test:
PYTHONPYCACHEPREFIX=$(PYTHONPYCACHEPREFIX) $(PYTHON) -m pytest -q \
-p no:cacheprovider --basetemp=$(PYTEST_BASETEMP)
lock:
$(UV) lock --check
dependencies:
$(NPM) ls --all
build:
$(UV) build
adoption-m4:
$(PYTHON) tools/milestone4_adoption.py
version-check:
$(PYTHON) tools/check_release_identity.py --mode smoke --tag-state ignore > /dev/null
release-artifacts:
$(PYTHON) tools/check_release_identity.py --mode full --require-clean \
--tag-state ignore --output /tmp/docforge-milestone5-release-identity.json > /dev/null
secret-scan:
$(GITLEAKS) git . --no-banner --redact
$(GITLEAKS) dir . --no-banner --redact
migration-m5:
$(PYTHON) tools/milestone5_migration.py \
--output /tmp/docforge-milestone5-migration.json > /dev/null
task-evidence-m5-smoke:
$(PYTHON) tools/milestone5_task_evidence.py --mode smoke \
--output /tmp/docforge-milestone5-task-evidence-smoke.json > /dev/null
task-evidence-m5:
$(PYTHON) tools/milestone5_task_evidence.py --mode full \
--output /tmp/docforge-milestone5-task-evidence.json > /dev/null
compatibility-m5:
PYTHONPYCACHEPREFIX=$(PYTHONPYCACHEPREFIX) $(PYTHON) -m pytest -q \
-p no:cacheprovider --basetemp=$(PYTEST_BASETEMP)-compatibility \
tests/test_public_contract.py \
tests/test_adapter_contract.py \
tests/test_adapter_sdk.py \
tests/test_policy.py \
tests/test_projection_policy.py \
tests/test_projection_policy_integration.py \
tests/test_projection_contract.py \
tests/test_projection_schemas.py \
tests/test_projection_worker.py \
tests/test_projection_fragments.py \
tests/test_retrieval.py \
tests/test_graph_projection.py \
tests/test_graph_rendering.py \
tests/test_graph_publication.py
concurrency-m5:
PYTHONPYCACHEPREFIX=$(PYTHONPYCACHEPREFIX) $(PYTHON) -m pytest -q \
-p no:cacheprovider --basetemp=$(PYTEST_BASETEMP)-concurrency \
tests/test_core.py::DocForgeCoreTests::test_stale_source_fails_closed_and_failed_rebuild_preserves_index \
tests/test_core.py::DocForgeCoreTests::test_query_rechecks_source_identity_before_returning \
tests/test_core.py::DocForgeCoreTests::test_source_set_change_during_load_fails_closed \
tests/test_adapter_contract.py::AdapterContractTests::test_fast_incremental_reads_reverify_a_changed_index_file \
tests/test_retrieval.py::TaskRetrievalTests::test_final_generation_change_rejects_the_whole_capsule \
tests/test_visualization.py::VisualizationTests::test_snapshot_source_never_mixes_pinned_graph_with_newer_canonical_text \
tests/test_graph_publication.py::GraphPublicationTests::test_status_detects_source_and_publication_races \
tests/test_changesets.py
recovery-m5:
PYTHONPYCACHEPREFIX=$(PYTHONPYCACHEPREFIX) $(PYTHON) -m pytest -q \
-p no:cacheprovider --basetemp=$(PYTEST_BASETEMP)-recovery \
tests/test_milestone5_recovery.py \
tests/test_core.py::DocForgeCoreTests::test_missing_or_corrupt_generation_falls_back_and_repairs \
tests/test_core.py::DocForgeCoreTests::test_stale_source_fails_closed_and_failed_rebuild_preserves_index \
tests/test_incremental_cache.py \
tests/test_python_reference_adapter.py \
tests/test_javascript_reference_adapter.py \
tests/test_cpp_reference_adapter.py \
tests/test_generation_diff.py \
tests/test_rendering.py \
tests/test_graph_publication.py \
tests/test_projection_fragments.py
command-reference-check:
$(PYTHON) tools/generate_command_reference.py \
--output docs/COMMAND_REFERENCE.md --check > /dev/null
docs-check: command-reference-check
$(PYTHON) tools/check_documentation.py \
--pending-inventory tools/docs_pending_m4_pages.txt
benchmark-smoke:
$(PYTHON) tools/milestone0_baseline.py --nodes 25 --samples 1 --cold-samples 1 \
--output /tmp/docforge-milestone0-smoke.json > /dev/null
benchmark:
$(PYTHON) tools/milestone0_baseline.py --nodes 1000 --samples 10 --cold-samples 3
benchmark-m1-smoke:
$(PYTHON) tools/milestone1_benchmark.py --nodes 25 --samples 1 \
--output /tmp/docforge-milestone1-smoke.json > /dev/null
benchmark-m1:
$(PYTHON) tools/milestone1_benchmark.py --nodes 1000 --samples 10
benchmark-m2-smoke:
$(PYTHON) tools/milestone2_benchmark.py --nodes 25 --samples 1 \
--output /tmp/docforge-milestone2-smoke.json > /dev/null
benchmark-m2:
$(PYTHON) tools/milestone2_benchmark.py --nodes 1000 --samples 10
benchmark-m3-smoke:
$(PYTHON) tools/milestone3_benchmark.py --mode smoke \
--output /tmp/docforge-milestone3-smoke.json > /dev/null
benchmark-m3:
$(PYTHON) tools/milestone3_benchmark.py --mode full
benchmark-m3-full: benchmark-m3
benchmark-m4-smoke:
$(PYTHON) tools/milestone4_benchmark.py --mode smoke \
--output /tmp/docforge-milestone4-smoke.json > /dev/null
benchmark-m4:
$(PYTHON) tools/milestone4_benchmark.py --mode full
benchmark-m4-full: benchmark-m4
benchmark-full: benchmark benchmark-m1 benchmark-m2 benchmark-m3-full benchmark-m4-full
gate: format-check lint type compile contract test accessibility lock dependencies build docs-check benchmark-smoke benchmark-m1-smoke benchmark-m2-smoke benchmark-m3-smoke benchmark-m4-smoke
release-gate: gate compatibility-m5 migration-m5 concurrency-m5 recovery-m5 task-evidence-m5 adoption-m4 version-check release-artifacts secret-scan benchmark-full
fresh-clone-m5:
$(PYTHON) tools/milestone5_fresh_clone.py \
--output /tmp/docforge-milestone5-fresh-clone.json > /dev/null
release-pretag: release-gate fresh-clone-m5
$(PYTHON) tools/check_release_identity.py --mode smoke --require-clean \
--tag-state absent > /dev/null
release-posttag:
$(PYTHON) tools/check_release_identity.py --mode smoke --require-clean \
--tag-state head > /dev/null

282
README.md
View file

@ -1,97 +1,271 @@
# DocForge
DocForge is a project-scoped documentation graph for people and AI agents. It validates canonical
documentation, builds a disposable search and relationship index, compiles bounded context, renders
declared manuals, visualizes project structure, and manages reviewable documentation changesets.
DocForge is a project-scoped documentation and source graph for people and AI agents. Canonical
project files remain authoritative; DocForge validates them, builds disposable search and
relationship indexes, compiles bounded context, manages reviewable changesets, renders declared
manuals and portable graph views, and serves one project-bound MCP surface.
## What it does
DocForge never treats indexed text as instructions. It does not run project build commands,
compilers, Git operations, deployments, or arbitrary renderers, and it does not select projects
globally.
- Validates stable Markdown/TOML nodes and typed relationships.
- Builds a deterministic SQLite search and graph index.
- Exposes project-bound CLI and MCP query surfaces.
- Creates, validates, diffs, and previews isolated changesets.
- Applies one explicitly approved changeset hash through CLI or gated MCP.
- Runs a managed loopback graph browser with neighborhood, semantic Flow, convergence Web,
source inspection, and branch-aware node hiding.
- Supports generic documentation projects and project-owned source adapters.
DocForge 2.0.0 is the current stable DocForge2 release. It gives the successor product an
unambiguous major-version identity, preserves the complete 1.4 adapter and recovery platform, and
adds explicit cross-identity proposal acceptance for project-owned canonical appliers. Annotated
tag `v2.0.0` identifies the synchronized release commit.
DocForge never treats indexed text as instructions. It does not run shell commands, mutate Git,
build applications, deploy, publish, or select projects globally.
## Start here
## Graph views
- New installation or first project: [New-project quickstart](docs/NEW_PROJECT_QUICKSTART.md)
- Mental model and authority: [Core concepts and authority](docs/CORE_CONCEPTS_AND_AUTHORITY.md)
- Complete configuration shape: [Project descriptor](docs/PROJECT_DESCRIPTOR.md)
- Fixed runnable examples: [Reference adapters](docs/REFERENCE_ADAPTERS.md)
- Agent and client setup: [Agent integration](docs/AGENT_INTEGRATION.md)
- Exact live command and tool inventory: [Generated command reference](docs/COMMAND_REFERENCE.md)
- Task-oriented operating guide: [User manual](docs/USER_MANUAL.md)
- Release changes and evidence: [Changelog](CHANGELOG.md) and [Milestone 5
baseline](docs/MILESTONE_5_BASELINE.md)
The browser presents the same indexed graph through three complementary views:
## Current capabilities
- **Nodes** shows a bounded, relation-neutral neighborhood around the focus. It is the broad
inspection view for seeing stored incoming and outgoing relationships without changing their
direction. Semantic cards distinguish structure, behavior, dependencies, execution, data,
evidence, context, and other relationships.
- **Flow** shows semantic origin-to-destination paths that terminate at the focus. DocForge
reverses prerequisite-style relationships for presentation, so imports, dependencies, reads,
inheritance, definitions, and tests flow toward the thing they help create or exercise.
- **Web** shows the larger convergence picture: Flow contributors plus contextual relationships,
callers, containers, and direct members or execution dependencies owned by the focus.
- Validates Markdown and TOML nodes, stable IDs, typed relationships, project limits, and confined
paths.
- Builds a deterministic, disposable SQLite graph and search index with integrity and generation
evidence.
- Exposes project-bound CLI and MCP read, proposal, application, rendering, and visualization
surfaces according to the startup policy.
- Compiles versioned, generation-bound task context with cited evidence, explicit gaps, bounded
output, and deterministic continuation.
- Creates isolated documentation changesets, validates complete projected graphs, and applies only
one explicitly approved changeset hash through a separately bound canonical applier.
- Supports complete-projection adapters and opt-in incremental adapters with reverse-dependency
invalidation, bounded extraction caches, and a clean full-build equivalence oracle.
- Publishes function-scoped Logic separately from the primary graph for Python, JavaScript,
TypeScript, and C++ integrations that provide it.
- Compiles manual and portable graph plans into immutable packages for fixed detached renderers,
then records bounded receipts.
- Runs a managed loopback graph viewer with Nodes, Flow, Web, lazy Logic, source inspection, and
branch-aware hiding.
- Generates deterministic Codex, Claude, and OpenClaw client fragments without copying ambient
environment values or secrets.
Graph cards show the node's readable leaf name and kind without clipping either value. The full
qualified identity remains available in the tooltip, compact descriptor, and full inspector.
## Adapter platform
**Hide node** removes noise without changing the index. In Flow and Web, hiding a contributor also
removes upstream ancestors that no longer have a path to the focus. Nodes between the hidden
contributor and the focus stay visible, and alternate ancestor paths remain intact. **Restore
hidden** restores the presentation.
Adapter authors use the public `docforge.adapter_sdk` surface for graph types, complete and
incremental contracts, validation helpers, and `verify_adapter_conformance()`. Existing adapters
that implement only `load_projection()` remain supported. Incremental adapters add a manifest and
source extraction while retaining `load_projection()` as the complete graph oracle. An adapter
that publishes Logic incrementally also supplies `load_complete_assembly()` so the complete oracle
covers both graph and Logic.
## Five-minute start
The repository includes bounded reference integrations for:
Requirements are Python 3.12+, `uv`, and Node.js/npm.
- Python, using the standard-library AST and publishing only project-local imports;
- JavaScript and TypeScript, using their distinct optional Tree-sitter grammars and publishing only
project-local static relative imports and re-exports;
- C++, using `compile_commands.json` as translation-unit inventory and fingerprint evidence, without
executing its commands or a compiler, and publishing only directly resolvable project-local
quoted includes.
These integrations demonstrate the adapter contract; they do not claim resolved calls,
inheritance, types, runtime behavior, macro expansion, compiler include semantics, or semantic
ownership. Python works from the base wheel. Install the `javascript`, `typescript`, or `cpp`
extra for the corresponding grammar, or `languages` for all three:
```bash
git clone forgejo@repo.andraxion.net:administrator/DocForge.git /absolute/path/DocForge
uv pip install "/absolute/path/DocForge[javascript]"
uv pip install "/absolute/path/DocForge[typescript]"
uv pip install "/absolute/path/DocForge[cpp]"
uv pip install "/absolute/path/DocForge[languages]"
```
Reference projects use the fixed `.docforge/reference-adapter.toml` descriptor and the installed
`docforge.reference_mcp` module. That server is read-only and exposes no proposal or application
surface.
Project-owned adapters use `AdapterLauncherV1` and
`generate_adapter_client_configuration()` to produce a client fragment from an explicitly
constructed adapter project. The launcher is immutable and contains no command, caller arguments,
working directory, environment, discovery rule, or callable selector. It resolves through isolated
Python to one installed, project-owned top-level module; the only trusted dotted exception is the
fixed `docforge.reference_mcp` binding. Generic `docforge configure` intentionally refuses custom
adapters because it cannot safely reconstruct project-owned composition.
See the [Adapter authoring guide](docs/ADAPTER_AUTHORING_GUIDE.md), [Reference
adapters](docs/REFERENCE_ADAPTERS.md), and [Agent integration](docs/AGENT_INTEGRATION.md) for the
supported routes.
## Authority and projections
Canonical Markdown, TOML, adapter-declared sources, and descriptor files own project facts. The
SQLite index, extraction cache, render packages, previews, portable artifacts, receipts, client
fragments, and viewer processes are derived and replaceable.
The primary graph contains project nodes and relationships. Logic is a separate, lazy,
function-scoped control-flow projection. Manual output, portable graph output, and the live viewer
are independent consumers of one validated generation:
```text
validated generation
├── ManualRenderPlanV1 → immutable package → detached manual renderer
├── GraphViewPlanV1 → immutable package → detached portable graph renderer
└── pinned index → managed read-only live viewer
```
The version-2 projection policy selects each consumer independently:
```text
manual: auto | explicit | disabled
portable_graph: explicit | disabled
live_viewer: on-demand | disabled
```
The process capability policy is separate. Capability mode controls the registered read,
proposal, application, or reserved operator surface; descriptor writers and the startup-bound
canonical applier determine whether mutations are actually authorized. `--no-ast` is a restrictive
binding policy over adapter evolution, Logic publication, and Logic retrieval. It is not a parser
inspection mechanism or a filesystem sandbox, and the AST/Tree-sitter reference integrations
should not be presented as no-AST adapters.
Read [Policy precedence](docs/POLICY_PRECEDENCE.md), [Rendering and
visualization](docs/RENDERING_AND_VISUALIZATION.md), and [Legacy and no-AST
operation](docs/LEGACY_AND_NO_AST.md) before changing those boundaries.
## Five-minute generic project
Requirements are Python 3.12 or newer and an installed DocForge environment. From a development
checkout, `uv sync --group dev` creates `.venv`:
```bash
git clone <repository-url> /absolute/path/DocForge
cd /absolute/path/DocForge
uv sync --group dev
npm ci
PROJECT=/absolute/path/MyProject
.venv/bin/docforge --project-root "$PROJECT" onboard
.venv/bin/docforge --project-root "$PROJECT" onboard \
--scaffold \
--project-id my-project \
--title "My Project"
```
The first command is read-only. Scaffolding is explicit and create-only: it writes a generic
descriptor, one canonical overview node, and a built-in manual template, then indexes and renders
them. Detected source languages remain `adapter_required` until a validated frontend is selected.
Operate the configured project:
```bash
.venv/bin/docforge --project-root "$PROJECT" validate
.venv/bin/docforge --project-root "$PROJECT" reindex
.venv/bin/docforge --project-root "$PROJECT" search architecture
.venv/bin/docforge --project-root "$PROJECT" visualize
```
Install the persistent per-user graph viewer once:
Install the persistent per-user graph viewer manager once when using the live viewer:
```bash
.venv/bin/docforge-viewer-manager install-user-service
```
Start an MCP server for one project:
Start one generic, read-only MCP server:
```bash
.venv/bin/docforge-mcp \
--project-root "$PROJECT" \
--proposal-writer project-editor
--capability-mode read
```
Add `--canonical-applier project-editor` only when that MCP integration should expose the
hash-bound `docforge_apply_changeset` tool.
Add a descriptor-authorized proposal writer only when the client should create proposals. Add a
matching `--canonical-applier` only when that integration should expose exact-hash application.
See the [quickstart](docs/NEW_PROJECT_QUICKSTART.md) for the reference-adapter and generated-client
paths.
## Documentation
- [User manual](docs/USER_MANUAL.md) — features, setup, visualization, CLI, MCP, apply, adapters,
and troubleshooting.
- [Core contract](docs/CONTRACT.md) — invariants and security boundary.
- [MCP contract](docs/MCP_CONTRACT.md) — exact tool and process boundary.
- [Viewer manager](docs/VIEWER_MANAGER.md) — native service setup and lifecycle.
- [Adapter decision](docs/APPLICATION_DECISION.md) — why custom adapters own canonical
serialization.
### Learn and operate
- [New-project quickstart](docs/NEW_PROJECT_QUICKSTART.md)
- [Core concepts and authority](docs/CORE_CONCEPTS_AND_AUTHORITY.md)
- [Project descriptor](docs/PROJECT_DESCRIPTOR.md)
- [Policy precedence](docs/POLICY_PRECEDENCE.md)
- [Reference adapters](docs/REFERENCE_ADAPTERS.md)
- [Agent integration](docs/AGENT_INTEGRATION.md)
- [User manual](docs/USER_MANUAL.md)
- [Generated command reference](docs/COMMAND_REFERENCE.md)
- [Project onboarding](docs/PROJECT_ONBOARDING.md)
- [Rendering and visualization](docs/RENDERING_AND_VISUALIZATION.md)
- [Recovery and performance](docs/RECOVERY_AND_PERFORMANCE.md)
- [Security](docs/SECURITY.md)
### Contracts and compatibility
- [Core contract](docs/CONTRACT.md)
- [MCP contract](docs/MCP_CONTRACT.md)
- [Compatibility contract](docs/COMPATIBILITY.md)
- [Adapter authoring guide](docs/ADAPTER_AUTHORING_GUIDE.md)
- [Incremental indexing](docs/INCREMENTAL_INDEXING.md)
- [Canonical application](docs/CANONICAL_APPLICATION.md)
- [Legacy and no-AST operation](docs/LEGACY_AND_NO_AST.md)
- [Migrating from version 1](docs/MIGRATING_FROM_V1.md)
- [Viewer manager](docs/VIEWER_MANAGER.md)
### Milestone evidence
- [Milestone 5 baseline](docs/MILESTONE_5_BASELINE.md)
- [Milestone 5 closeout](docs/MILESTONE_5_CLOSEOUT.md)
- [Milestone 4 baseline](docs/MILESTONE_4_BASELINE.md)
- [Milestone 4 closeout](docs/MILESTONE_4_CLOSEOUT.md)
- [Milestone 3 baseline](docs/MILESTONE_3_BASELINE.md) and [closeout](docs/MILESTONE_3_CLOSEOUT.md)
- [Milestone 2 baseline](docs/MILESTONE_2_BASELINE.md) and [closeout](docs/MILESTONE_2_CLOSEOUT.md)
- [Milestone 1 baseline](docs/MILESTONE_1_BASELINE.md) and [closeout](docs/MILESTONE_1_CLOSEOUT.md)
- [Milestone 0 baseline](docs/MILESTONE_0_BASELINE.md)
Historical milestone records preserve the facts and dependency observations of their frozen
candidates. Use the current guides and contracts for present behavior.
## Development
Run the repository-native gate:
```bash
npx pyright
npm run lint:web
uv run ruff check src tests tools
uv run ruff format --check src tests tools
uv run python -m compileall -q src tests tools
uv run pytest -q
make gate
```
See [AGENTS.md](AGENTS.md) before changing core boundaries.
Milestone 4 maintenance entry points include:
```bash
make adoption-m4
make benchmark-m4-smoke
make benchmark-m4
make benchmark-m4-full
make command-reference-check
make docs-check
```
`adoption-m4` builds and exercises a fresh base wheel without Tree-sitter packages and proves the
Python reference plus a real isolated read-only MCP retrieval. The optional-language integrations
have their own focused tests and extras. `benchmark-m4` runs the maintained full adapter workload;
the smoke target is for routine gate coverage, not final performance evidence.
Milestone 5 adds focused and aggregate release gates:
```bash
make compatibility-m5
make migration-m5
make concurrency-m5
make recovery-m5
make task-evidence-m5
make release-gate
make fresh-clone-m5
```
`release-gate` includes the complete repository, accessibility, adoption, artifact-reproducibility,
secret-scan, and full maintained benchmark sequence. `fresh-clone-m5` is the final remote-commit
rehearsal and must pass before the annotated tag and Forgejo release are created.
Pass `--diagnostics` to `docforge` or `docforge-mcp` for bounded request-local timings and compiler
work counters. Diagnostics are disabled by default and do not displace a primary result when the
configured output budget is tight.
Read [AGENTS.md](AGENTS.md) before changing core boundaries.

View file

@ -1,706 +1,108 @@
# Completed slices
## DFG-20 gated application and self-service graph operations
### Changed
- Released DocForge 0.13.0 with exact-hash canonical application through CLI and opt-in MCP.
- Added generic Markdown/TOML serialization, rollback, semantic verification, index refresh, and
declared render refresh. Custom adapters retain ownership of canonical serialization.
- Added CLI `reindex`, `visualize`, visualization status, and visualization stop commands.
- Added browser-side node hiding/restoration, bounded source inspection at anchors, and a
scrollable full inspector.
- Replaced the repository quick reference with a dedicated user manual covering setup, CLI, MCP,
visualization, application, adapters, and troubleshooting.
### Verification
- Application tests cover all four proposal operations, exact-hash rejection, MCP gating, derived
refresh, and CLI use.
- Visualization tests cover source confinement, browser script validity, hiding controls, and
inspector layout.
## DFG-19 cross-platform supervised viewer manager
### Changed
- Released DocForge 0.12.0 with an authenticated loopback viewer-manager protocol.
- Moved viewer process ownership out of the MCP stdio host and into an OS-supervised per-user
manager. Linux uses systemd user services, macOS LaunchAgents, and Windows Task Scheduler.
- Added `docforge_visualization_status` and changed worker lifetime to a one-hour browser-activity
policy with explicit per-project stop.
- Replaced Unix-socket and inherited-file-descriptor assumptions with loopback TCP and standard
input/output worker control, so the manager protocol works on Windows as well as Unix platforms.
### Verification
- Lifecycle tests cover manager worker reuse, browser activity renewal, idle reclamation, explicit
stop, strict MCP surface registration, and manager-mediated visualization startup.
## DFG-18 persistent visualization lifecycle
### Changed
- Released DocForge 0.11.0 with a persistent project-bound visualization worker.
- Replaced browser leases and MCP-owner-process shutdown with an explicit
`docforge_stop_visualization` read tool.
- Added a private, atomically written project-cache registry. It reuses a live worker only when
its authenticated loopback endpoint and exact index snapshot identity match the current request.
- Removed the parent-process `Popen` lifecycle dependency by spawning the session-isolated worker
directly, so no process cleanup warning or parent lifetime remains coupled to the browser.
### Verification
- A process-boundary test terminates the launcher, confirms the viewer remains live, confirms a
separate runner reuses its URL, and confirms the explicit stop tool terminates it.
- Focused warning-strict lifecycle and MCP contract tests pass.
## DFG-17 relationship-aware graph and upstream flow
### Changed
- Released DocForge 0.10.0 with the fixed `graph-browser@8` template.
- Assigned generic semantic families, distinct colors, line patterns, and directional endpoint
symbols to common structure, execution, data, dependency, evidence, and context relations.
- Added a static visible-relationship key shared by Nodes and Flow, including deterministic
fallback styling for project-defined relations.
- Renamed topology-derived navigation from ambiguous Children and Edge language to Focus node,
Outgoing paths, and Incoming & lateral.
- Implemented bounded upstream Flow layers. Calls, dispatches, launches, activations, and writes
keep declared direction; reads, imports, and dependencies reverse for lineage presentation;
structure, evidence, context, and unknown relations remain excluded.
- Separated relationship-line and context-node CSS classes to prevent style inheritance and DOM
selector collisions.
### Verification
- A deterministic JavaScript harness covers relation classification, fallback styling, semantic
direction, evidence exclusion, upstream membership, dependency reversal, and layered placement.
- Browser interaction QA verifies distinct line colors, dash patterns, endpoint markers, exact
relationship keys, Nodes-to-Flow switching, evidence exclusion, and the destination-on-right
layout without console or page errors.
- HTML, CSS, and JavaScript validation, strict Pyright, Ruff, formatting, compilation, dependency
locks, npm audit, the complete warning-strict suite, and diff checks pass.
### Limits
- Flow operates on the already bounded neighborhood returned for the current focus and depth.
- Unknown project-defined relations receive deterministic Nodes styling but do not enter Flow until
their semantic direction is declared in the fixed relation map.
- Cycles are bounded by visited-node traversal. A later gate may add explicit cycle-group rendering
if real project graphs demonstrate that need.
## DFG-16 browser asset quality gate
### Changed
- Added pinned ESLint, Stylelint, CSS-tree, and HTML Validate development tooling.
- Added one `npm run lint:web` gate that extracts the exact embedded viewer assets without writing
generated repository files.
- Validated a freshly rendered fixture manual in addition to the graph viewer.
- Corrected viewer landmark names, explicit input type, ARIA group semantics, inline legend styles,
and HTML doctype casing.
### Verification
- HTML Validate passes the served graph document and a freshly rendered manual.
- Stylelint and CSS-tree pass the embedded stylesheet with syntax and property-value validation.
- ESLint passes the embedded browser script with recommended browser rules and no inline disables.
- Strict Pyright, Ruff, formatting, compilation, all 52 warning-strict tests, dependency locks, and
diff checks pass.
## DFG-15 strict static typing gate
### Changed
- Made the existing strict Pyright configuration resolve DocForge's `.venv` automatically.
- Converted validated TOML, JSON, subprocess, socket, MCP, render, and visualization boundaries
from unknown dynamic values into explicit checked types.
- Kept runtime validation and fail-closed behavior at every untrusted input boundary.
- Made strict `pyright` an explicit repository development gate.
### Verification
- Pyright reports zero errors, warnings, or informational diagnostics across all source modules.
- Ruff lint and formatting, Python compilation, all 52 warning-strict tests, and diff checks pass.
## DFG-14.1 detached viewer lifecycle correction
### Changed
- Released DocForge 0.8.1 with the existing `graph-browser@5` interface.
- Moved the loopback listener into a detached worker so MCP transport teardown cannot kill an
active viewer.
- Bound the worker to the longer-lived MCP client host plus the existing browser lease and startup
grace.
### Verification
- Added a process-boundary regression test that exits the launching transport process, verifies the
viewer still responds, and then verifies lease expiry.
- Retained the in-process listener tests for token confinement, read-only behavior, stale-index
rejection, and lease renewal.
## DFG-14 durable graph navigation
### Changed
- Released the fixed `graph-browser@5` template and DocForge 0.8.0.
- Kept the loopback listener alive across short-lived MCP standard-input transactions with a
browser-renewed lease, while preserving explicit process termination and bounded abandoned-page
cleanup.
- Added a visible disconnected state instead of leaving stale controls to fail silently.
- Added pointer and keyboard resizing for both side panels.
- Made the unblurred modal natively resizable and draggable by its constrained title bar.
- Added generic topology-derived Primary focus, Children, and Edge & context navigation sections.
- Arranged neighborhoods by shortest-hop rings and applied distinct role palettes that darken
progressively by hop distance, capped at fifty percent.
- Kept all category and color decisions client-side without changing project graph facts.
### Verification
- Focused HTTP tests cover heartbeat renewal, bounded lease expiry, non-daemon listener ownership,
and the unchanged token/read-only boundary.
- A deterministic JavaScript harness proves topology roles, hop rings, and distance shading.
- Embedded JavaScript syntax and interaction-contract checks cover panel resizing, modal movement
and resizing, unblurred backdrop behavior, grouped navigation, and lease renewal.
- Ruff, formatting, compilation, the complete warning-strict suite, and live project-bound viewer
checks pass.
### Limits
- Panel widths, modal geometry, viewport position, and open dialog state are not persisted.
- Topology roles are presentation aids. They do not replace project-authored relationship meaning.
- Background-browser timer throttling is tolerated by the three-minute lease but may delay cleanup.
### Next gate
No further gate is planned. Measure use before adding saved layouts, minimaps, or export.
## DFG-13 graph activation reliability
### Changed
- Released the fixed `graph-browser@4` template.
- Delayed SVG pointer capture until movement crosses the four-pixel drag threshold so an ordinary
click remains targeted at the graph node and reaches the modal inspection handler.
- Preserved pointer capture and click suppression for actual canvas drags.
- Added an explicit hidden-state rule so rendered neighborhoods remove the empty-canvas instruction.
- Kept the token-bound HTTP surface, graph data, and listener lifetime unchanged.
- Released the compatible fix as DocForge 0.7.3.
### Verification
- Focused interaction-contract checks distinguish click setup from drag pointer capture and cover
empty-state hiding.
- Embedded JavaScript syntax validation, Ruff, formatting, compilation, and the complete
warning-strict 49-test DocForge suite pass.
### Limits
- The listener remains owned by the MCP process and closes when that process exits.
- Browser state remains client-local and is not persisted.
### Next gate
No further gate is planned. Measure graph-browser use before adding history, comparison, or editing
surfaces.
## DFG-12 modal node inspection
### Changed
- Released the fixed `graph-browser@3` template with a native modal node inspector.
- Made graph-node activation inspect full validated node metadata and content without replacing the
current neighborhood or viewport.
- Added mouse and keyboard activation plus Escape, explicit close controls, and backdrop dismissal.
- Added a separate Explore neighborhood action for intentional graph recentering.
- Kept the existing token-bound, read-only HTTP surface and exact-node endpoint unchanged.
- Released the compatible change as DocForge 0.7.2.
### Verification
- Focused HTTP interaction-contract checks cover the dialog, inspection handler, and explicit
neighborhood action.
- Embedded JavaScript syntax validation and the complete warning-strict DocForge suite pass.
### Limits
- Dialog state is session-local and is not persisted in the URL.
- Node content remains plain text and is not rendered as trusted HTML.
- The right sidebar continues to describe the current root neighborhood.
### Next gate
No further gate is planned. Measure graph-browser use before adding history, comparison, or editing
surfaces.
## DFG-11 graph viewport navigation
### Changed
- Released the fixed `graph-browser@2` template with pointer-centered mouse-wheel zoom.
- Added left-button drag pan with pointer capture and a four-pixel movement threshold.
- Preserved normal node activation by suppressing click navigation only after an actual drag.
- Added keyboard-operable zoom-in, zoom-out, and reset buttons plus a live zoom percentage.
- Reset the viewport whenever a new root neighborhood loads.
- Kept all viewport behavior client-side without adding HTTP endpoints or project authority.
- Released the compatible change as DocForge 0.7.1.
### Verification
- Focused HTTP tests and embedded JavaScript syntax validation cover button zoom, reset,
pointer-centered wheel zoom, left-drag pan, and preserved node-click handling. The current agent
runtime did not expose its rendered browser automation connection, so no rendered interaction
claim is made for this gate.
- The complete warning-strict DocForge suite passes.
### Limits
- Viewport position is session-local and is not persisted.
- The radial layout itself remains deterministic and fixed.
- A minimap, saved node positions, and alternate layouts remain outside the current contract.
### Next gate
No further gate is planned. Measure dense-graph use before adding more navigation or layout
features.
## DFG-10 project-bound graph visualization
### Changed
- Added the fixed `docforge_visualize` MCP read tool to generic, read-only adapter, and
proposal-enabled adapter servers.
- Added the built-in `graph-browser@1` HTML/CSS/JavaScript template with project overview, family
filtering, lexical search, exact node content, and bounded neighborhood traversal.
- Bound the ephemeral HTTP listener to `127.0.0.1` on an operating-system-selected port.
- Added an unguessable per-process URL token and rejected every non-token path.
- Exposed only fixed `GET` and `HEAD` endpoints. Rejected POST, PUT, PATCH, and DELETE.
- Validated the complete project and index once per MCP invocation, then served fast queries from
the exact validated SQLite snapshot.
- Rejected index replacement or alteration after launch and required reinvocation to refresh.
- Accepted no project root, database path, SQL, template path, bind address, command, or renderer.
- Released the capability as DocForge 0.7.0 without changing canonical-write policy.
### Verification
- Protocol tests exercised the new tool through the official in-memory MCP transport.
- HTTP tests proved token confinement, loopback binding, security headers, read-only methods,
deterministic results, exact node retrieval, and snapshot invalidation.
- Cross-project tests ran two simultaneous visualization servers and proved separate project data,
ports, tokens, and indexes.
- The complete warning-strict DocForge suite passed.
- Ani-web proof loaded 3,289 nodes and 6,292 edges. After one full validation, the graph overview
returned in approximately 0.30 seconds and a node neighborhood in approximately 0.03 seconds.
### Limits
- The browser is a validated index snapshot, not a live canonical-file watcher.
- It is reachable only from the machine running the MCP process.
- It does not persist, publish, or externally host a visualization.
- It does not infer relationships beyond the configured project's graph.
### Next gate
No further gate is planned. Measure actual graph-browser use before adding layout modes, exports,
remote access, or project-declared visualization templates.
## DFG-9 controlled application decision
### Decision
- Retained manual canonical integration as the permanent DocForge 0.x policy.
- Added no application command to the library, CLI, or MCP server.
- Kept project builders, tests, Git, deployment, and publication under developer or project-owner
control.
- Required a new approved gate with measured multi-project evidence before canonical application
can be reconsidered.
### Evidence
- DFG-8 produced one real content-only AssetForge proposal and one manual chapter replacement.
- Validation, conflicts, diffs, previews, and stale-source handling were already automated.
- Manual integration completed without an error, lost work, or meaningful repeated cost.
- Automating the remaining step would require canonical writers, developer authorization, atomic
rollback, failure recovery, and project-format ownership that the current evidence does not
justify.
- Existing exact-surface MCP tests prohibit application tools, and proposal tests preserve
canonical source bytes.
### Limits
- DocForge does not apply, commit, push, build, deploy, or publish canonical changes.
- Reopening the decision requires a separately approved, versioned contract and cannot add MCP
canonical application.
### Next gate
No further DFG gate is planned. Continue measured adoption through project-owned integrations.
## DFG-0 contract freeze and DFG-1 standalone read-only core
### Changed
- Created the standalone DocForge repository and versioned the project, node, edge, result, and
reserved changeset contracts.
- Added one-root project descriptors with confined canonical, authority, cache, and index paths.
- Added generic Markdown front matter and TOML node loading, stable IDs, typed relationships,
authority classes, limits, deterministic ordering, dependency-cycle validation, and hashes.
- Added atomic SQLite FTS5 indexes with project-root fingerprints, source revisions, logical row
validation, stale rejection, and preservation of the previous index when rebuilds fail.
- Added exact lookup, bounded search and filtering, backlinks, dependencies, impact traversal, and
cited token-budgeted context compilation with explicit omissions.
- Added deterministic JSON CLI commands for project information, validation, index operations,
retrieval, traversal, and context compilation.
- Added two unrelated generic fixtures. No Worldforge or AssetForge vocabulary entered the core.
### Verification
- Ruff lint and format checks passed.
- Python compilation passed.
- All 13 unit and integration tests passed.
- Tests covered root and symbolic-link escapes, unknown configuration, cache overlap, duplicate and
broken graph state, dependency cycles, source-set changes, stale indexes, tampered rows,
cross-project cache reuse, query-time source changes, deterministic retrieval, and bounded context.
- Installed CLI proof built and checked a temporary project index, returned the expected search
result, selected the required node and dependency, used 153 of 180 estimated tokens, and reported
the omitted proof node.
### Limits
- No MCP server exists yet.
- No changeset or write operation exists.
- No project adapter or renderer exists.
- The token estimator is deliberately conservative and lexical; measured project adoption remains a
later gate.
### Next gate
DFG-2: expose only the proven read operations through a project-bound local stdio MCP server.
## DFG-2 project-bound read-only MCP server
### Changed
- Pinned the official stable MCP Python SDK to the compatible `mcp>=1.28,<2` release line.
- Added a local standard input/output server bound to one immutable project root at startup.
- Exposed eleven read tools for project health, contract boundaries, exact lookup, search, metadata
filtering, backlinks, dependencies, impact, bounded context, source validation, and render status.
- Added project identity, root fingerprint, revision, source hash, adapter version, staleness, server
version, and an untrusted-content warning to tool results.
- Added structured domain failures for missing nodes, stale indexes, and oversized results without
returning partial content.
- Exposed no write, proposal, arbitrary file, shell, Git, build, deployment, publication, or
project-switching operation.
- Kept cache rebuilding as an explicit CLI integration action. MCP queries fail closed when the
derived index is missing or stale.
### Verification
- Ruff lint and format checks passed.
- Python compilation passed.
- All 19 core, CLI, and MCP tests passed with `ResourceWarning` treated as an error.
- Protocol tests called all eleven tools through the official in-memory MCP transport.
- A separate subprocess test initialized the server through real stdio transport and retrieved only
its configured fixture project.
- Tests proved the exact read-only tool surface, fixed project identity, structured missing and stale
failures, output limits, explicit omissions, safe fallback when passive Git revision detection is
unavailable, and the absence of canonical write tools.
### Limits
- The server cannot create changesets or proposals yet.
- The server cannot rebuild its own index.
- Render status reports `not_configured` until DFG-4 defines renderer orchestration.
- Worldforge and AssetForge adapters remain unopened.
### Next gate
DFG-3: add isolated, hash-bound proposal changesets without canonical write authority.
## DFG-3 isolated changesets
### Changed
- Added a confined changeset root and project-declared proposal writers with explicit family and
operation permissions.
- Bound proposal identity once at MCP server startup. Tools cannot select or impersonate a writer.
- Added ordered, project-bound JSON changesets with canonical base revision and source hash, root
fingerprint, creator, optimistic changeset hash, expected node hashes, rationales, and structured
relationship changes.
- Added create, update, same-format move, and delete proposals. Deletes require exact removal of every
incident relationship; required profile nodes cannot be deleted.
- Added deterministic projected graph validation and structured metadata, content, source, and
relationship diffs without changing canonical files.
- Added exact stale-base, stale-node, stale-changeset, ownership, family, operation, path, graph,
source, size, and cross-proposal conflict failures.
- Added process-safe file locking, atomic replacement, symbolic-link rejection, source confinement,
configured limits, and rollback if canonical inputs change during proposal storage.
- Added nine MCP proposal tools, including stale-safe proposal inspection and bounded listing.
Canonical application, previews, arbitrary commands, Git mutation, builds, deployment, and
publication remain absent.
### Verification
- Focused core tests cover all four operation types, deterministic diffs, canonical immutability,
simultaneous append serialization, overlapping changesets, stale identities, atomic failures,
family permissions, ownership, target confinement, symbolic links, and configuration validation.
- Protocol tests call all four mutation tools through the official in-memory MCP transport and prove
fixed writer identity, isolated output, validation, deterministic diff retrieval, and the disabled
mutation behavior of a server without a writer.
- Ruff formatting and lint checks, Python compilation, all five JSON schema parses, and the locked
dependency check passed.
- All 29 core, CLI, changeset, concurrency, in-memory MCP, and real stdio tests passed with
`ResourceWarning` treated as an error.
### Limits
- Changesets are proposals only. DocForge does not apply them to canonical project files.
- A changeset may operate on a node once; a later operation on the same node requires another
changeset after external integration.
- Moves preserve the canonical source format. Cross-format conversion belongs to a future adapter or
explicit migration contract.
- Preview generation and renderer orchestration remain unopened.
### Next gate
DFG-4: add deterministic previews and confined renderer orchestration without canonical application.
## DFG-4 deterministic previews and renderer orchestration
### Changed
- Added optional project-declared template, preview, view, and derived-output configuration with
strict root confinement, overlap rejection, stable view IDs, and configured size limits.
- Added an explicit renderer protocol backed by a closed built-in registry. Configuration cannot
name commands, modules, executable paths, or undeclared renderers.
- Added the `generic_html` renderer with pinned `markdown-it-py` CommonMark parsing, disabled raw
HTML, fixed safe template tokens, deterministic node ordering, navigation, metadata, content, and
relationship output.
- Added render identities covering canonical and proposal inputs, node and edge identities, view
configuration, template hash, renderer contract, and exact Markdown parser version.
- Added atomic per-view CLI rendering, non-writing render status, and isolated changeset previews.
Input changes detected before replacement preserve prior output.
- Added `docforge_preview_changeset` to MCP and made `docforge_render_status` report configured view
hashes and state. MCP cannot render declared project output or select a renderer or command.
- Split shared configuration validation, render configuration, renderer contract, and orchestration
into focused modules instead of expanding the project loader or MCP translation layer.
### Verification
- Renderer tests prove repeatable identities and bytes, current and stale status, isolated previews,
escaped raw HTML, CommonMark conversion, unchanged canonical and declared output, configured
limits, symbolic-link rejection, and preservation of prior output after invalid or changing input.
- Configuration tests reject command-like fields, unsupported renderer IDs, protected output paths,
undeclared views, oversized templates and output, and unsafe symbolic links.
- CLI tests cover declared render, render status, isolated preview, and structured unknown-view
failure. Protocol tests exercise preview through the official in-memory MCP transport and prove
declared output remains absent.
- Ruff formatting and lint checks, Python compilation, all five JSON schema parses, and the locked
dependency check passed.
- All 35 core, CLI, changeset, concurrency, renderer, in-memory MCP, and real stdio tests passed with
`ResourceWarning` treated as an error.
### Limits
- The first built-in renderer emits one self-contained HTML file per view. Multi-file asset bundles
and project-specific view models remain future adapter work.
- Preview generation validates proposals but does not apply them to canonical documentation.
- Declared project-output rendering is an explicit local CLI integration action, not an MCP tool.
- Worldforge and unrelated-project adapters remain unopened.
### Next gate
DFG-5: reproduce Worldforge semantics and generated output through a shadow-only adapter without
changing the live workflow.
## DFG-5A Worldforge non-AssetForge shadow proof
### Changed
- Added a reusable adapter contract with ordered project projections, adapter metadata, root and
identity validation, a standard read-index bridge, and byte-exact artifact comparison.
- Generalized the derived index boundary to accept any immutable project service without changing
generic project loading, proposals, rendering, or MCP behavior.
- Added a Worldforge-local shadow adapter that translates the existing normalized manual index into
core nodes and edges while retaining acceptance and relationship provenance as adapter metadata.
- Kept Worldforge-specific weighted search, backlink ordering, context profiles, and render-model
composition in the Worldforge adapter.
- Excluded the independently managed AssetForge family and combined manual output from this subgate.
### Verification
- The shadow graph matched 522 nodes and 805 edges exactly and built through DocForge's standard
disposable index.
- Exact lookup, three weighted searches, active-development filtering, Phase 5 backlinks, and Phase
5 dependency traversal matched the current Worldforge index.
- Active, Phase 3, and Phase 5 context packs were byte-repeatable. Active also matched the current
derived context cache.
- All 31 generated outputs that do not require AssetForge matched committed bytes. The proof wrote
only temporary derived files and removed them afterward.
- DocForge adapter-contract tests cover standard index use, graph and metadata rejection, identity
changes, cache confinement, and complete byte-exact artifact comparison.
### Limits
- The remaining 10 AssetForge nodes, 25 incident edges, AssetForge context profile, and combined
`manual/manual.html` output are not read or rebuilt by this proof.
- The shadow adapter is an explicit local command. It is not discoverable or executable through the
normal MCP server.
### Next gate
DFG-5B: complete the full-family shadow proof when AssetForge is explicitly authorized.
## DFG-5B Worldforge full-family shadow completion
### Changed
- Expanded the Worldforge-local adapter from the partial proof to all source families, including
the ten AssetForge nodes and their 25 incident edges.
- Added a Worldforge-owned AssetForge context profile with deterministic source ordering, stable
node and source citations, a hard token budget, and explicit omission records.
- Routed the shadow render comparison through the complete Worldforge builder output inventory,
including the combined `manual/manual.html` output.
- Released the adapter boundary as DocForge 0.4.0. Worldforge-specific context, query, and render
policy remains outside the generic core.
### Verification
- The shadow graph matched all 532 nodes and 830 edges exactly through DocForge's standard
disposable index.
- Exact lookup, five weighted searches, development and AssetForge filters, backlinks, and phase
and AssetForge dependency traversal matched the current Worldforge index.
- Active, Phase 3, Phase 5, and AssetForge contexts were byte-repeatable. Active matched the current
cache; AssetForge included all ten nodes under the normal budget.
- A reduced AssetForge budget retained the required root, stayed within budget, and recorded
omissions. An invalid budget failed before producing a context.
- All 32 generated outputs matched committed bytes. The proof wrote only temporary derived files
and removed them afterward.
### Limits
- The adapter remains an explicit local shadow command and is not loaded by the normal MCP server.
- Canonical application and public deployment remain outside DocForge.
- Reuse outside Worldforge is not yet proven.
### Next gate
DFG-6: prove the generic core with an unrelated project and simultaneous project-isolated servers.
## DFG-6 unrelated-project proof
### Changed
- Added an Awesome Ski Game fixture using the generic project descriptor, five unrelated node
families, six relationships, a bounded `ride-day` context, one proposal writer, and one declared
field-guide view.
- Added an end-to-end proof covering generic loading, indexing, exact graph counts, search, filters,
dependency traversal, deterministic bounded context, isolated updates, validation, diffs, and
escaped preview rendering.
- Added a source guard that rejects Worldforge, AssetForge, phase, or villager vocabulary in the
generic core.
- Added a live isolation proof with two simultaneous MCP server subprocesses bound to Awesome Ski
Game and Alpha Documentation.
### Verification
- Awesome Ski Game loaded through `adapter = "generic"` with five nodes, six edges, and trail,
riding, safety, session, and proof families.
- The 300-token context retained its required session node, stayed within budget, recorded
omissions, and reproduced exactly.
- The proposal changed only its isolated changeset and preview. Canonical sources and declared
output remained unchanged, and raw HTML was escaped.
- Both live servers returned their own project identity and nodes, rejected the other project's
stable IDs, and wrote same-named changesets and previews only under their bound roots.
- The complete DocForge suite passes with warnings treated as errors.
### Limits
- This proof does not adopt DocForge inside Worldforge or enable any canonical write path.
- The Worldforge adapter and generic Awesome Ski Game fixture remain separate ownership paths.
- HTTP transport, accounts, and web administration remain unopened.
### Next gate
DFG-7: adopt project-bound DocForge retrieval for real Worldforge read-only tasks with measured
quality and a documented rollback path.
## DFG-7 Worldforge read-only adoption
### Changed
- Added an explicit adapter-backed read-only MCP constructor that accepts one validated project
service and an optional project-owned context provider.
- Kept adapter discovery, session selection, family partitioning, and project context policy outside
the generic core.
- Added Worldforge-owned descriptors and separate Worldforge and AssetForge sessions with disjoint
derived indexes and the exact fixed read tool surface.
- Added durable retrieval, context-size, omission, latency, stale-state, and rollback evidence.
- Released the adapter-backed read-only boundary as DocForge 0.5.0.
### Verification
- Eight real Worldforge and AssetForge retrieval tasks retained every required node in the first
five results; seven matched the current manual index result set exactly.
- Active and Phase 5 contexts reduced the full structured Worldforge session by 97.3% and 98.2%.
Tight budgets reported every omitted candidate.
- Two simultaneous MCP processes retained separate identities, exposed only read tools, rejected
cross-family node access, and kept Worldforge stale-state failure isolated from AssetForge.
- Checked DocForge search measured 84.4 ms median in the adoption run versus 18.8 ms for the current
manual index. The additional validation cost remained below 0.1 seconds.
- The DocForge suite, Worldforge manual suite, integration tests, shadow proof, format, lint, and
generated-output checks passed.
### Limits
- DocForge read-only operations do not write canonical Worldforge files or replace its builder.
- Adapter-backed read-only service construction is explicit; the generic server does not discover
project adapters or sessions.
- AssetForge proposal access, canonical application, publication, and deployment remain closed.
### Next gate
DFG-8: adopt isolated AssetForge-only proposals with explicit review and the canonical Worldforge
build and verification workflow.
## DFG-8 AssetForge proposal adoption
### Changed
- Added confined adapter proposal settings for canonical sources, writer permissions, changesets,
templates, previews, and declared review output.
- Added a project-owned proposal-validation hook while retaining generic hash, permission, graph,
conflict, atomic-storage, diff, and preview enforcement in the core.
- Moved generic Markdown and TOML source-layout validation behind the generic project owner so an
adapter can enforce its own canonical format without weakening graph validation.
- Added an explicit full-surface server constructor for one configured project service and one
startup-bound writer.
- Marked base, content, source, adapter-source, and index conflicts as stale tool results.
- Released the proposal-enabled adapter boundary as DocForge 0.6.0.
### Verification
- OpenClaw was bound to the AssetForge-only session and update-only permission.
- One real proposal updated an existing AssetForge chapter, validated, produced a structured diff,
and rendered an isolated escaped preview before manual integration.
- The Worldforge canonical builder and manual index rebuilt after review. The original changeset
then failed with `base_conflict` against the new canonical source hash.
- Live tests rejected create, metadata, root-manifest, relationship, and cross-family access;
rejected an overlapping changeset; escaped raw HTML; preserved canonical bytes; and rejected
stale canonical sources.
- The complete DocForge and Worldforge manual suites, shadow proof, generated-output check, format,
and lint passed.
### Limits
- DFG-8 permits content-only updates to existing AssetForge chapters. Create, move, delete,
metadata, relationship, and manifest changes remain closed.
- DocForge does not apply canonical changes, run the Worldforge builder, use Git, deploy, or publish.
- A developer must review and manually integrate accepted prose.
### Next gate
DFG-9: decide from evidence whether a narrowly scoped developer-only application command is
justified or manual integration should remain permanent.
# DocForge2 completed milestones
This file records DocForge2 milestones only. Product behavior is defined by the current contracts
under `docs/`, not by milestone notes.
## 2026-08-02 - DocForge 2.0 release
- Verified the unreleased explicit accepted-writer policy at development commit
`7b21541ab37c35dc9f3d47f631fef0c1884f9525` against released `v1.4.0` commit
`f9a05f868eec6c35e2b74a69467459e6ff2190bf` in isolated locked environments.
- Preserved same-identity application as the default, kept contributor processes without an
application tool, and proved explicit allowlisted application with creator and applier receipts.
- The development repository gate passed 142 contract tests plus 272 subtests, 381 complete tests
plus 422 subtests, three accessibility flows, and all static, build, documentation, and smoke
benchmark checks.
- Compatibility passed 116 tests plus 263 subtests, concurrency and application passed 31 tests
plus 2 subtests, recovery passed 72 tests plus 62 subtests, and adoption, reproducible-artifact,
migration, secret-scan, and anonymous fresh-clone release gates passed.
- All full maintained benchmark tracks passed. The median absolute difference across 66
multi-sample operations was 0.71%, exact adapter graph and Logic evidence matched, and all six
representative task answers remained exact.
- Released the verified candidate as `2.0.0`, giving DocForge2 an unambiguous distribution and
executable identity while preserving its versioned schemas and compatibility contracts.
## 2026-07-31 - live-documentation boundary cleanup
- Replaced the historical application-decision memo with the current canonical-application
contract.
- Removed predecessor rollout chronology and transition-only repository notes from the live
documentation tree. Repository history remains the archive.
- Updated live documentation references and validation requirements.
- Formatting, lint, command-reference, web, and 33-page documentation checks passed. Focused
adapter and changeset tests passed 45 tests plus 2 subtests.
## Milestone 5 - stabilization and first release
- Released DocForge2 `1.4.0` from synchronized `main` and `dev` branches.
- Centralized version identity across package metadata, Python, CLI, MCP, generated client
configuration, and release evidence.
- Proved compatibility, deterministic output, recovery, concurrency, security, representative
task advantage, reproducible artifacts, fresh-wheel adoption, and fresh-clone installation.
- Hardened derived publication and exact-hash canonical application against measured race windows.
- Published the Forgejo release with reproducible wheel, source distribution, and
machine-readable identity evidence.
The final documentation-bearing candidate passed 378 tests plus 422 subtests, all maintained
quality and benchmark gates, three accessibility flows, secret scans, artifact reproduction, and
the anonymous fresh-clone rehearsal.
## Milestone 4 - adapter SDK and product documentation
- Added the stable `docforge.adapter_sdk` authoring surface and complete/incremental conformance
helper.
- Added bounded Python, JavaScript, TypeScript, and C++ reference integrations.
- Added a closed reference-adapter descriptor and fixed read-only reference MCP server.
- Added deterministic client configuration generation for custom project adapters.
- Generated CLI and MCP references from live implementation metadata.
- Added onboarding, authority, descriptor, policy, adapter, rendering, security, recovery, and
agent-integration documentation.
The candidate passed 347 tests plus 402 subtests, strict typing and lint gates, all maintained
benchmarks, package adoption, and three accessibility flows.
## Milestone 3 - independent projections
- Separated manual rendering, portable graph rendering, and live visualization policy.
- Added immutable render plans and packages, fixed detached workers, content-addressed publication,
receipts, recovery, and accessibility gates.
- Added semantic manual fragments with deterministic full-render equivalence.
- Pinned live viewer reads to one validated index generation.
- Replaced recursive cycle planning with an iterative traversal proven at 10,000 nodes.
The complete gate passed 281 tests plus 272 subtests and all maintained scale, response-size,
determinism, no-work, package, and browser checks.
## Milestone 2 - project policy and task-shaped context
- Added capability-aware bootstrap, effective policy, retrieval plans, bounded context capsules,
transition receipts, deterministic client fragments, and read-only diagnostics.
- Kept normal MCP work project-bound and explicit about enabled read, proposal, application,
rendering, and viewer capabilities.
- Added fixed configuration, policy, response-size, counter, and memory contracts.
The complete gate passed 205 tests plus 120 subtests, strict types and lint, package builds, and all
maintained milestone benchmarks.
## Milestone 1 - fast observable core
- Removed repeated whole-project parsing from routine warm reads.
- Added versioned generation receipts and generation-pinned SQLite retrieval.
- Kept complete loading and deep validation as recovery and equivalence oracles.
- Added structured work counters proving that retrieval and status calls do not hide builds,
parsing, rendering, or complete-project hashing.
The maintained 1,000-node benchmark kept exact retrieval, search, filtering, traversal, context,
render status, and visualization status within their recorded latency and response-size gates.
## Milestone 0 - successor foundation
- Established the public DocForge2 repository and its `main` and `dev` workflow.
- Froze package, CLI, MCP, adapter, schema, changeset, rendering, safety, and no-AST compatibility
contracts.
- Added repository-native aggregate, contract, build, and performance gates.
- Recorded cold and warm latency, startup, memory, rendering, response-size, incremental, and
pinned-SQLite baselines.
The aggregate gate passed formatting, Python and web lint, strict types, compilation, public
contract checks, tests, dependency validation, package builds, and the maintained benchmark.

108
benchmarks/README.md Normal file
View file

@ -0,0 +1,108 @@
# Benchmarks
Milestone 0 records measurements before changing compiler, storage, rendering, or response
contracts.
Run the maintained smoke benchmark:
```bash
make benchmark-smoke
```
Run the 1,000-node generic baseline:
```bash
make benchmark
```
Run the Milestone 1 warm-operation counter and latency smoke gate:
```bash
make benchmark-m1-smoke
```
Run the maintained 1,000-node Milestone 1 benchmark:
```bash
make benchmark-m1
```
Run the Milestone 2 agent-workflow smoke and full gates:
```bash
make benchmark-m2-smoke
make benchmark-m2
```
Run the Milestone 3 independent-projection smoke and maintained full gates:
```bash
make benchmark-m3-smoke
make benchmark-m3
make benchmark-m3-full
```
Run the Milestone 4 adapter SDK smoke, maintained full, and fresh-wheel adoption gates:
```bash
make benchmark-m4-smoke
make benchmark-m4-full
make adoption-m4
```
The benchmark creates canonical sources, derived state, changesets, rendered output, and caches
only in a disposable temporary directory. It does not read another project, self-host DocForge, or
mutate repository content.
`milestone0-2026-07-29.json` is the clean-tree baseline captured from commit
`fd4759096e90edb13a745621aae4872f23079357`. It uses compact sorted JSON for response sizes and
`time.perf_counter_ns()` for durations. The file is data, not a performance threshold. Later work
must explain fixture or environment changes before comparing results.
`milestone1-2026-07-29.json` is the clean-tree fast-core baseline captured from commit
`6253c45a5eca01efa8c73ea3dfe4d85c55878ada`. Unlike the historical baseline, the Milestone 1
harness enforces operation-specific p95 ceilings and fixed zero-work counter invariants. Its
human-readable interpretation is in
[`docs/MILESTONE_1_BASELINE.md`](../docs/MILESTONE_1_BASELINE.md).
`milestone2-2026-07-29.json` is the clean-tree agent-retrieval and client-integration baseline
captured from commit `fb0df5e4a1c591c2a84788fd4814d98550f11863`. It gates every warmup and
sample, reconstructs complete task-context and generation-diff collections across bounded pages,
records whether diagnostics were dropped for response budget, checks all hidden-work counters,
and measures isolated-process peak RSS. Its interpretation is in
[`docs/MILESTONE_2_BASELINE.md`](../docs/MILESTONE_2_BASELINE.md).
`milestone3-2026-07-29.json` is the clean-tree independent-projection baseline captured from commit
`f5dccb5e1c312121f1af63780162f593d9363b98`. It measures versioned manual and graph planning,
in-process and detached rendering, production cold/warm/forced-full behavior, fragment cache
sweeps, add/change/delete/reorder equivalence, portable publication, receipt-only status, traced
memory, detached worker peak memory, and response size. Its interpretation is in
[`docs/MILESTONE_3_BASELINE.md`](../docs/MILESTONE_3_BASELINE.md).
`milestone4-2026-07-29.json` is the clean-tree adapter SDK and reference-Python baseline captured
from commit `95271dcf2e48045b9d3aed9b9ea09c7fc155692c`. It measures cold and warm
incremental builds, complete/incremental graph-plus-Logic equivalence, corrupt extraction-cache
recovery, corrupt-index recovery, parser and extraction counters, traced and process memory, and
bounded response size. Its interpretation is in
[`docs/MILESTONE_4_BASELINE.md`](../docs/MILESTONE_4_BASELINE.md).
The generic fixtures expose whole-source and projection scaling. They do not replace incremental
adapter equivalence tests. Milestone 3's portable fixture contains 1,000 Nodes/Flow/Web nodes and
999 edges; portable version 1 deliberately excludes Logic.
The Milestone 1 harness treats wall time and structured work counters as separate gates. Warm
operations fail if they load a complete project, parse source files, reconstruct an adapter
projection, extract adapter sources, build an index, prepare a render, construct rendered output,
or hash complete rendered output. Its latency ceilings are the Milestone 1 targets, not claims
about all hardware.
Every warmup and measured invocation is validated. The recorded counter ranges also require one
index synchronization for the synchronization operation, no hidden synchronization for reads and
status, one index check for each retrieval snapshot, and exactly one manager request for viewer
status. The 1,000-node run records bounded semantic summaries for exact errors, search, filtering,
backlinks, both traversal directions, paged context, render receipt states, and visualization
freshness. The reported p95 uses the nearest-rank method; with ten samples it is the maximum.
`process_peak_rss_kib` is the cumulative main-process `RUSAGE_SELF` high-water mark and excludes the
detached viewer worker. It is diagnostic and not operation-local. Milestone 3 memory gates use
per-operation `tracemalloc` peaks and detached worker receipt peaks instead. Milestone 4 uses
per-operation `tracemalloc` peaks and a cumulative benchmark-process high-water gate.

View file

@ -0,0 +1,282 @@
{
"benchmark": "docforge2_milestone0",
"environment": {
"implementation": "CPython",
"machine": "x86_64",
"platform": "Linux-7.1.3-200.nobara.fc44.x86_64-x86_64-with-glibc2.43",
"python": "3.14.6"
},
"fixture": {
"context_budget_tokens": 32000,
"edge_count": 999,
"kind": "synthetic_generic",
"node_count": 1000,
"source_file_count": 1000,
"traversal_depth": 8
},
"known_gaps": [
"Generic warm reads still parse canonical source files.",
"Compiler stages are not separately instrumented.",
"Scaled incremental extraction is not measured by this generic fixture.",
"Manual planning is not separated from rendering.",
"Portable graph planning and rendering do not exist in Milestone 0.",
"Per-operation peak RSS requires an external process harness."
],
"method": {
"clock": "time.perf_counter_ns",
"cold_samples": 3,
"memory": "resource.getrusage(RUSAGE_SELF).ru_maxrss",
"response_size": "UTF-8 bytes of compact sorted JSON",
"samples": 10
},
"operations": {
"changeset_diff": {
"max_ms": 151.704,
"median_ms": 148.23,
"min_ms": 145.184,
"p95_ms": 151.704,
"response_bytes": 1575,
"samples": 10
},
"changeset_register": {
"max_ms": 215.227,
"median_ms": 212.281,
"min_ms": 210.251,
"p95_ms": 215.227,
"response_bytes": 1013,
"samples": 10
},
"changeset_validate": {
"max_ms": 162.789,
"median_ms": 148.522,
"min_ms": 145.624,
"p95_ms": 162.789,
"response_bytes": 969,
"samples": 10
},
"cli_exact_startup": {
"max_ms": 370.763,
"median_ms": 369.681,
"min_ms": 368.637,
"p95_ms": 370.763,
"response_bytes": 814,
"samples": 3
},
"cli_info_startup": {
"max_ms": 215.402,
"median_ms": 214.045,
"min_ms": 212.262,
"p95_ms": 215.402,
"response_bytes": 407,
"samples": 3
},
"cold_synchronize": {
"max_ms": 696.071,
"median_ms": 692.605,
"min_ms": 689.365,
"p95_ms": 696.071,
"response_bytes": 870,
"samples": 3
},
"context_32k": {
"max_ms": 453.055,
"median_ms": 436.897,
"min_ms": 427.958,
"p95_ms": 453.055,
"response_bytes": 258034,
"samples": 10
},
"dependencies_depth_8": {
"max_ms": 291.129,
"median_ms": 287.791,
"min_ms": 286.129,
"p95_ms": 291.129,
"response_bytes": 1650,
"samples": 10
},
"exact_hash_apply_and_refresh": {
"max_ms": 1253.231,
"median_ms": 1253.231,
"min_ms": 1253.231,
"p95_ms": 1253.231,
"response_bytes": 3509,
"samples": 1
},
"exact_node": {
"max_ms": 304.226,
"median_ms": 286.306,
"min_ms": 283.013,
"p95_ms": 304.226,
"response_bytes": 696,
"samples": 10
},
"full_index_build": {
"max_ms": 289.531,
"median_ms": 277.172,
"min_ms": 276.361,
"p95_ms": 289.531,
"response_bytes": 661,
"samples": 3
},
"full_index_check": {
"max_ms": 157.888,
"median_ms": 148.778,
"min_ms": 143.19,
"p95_ms": 157.888,
"response_bytes": 661,
"samples": 10
},
"impact_depth_8": {
"max_ms": 290.636,
"median_ms": 288.28,
"min_ms": 285.26,
"p95_ms": 290.636,
"response_bytes": 1650,
"samples": 10
},
"manual_render": {
"max_ms": 283.38,
"median_ms": 280.007,
"min_ms": 279.406,
"p95_ms": 283.38,
"response_bytes": 757,
"samples": 3
},
"manual_render_status": {
"max_ms": 155.833,
"median_ms": 150.591,
"min_ms": 148.979,
"p95_ms": 155.833,
"response_bytes": 760,
"samples": 10
},
"mcp_bootstrap": {
"max_ms": 277.901,
"median_ms": 273.41,
"min_ms": 271.248,
"p95_ms": 277.901,
"response_bytes": 1720,
"samples": 10
},
"mcp_context_32k": {
"max_ms": 447.168,
"median_ms": 434.853,
"min_ms": 430.848,
"p95_ms": 447.168,
"response_bytes": 258224,
"samples": 10
},
"mcp_exact_node": {
"max_ms": 298.913,
"median_ms": 287.094,
"min_ms": 284.413,
"p95_ms": 298.913,
"response_bytes": 886,
"samples": 10
},
"mcp_import_and_help": {
"max_ms": 300.449,
"median_ms": 299.278,
"min_ms": 297.873,
"p95_ms": 300.449,
"response_bytes": 536,
"samples": 3
},
"mcp_render_status": {
"max_ms": 164.194,
"median_ms": 150.758,
"min_ms": 147.522,
"p95_ms": 164.194,
"response_bytes": 950,
"samples": 10
},
"mcp_search_limit_20": {
"max_ms": 307.09,
"median_ms": 289.737,
"min_ms": 284.139,
"p95_ms": 307.09,
"response_bytes": 10502,
"samples": 10
},
"project_load": {
"max_ms": 132.255,
"median_ms": 128.65,
"min_ms": 127.489,
"p95_ms": 132.255,
"samples": 10
},
"project_open": {
"max_ms": 0.607,
"median_ms": 0.584,
"min_ms": 0.568,
"p95_ms": 0.607,
"samples": 10
},
"search_limit_20": {
"max_ms": 301.227,
"median_ms": 288.793,
"min_ms": 286.57,
"p95_ms": 301.227,
"response_bytes": 10312,
"samples": 10
},
"viewer_neighborhood_depth_8": {
"max_ms": 0.46,
"median_ms": 0.425,
"min_ms": 0.41,
"p95_ms": 0.46,
"response_bytes": 4891,
"samples": 10
},
"viewer_overview": {
"max_ms": 1.124,
"median_ms": 1.078,
"min_ms": 1.048,
"p95_ms": 1.124,
"response_bytes": 977,
"samples": 10
},
"viewer_search_limit_20": {
"max_ms": 1.389,
"median_ms": 1.363,
"min_ms": 1.344,
"p95_ms": 1.389,
"response_bytes": 10423,
"samples": 10
},
"viewer_snapshot_pin": {
"max_ms": 143.9,
"median_ms": 143.761,
"min_ms": 143.569,
"p95_ms": 143.9,
"samples": 3
},
"viewer_web_depth_8": {
"max_ms": 0.899,
"median_ms": 0.396,
"min_ms": 0.353,
"p95_ms": 0.899,
"response_bytes": 5811,
"samples": 10
},
"warm_no_change_synchronize": {
"max_ms": 145.926,
"median_ms": 142.479,
"min_ms": 140.151,
"p95_ms": 145.926,
"response_bytes": 779,
"samples": 10
}
},
"process_peak_rss_kib": 528228,
"schema_version": 1,
"sizes": {
"context_compact_bytes": 258034,
"manual_artifact_bytes": 583150,
"static_viewer_assets_bytes": 105244
},
"source": {
"dirty": false,
"revision": "fd4759096e90edb13a745621aae4872f23079357"
}
}

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,457 @@
{
"benchmark": "docforge2_milestone2",
"environment": {
"implementation": "CPython",
"machine": "x86_64",
"platform": "Linux-7.1.3-200.nobara.fc44.x86_64-x86_64-with-glibc2.43",
"python": "3.14.6"
},
"fixture": {
"edge_count": 999,
"kind": "synthetic_generic_focus_fan_in",
"max_tool_output_chars": 200000,
"node_count": 1000,
"source_file_count": 1000
},
"isolated_process_peak_rss_kib": 86448,
"method": {
"clock": "time.perf_counter_ns",
"memory": "isolated child-process resource.getrusage(RUSAGE_SELF).ru_maxrss",
"memory_limit_kib": 262144,
"memory_probe_samples": 10,
"percentile": "nearest-rank",
"response_size": "UTF-8 bytes of compact sorted JSON",
"samples": 10,
"warmups": 1,
"zero_work_counters": [
"project_loads",
"source_files_parsed",
"source_bytes_parsed",
"adapter_projection_loads",
"adapter_source_extractions",
"index_synchronizations",
"index_builds",
"render_prepare_calls",
"render_output_bytes_built",
"render_output_bytes_hashed",
"viewer_manager_requests"
]
},
"operations": {
"bootstrap_no_ast": {
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 1, "minimum": 1},
"index_synchronizations": {"maximum": 1, "minimum": 1},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 1, "minimum": 1},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"max_ms": 9.379,
"maximum_response_bytes": 8796,
"median_ms": 9.153,
"min_ms": 9.026,
"p95_limit_ms": 100,
"p95_ms": 9.379,
"response_bytes": 8795,
"response_limit_bytes": 32768,
"samples": 10,
"validated_invocations": 11
},
"bootstrap_read": {
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 1, "minimum": 1},
"index_synchronizations": {"maximum": 1, "minimum": 1},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 1, "minimum": 1},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"max_ms": 9.884,
"maximum_response_bytes": 7495,
"median_ms": 9.406,
"min_ms": 9.053,
"p95_limit_ms": 100,
"p95_ms": 9.884,
"response_bytes": 7491,
"response_limit_bytes": 32768,
"samples": 10,
"validated_invocations": 11
},
"configuration_preview": {
"claude": {
"artifact_format": "claude-json-fragment-v1",
"configuration_hash": "4dbb4ed0f38264fdba350de8904cc898493d620194b5105b7889c54bd5913c9c",
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 0, "minimum": 0},
"index_synchronizations": {"maximum": 0, "minimum": 0},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 0, "minimum": 0},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"max_ms": 364.365,
"maximum_response_bytes": 2748,
"median_ms": 314.326,
"min_ms": 314.278,
"p95_limit_ms": 500,
"p95_ms": 364.365,
"response_bytes": 2748,
"response_limit_bytes": 32768,
"samples": 10,
"validated_invocations": 11
},
"codex": {
"artifact_format": "codex-toml-fragment-v1",
"configuration_hash": "3e1dd5191021da1778cd1c4f4658768537775e5e16252e42d4c80e328841145b",
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 0, "minimum": 0},
"index_synchronizations": {"maximum": 0, "minimum": 0},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 0, "minimum": 0},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"max_ms": 364.383,
"maximum_response_bytes": 2627,
"median_ms": 314.365,
"min_ms": 314.248,
"p95_limit_ms": 500,
"p95_ms": 364.383,
"response_bytes": 2627,
"response_limit_bytes": 32768,
"samples": 10,
"validated_invocations": 11
},
"openclaw": {
"artifact_format": "openclaw-json-fragment-v1",
"configuration_hash": "6f269e90a55088c5d517f91761c53a3b90036d62fd80b5b9d094668267257b99",
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 0, "minimum": 0},
"index_synchronizations": {"maximum": 0, "minimum": 0},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 0, "minimum": 0},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"max_ms": 364.532,
"maximum_response_bytes": 2869,
"median_ms": 314.401,
"min_ms": 314.251,
"p95_limit_ms": 500,
"p95_ms": 364.532,
"response_bytes": 2869,
"response_limit_bytes": 32768,
"samples": 10,
"validated_invocations": 11
}
},
"doctor": {
"claude": {
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 0, "minimum": 0},
"index_synchronizations": {"maximum": 0, "minimum": 0},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 0, "minimum": 0},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"doctor_state": "degraded",
"max_ms": 0.446,
"maximum_response_bytes": 3669,
"median_ms": 0.364,
"min_ms": 0.352,
"p95_limit_ms": 100,
"p95_ms": 0.446,
"response_bytes": 3669,
"response_limit_bytes": 32768,
"samples": 10,
"summary": {"failed": 0, "passed": 11, "skipped": 1, "warning": 2},
"validated_invocations": 11
},
"codex": {
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 0, "minimum": 0},
"index_synchronizations": {"maximum": 0, "minimum": 0},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 0, "minimum": 0},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"doctor_state": "healthy",
"max_ms": 0.556,
"maximum_response_bytes": 3595,
"median_ms": 0.421,
"min_ms": 0.404,
"p95_limit_ms": 100,
"p95_ms": 0.556,
"response_bytes": 3595,
"response_limit_bytes": 32768,
"samples": 10,
"summary": {"failed": 0, "passed": 13, "skipped": 1, "warning": 0},
"validated_invocations": 11
},
"openclaw": {
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 0, "minimum": 0},
"index_synchronizations": {"maximum": 0, "minimum": 0},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 0, "minimum": 0},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"doctor_state": "healthy",
"max_ms": 0.484,
"maximum_response_bytes": 3602,
"median_ms": 0.384,
"min_ms": 0.353,
"p95_limit_ms": 100,
"p95_ms": 0.484,
"response_bytes": 3602,
"response_limit_bytes": 32768,
"samples": 10,
"summary": {"failed": 0, "passed": 13, "skipped": 1, "warning": 0},
"validated_invocations": 11
}
},
"generation_diff_complete": {
"max_ms": 425.315,
"maximum_response_bytes": 984,
"median_ms": 418.607,
"min_ms": 410.974,
"p95_limit_ms": 500,
"p95_ms": 425.315,
"response_bytes": 983,
"response_limit_bytes": 32768,
"result_summary": {
"aggregate_page_bytes": 664715,
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 0, "minimum": 0},
"index_synchronizations": {"maximum": 0, "minimum": 0},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 2, "minimum": 2},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"elapsed_ms": 419.04,
"item_count": 1000,
"maximum_cursor_bytes": 448,
"maximum_page_bytes": 66516,
"ordered_item_hash": "1ac48cc72532809ef5d3e949756e536eec819f348eaf06338c9b39b14e63b2c7",
"page_count": 10,
"receipt_hash": "6913c962972d8255f56966e5cfab5ac8293e41bdfb5f39ef91a4f33a3b092f88",
"status": "ok"
},
"samples": 10,
"validated_invocations": 11
},
"generation_diff_diagnostic_page": {
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 0, "minimum": 0},
"index_synchronizations": {"maximum": 0, "minimum": 0},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 2, "minimum": 2},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"max_ms": 41.65,
"maximum_response_bytes": 66516,
"median_ms": 40.961,
"min_ms": 40.239,
"p95_limit_ms": 100,
"p95_ms": 41.65,
"response_bytes": 66516,
"response_limit_bytes": 200000,
"samples": 10,
"validated_invocations": 11
},
"generation_diff_maximum_page": {
"diagnostics_dropped_for_budget": true,
"max_ms": 59.184,
"maximum_response_bytes": 199566,
"median_ms": 55.565,
"min_ms": 54.68,
"p95_limit_ms": 100,
"p95_ms": 59.184,
"response_bytes": 199566,
"response_limit_bytes": 200000,
"samples": 10,
"validated_invocations": 11
},
"task_context_complete": {
"max_ms": 721.847,
"maximum_response_bytes": 1325,
"median_ms": 703.561,
"min_ms": 688.975,
"p95_limit_ms": 2500,
"p95_ms": 721.847,
"response_bytes": 1324,
"response_limit_bytes": 32768,
"result_summary": {
"aggregate_page_bytes": 348845,
"capsule_hash": "20663ed685a255f7cb8a0e8d78262bf7bf26863d728f0159e0b29000f6a52b0a",
"collection_hash": "9dbc46eb5b1ac8c8340bb149205d3599fffaf67e3b363de2d035475da274857c",
"collection_hash_reconstructed": true,
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 1, "minimum": 1},
"index_synchronizations": {"maximum": 0, "minimum": 0},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 2, "minimum": 2},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"elapsed_ms": 719.55,
"evidence_count": 108,
"item_count": 1000,
"maximum_cursor_bytes": 1066,
"maximum_page_bytes": 151172,
"omission_count": 892,
"ordered_candidate_hash": "bdeb3a8f4018000f72a5ff1891aa800b6edf98070b9814f443c6bac4e52c38f3",
"ordered_evidence_hash": "ec07bac7f528f5a3afbc083ea5ad60541335c8789d42980fe7bae0476e0a5331",
"page_count": 11,
"plan_hash": "84dbe60267e5f8359adcf30d4d395f1ac5d5bea957beaa3888d2938120860a9c",
"status": "ok"
},
"samples": 10,
"validated_invocations": 11
},
"task_context_diagnostic_page": {
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 1, "minimum": 1},
"index_synchronizations": {"maximum": 0, "minimum": 0},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 2, "minimum": 2},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"max_ms": 93.115,
"maximum_response_bytes": 7190,
"median_ms": 73.326,
"min_ms": 72.264,
"p95_limit_ms": 500,
"p95_ms": 93.115,
"response_bytes": 7190,
"response_limit_bytes": 200000,
"samples": 10,
"validated_invocations": 11
},
"task_context_maximum_page": {
"counter_ranges": {
"adapter_projection_loads": {"maximum": 0, "minimum": 0},
"adapter_source_extractions": {"maximum": 0, "minimum": 0},
"index_builds": {"maximum": 0, "minimum": 0},
"index_checks": {"maximum": 1, "minimum": 1},
"index_synchronizations": {"maximum": 0, "minimum": 0},
"project_loads": {"maximum": 0, "minimum": 0},
"render_output_bytes_built": {"maximum": 0, "minimum": 0},
"render_output_bytes_hashed": {"maximum": 0, "minimum": 0},
"render_prepare_calls": {"maximum": 0, "minimum": 0},
"source_bytes_parsed": {"maximum": 0, "minimum": 0},
"source_files_parsed": {"maximum": 0, "minimum": 0},
"source_generation_checks": {"maximum": 2, "minimum": 2},
"viewer_manager_requests": {"maximum": 0, "minimum": 0}
},
"diagnostics_dropped_for_budget": false,
"max_ms": 87.372,
"maximum_response_bytes": 7192,
"median_ms": 84.933,
"min_ms": 83.652,
"p95_limit_ms": 500,
"p95_ms": 87.372,
"response_bytes": 7192,
"response_limit_bytes": 200000,
"samples": 10,
"validated_invocations": 11
}
},
"process_peak_rss_kib": 86072,
"schema_version": 1,
"source": {
"dirty": false,
"revision": "fb0df5e4a1c591c2a84788fd4814d98550f11863"
}
}

View file

@ -0,0 +1,442 @@
{
"benchmark": "docforge2_milestone3",
"environment": {
"implementation": "CPython",
"machine": "x86_64",
"platform": "Linux-7.1.3-200.nobara.fc44.x86_64-x86_64-with-glibc2.43",
"python": "3.14.6"
},
"equivalence": {
"manual_full_vs_fragment_assisted": true,
"manual_in_process_vs_detached": true,
"manual_production_cold_warm_full": true,
"manual_production_variants": {
"add": true,
"change": true,
"delete": true,
"reorder": true
},
"portable_graph_in_process_vs_detached": true
},
"fixture": {
"edge_count": 999,
"full_coverage": true,
"kind": "synthetic_generic_projection",
"manual_page_count": 1000,
"node_count": 1000,
"portable_graph_edge_count": 999,
"portable_graph_node_count": 1000
},
"memory": {
"manual_production_worker_peak_bytes": 104771584,
"manual_worker_peak_bytes": 88580096,
"portable_graph_worker_peak_bytes": 89583616,
"process_peak_rss_kib": 102692
},
"method": {
"clock": "time.perf_counter_ns",
"detached_peak_memory": "worker receipt resource peak RSS",
"determinism": "stable semantic summaries must match across samples; report JSON uses sorted keys",
"full_mode_node_requirement": 1000,
"in_process_peak_memory": "tracemalloc per measured invocation",
"maximum_worker_artifact_bytes": 20000000,
"process_peak_memory": "resource.getrusage(RUSAGE_SELF).ru_maxrss",
"response_size": "UTF-8 bytes of canonical compact sorted JSON",
"samples": 10
},
"mode": "full",
"operations": {
"fragment_assisted_equivalence": {
"max_ms": 666.119,
"maximum_response_bytes": 664,
"maximum_traced_peak_bytes": 5332792,
"median_ms": 638.999,
"min_ms": 633.157,
"p95_limit_ms": 15000,
"p95_ms": 666.119,
"response_limit_bytes": 128000,
"samples": 10,
"stable_result": {
"artifact_bytes": 583149,
"artifact_sha256": "278ff8cd6b092e7d8114527dba88051917aa827b22d4aed61e57acf004e5870c",
"package_bytes": 2364302,
"package_id": "ecc21e58106c420a9f78ffbba997778ca98b61586ecfdb23a15b08f431a52c42",
"plan_bytes": 1006393,
"plan_id": "47008f5177230e3ad5abbfee8bccaff0498963106cb33ed7cdd6f8bbc3c495de"
},
"traced_peak_limit_bytes": 268435456
},
"fragment_cache_hit_sweep": {
"max_ms": 514.324,
"maximum_response_bytes": 145,
"maximum_traced_peak_bytes": 6941174,
"median_ms": 498.911,
"min_ms": 495.447,
"p95_limit_ms": 5000,
"p95_ms": 514.324,
"response_limit_bytes": 32768,
"samples": 10,
"stable_result": {
"aggregate_content_bytes": 516921,
"fragment_count": 1000,
"ordered_record_hash": "99a95d575504238233a6367c884fa1d50231855f2f2ffe3dae90fa18c4136f72"
},
"traced_peak_limit_bytes": 268435456
},
"fragment_cache_miss_sweep": {
"max_ms": 106.888,
"maximum_response_bytes": 145,
"maximum_traced_peak_bytes": 146245,
"median_ms": 106.203,
"min_ms": 105.383,
"p95_limit_ms": 5000,
"p95_ms": 106.888,
"response_limit_bytes": 32768,
"samples": 10,
"stable_result": {
"aggregate_content_bytes": 516921,
"fragment_count": 1000,
"ordered_record_hash": "99a95d575504238233a6367c884fa1d50231855f2f2ffe3dae90fa18c4136f72"
},
"traced_peak_limit_bytes": 268435456
},
"fragment_cache_put_sweep": {
"max_ms": 539.212,
"maximum_response_bytes": 145,
"maximum_traced_peak_bytes": 509548,
"median_ms": 539.212,
"min_ms": 539.212,
"p95_limit_ms": 10000,
"p95_ms": 539.212,
"response_limit_bytes": 32768,
"samples": 1,
"stable_result": {
"aggregate_content_bytes": 516921,
"fragment_count": 1000,
"ordered_record_hash": "99a95d575504238233a6367c884fa1d50231855f2f2ffe3dae90fa18c4136f72"
},
"traced_peak_limit_bytes": 268435456
},
"manual_detached_worker": {
"child_peak_limit_bytes": 268435456,
"max_ms": 599.394,
"maximum_child_peak_bytes": 88580096,
"maximum_response_bytes": 667,
"maximum_traced_peak_bytes": 28203708,
"median_ms": 591.317,
"min_ms": 585.123,
"p95_limit_ms": 20000,
"p95_ms": 599.394,
"response_limit_bytes": 128000,
"samples": 10,
"stable_result": {
"artifact_bytes": 583149,
"artifact_sha256": "278ff8cd6b092e7d8114527dba88051917aa827b22d4aed61e57acf004e5870c",
"package_bytes": 1007297,
"package_id": "5fec20ace40b6db7e89083704af371245a5b5b73247f71edb6eab0cb2bd5e2c1",
"plan_bytes": 1006393,
"plan_id": "47008f5177230e3ad5abbfee8bccaff0498963106cb33ed7cdd6f8bbc3c495de"
},
"traced_peak_limit_bytes": 268435456
},
"manual_forced_full": {
"child_peak_limit_bytes": 268435456,
"max_ms": 978.87,
"maximum_child_peak_bytes": 104771584,
"maximum_response_bytes": 668,
"maximum_traced_peak_bytes": 30425714,
"median_ms": 944.135,
"min_ms": 934.773,
"p95_limit_ms": 20000,
"p95_ms": 978.87,
"response_limit_bytes": 128000,
"samples": 10,
"stable_result": {
"output_bytes": 583149,
"output_sha256": "278ff8cd6b092e7d8114527dba88051917aa827b22d4aed61e57acf004e5870c",
"render_identity": "73f9a175afd7bfea5af70d4c7df902c982f042c413d533817dc2bade9e54d4b0"
},
"traced_peak_limit_bytes": 268435456
},
"manual_full_render": {
"max_ms": 810.49,
"maximum_response_bytes": 664,
"maximum_traced_peak_bytes": 4412754,
"median_ms": 801.948,
"min_ms": 773.7,
"p95_limit_ms": 15000,
"p95_ms": 810.49,
"response_limit_bytes": 128000,
"samples": 10,
"stable_result": {
"artifact_bytes": 583149,
"artifact_sha256": "278ff8cd6b092e7d8114527dba88051917aa827b22d4aed61e57acf004e5870c",
"package_bytes": 1007297,
"package_id": "5fec20ace40b6db7e89083704af371245a5b5b73247f71edb6eab0cb2bd5e2c1",
"plan_bytes": 1006393,
"plan_id": "47008f5177230e3ad5abbfee8bccaff0498963106cb33ed7cdd6f8bbc3c495de"
},
"traced_peak_limit_bytes": 268435456
},
"manual_incremental_cold": {
"child_peak_limit_bytes": 268435456,
"max_ms": 2827.152,
"maximum_child_peak_bytes": 99454976,
"maximum_response_bytes": 668,
"maximum_traced_peak_bytes": 35160716,
"median_ms": 2827.152,
"min_ms": 2827.152,
"p95_limit_ms": 20000,
"p95_ms": 2827.152,
"response_limit_bytes": 128000,
"samples": 1,
"stable_result": {
"output_bytes": 583149,
"output_sha256": "278ff8cd6b092e7d8114527dba88051917aa827b22d4aed61e57acf004e5870c",
"render_identity": "73f9a175afd7bfea5af70d4c7df902c982f042c413d533817dc2bade9e54d4b0"
},
"traced_peak_limit_bytes": 268435456
},
"manual_incremental_warm": {
"child_peak_limit_bytes": 268435456,
"max_ms": 2206.54,
"maximum_child_peak_bytes": 104767488,
"maximum_response_bytes": 669,
"maximum_traced_peak_bytes": 34676043,
"median_ms": 2143.388,
"min_ms": 2125.466,
"p95_limit_ms": 20000,
"p95_ms": 2206.54,
"response_limit_bytes": 128000,
"samples": 10,
"stable_result": {
"output_bytes": 583149,
"output_sha256": "278ff8cd6b092e7d8114527dba88051917aa827b22d4aed61e57acf004e5870c",
"render_identity": "73f9a175afd7bfea5af70d4c7df902c982f042c413d533817dc2bade9e54d4b0"
},
"traced_peak_limit_bytes": 268435456
},
"manual_status_no_work": {
"max_ms": 111.381,
"maximum_response_bytes": 1542,
"maximum_traced_peak_bytes": 1792025,
"median_ms": 62.881,
"min_ms": 60.111,
"p95_limit_ms": 500,
"p95_ms": 111.381,
"response_limit_bytes": 256000,
"samples": 10,
"stable_result": {
"counters": {
"adapter_projection_loads": 0,
"adapter_source_extractions": 0,
"index_builds": 0,
"index_checks": 0,
"index_synchronizations": 0,
"project_loads": 0,
"render_output_bytes_built": 0,
"render_output_bytes_hashed": 0,
"render_prepare_calls": 0,
"source_bytes_parsed": 0,
"source_files_parsed": 0,
"source_generation_checks": 2,
"viewer_manager_requests": 0
},
"response": {
"adapter": "generic",
"configured": true,
"outputs": [
{
"actual_output_hash": "278ff8cd6b092e7d8114527dba88051917aa827b22d4aed61e57acf004e5870c",
"expected_output_hash": "278ff8cd6b092e7d8114527dba88051917aa827b22d4aed61e57acf004e5870c",
"path": ".docforge/rendered/manual.html",
"projection_receipt": {
"artifacts": [
{
"artifact_id": "manual.html",
"bytes": 583149,
"media_type": "text/html; charset=utf-8",
"sha256": "278ff8cd6b092e7d8114527dba88051917aa827b22d4aed61e57acf004e5870c"
}
],
"contract": "docforge.projection-receipt",
"diagnostics": {
"warnings": []
},
"kind": "manual",
"package_id": "b30c65390570537a16ce03afe6a992599c6ca0c4ef5ea2f06d14b33e248bc9aa",
"peak_memory_bytes": 104771584,
"plan_id": "47008f5177230e3ad5abbfee8bccaff0498963106cb33ed7cdd6f8bbc3c495de",
"receipt_id": "ead4c25b34ada8df674903a592e25dd6bbf9817df2b04e5ebc66cde9e41c1ed1",
"renderer": {
"renderer_id": "generic_html",
"renderer_version": "1+markdown-it-py-4.2.0"
},
"schema_version": 1,
"timing": {
"elapsed_ns": 114544865
}
},
"reason": null,
"receipt_schema_version": 1,
"render_identity": "73f9a175afd7bfea5af70d4c7df902c982f042c413d533817dc2bade9e54d4b0",
"renderer": "generic_html",
"renderer_version": "1+markdown-it-py-4.2.0",
"state": "current",
"template_hash": "ae0ebb3eadeb530e9d033c8fbd21321d406e29d04c0ae7d07a9a0f929f0ffa9f",
"verification": "receipt",
"view_id": "manual"
}
],
"project_id": "synthetic-1000",
"project_root_fingerprint": "0747acfd975703c0",
"revision": "unversioned",
"source_hash": "a314da4ffa8fcf291ef7a7b0fc87737caeaa89fa3c558ea442afeb0e5ae49c2d",
"state": "current",
"status": "ok",
"verification": "receipt"
}
},
"traced_peak_limit_bytes": 268435456
},
"portable_graph_detached_worker": {
"child_peak_limit_bytes": 268435456,
"max_ms": 268.428,
"maximum_child_peak_bytes": 89583616,
"maximum_response_bytes": 660,
"maximum_traced_peak_bytes": 27587227,
"median_ms": 265.448,
"min_ms": 263.761,
"p95_limit_ms": 20000,
"p95_ms": 268.428,
"response_limit_bytes": 128000,
"samples": 10,
"stable_result": {
"artifact_bytes": 718383,
"artifact_sha256": "91dd5bb31679225e5af40a1f94a9e75f72921e36416e3d8b00c45f46a37a3a30",
"package_bytes": 398715,
"package_id": "72f96d7ab92c1d6b59f4f32539558232ecf35deda099fc4887c77890da1d4309",
"plan_bytes": 398158,
"plan_id": "6dac35937d855f72c6670f87aace3c0cb1dbff67bb842c55370429584064f76b"
},
"traced_peak_limit_bytes": 268435456
},
"portable_graph_full_render": {
"max_ms": 323.69,
"maximum_response_bytes": 657,
"maximum_traced_peak_bytes": 3490976,
"median_ms": 303.736,
"min_ms": 301.215,
"p95_limit_ms": 10000,
"p95_ms": 323.69,
"response_limit_bytes": 128000,
"samples": 10,
"stable_result": {
"artifact_bytes": 718383,
"artifact_sha256": "91dd5bb31679225e5af40a1f94a9e75f72921e36416e3d8b00c45f46a37a3a30",
"package_bytes": 398715,
"package_id": "72f96d7ab92c1d6b59f4f32539558232ecf35deda099fc4887c77890da1d4309",
"plan_bytes": 398158,
"plan_id": "6dac35937d855f72c6670f87aace3c0cb1dbff67bb842c55370429584064f76b"
},
"traced_peak_limit_bytes": 268435456
},
"portable_graph_status_no_work": {
"max_ms": 59.331,
"maximum_response_bytes": 879,
"maximum_traced_peak_bytes": 685012,
"median_ms": 58.467,
"min_ms": 58.052,
"p95_limit_ms": 500,
"p95_ms": 59.331,
"response_limit_bytes": 256000,
"samples": 10,
"stable_result": {
"counters": {
"adapter_projection_loads": 0,
"adapter_source_extractions": 0,
"index_builds": 0,
"index_checks": 0,
"index_synchronizations": 0,
"project_loads": 0,
"render_output_bytes_built": 0,
"render_output_bytes_hashed": 0,
"render_prepare_calls": 0,
"source_bytes_parsed": 0,
"source_files_parsed": 0,
"source_generation_checks": 2,
"viewer_manager_requests": 0
},
"response": {
"adapter": "generic",
"configured": true,
"outputs": [
{
"artifact": {
"artifact_id": "portable-graph.html",
"bytes": 718383,
"media_type": "text/html; charset=utf-8",
"sha256": "91dd5bb31679225e5af40a1f94a9e75f72921e36416e3d8b00c45f46a37a3a30"
},
"package_id": "72f96d7ab92c1d6b59f4f32539558232ecf35deda099fc4887c77890da1d4309",
"path": ".docforge/portable-graph/architecture.html",
"plan_id": "6dac35937d855f72c6670f87aace3c0cb1dbff67bb842c55370429584064f76b",
"publication_id": "2620ce106593d5499931b145a42fc1eccd31f5e356b19bf1d76687f8d45bcad2",
"reason": null,
"renderer": "portable_graph_html",
"renderer_version": "1",
"state": "current",
"verification": "manifest",
"view_id": "architecture"
}
],
"project_id": "synthetic-1000",
"project_root_fingerprint": "0747acfd975703c0",
"revision": "unversioned",
"source_hash": "a314da4ffa8fcf291ef7a7b0fc87737caeaa89fa3c558ea442afeb0e5ae49c2d",
"state": "current",
"status": "ok"
}
},
"traced_peak_limit_bytes": 268435456
}
},
"schema_version": 1,
"sizes": {
"fragment_assisted_manual": {
"aggregate_fragment_content_bytes": 516921,
"artifact_bytes": 583149,
"artifact_sha256": "278ff8cd6b092e7d8114527dba88051917aa827b22d4aed61e57acf004e5870c",
"fragment_count": 1000,
"package_bytes": 2364302,
"package_id": "ecc21e58106c420a9f78ffbba997778ca98b61586ecfdb23a15b08f431a52c42",
"plan_bytes": 1006393,
"plan_id": "47008f5177230e3ad5abbfee8bccaff0498963106cb33ed7cdd6f8bbc3c495de",
"receipt_bytes": 664
},
"manual": {
"artifact_bytes": 583149,
"artifact_sha256": "278ff8cd6b092e7d8114527dba88051917aa827b22d4aed61e57acf004e5870c",
"package_bytes": 1007297,
"package_id": "5fec20ace40b6db7e89083704af371245a5b5b73247f71edb6eab0cb2bd5e2c1",
"plan_bytes": 1006393,
"plan_id": "47008f5177230e3ad5abbfee8bccaff0498963106cb33ed7cdd6f8bbc3c495de",
"receipt_bytes": 664
},
"manual_status_response_bytes": 1542,
"portable_graph": {
"artifact_bytes": 718383,
"artifact_sha256": "91dd5bb31679225e5af40a1f94a9e75f72921e36416e3d8b00c45f46a37a3a30",
"package_bytes": 398715,
"package_id": "72f96d7ab92c1d6b59f4f32539558232ecf35deda099fc4887c77890da1d4309",
"plan_bytes": 398158,
"plan_id": "6dac35937d855f72c6670f87aace3c0cb1dbff67bb842c55370429584064f76b",
"receipt_bytes": 657
},
"portable_graph_status_response_bytes": 879
},
"source": {
"dirty": false,
"revision": "f5dccb5e1c312121f1af63780162f593d9363b98"
}
}

View file

@ -0,0 +1,205 @@
{
"benchmark": "docforge2_milestone4",
"environment": {
"implementation": "CPython",
"machine": "x86_64",
"platform": "Linux-7.1.3-200.nobara.fc44.x86_64-x86_64-with-glibc2.43",
"python": "3.14.6"
},
"evidence": {
"cache_recovery_exact": true,
"complete_assembly_hash": "2888182ab765fbffe3ba873c1613345640e8e6d89be74ddfca7452c0a5056345",
"edge_count": 1001,
"edge_hash": "00cc90998b6783afc8c9d1fd900409e5e3ba352868c4ba30e9bb2cbd59c52d35",
"full_incremental_graph_and_logic_exact": true,
"index_recovery_exact": true,
"logic_edge_count": 2338,
"logic_hash": "c86a3ae74777c2cec3a82c83e6e5bcca0196772ccea63cb13340fa9141593b0c",
"logic_node_count": 2338,
"logic_projection_count": 334,
"node_count": 1002,
"node_hash": "f7705dedf8dd388857a20d11f459dc74de797bcd12d3ed36e7f3aa75d67c328f",
"source_count": 334,
"source_hash": "30ef23aa41061de2d4a7c995fe109d7a41518d9ee5493d805dd256581b47dae2",
"warm_zero_ast_parse": true,
"warm_zero_source_extraction": true
},
"evidence_sha256": "4a8db461a311b5df4abd5aa00063e9a347d8b9dba19b30e35684f561d5271549",
"fixture": {
"expected_node_count": 1002,
"kind": "synthetic_python_reference_adapter",
"nodes_per_source": 3,
"source_count": 334
},
"memory": {
"process_peak_bytes": 78798848,
"process_peak_limit_bytes": 536870912
},
"method": {
"ast_instrumentation": "temporary counter around stdlib ast.parse",
"clock": "time.perf_counter_ns",
"extraction_instrumentation": "reference adapter extract_source entry counter",
"full_mode_source_requirement": 334,
"in_process_peak_memory": "tracemalloc per measured invocation",
"process_peak_memory": "resource.getrusage(RUSAGE_SELF).ru_maxrss",
"response_size": "UTF-8 bytes of compact sorted JSON",
"samples": 3,
"threshold_basis": "Regression tripwires leave several times the Milestone 3 1,000-node allowances for AST, Logic, extraction-cache, and SQLite work."
},
"mode": "full",
"operations": {
"cold_incremental_build": {
"max_ms": 1599.76,
"maximum_response_bytes": 14580,
"maximum_traced_peak_bytes": 68573540,
"median_ms": 1599.76,
"min_ms": 1599.76,
"p95_limit_ms": 30000.0,
"p95_ms": 1599.76,
"response_limit_bytes": 262144,
"samples": 1,
"stable_result": {
"ast_parse_calls": 668,
"cache_hits": 0,
"deleted_sources": 0,
"edge_count": 1001,
"edge_hash": "00cc90998b6783afc8c9d1fd900409e5e3ba352868c4ba30e9bb2cbd59c52d35",
"extraction_calls": 334,
"invalidated_sources": 334,
"logic_edge_count": 2338,
"logic_hash": "c86a3ae74777c2cec3a82c83e6e5bcca0196772ccea63cb13340fa9141593b0c",
"logic_node_count": 2338,
"logic_projection_count": 334,
"node_count": 1002,
"node_hash": "f7705dedf8dd388857a20d11f459dc74de797bcd12d3ed36e7f3aa75d67c328f",
"reparsed_sources": 334,
"revision": "30ef23aa4106",
"source_hash": "30ef23aa41061de2d4a7c995fe109d7a41518d9ee5493d805dd256581b47dae2",
"status": "ok",
"total_sources": 334
},
"traced_peak_limit_bytes": 268435456
},
"corrupt_extraction_cache_recovery": {
"max_ms": 1695.406,
"maximum_response_bytes": 14580,
"maximum_traced_peak_bytes": 69776923,
"median_ms": 1695.406,
"min_ms": 1695.406,
"p95_limit_ms": 30000.0,
"p95_ms": 1695.406,
"response_limit_bytes": 262144,
"samples": 1,
"stable_result": {
"ast_parse_calls": 668,
"cache_hits": 0,
"deleted_sources": 0,
"edge_count": 1001,
"edge_hash": "00cc90998b6783afc8c9d1fd900409e5e3ba352868c4ba30e9bb2cbd59c52d35",
"extraction_calls": 334,
"invalidated_sources": 334,
"logic_edge_count": 2338,
"logic_hash": "c86a3ae74777c2cec3a82c83e6e5bcca0196772ccea63cb13340fa9141593b0c",
"logic_node_count": 2338,
"logic_projection_count": 334,
"node_count": 1002,
"node_hash": "f7705dedf8dd388857a20d11f459dc74de797bcd12d3ed36e7f3aa75d67c328f",
"reparsed_sources": 334,
"revision": "30ef23aa4106",
"source_hash": "30ef23aa41061de2d4a7c995fe109d7a41518d9ee5493d805dd256581b47dae2",
"status": "ok",
"total_sources": 334
},
"traced_peak_limit_bytes": 268435456
},
"corrupt_index_recovery": {
"max_ms": 1707.879,
"maximum_response_bytes": 14779,
"maximum_traced_peak_bytes": 68561924,
"median_ms": 1707.879,
"min_ms": 1707.879,
"p95_limit_ms": 30000.0,
"p95_ms": 1707.879,
"response_limit_bytes": 262144,
"samples": 1,
"stable_result": {
"action": "rebuilt",
"ast_parse_calls": 0,
"cache_hits": 334,
"edge_count": 1001,
"edge_hash": "00cc90998b6783afc8c9d1fd900409e5e3ba352868c4ba30e9bb2cbd59c52d35",
"extraction_calls": 0,
"initial_error_code": "invalid_index",
"logic_hash": "c86a3ae74777c2cec3a82c83e6e5bcca0196772ccea63cb13340fa9141593b0c",
"logic_projection_count": 334,
"node_count": 1002,
"node_hash": "f7705dedf8dd388857a20d11f459dc74de797bcd12d3ed36e7f3aa75d67c328f",
"reparsed_sources": 0,
"revision": "30ef23aa4106",
"source_hash": "30ef23aa41061de2d4a7c995fe109d7a41518d9ee5493d805dd256581b47dae2",
"status": "ok"
},
"traced_peak_limit_bytes": 268435456
},
"full_incremental_equivalence": {
"max_ms": 1897.919,
"maximum_response_bytes": 230,
"maximum_traced_peak_bytes": 64305919,
"median_ms": 1897.919,
"min_ms": 1897.919,
"p95_limit_ms": 30000.0,
"p95_ms": 1897.919,
"response_limit_bytes": 262144,
"samples": 1,
"stable_result": {
"ast_parse_calls": 1336,
"edge_count": 1001,
"extraction_calls": 668,
"logic_projection_count": 334,
"node_count": 1002,
"project_id": "milestone4-python-benchmark",
"revision": "30ef23aa4106",
"source_hash": "30ef23aa41061de2d4a7c995fe109d7a41518d9ee5493d805dd256581b47dae2",
"status": "ok"
},
"traced_peak_limit_bytes": 268435456
},
"warm_incremental_build": {
"max_ms": 1265.387,
"maximum_response_bytes": 14578,
"maximum_traced_peak_bytes": 69997166,
"median_ms": 1244.427,
"min_ms": 1239.941,
"p95_limit_ms": 20000.0,
"p95_ms": 1265.387,
"response_limit_bytes": 262144,
"samples": 3,
"stable_result": {
"ast_parse_calls": 0,
"cache_hits": 334,
"deleted_sources": 0,
"edge_count": 1001,
"edge_hash": "00cc90998b6783afc8c9d1fd900409e5e3ba352868c4ba30e9bb2cbd59c52d35",
"extraction_calls": 0,
"invalidated_sources": 0,
"logic_edge_count": 2338,
"logic_hash": "c86a3ae74777c2cec3a82c83e6e5bcca0196772ccea63cb13340fa9141593b0c",
"logic_node_count": 2338,
"logic_projection_count": 334,
"node_count": 1002,
"node_hash": "f7705dedf8dd388857a20d11f459dc74de797bcd12d3ed36e7f3aa75d67c328f",
"reparsed_sources": 0,
"revision": "30ef23aa4106",
"source_hash": "30ef23aa41061de2d4a7c995fe109d7a41518d9ee5493d805dd256581b47dae2",
"status": "ok",
"total_sources": 334
},
"traced_peak_limit_bytes": 268435456
}
},
"schema_version": 1,
"source": {
"dirty": false,
"revision": "95271dcf2e48045b9d3aed9b9ea09c7fc155692c"
}
}

View file

@ -0,0 +1,549 @@
# Language Adapter Authoring Guide
This guide explains how to turn compiler, parser, build-system, or documentation evidence into a
deterministic DocForge project graph. It covers the design work that is easy to miss when a small
fixture is expanded into a complete repository.
Use this guide after the repository assessment in
[Project Onboarding](PROJECT_ONBOARDING.md). The onboarding checklist decides whether an adapter
is needed. This guide defines how to build and prove one.
## Public authoring surface
Adapter authors should import typed contracts, validation helpers, graph models, and the
conformance helper from `docforge.adapter_sdk`:
```python
from docforge.adapter_sdk import (
AdapterAssembly,
AdapterImplementation,
AdapterManifest,
AdapterProject,
AdapterProjectSettings,
AdapterProjection,
AdapterSource,
AdapterSourceProjection,
LogicProjection,
verify_adapter_conformance,
)
```
Language-specific modules under `docforge.adapters` are repository reference implementations, not
the general authoring namespace. Their exact, deliberately narrow behavior is documented in
[Reference Adapters](REFERENCE_ADAPTERS.md).
## Required outcome
A production adapter must provide one reproducible public graph from authoritative project
evidence. It must not guess facts from filenames, preserve unstable parser identities, publish the
same declaration from several extraction units, or let an incremental cache become a second
source of truth.
The complete and incremental paths must publish exactly the same:
- project identity, adapter identity, revision, and source hash;
- nodes and stable node identities;
- relationships and their deterministic metadata;
- function Logic projections and owners;
- source paths and anchors;
- validation failures for invalid input.
Performance does not relax this requirement. A fast graph that sometimes retains stale or
translation-unit-dependent facts is invalid.
## The four adapter layers
Keep these layers separate:
1. **Evidence discovery** finds the authoritative build and source inputs.
2. **Source extraction** converts one extraction unit into raw, deterministic evidence.
3. **Assembly** resolves overlap and assigns each published fact to one owner.
4. **DocForge publication** validates, caches, indexes, queries, and visualizes the assembled graph.
DocForge supplies the publication contracts. A language frontend owns the first three layers
because only the frontend understands the language's build semantics, identity rules, generated
evidence, and source ownership.
The contracts are:
```python
class MyAdapter:
def load_manifest(self) -> AdapterManifest: ...
def extract_source(self, source: AdapterSource) -> AdapterSourceProjection: ...
def assemble_projection(
self,
manifest: AdapterManifest,
contributions: tuple[AdapterSourceProjection, ...],
) -> AdapterAssembly: ...
def load_complete_assembly(self) -> AdapterAssembly: ...
def load_projection(self) -> AdapterProjection: ...
```
`assemble_projection()` is optional only when extraction units already have disjoint ownership.
`load_projection()` is always required. It is the clean primary-graph compatibility oracle.
`load_complete_assembly()` supplies the cache-independent complete graph-plus-Logic oracle. It is
required when incremental contributions publish Logic; without it, DocForge cannot prove Logic
parity.
## What the conformance helper proves
Run the public helper against a fresh confined cache root:
```python
report = verify_adapter_conformance(
MyAdapter(project_root),
cache_root=project_root / ".docforge" / "cache" / "adapter-conformance",
)
```
`verify_adapter_conformance()` proves:
- two repeated complete assemblies are exactly deterministic, including Logic;
- when `load_complete_assembly()` exists, its primary graph exactly matches
`load_projection()`; and
- when the incremental methods exist, the assembled incremental graph and Logic exactly match the
independent complete oracle.
The report records stable identity, counts, and an assembly hash. This helper does not by itself
prove process confinement, implementation restart behavior, corrupt-cache recovery, warm zero
parsing, retrieval, or every case in the proof matrix below. Keep those as separate focused and
integration tests.
## Step 1: define authority before parsing
Write down which tool owns each fact before implementing extraction.
Typical authorities include:
| Fact | Suitable authority |
|---|---|
| Source inventory | Build graph, workspace manifest, or declared source roots |
| Active flags and features | Build-system output |
| Declaration identity | Compiler or language-service semantic identity |
| Definition location | Compiler or parser source location |
| Calls and inheritance | Resolved semantic evidence |
| Include or module dependencies | Compiler or build graph |
| Function control-flow shape | A language-aware analyzer |
| Documentation meaning | Canonical documentation sources |
Do not let a syntax parser overrule a compiler on semantics. Do not infer a resolved call merely
because names match. When the selected frontend cannot prove a fact, omit it or label it as
explicitly derived or proposed.
Record:
- frontend and compiler versions;
- build-system and feature configuration;
- supported source and generated-source roots;
- supported declaration and relationship kinds;
- unsupported facts;
- any normalization applied to frontend output.
Changing one of these rules normally requires an extractor or adapter version change.
## Step 2: define the extraction unit
An extraction unit is the smallest input that can be fingerprinted, invalidated, and re-extracted
without hidden state.
Examples include:
- one compiler translation unit;
- one module or crate target;
- one Java source set or compilation unit;
- one canonical documentation file;
- one generated API registry snapshot.
Use the build system's real unit rather than an arbitrary file grouping. One source file is not
necessarily one semantic unit when features, generated files, module resolution, or compiler flags
change its meaning.
Every unit needs:
- a stable source ID;
- a project-confined relative path;
- a content or semantic fingerprint;
- an extractor version;
- direct dependencies whose changes can alter its evidence.
Do not include output paths, temporary directories, wall time, process IDs, pointer values, or
unordered container iteration in any identity or fingerprint.
## Step 3: prove one bounded extraction
Begin with one representative unit. Include enough language behavior to expose identity and
ownership problems:
- declaration and definition;
- nested and out-of-line ownership;
- resolved call;
- inheritance or interface implementation;
- an internal or private symbol;
- one function suitable for Logic extraction;
- one shared declaration imported by another unit.
The first proof should establish:
- repeated byte-stable extraction;
- project-root confinement;
- stable source and symbol IDs;
- exact source paths and anchors;
- relationship endpoint validity;
- exact Logic ownership;
- explicit rejection of unsafe or ambiguous build input.
Do not activate the adapter in a routine project session at this point. A correct single-unit
projection does not prove whole-project ownership.
## Step 4: design stable identities
Stable identity is a semantic design decision, not a serialization detail.
Prefer, in order:
1. a stable compiler or language-tool symbol identity;
2. a normalized semantic key built from qualified ownership, symbol kind, and signature;
3. a source-qualified identity only for symbols whose language visibility is source-local.
Avoid:
- parser object addresses or transient declaration IDs;
- traversal order;
- result-set position;
- source line as the entire identity;
- a display name without namespace, owner, or signature;
- one identity policy for both externally shared and source-local symbols.
Definitions and declarations of the same externally visible symbol must converge. Anonymous,
private-to-unit, or internal-linkage symbols must remain distinct when the language makes them
distinct.
Every normalization used inside an identity must be tested against more than one extraction unit.
Some compiler fields change only after a symbol is instantiated, referenced, or fully evaluated.
If such a field is not part of authored identity, remove or normalize it before hashing.
## Step 5: separate raw evidence from the public graph
Many semantic tools repeat the same declaration in every unit that imports a header, module,
crate, package, or generated interface. That repetition is valid raw evidence but invalid public
ownership.
Do not force raw extraction to guess the final owner before the complete dependency inventory is
known. Instead:
1. extract deterministic raw contributions;
2. inventory the complete current contribution set;
3. assign each shared source or symbol to one deterministic owner;
4. publish each node, relationship, and Logic record once;
5. reject conflicting evidence instead of silently choosing incompatible values.
The raw contribution cache may contain overlap. The assembled DocForge graph may not.
### Ownership rules
Define ownership for every published fact. A common policy is:
| Fact | Recommended owner |
|---|---|
| Shared source declaration | Deterministically selected dependent unit or declared module owner |
| Unit-private symbol | Its extraction unit |
| Function Logic | The unit owning the exact function definition |
| Containment | The owner of the contained member |
| Call relationship | The owner of the caller |
| Inheritance relationship | The owner of the derived type |
| Source declaration/definition edge | The owner of the source or symbol selected by the adapter |
The correct policy depends on the language. What matters is that it is explicit, deterministic,
and identical in complete and incremental assembly.
Relationship metadata also needs deterministic reduction. Repeated evidence locations must not
grow with the number or order of extraction units. Select one stable evidence record, or define a
bounded ordered representation with a documented reason.
## Step 6: normalize frontend-generated noise
Whole-project extraction exposes facts that a single fixture will not. Compilers and language
services may emit:
- implicit template or generic instantiations;
- synthesized methods and bridge functions;
- default constructors or defaulted functions;
- inferred exception or effect annotations;
- generated annotation-processor output;
- macro expansions;
- duplicate declarations with different amounts of semantic completion;
- declarations from libraries outside the project root.
For each category, choose one policy:
- publish as authored evidence;
- publish as derived evidence with a stable identity;
- attach to an authored owner without becoming a primary node;
- omit as compiler-use noise.
Do not keep a field merely because the frontend emits it. If it changes based on whether another
unit uses the symbol, it will break determinism or complete/incremental parity unless normalized.
Add a regression fixture for every normalization rule. The fixture should demonstrate the
unstable input and the expected stable output.
## Step 7: compose the complete reference graph
Run the frontend over the complete supported source inventory and apply the final ownership
partition.
The composition must:
- sort units and outputs deterministically;
- converge declarations and definitions;
- preserve source-local identity;
- resolve semantic parents rather than relying only on visual parser nesting;
- reject duplicate node identities with conflicting semantic metadata;
- reject conflicting Logic for one owner;
- reject duplicate source IDs;
- reject missing relationship endpoints;
- retain explicit status when optional Logic analysis cannot parse a function.
Run two independent complete extractions and compare exact serialized projections or their
canonical fingerprints.
Record:
- extraction-unit count;
- node, relationship, and Logic counts;
- relationship counts by important type;
- duration and peak memory;
- output or index size;
- exact fingerprint.
These measurements establish the reference shape. They are not performance promises across
machines.
## Step 8: add dependency-aware incremental extraction
Use authoritative dependency evidence whenever possible:
- compiler dependency output;
- module or crate graph;
- Maven or Gradle compilation graph;
- generated-source and annotation-processor inputs;
- explicitly declared documentation dependencies.
The manifest must include every unit that can invalidate a cached contribution. A dependency-only
unit may publish an empty contribution; it still needs a fingerprint and stable ID so reverse
dependents are invalidated.
The incremental path is:
1. build the current manifest;
2. compare fingerprints, extractor versions, additions, and deletions;
3. invalidate changed units and their reverse dependents;
4. reuse only valid raw contributions;
5. extract invalidated units;
6. assemble the complete current contribution set;
7. validate the candidate graph;
8. reread the manifest to detect concurrent source changes;
9. atomically publish the cache and index.
Missing, incompatible, or corrupt cache data is a cache miss. It must never become a partial graph
or replace the last valid index.
### Declare the process-stable adapter implementation boundary
An adapter object is loaded once when its project-bound process starts. Source synchronization can
refresh the graph, but it cannot safely replace already imported adapter code in place.
`AdapterProject` automatically fingerprints a project-local Python package containing the loader
class and a declared descriptor file. Declare a broader or non-Python boundary explicitly when the
adapter uses helpers, configuration, schemas, or templates outside that inferred package:
```python
settings = AdapterProjectSettings(
implementation=AdapterImplementation(
roots=(project_root / "docforge_adapter",),
files=(project_root / ".docforge" / "project.toml",),
suffixes=(".py", ".toml"),
),
)
```
The boundary is confined to the project root and limited to 4,096 files and 64,000,000 bytes.
DocForge fingerprints relative paths and bytes. An added, changed, deleted, missing, or symlinked
implementation file produces `adapter_restart_required` before another MCP operation. Restart the
project-bound server; do not use Python module reloading to mutate a live adapter graph.
The manifest remains a current source snapshot, not a Git-index snapshot. A Git-backed adapter must
omit a deleted source whether its deletion is unstaged or staged. Staging is never a required
DocForge synchronization step.
### Current aggregate bounds
DocForge limits one extraction-cache generation to 10,000 source contributions and 64,000,000
encoded bytes. Malformed, incompatible, missing, oversized, or unsafe cache data is treated as a
cache miss.
The assembled graph is bounded by the effective project `Limits.max_nodes`. When adapter settings
do not override limits, DocForge selects at least 10,000 nodes and raises that bound to the
manifest's `estimated_nodes` when larger. Aggregate assembly ceilings are:
- nodes: `max_nodes`;
- relationships: `max_nodes * 32`;
- Logic nodes across all functions: `max_nodes * 32`; and
- Logic edges across all functions: `max_nodes * 64`.
Adapters should set realistic limits and fail before retaining unbounded frontend evidence. The
cache and assembly bounds do not replace each adapter's own bounded source, command, parser, or
per-function limits.
## Step 9: keep the complete path independent
The full rebuild must not read the incremental extraction cache. Otherwise equivalence compares
the cache with itself and cannot detect stale or incorrectly owned facts.
The complete path must independently:
- rediscover the supported source inventory;
- extract every semantic unit;
- apply the same normalization rules;
- apply the same public ownership partition;
- generate the same project metadata and source hash;
- publish the same nodes, relationships, and Logic.
Raw unpartitioned frontend output is not the equivalence oracle when the incremental assembler
publishes a partitioned graph. Both paths must compare the same public projection shape.
## Step 10: integrate manual and source projections
Keep canonical manual and derived source projections independently rebuildable. Compose them at a
session boundary rather than making source extraction rewrite documentation.
Decide:
- which sessions receive source nodes;
- which documentation families remain isolated;
- whether active context includes source nodes or manual guidance only;
- how manual nodes link to implementation nodes;
- what happens when required build evidence is absent;
- whether source support is required, optional with a null fallback, or disabled.
The source graph has no authority to edit runtime code or canonical documentation. MCP, viewer,
cache, and index operation remain bound to one explicit project root.
## Required proof matrix
An adapter is not complete until these cases pass:
| Case | Required result |
|---|---|
| Repeated single-unit extraction | Exact stable output |
| Two independent complete builds | Exact public projection equality |
| Cold incremental build | Every current unit extracted once |
| Unchanged warm build | Zero reparses |
| Implementation/source change | Only the unit and declared dependents reparse |
| Shared header/module change | Every reverse dependent reparses |
| Added source | New contribution appears without stale duplicates |
| Renamed source | Old contribution disappears and new identity follows policy |
| Unstaged and staged deleted source | Owned nodes, relationships, and Logic disappear identically |
| Adapter implementation edit/add/delete | `adapter_restart_required` before synchronization |
| Adapter descriptor/configuration edit | `adapter_restart_required` before synchronization |
| Build flags/features change | Affected units invalidate |
| Extractor version change | Old contributions invalidate |
| Corrupt cache | Clean recovery without partial publication |
| Interrupted extraction | Last validated index remains active |
| Complete versus incremental | Exact equality |
| Cross-session isolation | Unconfigured sessions cannot see the source graph |
| Viewer and MCP retrieval | Exact symbols, relationships, and Logic are retrievable |
Fixtures must include overlapping shared declarations. A single-file fixture cannot prove
assembly.
## Performance review
Measure before and after adding incremental extraction:
- complete extraction wall time and peak resident memory;
- warm manifest, assembly, validation, and index time;
- raw cache and final index size;
- cache-unit count and cache-hit count;
- node, relationship, and Logic counts;
- checked query latency.
If warm operation remains expensive, identify whether the cost is dependency discovery,
fingerprinting, assembly, validation, or indexing. Do not hide staleness behind an arbitrary time
window. An optimization must retain same-operation source-change detection or replace it with an
equally explicit freshness contract.
## Troubleshooting by symptom
### Node counts change between identical complete builds
Check for unstable frontend IDs, unordered output, generated declarations, inferred type or
exception information, and source paths containing temporary directories.
### The same header or module node appears in many contributions
Keep the overlap in raw evidence and add deterministic assembly ownership. Do not publish every
copy and rely on index deduplication.
### Complete and incremental graphs differ
Compare the public ownership partition first. Confirm that the complete path does not compare
unpartitioned raw output with assembled incremental output. Then compare normalization versions,
dependency inventories, deleted units, and relationship ownership.
### Warm builds consume nearly complete-build memory
Check whether cached raw contributions are too verbose, whether the assembler retains all
frontend AST data, and whether dependency discovery reparses semantic source. Cache only the
bounded projection required for deterministic assembly.
### Relationship counts grow when another unit includes the same source
Define a relationship owner and deterministic evidence reduction. Do not concatenate repeated
evidence from every unit.
### A function has conflicting Logic
Attach Logic only to the exact semantic definition owner. A syntax analyzer may find a similar
function, but line or symbol agreement with the semantic frontend is required before publication.
### An unchanged query is slow
Measure manifest revalidation separately from SQLite lookup. Preserve freshness; optimize the
authoritative dependency and fingerprint path rather than skipping it silently.
## Change and release rules
Change the extractor version when parsing, normalization, identity, relationship, Logic, or
dependency behavior can alter a source contribution.
Change the adapter version when assembly, project metadata, session composition, or public graph
policy changes.
Change the cache schema version when older serialized contributions cannot be read safely.
For every such change:
1. add a fixture for the behavior;
2. run the complete proof matrix;
3. prove clean recovery from the previous disposable cache;
4. record before-and-after graph counts and fingerprints;
5. update the adapter's operating guide and unsupported-fact list.
## Completion checklist
- [ ] Authority for every extracted fact is recorded.
- [ ] Build evidence and source inventory are reproducible.
- [ ] Stable identities distinguish shared and source-local symbols correctly.
- [ ] One representative unit extracts deterministically.
- [ ] Shared declarations and relationships have explicit public owners.
- [ ] Frontend-generated noise has documented normalization rules.
- [ ] Two independent complete builds match exactly.
- [ ] Incremental extraction uses authoritative dependencies.
- [ ] The complete oracle is independent of the cache.
- [ ] Complete and incremental public graph and Logic projections match exactly.
- [ ] `verify_adapter_conformance()` passes, and the separate proof-matrix cases also pass.
- [ ] Corrupt, missing, and interrupted cache cases fail safely.
- [ ] Session composition and family isolation are proven.
- [ ] Viewer, query, context, and Logic retrieval are proven.
- [ ] Performance, graph shape, unsupported facts, and version rules are recorded.

105
docs/AGENT_INTEGRATION.md Normal file
View file

@ -0,0 +1,105 @@
# Agent integration
DocForge generates deterministic, project-bound MCP client fragments for Codex, Claude, and
OpenClaw. Generic projects and custom adapters use different configuration APIs, but both produce
fixed standard-input/output bindings with empty generated environments and explicit capability
policy.
## Fixed reference binding
Configure one of the in-repository adapters with
`.docforge/reference-adapter.toml` as described in
[Reference Adapters](REFERENCE_ADAPTERS.md), then start:
```bash
python -I -m docforge.reference_mcp \
--project-root /absolute/path/to/project \
--capability-mode read
```
`docforge.reference_mcp` is a fixed trusted module in the installed DocForge distribution. It
selects a provider only from the validated reference configuration and reports binding metadata
containing:
- `server_module = "docforge.reference_mcp"`;
- `adapter_mode = "reference"`;
- the selected `reference_language`; and
- the exact `reference_config_hash`.
The reference binding registers exactly the 21 read tools listed in
[MCP Boundary](MCP_CONTRACT.md#read-tools). It has no proposal or canonical-application surface.
## Generate a reference client fragment
The public launcher and generator APIs are:
```python
from pathlib import Path
from docforge.adapter_launcher import AdapterLauncherV1
from docforge.client_config import generate_adapter_client_configuration
from docforge.reference_mcp import REFERENCE_MCP_MODULE, create_reference_project
root = Path("/absolute/path/to/project")
project = create_reference_project(root)
launcher = AdapterLauncherV1.for_project(project, module=REFERENCE_MCP_MODULE)
plan = generate_adapter_client_configuration(
project,
launcher,
"codex", # "codex", "claude", or "openclaw"
capability_mode="read",
)
print(plan["artifact"]["content"])
```
Pass `output=Path(...)` only when the caller has selected an exact destination. Publication is an
atomic create-or-exact-match operation; it does not merge or replace different existing content.
The result includes the launcher, source-availability, project, policy, artifact, and configuration
hashes needed to inspect the binding before use.
The fixed reference module accepts read mode only. Do not request proposal or application mode for
it.
## Custom project-owned adapter launchers
Generic `docforge configure` intentionally refuses a custom adapter project. Construct the
project-owned `ProjectService`, then use `AdapterLauncherV1` and
`generate_adapter_client_configuration()` as the custom-adapter route.
`AdapterLauncherV1` schema version 1 binds:
- one project ID, canonical absolute project root, adapter identity, and descriptor hash;
- `entry_point = "python-module"`; and
- one installed project-owned top-level Python module.
The only trusted dotted-module exception is the fixed `docforge.reference_mcp` binding. A
project-owned module is resolved through isolated Python without importing or executing it during
the probe, and it must resolve to one canonical regular `.py` file inside the project root.
The launcher contract has no arbitrary command, command arguments, shell string, working
directory, environment, callable selector, discovery rule, or module-reload mechanism. Generated
bindings invoke the current Python executable as `python -I -m <module>` with only DocForge's
validated project, capability, render-policy, authority, and no-AST options. Source identity and
the project binding are revalidated before publication; drift fails closed.
Proposal and application modes are available only to a custom module that implements those fixed
server arguments and only when the project descriptor declares the matching writer and canonical
applier authority. Generating a mode does not manufacture that authority.
## Session workflow
After registering the generated fragment in the selected client:
1. Start a new MCP process or client session.
2. Call `docforge_bootstrap`.
3. Verify project ID, root fingerprint, adapter identity, revision, source hash, binding metadata,
effective policy, and registered tools.
4. Retrieve exact or bounded project evidence through the fixed read tools.
5. Restart the process if `adapter_restart_required` reports an implementation or configuration
change.
Document text returned by DocForge is untrusted project content. It never overrides client, user,
or project authority. See [MCP Boundary](MCP_CONTRACT.md) for synchronization, pagination,
retrieval, proposal, and application rules, and
[Legacy Adapters and No-AST Policy](LEGACY_AND_NO_AST.md) before selecting `no_ast=True`.

View file

@ -1,8 +1,4 @@
# Canonical application decision
**Status:** Superseded by the DocForge 0.13 hash-bound application contract.
## Decision
# Canonical application
DocForge may apply one isolated changeset to canonical project sources through an explicit,
project-bound canonical applier. Application is available through both CLI and MCP. It is never an
@ -17,7 +13,11 @@ files.
- CLI requires `apply CHANGESET_ID --changeset-hash SHA256 --applier WRITER_ID`.
- MCP registers `docforge_apply_changeset` only when the server starts with an explicit canonical
applier identity and compatible applier implementation.
- The changeset creator and applier identity must match a configured proposal writer.
- The changeset creator must be a configured proposal writer.
- Application defaults to changesets created by the applier identity. A project-owned server may
explicitly bind additional configured proposal writers that its applier is authorized to accept.
- Cross-identity acceptance does not let the applier edit the contributor's proposal and does not
give the contributor an application tool.
- The exact final changeset hash is required. Any proposal mutation invalidates an earlier
approval.
@ -31,17 +31,15 @@ checks the derived index and regenerates declared render views.
Application does not run project commands, tests, shell operations, Git, deployment, publication,
or arbitrary renderers. Those remain with the owning project workflow.
## Why the earlier decision changed
## Safety contract
The earlier DFG-9 decision preserved manual integration because there was not yet repeated evidence
for canonical application. Later multi-project use produced recurring proposal application work,
stale-index round trips, and an explicit user requirement for faster approved integration. The new
contract addresses the original safety concerns with:
The application boundary requires:
- exact changeset-hash approval;
- startup-bound applier identity;
- an explicit accepted-writer allowlist for any cross-identity application;
- project-owned serializers for custom adapters;
- canonical path and symlink confinement;
- canonical path and symbolic-link confinement;
- rollback and semantic round-trip verification;
- deterministic derived-state refresh; and
- complete separation from Git, builds, deployment, and publication.

79
docs/COMMAND_REFERENCE.md Normal file
View file

@ -0,0 +1,79 @@
# DocForge command reference
> Generated from the live CLI parser and MCP registrations. Do not edit this file by hand.
Global CLI options are documented in `docforge --help` and are not repeated in each command row. The MCP table is the complete generic application-enabled surface; the fixed reference-adapter server exposes only its read rows.
## CLI commands
| Command | Invocation |
|---|---|
| `apply` | `docforge apply [-h] --changeset-hash CHANGESET_HASH --applier APPLIER changeset_id` |
| `backlinks` | `docforge backlinks [-h] [--relation RELATION] [--limit LIMIT] node_id` |
| `build` | `docforge build [-h]` |
| `check` | `docforge check [-h]` |
| `configure` | `docforge configure [-h] --project PROJECT [--name NAME] [--capability-mode {read,proposal,application}] [--proposal-writer PROPOSAL_WRITER] [--canonical-applier CANONICAL_APPLIER] [--no-ast] [--manual-render-policy {auto,explicit,disabled}] [--portable-graph-policy {explicit,disabled}] [--live-viewer-policy {on-demand,disabled}] [--startup-timeout STARTUP_TIMEOUT] [--tool-timeout TOOL_TIMEOUT] [--output OUTPUT] {codex,claude,openclaw}` |
| `context` | `docforge context [-h] [--budget BUDGET] [--limit LIMIT] [--cursor CURSOR] profile` |
| `dependencies` | `docforge dependencies [-h] [--depth DEPTH] [--limit LIMIT] node_id` |
| `doctor` | `docforge doctor [-h] --client {codex,claude,openclaw} [--project PROJECT] [--config CONFIG] [--server-name SERVER_NAME]` |
| `filter` | `docforge filter [-h] [--family FAMILY] [--authority AUTHORITY] [--status STATUS] [--tag TAG] [--limit LIMIT]` |
| `generation-diff` | `docforge generation-diff [-h] [--limit LIMIT] [--cursor CURSOR]` |
| `graph-plan` | `docforge graph-plan [-h] view_id` |
| `graph-render` | `docforge graph-render [-h] view_id` |
| `graph-render-status` | `docforge graph-render-status [-h] [view_id]` |
| `impact` | `docforge impact [-h] [--depth DEPTH] [--limit LIMIT] node_id` |
| `info` | `docforge info [-h]` |
| `onboard` | `docforge onboard [-h] [--language LANGUAGE] [--scaffold] [--project-id PROJECT_ID] [--title TITLE] [--content-root CONTENT_ROOT]` |
| `preview` | `docforge preview [-h] changeset_id view_id` |
| `reindex` | `docforge reindex [-h]` |
| `render` | `docforge render [-h] view_id` |
| `render-status` | `docforge render-status [-h] [--deep] [view_id]` |
| `search` | `docforge search [-h] [--limit LIMIT] query` |
| `show` | `docforge show [-h] node_id` |
| `sync` | `docforge sync [-h]` |
| `validate` | `docforge validate [-h]` |
| `validate-index` | `docforge validate-index [-h]` |
| `visualization-status` | `docforge visualization-status [-h]` |
| `visualization-stop` | `docforge visualization-stop [-h]` |
| `visualize` | `docforge visualize [-h] [--node NODE \| --query QUERY] [--depth DEPTH] [--no-open]` |
## MCP tools
| Surface | Tool | Required arguments | Optional arguments | Input schema SHA-256 | Description |
|---|---|---|---|---|---|
| read | `docforge_backlinks` | `node_id` | `limit`, `relation` | `bf9a5936e2d1ee70897dc0795ec9d716be63c9447e6fdcd3890d5dd92f45f690` | Return bounded incoming relationships for one exact stable node. |
| read | `docforge_bootstrap` | — | — | `af6c17a59a7713cfb4497f0f66b05efc091abbfe109573bb86d673e39b9b7413` | Synchronize and report the complete fixed project binding and workflow. |
| read | `docforge_dependencies` | `node_id` | `depth`, `limit` | `1283a8a030cf9369493a13e2aa87317d371470904a397910bc0b52997b3ec3b8` | Traverse declared depends_on relationships within the configured depth limit. |
| read | `docforge_filter_nodes` | — | `authority`, `family`, `limit`, `status`, `tag` | `2a51c4d8f4d76d89a2e7f90b3b06d14c7e16303b9c81e62af6747286b58cf68b` | Filter current nodes deterministically by validated metadata. |
| read | `docforge_get_context` | `profile` | `budget`, `cursor`, `limit` | `2f164828c2891a88d0ea7173790a72cc6f3da1279f14da2ca495e8989880b298` | Compile bounded cited context from one configured profile with explicit omissions. |
| read | `docforge_get_contract` | — | — | `19eb6785523ed9db645d17f61df2b8b584ac10ad308199e9a96e6e176a87169e` | Report canonical and derived boundaries plus allowed and excluded operations. |
| read | `docforge_get_generation_diff` | — | `cursor`, `limit` | `0a2fccfea4bf5d78481d40b49e7c812ded9fbe9c93cb79296ede78f22b259fcf` | Return the latest bounded primary-graph generation transition. |
| read | `docforge_get_logic` | `owner_node_id` | — | `56909ee4d72143b470528ea2b3544325e1e1920dbcc758f3caee4a4924f904a1` | Return the lazy control-flow projection owned by one function or method. |
| read | `docforge_get_node` | `node_id` | — | `52408b0c3a297e7b5cf96b293fd6d4625f05c4b090c02bfe806d25ce6c2d1c68` | Return one exact stable node from the current validated project index. |
| read | `docforge_get_task_context` | `task`, `task_kind` | `budget`, `cursor`, `focus_node_id`, `limit` | `0ffbc030bbe5cca2ae24920ec812ac976be9090da4fa2887d87526aba64a21f7` | Return one bounded task-shaped context capsule with explicit evidence gaps. |
| read | `docforge_graph_plan` | `view_id` | — | `7a014ba692f57badca9ca2098b88f82f4749ada6d71f5a3b8bc141abc86645f9` | Plan one declared portable graph without publishing derived output. |
| read | `docforge_graph_render_status` | — | `view_id` | `1d2f3604b0151ba2e228cc6ce1829d7f4244193b53d85df46eb6a3e608c59564` | Report portable-graph publication state without rendering. |
| read | `docforge_impact` | `node_id` | `depth`, `limit` | `1c71db5bb4e1afe180d98dd5000a757b7e3fd1a048b5f68af13d9b7ac56e2ce2` | Traverse bounded incoming relationships and report exact paths. |
| read | `docforge_project_info` | — | — | `a42973543d5e6f4147953d175f6aec43d228cd6bd1a63f3ca4e84a51ad7e8a64` | Report the fixed project identity, revision, source hash, and index health. |
| read | `docforge_render_status` | — | `deep`, `view_id` | `5ea5683bd74e685d4cabb1c960a66eff23a001bd2c68e1af350c6e46e936b9fd` | Report receipt state, or explicitly recompute the side-effect-free render oracle. |
| read | `docforge_search` | `query` | `limit` | `735fff8bd82b20730b0907db8bf61eb0de9fff240ef4fa239ce555779ae49047` | Run bounded lexical search over the current validated project index. |
| read | `docforge_stop_visualization` | — | — | `066a6e34019f4abcd3523f24d9e447775e1479adbea1134341f434a6bb5f8d35` | Explicitly stop this project's persistent read-only graph browser. |
| read | `docforge_sync` | — | — | `a2ac50c03b9b0e08b271216f76920279563ed264bef06d6791695c68dcb3b2a2` | Ensure the disposable project index matches current canonical sources. |
| read | `docforge_validate_project` | — | — | `3192ba97a6d7fff13008c5ed3bfa549d1eeea95332579592d96b8fd326ad06f7` | Validate current canonical sources and graph without writing any project file. |
| read | `docforge_visualization_status` | — | — | `03751f891e0a9318355a6b28be551eec6009c0f96f05ff99536e2f401b76f87f` | Report this project's managed graph browser lifecycle state. |
| read | `docforge_visualize` | — | `depth`, `node_id`, `query` | `cdc9669ba934e591379ea684934d99e2e80ba3f181d6c58867774a8064c32f4b` | Start the fixed read-only graph browser for this configured project. |
| proposal | `docforge_abandon_changeset` | `changeset_id`, `expected_changeset_hash`, `reason` | — | `8813e40ffc12b1ae3d9abe73db8de214085a4a8042323146283fffa8f7476a62` | Mark one proposal abandoned while preserving its audit record. |
| proposal | `docforge_create_changeset` | `changeset_id` | — | `0de93025b9caaa9e33b4b2b3bcfabdde087c944daba48dd77ae3c8f757c52ab8` | Create an empty hash-bound proposal under the configured isolated changeset root. |
| proposal | `docforge_get_changeset` | `changeset_id` | `cursor`, `limit` | `919737a8e732cf95305a2964cddf98f1fdce9ce2cfb52ff659436c244a38495d` | Inspect a stored proposal even when its canonical base has become stale. |
| proposal | `docforge_get_changeset_diff` | `changeset_id` | `cursor`, `limit` | `5cc1d2b035f37e3866766cfa0c3541b144762b764bf455a641ab459df03f8a59` | Return a deterministic structured and textual diff without applying the proposal. |
| proposal | `docforge_list_changesets` | — | `cursor`, `include_history`, `limit`, `status` | `b96519477781cb20acbfc9502a4b7d64a9119e0bd8d71e27290e90677aaf9ced` | List active proposals by default, with optional lifecycle history. |
| proposal | `docforge_preview_changeset` | `changeset_id`, `view_id` | — | `d84e1710a5cac52607037a7e581fcb221c42d3776f510eefe42fb6779fee7fb3` | Render one validated changeset through a declared view into its isolated preview path. |
| proposal | `docforge_propose_node_create` | `changeset_id`, `content`, `expected_changeset_hash`, `metadata`, `node_id`, `rationale`, `relationship_changes`, `target_source` | — | `8953b502b9d8d9c097935bae51fce4e830286eace9df3cb6c37b7e0542be5a4e` | Append one validated node creation without writing its canonical target. |
| proposal | `docforge_propose_node_delete` | `changeset_id`, `expected_changeset_hash`, `expected_content_hash`, `node_id`, `rationale`, `relationship_changes` | — | `8fb6a22e5c1ead3793384652e4b61e75cc23c2725906542533a121401f7faaee` | Append one validated deletion with explicit incident relationship removals. |
| proposal | `docforge_propose_node_move` | `changeset_id`, `expected_changeset_hash`, `expected_content_hash`, `node_id`, `rationale`, `target_source` | — | `32ef6fc4b9795a6a8e30f4f81f63d84d7350e983773f796524382d71cb870a67` | Append one validated same-format node move without moving a canonical file. |
| proposal | `docforge_propose_node_update` | `changeset_id`, `content`, `expected_changeset_hash`, `expected_content_hash`, `metadata`, `node_id`, `rationale`, `relationship_changes` | — | `6a29a73b9e50c9fe22cb6d056bbcdacef30c93c0822c8a9276ef33f9e51e7728` | Append one validated node update without changing canonical content. |
| proposal | `docforge_propose_relationship_update` | `changeset_id`, `expected_changeset_hash`, `expected_content_hash`, `node_id`, `rationale`, `relationship_changes` | — | `8e21fc7ba35289940dd7f12294b14e8553163ab901002dcd4a3998cdb33556c3` | Queue hash-bound relationship changes without rewriting node content. |
| proposal | `docforge_rebase_changeset` | `changeset_id`, `expected_changeset_hash` | — | `505434b2f9f611649221b95d0251db347c44752d1c73e5d69e1abfd916966788` | Safely rebase a proposal when every touched fact remains unchanged. |
| proposal | `docforge_register_changes` | `changeset_id`, `operations` | — | `c5a7d6ac07d3d3b95c13b0c57ad9b69360f2f063b1c8e22dd5f64f6d3a83dc5b` | Atomically register and validate a complete hash-bound proposal. |
| proposal | `docforge_validate_changeset` | `changeset_id` | `cursor`, `limit` | `cb9df5104dbefc3382900ec812ea581a43612d1577619071f1f03674041652c7` | Validate a proposal against its exact canonical base and other active proposals. |
| application | `docforge_apply_changeset` | `changeset_id`, `expected_changeset_hash` | — | `80259f5001f36b14f960c10bd94751d1431077b72e081fb5a5c806b79073b085` | Apply one exact validated changeset and refresh declared derived state. |

308
docs/COMPATIBILITY.md Normal file
View file

@ -0,0 +1,308 @@
# DocForge compatibility contract
Milestone 0 establishes DocForge2 as the successor repository without renaming or replacing the
working DocForge interfaces. Compatibility changes require an explicit decision, a contract-test
update, and migration guidance. DocForge 2.0.0 preserves that baseline and the complete 1.4
adapter, rendering, recovery, and release surfaces recorded below.
The compatibility gate is:
```bash
make contract
```
The complete repository gate is:
```bash
make gate
```
Milestone 5 also maintains `make compatibility-m5` for the frozen public, adapter, policy,
projection, rendering, and retrieval matrix. `make release-gate` aggregates that matrix with
migration, concurrency, recovery, task-evidence, adoption, version, artifact, secret-scan, browser,
and benchmark proofs.
## Distribution and Python imports
The Python distribution and import package remain `docforge`.
The installed executable names remain:
- `docforge`
- `docforge-mcp`
- `docforge-viewer-manager`
The top-level imports recorded by `docforge.__all__` remain supported. The documented adapter,
model, index, rendering, application, and MCP factory names imported from these submodules also
remain supported:
- `docforge.adapter_contract`
- `docforge.adapter_sdk`
- `docforge.adapter_launcher`
- `docforge.application`
- `docforge.client_config`
- `docforge.index`
- `docforge.mcp_server`
- `docforge.models`
- `docforge.policy`
- `docforge.render_contract`
- `docforge.reference_config`
- `docforge.reference_mcp`
Names beginning with an underscore are implementation details. New public names may be added
without breaking this contract.
The repository Python, JavaScript/TypeScript, and C++ adapters are supported reference
implementations. Their documented configuration, evidence limits, and unsupported-fact reports are
compatibility surfaces; their internal parser helpers are not adapter-authoring imports.
Milestone 4 adds three version-1 schemas without changing existing descriptor or result schemas:
- `schemas/adapter-launcher.schema.json`
- `schemas/adapter-client-configuration.schema.json`
- `schemas/reference-adapter.schema.json`
## CLI and MCP surfaces
Existing `docforge` command names and arguments remain supported. Existing `docforge-mcp` tool
names and arguments remain supported. Additive commands, tools, and response fields are allowed.
Removing or changing an existing name, required argument, stable error code, or safety boundary
requires an explicit compatibility decision.
Version `2.0.0` comes from one `docforge._version` authority. The four maintained executable
surfaces report `docforge 2.0.0`, `docforge-mcp 2.0.0`,
`python -m docforge.reference_mcp 2.0.0`, and `docforge-viewer-manager 2.0.0` for `--version`.
Generated generic and adapter client configurations include and hash-bind the same
`docforge_version`.
MCP results retain:
- A structured `status`.
- Project and source identity when available.
- Stable structured domain errors.
- A bounded content warning.
- Staleness information.
- The configured output-size limit.
The result schema describes the common envelope. Operation-specific fields are additive and remain
bounded by the configured tool-output limit.
The following Milestone 2 CLI additions do not change existing command signatures:
- `docforge configure codex|claude|openclaw --project ROOT`
- `docforge doctor --client codex|claude|openclaw`
Configuration output is a new version-1 machine-local contract. It preserves the `docforge`
package and executable names and emits the existing `docforge.mcp_server` module entrypoint.
Existing hand-written client configurations remain valid and are never rewritten automatically.
Doctor is inspection-only and does not become a hidden bootstrap, synchronization, or migration
path.
The following Milestone 3 CLI additions are also additive:
- `docforge graph-plan VIEW_ID`
- `docforge graph-render VIEW_ID`
- `docforge graph-render-status [VIEW_ID]`
- `--manual-render-policy auto|explicit|disabled`
- `--portable-graph-policy explicit|disabled`
- `--live-viewer-policy on-demand|disabled`
MCP adds the read-only `docforge_graph_plan` and `docforge_graph_render_status` tools. Portable
graph publication remains an explicit local CLI integration action. Existing manual render,
preview, visualization, and status names remain supported.
## Versioned data contracts
Milestone 0 preserves:
- Project descriptor schema version 1.
- Node schema version 1.
- Edge schema version 1.
- Changeset schema version 1.
- Result-envelope schema version 1.
- SQLite index schema version 3. Version 2 indexes remain disposable and automatically rebuild;
version 3 adds a source-ordered incoming-edge index for bounded impact traversal.
- Index-attestation schema version 1.
- Incremental extraction-cache schema version 1.
- Effective process-policy schema version 1. The project descriptor remains schema version 1;
machine-specific capability selection is a startup binding, not canonical project content.
- Read-pagination schema version 1. Existing tool names and required arguments are unchanged.
Context and changeset MCP reads accept optional limits and opaque generation-bound cursors.
Direct Python changeset methods and the ordinary CLI context command retain full legacy results
when pagination is not requested.
- Latest-generation-diff receipt schema version 1. The additive `generation-diff` CLI command and
`docforge_get_generation_diff` MCP read accept only optional pagination fields. They record one
primary-graph transition and do not create a history store or expose Logic details.
- Latest-generation-diff page schema version 1. Pages use one top-level pagination object and a
nested `receipt_header`. `stored_receipt_hash` names the complete stored receipt. Opaque cursors
may be restarted after a server or receipt change and are not durable public identifiers.
- Manual render-plan schema version 1.
- Graph view-plan schema version 1.
- Projection-package schema version 1.
- Projection-receipt schema version 1.
- Independent projection-policy schema version 2. Effective policy version 1 remains frozen.
Indexes, attestations, extraction caches, previews, and rendered artifacts are disposable. A schema
change may rebuild them. Canonical project content and stored proposals may not be silently
rewritten to satisfy a new implementation.
## Adapter compatibility
An adapter implementing only:
```python
load_projection()
```
remains first-class. Incremental manifests, source extraction, deterministic assembly, Logic
projection, and proposal or application support are optional capabilities. Incremental adapters
must retain `load_projection()` as their independent clean-build and equivalence oracle.
Project adapters remain explicitly composed. Generic DocForge does not discover arbitrary adapter
modules or choose a project globally.
The supported generation-diff Python boundary is `ProjectIndex.generation_diff()`. Helpers in the
`docforge.generation_diff` module implement the disposable publication contract and are internal;
they are not frozen as adapter-authoring imports.
## Preserved no-AST binding
`docforge-mcp --project-root /project --no-ast` is a stable shorthand for the
`preserve-no-ast` binding policy.
The binding:
- Keeps one-method complete-projection adapters working.
- Keeps non-AST incremental fingerprinting and caching working.
- Rejects nonempty function-Logic publication.
- Rejects a pre-existing index containing function Logic.
- Blocks `docforge_get_logic`.
- Prevents the live viewer from pinning an index containing Logic.
- Applies the same restriction during hash-bound canonical-application refresh.
- Reports the effective policy through bootstrap and contract results.
The legacy `adapter_policy` payload and error codes remain unchanged. The version-1
`effective_policy` is additive and makes precedence, capability mode, render behavior, blocked
tools, and prohibitions machine-readable.
DocForge does not inspect arbitrary adapter source to prove which parser implementation it uses.
The no-AST binding is an owner-selected process policy backed by Logic publication and retrieval
enforcement. It is not a filesystem sandbox and cannot stop an unrelated process with repository
write access from changing adapter code.
## Changesets and application
The following guarantees remain stable:
1. Registration writes one complete proposal atomically.
2. Proposal identity includes its project, root, base revision, canonical source hash, writer, and
ordered operations.
3. Validation and diff inspection precede application.
4. Append, rebase, abandonment, and application use exact current hashes.
5. Stale, conflicting, unauthorized, unsafe, or invalid proposals fail closed.
6. Canonical application is absent unless one startup-bound applier is configured.
7. Derived refresh failures produce an explicit degraded receipt after canonical application. They
do not make an applied proposal safe to apply twice.
8. Generic canonical publication compares exact target identity at the commit boundary. Concurrent
create, update, and delete mutations fail closed, roll back when exact state remains provable, or
retain recovery evidence without overwriting foreign data.
9. Per-file publication is atomic and in-process rollback covers earlier publications, but
canonical application has no process-death journal and does not promise multi-file crash
atomicity.
10. Cleanup degradation after semantic commit closes the proposal as `applied` and persists bounded
`application_recovery` lifecycle metadata instead of returning a retryable ordinary failure.
## Rendering and visualization
The `generic_html` renderer remains the supported version-1 manual projection. Its public
`GenericHtmlRenderer.prepare()` signature, renderer identity, frozen alpha bytes, confined paths,
raw-HTML suppression, fixed template tokens, deterministic identities, atomic replacement, and
side-effect-free status remain compatible. It now delegates through a versioned manual plan,
immutable package, and independent renderer.
The `portable_graph_html` renderer and `graph_render` descriptor table are additive. Manual and
portable graph declarations, plans, policies, publication receipts, and status remain separate.
The portable renderer does not replace the existing live viewer or `docforge_visualize`.
Existing project descriptors may retain any positive `max_render_bytes` accepted by schema version
1. A value above 20,000,000 bytes does not make the descriptor invalid, and a smaller actual
artifact still renders. Actual detached worker transfer is a separate fixed 20,000,000-byte
runtime boundary.
The live graph viewer remains a read-only consumer of a generation-pinned validated index. It does
not become project authority or MCP retrieval authority. Source reads use the pinned index
generation instead of reopening mutable canonical files behind that generation.
## Task-context compatibility
`docforge_get_task_context` is an additive MCP read tool. The legacy `docforge_get_context`
signature, profile compiler, direct Python results, and custom three-argument context-provider
contract remain unchanged.
The new `ContextCapsuleV1` and `RetrievalPlanV1` types live in the public
`docforge.retrieval` submodule. Version 1 guarantees:
- A closed task-kind vocabulary and core-derived plan. Callers cannot inject arbitrary operations,
SQL, paths, relations, or Logic requests.
- One immutable index transaction and one exact project, adapter, revision, source, policy,
request, plan, collection, and capsule identity.
- Deterministic bounded focus, traversal, hydration, token accounting, response packing, and
continuation, with fixed version-1 ceilings of 1,000 evidence items, 100,000 candidate edges, and
10,000 task-query characters.
- Raw preservation of project-owned relation names. Only the documented versioned alias map gains
task semantics; all other relations remain `unclassified`.
- Separate missing, incomplete, blocked, and provenance-limitation evidence.
- No-AST bindings retain task context but never add a Logic retrieval step or weaken the existing
Logic prohibition.
An integration that replaces the legacy context provider does not silently receive the core task
planner. Version 1 has no custom task-planner protocol. The task-context tool remains registered
for additive name compatibility but returns `task_context_unavailable` without loading or
synchronizing the custom projection.
The exact version-1 relation aliases are frozen by the MCP contract and repository contract tests.
Changing an alias category requires a new planner version; it is not a silent implementation
detail.
Milestone 3 adds `ManualRenderPlanV1`, `GraphViewPlanV1`, projection package and receipt version 1,
and projection policy version 2. These are additive submodule and schema contracts. They do not
change the legacy task-context, adapter, changeset, or effective-policy contracts described above.
Milestone 4 adds `docforge.adapter_sdk`, the fixed reference configuration and read-only MCP
binding, immutable adapter launchers, and generated adapter client fragments. A legacy adapter with
only `load_projection()` remains first-class and need not adopt incremental extraction, Logic, a
reference configuration, or launcher metadata.
## Safety boundary
DocForge remains bound to one explicit project root. It rejects absolute paths, root escapes, and
symbolic-link escapes. Documentation text remains untrusted data. Normal MCP operation exposes no
arbitrary filesystem access, renderer execution, shell command, Git mutation, deployment,
publication, or project switching.
DocForge is not a filesystem sandbox. Mode-0700 canonical transaction directories protect against
other users and ordinary path access; deliberate arbitrary tampering by another process with the
same operating-system UID is outside the compatibility boundary.
The historical `v1.0.0` release carried distribution metadata `1.0.0` while its module and MCP
runtime reported `0.15.0`. Version 2.0.0 records that inherited mismatch in its maintained
migration proof and resolves current identity through one authority. See
[migrating from v1](MIGRATING_FROM_V1.md).
## Recorded weaknesses, not compatibility promises
Milestone 0 records rather than redesigns these areas:
- Generic warm reads still repeat whole-project discovery, parsing, and validation.
- The base wheel intentionally omits Tree-sitter. JavaScript, TypeScript, and C++ syntax evidence
requires the matching `docforge[javascript]`, `docforge[typescript]`, or `docforge[cpp]` extra.
Python reference evidence uses the standard library and remains available in the base wheel.
- One individually oversized context entry is represented as explicit bounded omission evidence;
callers use targeted retrieval for that node.
- One individually oversized changeset diff is transported as reconstructable canonical-JSON
chunks. Cursors are corruption-detecting read tokens, not authenticated authorization tokens.
- Production fragment validation is currently slower than forced-full rendering at the maintained
1,000-page fixture. Full rendering remains the equivalence and recovery oracle.
- Remote render services, render farms, third-party renderer ecosystems, and a separate render MCP
remain deferred.
- DocForge2 does not self-host its bootstrap documentation.

View file

@ -1,4 +1,4 @@
# DocForge 0.14 contract
# DocForge 2.0 contract
## Authority boundary
@ -18,13 +18,35 @@ commit when Git is available; it cannot change repository state.
- Edge schema: `schemas/edge.schema.json`, version 1.
- Result envelope: `schemas/result.schema.json`, version 1.
- Changeset schema: `schemas/changeset.schema.json`, version 1.
- Index schema: version 1, disposable and reproducible.
- Core, CLI, and MCP server: version 0.15.0.
- Effective policy: `schemas/policy.schema.json`, version 1.
- Task context capsule: `schemas/context-capsule.schema.json`, version 1.
- Latest generation diff: `schemas/generation-diff.schema.json`, version 1.
- Latest generation-diff page: `schemas/generation-diff-page.schema.json`, version 1.
- Generated client configuration: `schemas/client-configuration.schema.json`, version 1.
- Client doctor result: `schemas/doctor-result.schema.json`, version 1.
- Manual render plan: `schemas/manual-render-plan.schema.json`, version 1.
- Graph view plan: `schemas/graph-view-plan.schema.json`, version 1.
- Projection package: `schemas/projection-package.schema.json`, version 1.
- Projection receipt: `schemas/projection-receipt.schema.json`, version 1.
- Independent projection policy: `schemas/projection-policy.schema.json`, version 2.
- Adapter launcher: `schemas/adapter-launcher.schema.json`, version 1.
- Adapter client configuration: `schemas/adapter-client-configuration.schema.json`, version 1.
- Reference adapter configuration: `schemas/reference-adapter.schema.json`, version 1.
- Index schema: version 3, disposable and reproducible.
- Index attestation: schema version 1, disposable and reproducible.
- Distribution, Python package, CLI, generic MCP, reference MCP, and viewer manager: version 2.0.0.
- Incremental extraction cache: version 1, disposable and reproducible.
Schema files describe the generic interchange contract. Runtime validation remains responsible for
path confinement, source hashing, relationship resolution, dependency cycles, project limits, stale
state, and adapter-specific rules that JSON Schema cannot prove by itself.
`src/docforge/_version.py` is the sole package-version authority. The maintained executable
surfaces report exactly `docforge 2.0.0`, `docforge-mcp 2.0.0`,
`python -m docforge.reference_mcp 2.0.0`, and `docforge-viewer-manager 2.0.0` for `--version`.
Generated generic and adapter client configurations bind `docforge_version` into their validated
hashes.
## Generic node storage
Markdown nodes begin with a TOML metadata block delimited by `+++`. The remaining Markdown is the
@ -38,8 +60,100 @@ gives special acyclic validation to `depends_on`; adapters may add stricter rule
## Result identity
Successful operations identify the project, adapter, current revision when available, and canonical
source hash. Errors use a stable code, direct message, and structured details. Query operations fail
if canonical source no longer matches the derived index.
source hash. Errors use a stable code, direct message, structured details, and a bounded remediation
tool when recovery is safe. MCP operations synchronize disposable index state under a project lock
before reading or proposing. Canonical source validation remains fail-closed.
Task-context retrieval derives a closed version-1 plan from a bounded task kind and the effective
process policy. It executes against one immutable index transaction and returns generation-bound,
hash-identified evidence, gaps, omissions, and provenance limitations. Project relation names
remain authoritative. The core applies task semantics only to its versioned alias set and preserves
every other allowed relation as unclassified.
An atomic index build writes a whole-file SHA-256 attestation after complete graph, row, FTS, and
SQLite integrity verification. A fresh process may use that receipt to verify an unchanged index
without reconstructing all graph rows. A missing, malformed, or mismatched receipt falls back to
complete verification and is repaired only after that verification succeeds.
Index replacement is the derived publication commit point. Attestation, cheap source-generation,
and latest-generation-diff receipts are independent post-commit evidence. Their failure produces
bounded degraded success and never falsely reports that a committed index mutation failed.
Derived publication stages and flushes one complete bounded artifact before atomic replacement or
no-clobber publication, then flushes the containing directory. An interruption therefore leaves
the preceding complete artifact, the new complete artifact, or explicit degraded post-commit
evidence. Index attestations, manual render receipts, generation-diff baselines, and portable-graph
manifests have maintained exact-oracle corruption-and-repair proofs. Canonical project files remain
unchanged throughout those recoveries.
Before replacement, a build accepts a predecessor only when its exact main-file inode has a
matching whole-file attestation, has no WAL, journal, or shared-memory sidecar, and passes the
published SQLite identity, row, hash, FTS, integrity, and policy checks. It uses an immutable
main-file read and never repairs predecessor evidence. The build then revalidates the new source
snapshot including exact node and edge equality and rejects a stable source identity that produces
different graph content as `generation_collision`.
The version-1 generation-diff receipt stores one bounded latest primary-graph transition. It is
not history and contains no Logic details or source text. Public pages carry one
`receipt_header`; its `stored_receipt_hash` identifies the complete persisted receipt rather than
the header alone. One top-level pagination object carries the only continuation cursor.
## Machine-local client integration
Generated Codex, Claude, and OpenClaw fragments are machine-local projections. They are not
canonical project content. Version 1 binds the selected project, exact isolated Python
interpreter, canonical argument layout, effective policy, no-AST projection, render policy,
timeouts, artifact bytes, and configuration hash. Milestone 3 adds the version-2 projection policy,
its hash, projection availability, and the exact descriptor hash to that attested configuration
evidence. Omitted default selectors are recomposed against the bound descriptor.
Preview is side-effect free. Explicit publication creates only one new private standalone
fragment in an existing real directory. It never merges or replaces different content. Descriptor,
parent, target, content, ownership, permission, and link identities are checked before and after
the directory durability boundary. A failure rolls back when that can be proven and otherwise
returns bounded unconfirmed publication evidence.
Doctor is a bounded read-only inspector with one fixed check inventory. It uses stable no-follow
descriptor and configuration reads plus stat-only derived-index evidence. It never loads a
complete projection, opens SQLite, starts MCP, executes the configured command, synchronizes,
builds, renders, starts a viewer, or writes configuration. Unprovable client behavior is a warning,
not an invented success.
## Public adapter SDK and reference binding
`docforge.adapter_sdk` is the stable adapter-authoring import boundary. It exposes the typed
projection, manifest, source contribution, complete assembly, project wrapper, graph model, and
conformance contracts needed by an adapter without requiring authors to import core implementation
modules.
Complete evidence includes the primary graph and function Logic. An incremental adapter that
publishes Logic must implement `load_complete_assembly()` as an independent complete oracle.
`verify_adapter_conformance()` proves repeated complete determinism, equality between the complete
assembly and `load_projection()`, and exact complete/incremental graph-plus-Logic parity. Separate
tests remain responsible for confinement, restart behavior, no-AST behavior, cache recovery, and
retrieval.
Adapter assemblies are bounded before publication. Primary nodes use the descriptor `max_nodes`
limit. Primary edges, Logic nodes, and Logic edges use fixed deterministic multipliers over that
limit. Version-1 extraction caches are regular-file-only, bounded to 10,000 sources and
64,000,000 bytes, and are treated as misses when corrupt, oversized, foreign, or incompatible.
`.docforge/reference-adapter.toml` is a closed version-1 selection among `python`, `javascript`,
`typescript`, and `cpp`. It declares one project identity and explicit non-overlapping source
roots. C++ additionally requires a confined `compile_commands.json`. It cannot declare a command,
module, environment, writer, applier, or remote endpoint.
`python -m docforge.reference_mcp --project-root ROOT` constructs only the selected fixed
repository reference adapter and exposes the read surface. It never registers proposal or
application tools.
`AdapterLauncherV1` is an immutable project-bound Python-module declaration. It accepts no
arbitrary command, arguments, working directory, environment, discovery, callable selector, or
module reload. Custom launchers resolve one installed top-level module through isolated Python and
require its origin inside the project root. The fixed `docforge.reference_mcp` module is the only
trusted dotted exception. `generate_adapter_client_configuration()` binds generated Codex,
Claude, and OpenClaw fragments to that launcher, current source availability, effective policy,
descriptor, interpreter, and exact artifact bytes.
## Isolated proposal model
@ -48,10 +162,17 @@ operation names its expected base hash. A move preserves the stable node ID. A d
every incident relationship. Proposal validation and storage are atomic. Application requires the
exact final changeset hash; prose is never auto-merged.
Relationship-only additions and removals use the validated update operation without changing node
metadata or content. They remain bound to the complete changeset base hash and the anchor node's
expected content hash.
The MCP process binds to one configured writer identity at startup. The project descriptor grants
that writer explicit families and operation types. A changeset records its creator, project root
fingerprint, base revision, canonical source hash, and ordered operations. Every append requires the
current changeset hash, so simultaneous writers cannot silently lose an operation.
The atomic registration operation captures a complete operation list against one current base,
fills omitted existing-node hashes from that synchronized snapshot, validates once, and writes one
final changeset.
Changesets from the same canonical base may coexist only when their touched node and source sets do
not overlap. Exact overlaps return structured conflicts naming the other changesets, nodes, and
@ -59,6 +180,37 @@ sources. A stale canonical base, stale node hash, stale changeset hash, unauthor
path, invalid graph, dependency cycle, unresolved delete relationship, or configured limit fails
before the proposal file changes.
A stale proposal may be rebased only when its stored node hashes, source targets, relationship
preconditions, permissions, conflict set, and complete projected graph still validate against the
current project. Application and explicit abandonment create derived lifecycle receipts. The
default active listing contains only draft and ready work. Stale, applied, and abandoned proposals
remain queryable by explicit status or history request. Terminal proposals do not block new
proposals.
## Canonical application durability
Generic canonical application stages replacements and backups in a mode-0700
`.docforge/application/transaction-*` directory. Before each canonical create, update, or delete,
it compares exact file identity at the publication boundary. Creates use no-clobber publication.
Updates and deletes use atomic exchange and no-replace detachment. A concurrent canonical-target
mutation fails closed, is rolled back only when exact displaced state remains provable, or is
retained without overwriting foreign data.
This is per-file compare-and-swap publication, not multi-file crash atomicity. In-process failures
run rollback across already published files, but there is no process-death journal. Process or host
death between publications may leave a partial canonical application and requires operator
inspection before a new proposal or restoration.
The private transaction namespace is integrity-confined against ordinary path access. It is not a
filesystem sandbox, and deliberate arbitrary tampering by another process with the same operating-
system UID is outside the contract. DocForge identity-checks private files before consuming or
removing them.
After the serializer reproduces the approved graph, canonical success is final. If private cleanup
then degrades, application still returns `applied`, closes the proposal, and persists bounded
`application_recovery` lifecycle metadata with status `cleanup_required`, retained paths, and
remediation. The reviewed changeset must not be applied twice.
## Declared rendering and previews
Render configuration is optional. A configured project declares one template root, one isolated
@ -66,15 +218,68 @@ preview root, and one or more stable view IDs. Each view names a built-in render
derived output file, title, and optional family filter. Paths are resolved under the project root
and may not overlap canonical content, authority files, changesets, templates, or previews.
The initial `generic_html` renderer uses pinned CommonMark parsing with raw HTML disabled. Templates
are UTF-8 files with a fixed token vocabulary; they cannot name commands, modules, or executable
The `generic_html` renderer uses pinned CommonMark parsing with raw HTML disabled. Templates are
UTF-8 files with a fixed token vocabulary; they cannot name commands, modules, or executable
renderers. Render identity covers the canonical source hash, optional changeset hash, selected node
and edge identities, view configuration, template hash, renderer contract, and exact parser version.
and edge identities, view configuration, template hash, renderer contract, and exact parser
version. The frozen version-1 API and alpha bytes are preserved by a compatibility wrapper over the
manual plan/package/renderer path.
An explicit CLI render atomically replaces one declared derived output. MCP can render a validated
changeset only to its isolated preview path. Status recomputes expected output without writing and
reports `current`, `stale`, `missing`, `unsafe`, or `oversized`. Input changes detected before atomic
replacement fail without replacing the prior output.
changeset only to its isolated preview path. Normal status verifies bounded source, configuration,
template, output, renderer, and publication-receipt identities without reconstructing the output.
Explicit deep status remains the side-effect-free full-render oracle. Input changes detected before
atomic replacement fail without publishing a current receipt for stale output.
## Independent projection boundary
Manual and portable graph plans are separate version-1 contracts over one immutable validated
generation. They use canonical JSON, deterministic ordering, fixed structural and serialized-size
bounds, and content-derived identities. Plans contain selected graph facts and bounded content.
They contain no live project object, database handle, absolute project or index path, arbitrary
query, command, executable path, or caller-selected module.
Projection packages bind one plan to inert assets, a closed built-in renderer identity, declared
component versions, and an artifact inventory with a byte allowance. Receipts bind the exact
package, plan, renderer, artifact hashes and sizes, diagnostics, timing, and detached peak memory.
Manual and graph renderer modules accept only validated packages. They cannot select nodes, invent
relationships, read project state, choose publication paths, or write canonical files.
Detached execution uses one fixed private Python module, isolated mode, a trusted working
directory, a sanitized environment, exactly one canonical newline-terminated JSON request and
response, a closed renderer allowlist, a 30-second timeout, disk-spooled stdout, and bounded reads. The package
contract is capped at 24,000,000 bytes and actual detached artifact transfer at 20,000,000 bytes.
Project descriptors may retain a larger `max_render_bytes` compatibility allowance, but an actual
detached transfer above the fixed worker boundary fails closed.
Portable graph configuration is independent of manual render configuration. One view selects
either an exact root or a bounded metadata-only lexical query plus closed filters and node, edge,
depth, and work limits. Logic is excluded. The renderer emits a complete static Nodes, Flow, or Web
artifact and uses JavaScript only as progressive enhancement.
Portable publication commits a content-addressed artifact, renderer receipt, and one bounded
generation/view manifest in that order. The manifest is the publication commit. Status reads only
bounded manifest and receipt evidence and never plans or renders. Repair restores declared output
only from validated content-addressed evidence. A failure after a replacement that cannot be
proven rolled back returns explicit degraded committed evidence.
Manual fragment records are disposable semantic cache entries. Their keys bind the renderer,
component version, and complete page semantics. The detached renderer recomputes the expected page
fragment before accepting cached bytes. Cold creation is compared with a full detached render
before cache publication. Invalid, corrupt, forged, stale, individually oversized, or
aggregate-oversized records fall back to the full oracle. The dedicated cache retains only current
keys and is capped at 10,000 entries and 64,000,000 bytes.
Projection policy version 2 composes manual `auto|explicit|disabled`, portable graph
`explicit|disabled`, and live viewer `on-demand|disabled` independently. Active plan, render,
application, onboarding, and viewer-start operations enforce the relevant policy before hidden
work. Receipt-only status and explicit viewer stop remain available. Effective policy version 1
and its legacy projections remain unchanged.
The live viewer remains separate from portable graph publication. It consumes one
generation-pinned validated index through the viewer manager. Source reads come from that pinned
generation and do not reopen mutable canonical files behind an older snapshot. Neither live nor
portable visualization is retrieval or canonical authority.
Normal MCP access does not expose canonical application. An explicitly configured canonical
applier registers one hash-bound application tool. No MCP mode exposes arbitrary renderer
@ -99,13 +304,14 @@ random token is part of every accepted URL path. Only `GET` and `HEAD` are suppo
no-store caching, a restrictive content-security policy, frame denial, MIME sniffing protection,
and no-referrer policy. The built-in template uses only same-origin JSON endpoints for graph
overview, bounded search, exact descriptor-category filtering, exact node content, bounded
incoming-and-outgoing neighborhoods, semantic Flow ancestry, convergence Web context, and one
node's bounded project-confined source file.
Descriptor filtering accepts only
family, authority, status, or tag plus one exact value. There is no write endpoint, arbitrary query
endpoint, static filesystem handler, external asset, or project-selection control.
incoming-and-outgoing neighborhoods, semantic Flow ancestry, convergence Web context, lazy
function-scoped Logic, and one node's bounded project-confined source file.
Descriptor filtering accepts only family, authority, status, or tag plus one exact value.
Search filtering accepts only family, indexed kind or callable, indexed language tag, and the
fixed `logic` or `source` capability. There is no write endpoint, arbitrary query endpoint, static
filesystem handler, external asset, or project-selection control.
The `graph-browser@15` template provides mouse-wheel zoom centered on the pointer, left-button drag
The `graph-browser@17` template provides mouse-wheel zoom centered on the pointer, left-button drag
pan, explicit zoom-in and zoom-out buttons, a reset-view button, and a live zoom percentage. A
four-pixel drag threshold defers pointer capture and preserves node activation for ordinary clicks.
Loading another root node fits the viewport to the returned neighborhood, including a useful
@ -125,6 +331,10 @@ inspectors. This display shortening is presentation-only and never changes index
Left-clicking or pressing Enter on a graph node opens a compact descriptor card containing the
validated metadata and content previously shown in the details panel. Its family, authority,
status, and tag pills are buttons that replace the left result list with exact matching nodes.
The left result panel also exposes composable family, node-kind, language, and capability filters
plus fixed convenience presets. These filters are bounded read-only queries over indexed
attributes and stored Logic ownership. Selecting a canvas node emphasizes only its incident edges
and directly connected nodes; unrelated visible paths are muted but remain present.
Right-clicking or pressing Shift+Enter opens the complete inspector. Inspection does not replace
the current neighborhood or reset the viewport. Both dialogs support Escape, explicit close
controls, and backdrop dismissal. Loading the inspected node as the new root requires the separate
@ -136,7 +346,7 @@ supported line, TOML, heading, or text anchors. Both side panels support pointer
resizing. The unblurred full inspector supports native resizing, constrained title-bar dragging,
and a fixed header/footer surrounding a scrollable body.
The header exposes a Nodes/Flow/Web segmented selector. Nodes displays the complete bounded
The header exposes a Nodes/Flow/Web/Logic segmented selector. Nodes displays the complete bounded
neighborhood. Flow displays semantic ancestry ending at the current root. Structural and execution
edges retain their declared source-to-target direction. Reads, imports, dependencies, inheritance,
and `tested_by` reverse because their declared target feeds or qualifies the source. Documentation
@ -146,10 +356,18 @@ direct root-owned members and execution dependencies into adjacent contributor b
does not fan back out through unrelated siblings. These are presentation transforms over the
validated snapshot; they do not add or change project relationships.
All three views color edges by relationship semantics and retain direction with visible SVG endpoint
Logic is available only when the focused node owns a stored `LogicProjection`. The browser
retrieves that projection through a bounded, exact-owner endpoint. Entry, condition, action,
control, convergence, return, raise, and exit nodes remain outside primary graph search and
traversal.
Logic edges retain their declared `TRUE`, `FALSE`, `NEXT`, `CASE`, `LOOP`, `EXCEPTION`, `RETURN`,
`RAISE`, `BREAK`, and `CONTINUE` labels. Hiding a logic node creates a visible omitted-path bridge
between retained predecessors and successors instead of pruning valid downstream control flow.
Nodes, Flow, and Web color edges by relationship semantics and retain direction with visible SVG endpoint
symbols. Line patterns provide a non-color cue. A static canvas key shows the exact symbol, color,
label, and visible count for each displayed relation, including a deterministic fallback for
project-defined relations. Nodes, Flow, and Web use the same map.
project-defined relations. Logic uses a separate fixed control-flow map.
The browser derives node presentation roles only from the returned bounded graph. The current root
is the focus. In Nodes, nodes reachable through outgoing edges are shown as outgoing paths; the
@ -159,6 +377,11 @@ distinct palettes and navigation sections. An undirected shortest-hop calculatio
distance rings; Flow and Web use left-to-right distance layers with the destination on the right.
Each role palette darkens progressively by distance, capped at fifty percent.
Logic uses a layered left-to-right layout with explicit horizontal clearance and vertical
separation between siblings. Control-flow edges use routed curves and distinct lanes, including
raised return and loop-back routes, to avoid drawing one path directly over another whenever the
bounded topology permits.
Each invocation creates or reuses one worker through the separately supervised, per-user viewer
manager. The manager is outside the short-lived MCP transport and owns all child workers as one OS
service unit. It accepts only authenticated loopback requests and a validated immutable snapshot.
@ -200,3 +423,21 @@ An explicit integration may construct the full fixed MCP surface for a configure
and one startup-bound writer. Canonical application is registered only when the integration also
supplies a startup-bound applier identity and project-owned `CanonicalApplier`. An adapter without
proposal settings or validation remains read-only.
An adapter may additionally implement the opt-in incremental contract. Its manifest inventories
stable source IDs, fingerprints, extractor versions, and source dependencies without parsing the
complete project. Each extraction owns deterministic nodes, relationships, and optional
function-scoped logic. Added, changed, deleted, and reverse-dependent sources are invalidated.
Cached and refreshed facts are always assembled into a complete projection and pass normal graph
validation before publication. The full projection loader remains the fallback and equivalence
oracle.
The Release 1 `AdapterLoader` contract remains valid. A loader that supplies only
`load_projection()` stays on the complete-projection path. Incremental capability detection is
additive and cannot make the new methods mandatory for an existing adapter. An incremental loader
must also implement `load_projection()` so a clean rebuild and equivalence check remain possible.
Logic projections are not primary graph nodes. They remain source-scoped, function-owned,
independently cached control-flow data so ordinary search, Nodes, Flow, and Web do not become
statement graphs. Index schema 2 stores them in dedicated owner, node, and edge tables. Reads are
bounded to one exact function or method owner.

View file

@ -0,0 +1,165 @@
# Core concepts and authority
DocForge is a project-bound knowledge compiler. It turns explicit canonical project facts into
validated graphs and bounded derived views without transferring authority to the index, an agent,
or a renderer.
## One project, one explicit root
Every operation is bound to one canonical real project directory. Paths in descriptors and adapter
configuration are project-relative and confined beneath that root. A CLI or MCP process does not
discover or switch projects after startup.
The project root determines:
- which descriptor and canonical sources may be read;
- where derived cache and changeset roots may exist;
- which project identity, revision, source hash, and generation appear in results;
- which writer, applier, rendering, and viewer policies can be selected.
Generated client fragments preserve that binding. They are machine-local configuration
projections, not portable project authority.
## Canonical facts and derived evidence
Canonical inputs own the facts:
- generic Markdown or TOML node sources;
- authority files named by a generic descriptor;
- an adapter's declared canonical sources and implementation boundary;
- `.docforge/project.toml` for a generic project;
- `.docforge/reference-adapter.toml` for a fixed reference integration.
Everything DocForge builds from those inputs is derived:
- SQLite indexes, attestations, generation receipts, and generation diffs;
- incremental extraction caches;
- task-context capsules and query responses;
- changeset previews;
- render plans, immutable packages, fragments, artifacts, and receipts;
- portable graph publications and live-viewer processes;
- generated Codex, Claude, and OpenClaw fragments.
Derived state may be discarded and rebuilt. A derived artifact can prove what it was bound to, but
it cannot override current canonical content.
## Nodes, relationships, and Logic
The primary graph contains nodes and directed relationships.
A node has a stable project-wide ID, title, family, authority, status, tags, summary, content,
source identity, and content hash. A generic Markdown file contains one node beneath a TOML
metadata block; a generic TOML source may contain multiple nodes. Adapter nodes use the same public
graph contract.
A relationship is an exact `(source, relation, target)` triple. The project descriptor defines the
allowed relation names. DocForge gives `depends_on` special acyclic validation, but it does not
invent domain meaning for a project's other relation names. Retrieval recognizes only a versioned
alias set for task planning and reports unknown allowed relations as `unclassified`.
Logic is deliberately separate. It is a lazy function-scoped control-flow projection owned by one
primary node. Decisions, actions, loops, convergence points, returns, and exceptions connect
through explicit branch edges. Logic does not add statement-level nodes to ordinary Nodes, Flow,
Web, search, or generation-diff results. Python, JavaScript, TypeScript, and C++ integrations may
publish Logic when their adapter contract supports it.
## Authority, status, family, and tags
These fields answer different questions:
- `authority` describes the role of the content. The generic vocabulary is `authoritative`,
`approved_plan`, `derived`, `proposal`, and `historical`.
- `status` is project-defined lifecycle state such as `current`, `active`, or `verified`.
- `family` is a project-defined content grouping used for filtering, profiles, rendering, and
writer permissions.
- `tags` are exact project labels for retrieval and presentation.
An `authoritative` node can still become stale; authority is not a freshness claim. A `derived`
node is still canonical if it is stored in a declared canonical source; the label describes its
role, not whether DocForge may silently regenerate it. Status and authority never grant an MCP
writer permission.
## Validation, synchronization, and generations
Validation loads the complete canonical graph and rejects unsafe paths, invalid source formats,
duplicate IDs, unresolved relationships, prohibited cycles, violated project limits, and
adapter-specific contract failures.
The disposable index is published atomically only after the complete graph and SQLite integrity
checks pass. Its attestation binds the whole index file. A successful replacement is the derived
publication commit point; later receipt-writing trouble is reported as degraded evidence rather
than as a false claim that the replacement failed.
MCP operations automatically synchronize derived state under a project lock before normal work.
A graph generation identifies one validated indexed snapshot. Generation-pinned retrieval and
rendering do not reopen mutable sources behind an older snapshot.
The latest generation diff is one bounded primary-graph transition, not a history database. It
contains no Logic details or source text.
## Complete and incremental adapters
`load_projection()` is the compatibility baseline and clean graph oracle. An incremental adapter
adds:
- `load_manifest()` for cheap project identity, source inventory, fingerprints, and dependencies;
- `extract_source()` for one cacheable source contribution;
- optionally `assemble_projection()` to normalize overlapping contributions.
When incremental contributions contain Logic, `load_complete_assembly()` supplies a
cache-independent complete graph-plus-Logic oracle. Warm cache behavior is an optimization, never a
different authority path. Corrupt or incompatible extraction caches are treated as misses, and a
clean complete build remains the equivalence and recovery boundary.
The public types, validators, and conformance helper are exported from `docforge.adapter_sdk`. See
the [Adapter authoring guide](ADAPTER_AUTHORING_GUIDE.md) and [Incremental
indexing](INCREMENTAL_INDEXING.md).
## Proposals are not canonical changes
A changeset is an isolated, ordered proposal over an exact canonical base. Each operation names
preconditions, and the final changeset has a content-derived hash. Validation projects the complete
resulting graph before application.
Canonical application requires:
1. a writer declared in the descriptor;
2. a process started with the matching proposal and application authority;
3. one explicit final changeset hash;
4. unchanged source, relationship, permission, and graph preconditions.
Application does not perform Git mutation, build, deployment, or publication. Until exact-hash
application succeeds, canonical project files remain unchanged.
## Three independent output projections
Manual rendering, portable graph publication, and the live viewer are separate:
- a manual is a declared derived HTML view over selected nodes;
- a portable graph is a content-addressed static Nodes, Flow, or Web artifact;
- the live viewer is a managed loopback process pinned to one validated index generation and can
request lazy Logic.
Their policies compose independently. Disabling one does not transfer its authority to another.
None is canonical documentation or a retrieval authority. See [Rendering and
visualization](RENDERING_AND_VISUALIZATION.md).
## Binding policy is not project truth
Capability mode, no-AST preservation, diagnostics, and projection modes describe one running
process or generated client binding. They do not rewrite the descriptor or canonical graph.
`--no-ast` forbids AST-family adapter evolution and Logic publication/retrieval for that binding.
It does not inspect parser implementation, sandbox the filesystem, or convert an existing
AST/Tree-sitter adapter into a no-AST adapter. Read [Policy precedence](POLICY_PRECEDENCE.md) and
[Legacy and no-AST operation](LEGACY_AND_NO_AST.md).
## Trust the narrowest evidence
DocForge reports stable IDs, hashes, generations, omissions, truncation, and provenance limits so a
consumer can distinguish proof from inference. A source inventory is not a semantic graph; a
syntax-level relationship is not compiler resolution; a configuration doctor is not a connection
test; a viewer snapshot is not continuous monitoring; and a benchmark is not a release.
Continue with the [Project descriptor](PROJECT_DESCRIPTOR.md), [New-project
quickstart](NEW_PROJECT_QUICKSTART.md), or [Core contract](CONTRACT.md).

View file

@ -0,0 +1,214 @@
# Incremental Adapter Indexing
DocForge Release 1 adapters return one complete immutable projection. That contract remains
supported. The incremental compiler adds an opt-in source-scoped contract that avoids reparsing
unchanged files while preserving the same validated, atomically published graph.
Import these contracts from the public `docforge.adapter_sdk` facade. See
[Legacy Adapters and No-AST Policy](LEGACY_AND_NO_AST.md) for the preserved one-method contract and
[Reference Adapters](REFERENCE_ADAPTERS.md) for the four maintained implementations.
## Release 1 compatibility
The incremental interface is additive:
- An existing adapter implementing only `load_projection()` continues to work unchanged.
- Existing generic projects, descriptors, canonical sources, changesets, and indexes require no
migration.
- Only adapters implementing both `load_manifest()` and `extract_source()` use the incremental
path.
- Incremental adapters must still implement `load_projection()` for clean rebuilds and equivalence
testing.
- Existing adapters receive identical correctness behavior but no incremental speedup until they
opt in.
## Safety model
Incremental indexing is an extraction optimization. It does not weaken publication:
1. The adapter returns a cheap, deterministic `AdapterManifest`.
2. DocForge compares every source fingerprint and extractor version with the last cache generation.
3. Added, changed, deleted, and reverse-dependent sources are invalidated.
4. The adapter reparses only invalidated sources.
5. DocForge assembles cached and refreshed contributions into a complete candidate projection.
6. The complete graph passes the same validation as a full adapter projection.
7. DocForge rereads the manifest to prove sources remained stable.
8. The extraction cache and SQLite graph are published with atomic file replacement.
An interrupted extraction never replaces the last validated SQLite index. A malformed,
incompatible, or missing cache is a cache miss, not a partial graph.
One cache generation is bounded to 10,000 source contributions and 64,000,000 encoded bytes.
Aggregate graph and Logic assembly limits are described in the
[Language Adapter Authoring Guide](ADAPTER_AUTHORING_GUIDE.md#current-aggregate-bounds).
## Adapter contract
An incremental loader implements the first, second, and fourth methods. It implements
`load_complete_assembly()` as well when it publishes Logic:
```python
class MyAdapter:
def load_manifest(self) -> AdapterManifest: ...
def extract_source(self, source: AdapterSource) -> AdapterSourceProjection: ...
def load_complete_assembly(self) -> AdapterAssembly: ...
def load_projection(self) -> AdapterProjection: ...
```
`load_projection()` remains the deterministic full-rebuild primary-graph oracle. An incremental
adapter that publishes Logic must additionally implement `load_complete_assembly()` as the
cache-independent complete graph-plus-Logic oracle.
Each `AdapterSource` declares:
- A stable source ID.
- A safe project-relative source path.
- A SHA-256 content fingerprint.
- An extractor version.
- Other source IDs whose changes can alter this source's extracted facts.
Each `AdapterSourceProjection` owns:
- Its primary graph nodes.
- Its primary graph relationships, including cross-source relationships owned by that source.
- Optional function-scoped logic projections.
Ownership must be deterministic. Two sources may not produce the same primary node or the same
function logic projection.
Some language tools emit overlapping raw evidence before ownership can be resolved. An incremental
loader may additionally implement:
```python
def assemble_projection(
manifest: AdapterManifest,
contributions: tuple[AdapterSourceProjection, ...],
) -> AdapterAssembly: ...
```
DocForge caches and invalidates the source contributions normally, then passes the complete current
contribution set to this pure assembly step. The assembler must deterministically select or merge
overlapping evidence and return one valid final graph and Logic set. It may not read hidden source
state or create a second extraction cache. The final identity, revision, and source hash must match
the manifest exactly.
Without an assembler, the stricter default remains in force: two contributions may not publish the
same primary node or Logic owner.
## Invalidation
DocForge invalidates a source when:
- It is new.
- Its fingerprint changed.
- Its extractor version changed.
- A declared dependency was added, changed, or deleted.
- Any source in its reverse-dependency chain was invalidated.
Deleted sources are omitted from the candidate projection. Their cached dependency declarations
remain available long enough to invalidate surviving dependents.
The manifest describes the current supported filesystem snapshot. It must not retain a missing file
only because Git still tracks it, and it must not require staging before a deletion disappears.
Git-backed adapters must prove that staged and unstaged deletions produce the same current source
set. DocForge then removes the omitted contribution and invalidates its surviving reverse
dependents.
If an adapter cannot precisely describe the affected sources, it should declare broader
dependencies or change its adapter/extractor version. Incorrectly retaining a stale relationship
is never an acceptable optimization.
## Adapter implementation lifecycle
Graph source changes are synchronizable. Changes to the code or configuration implementing the
adapter are not.
`AdapterProject` fingerprints the inferred or explicitly declared implementation boundary when the
project process starts. Every MCP operation validates that boundary before work begins. Changes to
implementation paths or bytes return `adapter_restart_required` with bounded added, changed, and
deleted path evidence. The error is intentionally not auto-repaired through `docforge_sync`; a
fresh process must import and validate the current adapter.
## Build reporting
`build` and `reindex` include an extraction report:
```json
{
"build": {
"mode": "incremental",
"cache_hits": 391,
"reparsed_sources": 3,
"invalidated_sources": 3,
"deleted_sources": 0,
"total_sources": 394,
"cache_hit_ids": ["..."],
"reparsed_source_ids": ["..."]
}
}
```
Adapters can call `AdapterProject.verify_incremental_equivalence()` in release and contract tests.
The check compares project identity, revision, source hash, nodes, relationships, and Logic against
the independent complete assembly. It also requires the complete assembly's primary graph to match
`load_projection()`.
## Manual changes and relationships
Changesets remain an approval queue, not a compiler queue. Compilation never silently applies a
proposal.
After explicit application:
1. The project-owned applier updates canonical files.
2. Changed manual files receive new fingerprints.
3. Incremental extraction reparses those files and affected dependents.
4. The complete candidate graph is validated and published.
5. Declared renders are regenerated.
`docforge_propose_relationship_update` queues relationship-only additions and removals without
rewriting node content. It is still bound to the changeset's complete base source hash and the
anchor node's expected content hash.
## Lazy logic boundary
`LogicProjection` stores control flow separately from the primary architecture graph. It is owned
by one function or method node and one source extraction.
Logic nodes can represent entries, conditions, basic blocks, calls, convergence points, loops,
returns, and raises. Logic edges retain relation, display label, and deterministic ordinal.
Adapters may leave
logic empty until they implement a language analyzer.
This boundary prevents thousands of boolean expressions and basic blocks from polluting Nodes,
Flow, Web, ordinary search, or architectural traversal. The Logic tab and `docforge_get_logic`
request one function-scoped projection on demand. The built-in analyzers cover Python,
JavaScript, TypeScript, and C++. Python uses the standard-library AST. JavaScript, TypeScript, and
C++ use distinct optional Tree-sitter grammars with thin language-aware control-flow profiles.
Ordinary graph reads use stored projections and do not load or execute these parsers. A grammar
alone supplies syntax, not control-flow meaning, so each new language still needs a profile for
its branch, loop, case, exception, and termination constructs. All analyzers report possible
static paths; they do not claim runtime branch outcomes.
## Manifest and warm-parser scope
Parser work is language-specific and must be measured at the correct boundary:
- Python manifest construction fingerprints source and tokenizes local imports without calling
`ast.parse`.
- JavaScript and TypeScript manifest construction lexes static relative module specifiers without
invoking their distinct Tree-sitter extraction parsers. Focused tests prove this behavior on an
unchanged warm build.
- The C++ reference manifest parses inventoried sources with `tree-sitter-cpp` to discover quoted
include dependencies. A warm C++ extraction-cache hit is not a zero-parser claim.
The maintained Python benchmark instruments the unchanged warm path and requires zero
`ast.parse` calls and zero `extract_source` calls. That exact zero-parser benchmark claim is Python
only. JavaScript and TypeScript retain focused parser-free-manifest tests; C++ deliberately does
not.
## Full rebuilds
Full rebuilds remain mandatory as a fallback and equivalence oracle. Change the adapter version,
extractor version, or cache schema whenever old cached facts are no longer valid. Removing the
confined extraction cache also forces a clean reparse without affecting canonical files.

80
docs/LEGACY_AND_NO_AST.md Normal file
View file

@ -0,0 +1,80 @@
# Legacy adapters and no-AST policy
DocForge preserves the original one-method adapter contract while offering incremental extraction,
complete graph-plus-Logic assemblies, and an independently selectable no-AST MCP policy. These are
separate compatibility boundaries.
## One-method adapters remain valid
An existing adapter that implements only:
```python
def load_projection(self) -> AdapterProjection: ...
```
continues to work. It receives the same projection validation, indexing, querying, visualization,
and MCP behavior as before. It does not receive extraction-cache speedups and publishes no Logic
through the complete assembly contract.
Incremental adoption is additive. Implement `load_manifest()` and `extract_source()` while keeping
`load_projection()` as the independent complete graph oracle. If incremental contributions publish
Logic, also implement `load_complete_assembly()` so complete and incremental graph-plus-Logic
output can be compared exactly. The public authoring surface is `docforge.adapter_sdk`; see the
[Language Adapter Authoring Guide](ADAPTER_AUTHORING_GUIDE.md).
## What no-AST means
`--no-ast` is an immutable policy on one MCP process binding. It preserves the configured adapter
strategy but forbids AST, Tree-sitter, compiler-AST, and function-Logic evolution for that binding.
It also rejects nonempty Logic before index publication and rejects a pre-existing index that
contains Logic before reads or live visualization.
Start the generic server with:
```bash
docforge-mcp --project-root /absolute/path/to/project --no-ast
```
Project-owned integrations select the same policy with
`create_project_server(..., no_ast=True)` or `create_read_only_server(..., no_ast=True)`.
`docforge_bootstrap` and `docforge_get_contract` report the effective policy, and
`docforge_get_logic` returns `adapter_policy_forbids_logic`.
Changing this policy requires a new process. It does not rewrite or reload an adapter in place.
## What no-AST does not mean
No-AST is not:
- an inspection mechanism that proves which parser an arbitrary adapter uses;
- a filesystem, Python-import, or process sandbox;
- a promise that the adapter performs no non-AST fingerprinting, dependency discovery, caching, or
complete-projection loading; or
- permission to change project files, adapter code, or the configured capability surface.
Repository permissions and project instructions remain responsible for processes that have direct
filesystem access. The MCP boundary still excludes shell execution, arbitrary file operations,
Git mutation, builds, deployment, publication, project switching, and cross-project retrieval.
## Reference adapters are not no-AST adapters
The built-in Python reference extracts syntax and Logic with the standard-library AST. The
JavaScript, TypeScript, and C++ references extract syntax and Logic with their optional
Tree-sitter grammars. Their normal projections contain Logic, so they must not be presented or
configured as no-AST adapters.
The narrower performance statements in [Reference Adapters](REFERENCE_ADAPTERS.md) concern
specific manifest and warm-cache paths. A parser-free manifest does not turn an adapter that
publishes AST-derived Logic into a no-AST adapter.
## Migration choices
Keep a one-method adapter when its complete rebuild is acceptable and no function Logic is needed.
Adopt incremental extraction when measured source parsing dominates synchronization. Adopt the
complete assembly contract when Logic is published or overlapping raw evidence needs deterministic
ownership.
Choose no-AST only when the project binding must forbid AST-derived publication and future adapter
upgrades of that kind. It is a binding policy, not a substitute for adapter confinement,
conformance tests, restart detection, cache recovery, or the full proof matrix in
[Incremental Adapter Indexing](INCREMENTAL_INDEXING.md).

View file

@ -6,29 +6,181 @@ configured `--proposal-writer`. It opens no network listener at startup. The exp
`docforge_visualize` read tool may start one token-protected loopback-only HTTP listener for the
same immutable project binding.
The additive `--diagnostics` startup option attaches a bounded version-1 request-local aggregate
to successes and structured errors. Fixed stage timings and counters expose source parsing,
adapter projection/extraction, index work, rendering work, and viewer-manager requests without
including content, paths, queries, node IDs, or SQL. Diagnostics are disabled by default. They are
the first response field discarded when the configured output limit would otherwise be exceeded.
Canonical application is a second independent startup gate. The generic server accepts
`--canonical-applier WRITER_ID`. A project adapter must also supply a compatible project-owned
canonical applier implementation.
The additive `--capability-mode read|proposal|application|operator` option selects a versioned
effective process policy. Existing factory defaults and tool ordering remain unchanged: the
read-only factory exposes the read surface, the ordinary project factory exposes the proposal
surface, and an application-enabled factory adds exact-hash application. `application` mode fails
closed unless a canonical applier is bound. `operator` is reserved for explicitly selected
operator-only tools and adds none in the current contract.
Bootstrap and contract results include `effective_policy` schema version 1 plus a separate
`capabilities` record. Policy states the requested process behavior. Capabilities state the actual
registered surface and startup-bound proposal/application access. The project descriptor remains
schema version 1 and does not silently acquire machine-specific process policy.
## Fixed reference binding
`docforge.reference_mcp` is the fixed runnable binding for the in-repository Python, JavaScript,
TypeScript, and C++ reference adapters:
```bash
python -I -m docforge.reference_mcp \
--project-root /absolute/project \
--capability-mode read
```
It loads only `.docforge/reference-adapter.toml`, selects only the fixed provider for the declared
language, and constructs the same project-bound read surface through `create_read_only_server()`.
Bootstrap binding metadata includes `server_module = "docforge.reference_mcp"`,
`adapter_mode = "reference"`, the selected `reference_language`, and the exact
`reference_config_hash`.
The reference command accepts read capability mode only. It registers all 21 read tools below and
no isolated proposal or canonical-application tools. The configuration cannot select a provider,
module, command, arguments, working directory, environment, or arbitrary discovery behavior. See
[Reference Adapters](REFERENCE_ADAPTERS.md) for language scope and
[Agent Integration](AGENT_INTEGRATION.md) for the immutable launcher and generated client
fragments.
## Read tools
- `docforge_bootstrap`
- `docforge_sync`
- `docforge_project_info`
- `docforge_get_contract`
- `docforge_get_node`
- `docforge_get_logic`
- `docforge_search`
- `docforge_filter_nodes`
- `docforge_backlinks`
- `docforge_dependencies`
- `docforge_impact`
- `docforge_get_context`
- `docforge_get_task_context`
- `docforge_validate_project`
- `docforge_render_status`
- `docforge_graph_plan`
- `docforge_graph_render_status`
- `docforge_visualize`
- `docforge_stop_visualization`
- `docforge_visualization_status`
- `docforge_get_generation_diff`
Each response states that document text is project content, not higher-priority instructions. Each
response includes project identity, revision, source hash, adapter version, and staleness state.
Every normal tool call first checks current source identity and atomically rebuilds disposable index
state when it is missing, stale, or invalid. `docforge_bootstrap` performs that synchronization and
returns the complete fixed binding, active index path, effective policy, proposal and application
capabilities, and a version-1 session contract. Bootstrap reuses the identity proven by
synchronization instead of loading the project again. Its first operation and workflow guidance
mention proposal or application tools only when the corresponding startup access is enabled.
`docforge_sync` exposes the same idempotent synchronization explicitly. Neither operation changes
canonical sources.
Search, filter, backlinks, dependencies, and impact accept explicit result limits bounded by the
project `max_results` policy. Omitted limits are still capped. Collection responses report whether
they were truncated. Traversal also reports whether truncation came from the result limit or its
deterministic candidate-edge work budget; it does not scan or materialize the complete edge table.
`docforge_get_context` accepts optional `limit` and `cursor` arguments. Its page is one deterministic
stream containing selected entries first and explicit omission evidence second. The page receipt
reports the returned count, total evidence count, whether another page exists, and an opaque
generation-bound cursor. Packing also observes the configured MCP response limit. An individually
oversized entry advances as a hash-identified `response size limit` omission so pagination cannot
loop; targeted retrieval remains available for that node. The existing three-argument custom
context-provider contract is unchanged because pagination is applied after provider selection.
`docforge_get_task_context` accepts one closed task kind (`change`, `implementation`, `failure`,
`ownership`, `test`, `operation`, or `release`), a bounded task description, and optional
`focus_node_id`, token `budget`, page `limit`, and opaque `cursor`. It derives, rather than accepts,
a version-1 retrieval plan. The plan contains only exact or lexical focus, bounded outgoing and
incoming graph traversal, and metadata hydration. It cannot request arbitrary SQL, paths, relation
names, or Logic extraction. Task context applies fixed internal ceilings of 1,000 evidence items,
100,000 examined candidate edges, and 10,000 task-query characters even when broader project
limits are configured. Traversal steps bind the complete project-owned relation vocabulary by hash
rather than copying an unbounded name list into every response.
The plan and returned context capsule are bound to the effective policy and one immutable index
generation. Every evidence item identifies its indexed source path, content hash, graph path,
additional qualifying relationship reasons, and the provenance facts that the current graph
cannot prove. Required evidence gaps distinguish an undeclared relation category, a completed
bounded search with no selected evidence, and an incomplete proof caused by a work, result, token,
or response limit. Unknown project relations remain present with their raw names and an
`unclassified_relation` limitation; DocForge never infers semantics from spelling outside the
versioned alias map.
Path relationship direction is relative to the preceding traversal node. Additional
`relationship_reasons` direction is relative to the evidence item itself: `outgoing` when that
evidence node is the stored source and `incoming` when it is the stored target.
The exact version-1 aliases are: structure (`contains`, `defined_in`, `defines`, `owns`);
implementation (`implemented_by`, `implements`, `inherits`, `inherits_from`); dependency
(`depends_on`, `imports`); execution (`activates`, `calls`, `dispatches_to`, `launches`); data
(`reads`, `writes`); evidence (`documents`, `governs`, `proves`, `tested_by`, `verifies`); and
context (`relates_to`). Every other allowed relation is `unclassified`.
Task-context continuation partitions the immutable evidence stream without changing its
`request_hash`, `plan_hash`, `collection_hash`, or `capsule_hash`. Its cursor additionally binds
the effective policy and task request. One evidence item that cannot fit advances exactly once as
a hash-identified `response_limit` omission. A changed generation, policy, plan, or collection
fails as `stale_cursor`.
`docforge_get_generation_diff` accepts only optional `limit` and `cursor` fields. It reads the one
latest version-1 primary-graph transition receipt; it does not accept arbitrary generations,
paths, or history selectors. Exact summary counts and the full item-collection hash cover the
complete transition. Pagination covers only the deterministically ordered retained details and
states separately when the fixed 1,000-item or 1 MiB publication limit permanently omitted
details.
The receipt binds project, root, adapter, index schema, from/to source identity, node and edge
hashes and counts, the committed index file identity, retained and full collection hashes, and its
own canonical hash. Node changes compare every core `Node` field. Edge identity is the exact
`(source_id, relation, target_id)` triple. Logic is excluded from public diff details.
Current pages use `page_schema_version = 1`. The nested `receipt_header` contains every stored
receipt field except `items`; its `stored_receipt_hash` is the hash of the complete stored receipt,
not of the header. Page items and hash-identified response-limit omissions are siblings of that
header. The only pagination object is at the top level, and its `next_cursor` is the only cursor
copy. The page hash covers the complete header, page items, omissions, receipt state, and
pagination receipt.
Receipt states are fail-closed: `current` is proven against cheap source identity and exact index
and receipt inodes; `stale` is a proven generation mismatch; `missing` means no receipt;
`unsafe` means confinement or file-type checks failed; `unverified` covers corrupt, foreign,
oversized, or concurrently changed evidence; and `unknown` means the project cannot provide a
cheap generation identity. Only `current` returns a page.
This status boundary is non-repairing. It never opens SQLite, loads or extracts an adapter
projection, parses source, synchronizes, builds, or writes a receipt. Cheap source identity and
stable receipt/index file identities can establish `current`; legacy projects without cheap
identity report `unknown`. Invalid or unavailable disposable evidence remains an explicit status
instead of triggering hidden recovery.
The legacy `docforge_get_context` tool and its custom three-argument provider contract remain
unchanged. A server with a custom context provider does not silently inherit the core task planner;
version 1 exposes no custom task-planner extension point. `docforge_get_task_context` returns
`task_context_unavailable` without synchronizing or loading the custom projection.
Version-1 cursors are canonical JSON encoded as base64url with a domain-separated SHA-256
corruption checksum. They are opaque and fail closed, but are not authenticated authorization
tokens. Cursors bind the project, adapter, source generation, operation parameters, collection
hash, and position. A changed generation or collection returns `stale_cursor` with
`restart_pagination`; DocForge never silently restarts at page one or combines generations.
Adapter-backed servers also validate their process-start implementation fingerprint before every
tool. `adapter_restart_required` is stale but not synchronizable. Its remediation is
`restart_project_server`; the current process does not reload project code, update Git staging, or
continue with a newly changed validator.
The normal command binds the generic project loader. An explicit project integration may instead
construct the same read-only surface from a validated `ProjectService` and project-owned context
@ -39,16 +191,23 @@ An explicit project integration may construct the full fixed surface only after
confined proposal policy and startup-bound writer. Adapter proposal validators may narrow the
writer's declared operations further. They cannot add arbitrary tools or weaken core changeset
validation. The fixed application tool is registered only through the separate canonical applier
gate.
gate. Application accepts only changesets created by the applier identity unless the project-owned
factory explicitly supplies `accepted_proposal_writers`. Every accepted identity must already be a
configured proposal writer. This allowlist permits review and acceptance across process identities;
it does not grant proposal mutation or application tools to a contributor process.
## Isolated proposal tools
- `docforge_create_changeset`
- `docforge_register_changes`
- `docforge_list_changesets`
- `docforge_get_changeset`
- `docforge_rebase_changeset`
- `docforge_abandon_changeset`
- `docforge_propose_node_create`
- `docforge_propose_node_update`
- `docforge_propose_node_move`
- `docforge_propose_relationship_update`
- `docforge_propose_node_delete`
- `docforge_validate_changeset`
- `docforge_get_changeset_diff`
@ -58,6 +217,38 @@ Proposal tools may write only below the configured changeset or isolated preview
change canonical files or declared project output. Without `--proposal-writer`, changeset mutation
tools return `proposal_access_disabled`. Validation, diff retrieval, and preview remain available
for existing changesets. A preview accepts a declared view ID, not a renderer name or command.
The relationship-update tool queues additions and removals without rewriting node content and
rejects an empty relationship list.
`docforge_register_changes` is the preferred write entry point. It creates, populates, projects,
conflict-checks, and validates one complete changeset in a single locked operation. Existing-node
operations may omit `expected_content_hash`; the server captures the current synchronized node hash
inside that transaction. The stored changeset remains fully hash-bound.
`docforge_rebase_changeset` moves a stale proposal to the current project base only when all
touched nodes, sources, relationships, permissions, and graph invariants still validate. It never
merges prose. `docforge_abandon_changeset` preserves an audit receipt without deleting the proposal.
Changeset listing returns draft and ready work by default. Stale, applied, and abandoned proposals
remain available through an explicit status or history request. Applied and abandoned proposals no
longer participate in overlap conflict detection.
Changeset list, inspection, validation, and diff reads accept optional `limit` and `cursor`
arguments. Direct Python and CLI methods still return their complete legacy result when pagination
is not requested. MCP defaults to bounded pages while preserving the exact changeset hash and
ordered operation sequence. Pages may contain fewer records than requested to remain inside the
response policy. Oversized inspection or validation pages return deterministic operation summaries
with hashes and character counts. An individually oversized diff becomes a sequence of
`canonical_json_chunk` pages; concatenating the ASCII chunks, decoding the JSON, and verifying its
payload hash reconstructs the exact `operations` and `changes` arrays without duplication.
Successful mutations return their existing full result while it fits the configured output limit.
Before any proposal, preview, or canonical mutation, the server verifies that a minimum exact
success receipt can fit. An impossible receipt fails with `result_too_large`,
`stage = "preflight"`, and `mutation_committed = false` before calling the mutation. If a successful
full result is too large, the server returns a version-1 compact receipt containing the exact
changeset ID and hash plus the operation outcome. It may fall back to a preflight-guaranteed
minimum receipt, but it never replaces a committed mutation with a failure response. Direct Python
and CLI integrations retain their detailed return values.
## Canonical application tool
@ -70,31 +261,46 @@ configured serializer.
The generic serializer confines staged Markdown/TOML writes to declared content roots and verifies
that the applied files reproduce the approved graph projection. A mismatch rolls canonical files
back. A successful apply rebuilds and checks the derived index and regenerates all declared render
views. It does not run project commands, shell, Git, builds, deployment, or publication.
back. Canonical success records an `applied` lifecycle receipt bound to the reviewed changeset hash
before refreshing derived state. Index or render refresh failures return a successful canonical
application with a degraded derived-refresh report and explicit remediation; they never invite the
caller to apply the same canonical change twice. DocForge does not run project commands, shell,
Git, builds, deployment, or publication.
When the full application result exceeds the tool-output limit, its compact success receipt retains
the applied lifecycle, exact hash, changed-source counts, and a derived-refresh summary. Detailed
index, render, and error payloads remain available through the corresponding read and status tools.
## Render boundary
`docforge_render_status` recomputes expected hashes without writing. `docforge_preview_changeset`
runs only a project-declared view through DocForge's fixed built-in renderer registry and writes one
atomic HTML file below the configured preview root. Rendering declared project output is available
only through the explicit local CLI integration command.
`docforge_render_status` reads bounded publication receipts and cheap file/source identities by
default. It does not parse canonical nodes, prepare Markdown, construct HTML, hash the complete
output, rebuild the index, or write state. Missing or corrupt receipts are conservative
`unverified` results. Callers may pass `deep = true` to explicitly request the side-effect-free
full-render equivalence oracle. `docforge_preview_changeset` runs only a project-declared view
through DocForge's fixed built-in renderer registry and writes one atomic HTML file below the
configured preview root. Rendering declared project output is available only through the explicit
local CLI integration command.
## Visualization boundary
`docforge_visualize` starts the fixed built-in `graph-browser@15` template against the currently
`docforge_visualize` starts the fixed built-in `graph-browser@17` template against the currently
validated derived index. It may focus one stable node, run one bounded lexical query, or open the
project overview. The tool returns a loopback URL and exact snapshot identity.
The tool cannot select a project, database, template, host, port, filesystem path, or SQL
expression. Its HTTP surface is token-bound, read-only, same-origin, and limited to overview,
search, exact family/authority/status/tag filtering, node-neighborhood JSON, semantic Flow,
convergence Web, and a bounded project-confined source read for one indexed node. The browser
search, exact family/authority/status/tag filtering, composable node-kind/language/capability
filtering, node-neighborhood JSON, semantic Flow,
convergence Web, lazy function-scoped Logic, and a bounded project-confined source read for one
indexed node. `docforge_get_logic` and the browser Logic endpoint accept one exact owner node ID and
return only that bounded stored projection. The browser
exposes an exact validated index snapshot. It rejects index
replacement or alteration and requires another MCP invocation to refresh.
Viewport interaction is entirely client-side: fitted neighborhood framing, wheel zoom, left-button
drag pan, explicit zoom buttons, reset, and Space-to-center selection never request or mutate
project data. Left activation visibly selects the node and opens a compact descriptor card.
project data. Left activation visibly selects the node, highlights its incident edges and direct
neighbors, mutes unrelated visible paths, and opens a compact descriptor card.
Right-click opens the full inspector. Descriptor-pill activation fills the fixed left panel with an
exact bounded category result set. The fixed right panel contains neighborhood navigation.
Replacing the current root requires an explicit Explore neighborhood action. Users may hide
@ -103,17 +309,29 @@ disconnected by a hidden node while retaining descendants that still lead to the
actions open the indexed source path
and navigate to recognized anchors. Nodes presents the
bounded neighborhood with relation-specific colors, line patterns, directional symbols, and a
visible key. Its navigation groups the focus, nodes reachable through outgoing edges, and remaining
incoming or lateral context. Flow presents semantic ancestry with relation-aware direction.
visible key. Semantic cards identify the focus and relation-derived Structure, Behavior,
Dependency, Execution, Data, Evidence, Context, and Related contributors. Cards display readable
leaf names and node kinds without truncation; complete qualified identities remain available in
tooltips and inspectors. Flow presents semantic ancestry with relation-aware direction.
Web follows bounded structural, dependency, execution, evidence, and contextual contributors into
the focus. Direct focus-owned members and execution dependencies become adjacent contributor
branches without expanding unrelated siblings. The relationship key is regenerated from each
visible view. The browser runs in a project-bound worker owned by
visible view. Logic displays possible static control paths for a focused function or method without
adding its statement-level nodes to primary search or architectural traversal. Hiding a Logic step
bridges its retained predecessors and successors with an explicit omitted path. The browser runs in
a project-bound worker owned by
the separately supervised per-user viewer manager. Standard-input transaction completion and MCP
host exit do not close the listener. Repeated visualization requests reuse the current worker while
its exact snapshot remains valid. `docforge_visualization_status` reports lifecycle state, and
`docforge_stop_visualization` explicitly stops the current project's worker. The manager reclaims a
worker only after one hour with no browser activity.
its exact snapshot remains valid. The version-2 manager protocol binds each worker to the exact
validated five-field index publication signature. Health checks compare that signature without
opening SQLite. A stale worker remains `state = running` but is never reused.
`docforge_visualization_status` reports lifecycle and freshness independently. `snapshot_state` and
top-level `staleness` are `stale` when either the index or cheap source identity is proven stale,
`current` only when both are proven current, and `unknown` otherwise. The `freshness` object exposes
the separate index and source states. Status never checks, synchronizes, or rebuilds the index and
never performs a complete project load. `docforge_stop_visualization` explicitly stops the current
project's worker. The manager reclaims a worker only after one hour with no browser activity.
## Excluded tools
@ -124,3 +342,42 @@ does not expose canonical application.
DocForge pins the official stable Python MCP SDK to the compatible `mcp>=1.28,<2` release line.
Migration to a later major release requires a separate contract and protocol compatibility review.
## Preserved no-AST bindings
An owner may start the generic MCP server with `--no-ast`:
```bash
docforge-mcp --project-root /absolute/project --no-ast
```
Project-owned integrations select the same immutable process policy with
`create_project_server(..., no_ast=True)` or `create_read_only_server(..., no_ast=True)`.
The policy preserves the current adapter extraction strategy. It does not require an adapter API
migration and does not disable complete-projection loading or non-AST incremental fingerprinting,
invalidation, and caching.
Both `docforge_bootstrap` and `docforge_get_contract` report the exact policy. MCP server
instructions tell clients not to add Python AST, Tree-sitter, compiler-AST, or function-Logic
extraction. Under this binding:
- `docforge_get_logic` returns `adapter_policy_forbids_logic`;
- a nonempty Logic projection is rejected before index publication;
- a pre-existing index containing Logic is rejected before any read or live-viewer snapshot;
- hash-bound canonical application refresh uses the same policy-bound index;
- `adapter_ast_upgrade` and `function_logic_extraction` appear as excluded operations; and
- changing the policy requires changing the process configuration and starting a new MCP process.
The policy governs the DocForge binding and conforming MCP clients. DocForge can enforce published
and indexed Logic, but it does not inspect arbitrary adapter source to prove which parser
implementation the adapter uses. DocForge still exposes no filesystem sandbox and cannot prevent
an unrelated process with direct repository write access from editing adapter files. Repository
permissions and project instructions remain responsible for that broader boundary.
The legacy `adapter_policy` object remains byte-compatible. It is now a projection of the
versioned `effective_policy`; `--no-ast` restrictively overrides adapter evolution, AST analysis,
and Logic indexing without widening any other capability.
See [Legacy Adapters and No-AST Policy](LEGACY_AND_NO_AST.md) for the compatibility boundary and
why the AST and Tree-sitter reference adapters are not no-AST adapters.

109
docs/MIGRATING_FROM_V1.md Normal file
View file

@ -0,0 +1,109 @@
# Migrating from DocForge v1
DocForge2 preserves the `docforge` distribution, Python package, CLI command, MCP tool prefix,
generic project descriptor version, and legacy adapter entry point. Migration is an additive
validation exercise, not a canonical-content rewrite.
Read [compatibility](COMPATIBILITY.md), [legacy and no-AST operation](LEGACY_AND_NO_AST.md), and
[recovery](RECOVERY_AND_PERFORMANCE.md) before changing a production binding.
## What remains compatible
- Generic Markdown and TOML projects keep `.docforge/project.toml` schema version 1.
- A custom adapter implementing only `load_projection()` remains valid.
- Existing canonical nodes, stable IDs, source paths, relationship vocabulary, changesets, and
reviewed hashes are not silently rewritten.
- Existing CLI and MCP names remain available. New fields, commands, tools, plans, policies, and
reference integrations are additive.
- `docforge-mcp --no-ast` remains the shorthand for the preserved no-AST binding.
Disposable index and cache schemas may change. Rebuild them rather than copying them as authority.
## Maintained v1.0.0 migration evidence
`make migration-m5` archives and executes the actual annotated `v1.0.0` release, then opens its
fixture, index, and active proposal through DocForge 2.0.0. The frozen tag object is
`2d7d306a37da89f1c860c7f0be161c45386acf61`; it identifies commit
`593c173b453236a6872d0a4e88e7a51a67a21cde`.
The tagged release contains an inherited identity mismatch that the migration proof records
rather than hiding: distribution metadata says `1.0.0`, while `docforge.__version__` and the MCP
server report `0.15.0`. DocForge 2.0.0 replaces that duplicated state with one authoritative
version and requires its package and server values to agree.
The maintained fixture evidence is exact:
- Canonical collection hash:
`9fde91b6b08669177d690cdf9f91b162120baee1f7e24b05fec67f56617f286a`.
- Graph snapshot hash:
`45bef8b0e1a4dac976e096dcf8f9048e1268211cd7f7638cdf99951428ec0500`.
- Source hash: `0aa6ad13a95355102300a69b2f9d06883c301d63e2624b083e45f15102dab504`.
- Active proposal hash:
`2c055dfae45443b4a4d9d4087ef70959e7293beae8d27acb14ffffe52eca7111`.
- Proposal-file hash:
`4658d494b43bd7c6cc3e3f5933a2c878c817b52bc4566e9e8429e7b8e43ca007`.
Current loading preserves all five identities and every canonical byte. It rebuilds the disposable
index from schema 1 to schema 3 without changing the graph or active proposal. All 20 tagged CLI
commands remain in the current 28-command surface, and all 25 tagged MCP tools remain in the
current 36-tool application-enabled surface. These counts describe the maintained rehearsal, not
a promise that every additive current command belongs in a legacy binding.
## Recommended migration
1. Record the v1 package version, adapter identity, project descriptor, canonical source hash,
active changesets, generated client configuration, and current rendered outputs.
2. Back up canonical sources and active proposal files. Derived `.docforge/cache`, preview,
portable-graph, and viewer state need not be authoritative backups.
3. Install the DocForge2 candidate in a separate environment. Do not repoint the production MCP
binding yet.
4. Run the existing adapter through complete loading and `ProjectIndex.check()`. For incremental
adapters, verify complete/incremental equivalence. Logic-producing incremental adapters must
implement the independent complete assembly oracle.
5. Compare project ID, adapter, revision, source hash, primary node and edge hashes, Logic policy,
and retrieval results with the v1 evidence.
6. Recompose effective capability and projection policy. Do not assume that a new default grants
proposal, application, rendering, or viewer access.
7. Rebuild disposable indexes, extraction caches, fragments, receipts, previews, and portable
artifacts from the validated candidate.
8. Generate a new client fragment. Generic projects use `docforge configure`; custom adapters use
`AdapterLauncherV1` with `generate_adapter_client_configuration()`.
9. Run doctor, real MCP bootstrap, representative retrieval, rendering status, recovery, and
exact-hash proposal/application checks in a disposable or shadow environment.
10. Repoint one production binding only after the candidate and rollback procedure pass.
## Adopting reference adapters
The repository reference adapters are narrow syntax evidence, not automatic semantic replacements
for mature v1 integrations. Python publishes modules, classes, functions, and local imports.
JavaScript and TypeScript publish syntax plus static project-relative imports and re-exports.
C++ uses a confined `compile_commands.json` as inert translation-unit inventory and publishes
directly resolvable project-local quoted includes.
Do not replace a compiler-, language-server-, or project-owned semantic adapter if the task depends
on resolved calls, types, inheritance, macros, compiler include semantics, runtime facts, or
ownership. Use [reference adapters](REFERENCE_ADAPTERS.md) as a measured starting point.
## Changesets
Existing changesets remain bound to their original project, root, base revision, source hash,
writer, operations, and exact hash. Do not edit a changeset to make it look current. Retrieve and
validate it through the candidate. Rebase only when every precondition remains true; otherwise
create and review a new proposal.
Before enabling canonical application, prove the exact serializer round trip and recovery behavior
for the selected adapter. The fixed reference MCP server is intentionally read-only.
## Rollback
Keep the v1 environment, configuration fragment, and service definition until the DocForge2
binding has passed live validation. To roll back:
1. Stop the DocForge2 MCP/viewer processes.
2. Restore the previous client or service binding.
3. Restore canonical sources only if a verified application receipt says they changed and the
adapter-specific rollback requires it.
4. Discard DocForge2-derived indexes, receipts, fragments, previews, and portable artifacts.
5. Restart v1 and verify its recorded project/source identity.
Never move a published tag or reuse a version identity for a corrected release.

View file

@ -0,0 +1,219 @@
# DocForge2 Milestone 0 baseline
Milestone 0 measures the inherited implementation before redesign. The evidence supports keeping
SQLite and targeting repeated source discovery, parsing, graph validation, and index verification
in later milestones. It does not support a speculative storage rewrite.
The maintained machine-readable result is
[`benchmarks/milestone0-2026-07-29.json`](../benchmarks/milestone0-2026-07-29.json). The harness is
[`tools/milestone0_baseline.py`](../tools/milestone0_baseline.py).
## Environment and method
- Repository revision: `fd4759096e90edb13a745621aae4872f23079357`.
- Working tree during the recorded run: clean.
- Platform: x86-64 Linux 7.1.3 with glibc 2.43.
- Python: CPython 3.14.6.
- Pytest: 9.1.1.
- Ruff: 0.16.0.
- Pyright: 1.1.411.
- Node.js: 22.22.2.
- npm: 10.9.7.
- Duration clock: `time.perf_counter_ns()`.
- Response size: UTF-8 bytes of compact, sorted JSON.
- Standalone memory: GNU `/usr/bin/time -v` maximum resident set size.
- Warning policy: repository-configured warnings as errors.
Two generic fixtures and the existing incremental contract fixture were measured. The small
`alpha` fixture has three nodes and two edges. The generated scale fixture has 1,000 Markdown
files, 1,000 nodes, 999 dependency edges, one depth-8 context profile with a 32,000-token budget,
and one manual view. All fixtures and derived artifacts were disposable and confined to `/tmp`.
No WorldForge, ScrapeStation, production project, or self-hosted DocForge data was used.
The committed 1,000-node run used ten warm samples and three cold samples. The supplementary audit
used more repetitions for short operations and separately launched processes for representative
memory and startup measurements.
## Repository-native gates
The Milestone 0 aggregate is:
```bash
make gate
```
It composes formatting, Python lint, HTML/CSS/JavaScript lint, strict types, compilation, public
contract tests, the complete warning-strict test suite, lock validation, npm dependency validation,
package building, and a disposable benchmark smoke run.
The candidate gate passed with:
- Ruff formatting and lint clean across 50 files.
- Web HTML, rendered-manual HTML, CSS, and JavaScript lint clean.
- Pyright reporting zero errors, warnings, or informational diagnostics.
- Python compilation clean.
- Public-contract gate: 8 tests and 42 schema subtests passed.
- Complete suite: 95 tests and 44 subtests passed.
- `uv lock --check` and `npm ls --all` passed.
- Wheel and source distribution built successfully.
Additional audit checks passed: `git diff --check`, parse validation for all five published JSON
schemas, and `git fsck --full`.
## Three-node baseline
These measurements show fixed overhead. They are not evidence of scale behavior.
| Operation | Median | p95 | Compact response |
|---|---:|---:|---:|
| Project open | 0.624 ms | 0.647 ms | — |
| Load, parse, validate, and fingerprint | 1.367 ms | 1.540 ms | — |
| Full index build | 4.177 ms | 4.409 ms | 665 B |
| Full index check | 1.794 ms | 1.868 ms | 665 B |
| Warm no-change synchronize | 1.812 ms | 2.075 ms | 783 B |
| Exact node | 3.800 ms | 4.095 ms | 704 B |
| Search | 3.903 ms | 4.375 ms | 1,240 B |
| Context | 5.080 ms | 5.452 ms | 1,847 B |
| Atomic manual render | 3.542 ms | 3.825 ms | 753 B |
| Render status | 1.941 ms | 2.190 ms | 756 B |
| MCP bootstrap | 3.174 ms | 3.458 ms | 2,458 B |
| MCP exact node | 3.920 ms | 4.419 ms | 894 B |
| MCP context | 5.326 ms | 5.616 ms | 2,037 B |
| Changeset registration | 2.572 ms | 3.351 ms | 1,047 B |
| Changeset validation | 2.642 ms | 2.928 ms | 1,003 B |
| Changeset diff | 2.751 ms | 3.167 ms | 1,745 B |
| Exact-hash apply and refresh | 15.989 ms | 17.558 ms | 3,455 B |
The rendered manual was 2,043 bytes. The multi-operation process peaked at 68,644 KiB RSS.
## 1,000-node maintained baseline
| Operation | Median | p95 | Compact response |
|---|---:|---:|---:|
| Project open | 0.584 ms | 0.607 ms | — |
| Load, parse, validate, and fingerprint | 128.650 ms | 132.255 ms | — |
| Cold synchronize from missing index | 692.605 ms | 696.071 ms | 870 B |
| Full index build | 277.172 ms | 289.531 ms | 661 B |
| Full index check | 148.778 ms | 157.888 ms | 661 B |
| Warm no-change synchronize | 142.479 ms | 145.926 ms | 779 B |
| Exact node | 286.306 ms | 304.226 ms | 696 B |
| Search, limit 20 | 288.793 ms | 301.227 ms | 10,312 B |
| Dependencies, depth 8 | 287.791 ms | 291.129 ms | 1,650 B |
| Impact, depth 8 | 288.280 ms | 290.636 ms | 1,650 B |
| Context, 32,000-token budget | 436.897 ms | 453.055 ms | 258,034 B |
| Atomic manual render | 280.007 ms | 283.380 ms | 757 B |
| Current manual render status | 150.591 ms | 155.833 ms | 760 B |
| MCP bootstrap | 273.410 ms | 277.901 ms | 1,720 B |
| MCP exact node | 287.094 ms | 298.913 ms | 886 B |
| MCP search, limit 20 | 289.737 ms | 307.090 ms | 10,502 B |
| MCP context | 434.853 ms | 447.168 ms | 258,224 B |
| MCP render status | 150.758 ms | 164.194 ms | 950 B |
| Changeset registration | 212.281 ms | 215.227 ms | 1,013 B |
| Changeset validation | 148.522 ms | 162.789 ms | 969 B |
| Changeset diff | 148.230 ms | 151.704 ms | 1,575 B |
| Exact-hash apply and refresh | 1,253.231 ms | 1,253.231 ms | 3,509 B |
The generated manual was 583,150 bytes. The static viewer HTML, CSS, and JavaScript totaled
105,244 bytes. The 258,224-byte MCP context result exceeds the normal 200,000-character project
limit. Under the normal policy it correctly becomes a structured `result_too_large` error rather
than a partial response.
The maintained harness reports a cumulative process high-water mark of 528,228 KiB. This includes
the entire multi-operation run and its child-process startup samples. The operation-isolated audit
is more useful for steady-state memory:
| Standalone operation | Peak RSS |
|---|---:|
| MCP import and `--help` | 65,568 KiB |
| Cold synchronization | 43,004 KiB |
| Exact CLI lookup | 39,504 KiB |
| Manual render | 39,632 KiB |
| Context compilation | 41,164 KiB |
| In-memory MCP connection and bootstrap | 79,516 KiB |
| Full 1,000-node operation harness | 78,308 KiB |
## Startup baseline
| Fresh-process operation | Median | Output |
|---|---:|---:|
| CLI `info`, 3 nodes | 80.139 ms | 386 B |
| CLI `info`, 1,000 nodes | 214.045 ms | 407 B |
| CLI exact lookup, 1,000 nodes | 369.681 ms | 814 B |
| `docforge-mcp --help` | 299.278 ms | 536 B |
| In-memory MCP create, connect, and list | 24.919 ms | 31 tools |
| In-memory MCP bootstrap, 1,000 nodes | 273.410 ms | 1,720 B |
## Incremental adapter baseline
The repository's existing two-source incremental test loader is contract evidence, not a scale
benchmark.
| Operation | Median | Maximum | Response |
|---|---:|---:|---:|
| Cold incremental build | 2.567 ms | 2.567 ms | 892 B |
| Warm build | 1.785 ms | 1.864 ms | 892 B |
| Warm no-change synchronize | 0.192 ms | 0.328 ms | 802 B |
| Exact node | 0.463 ms | 0.512 ms | 574 B |
| Full/incremental equivalence oracle | 0.203 ms | 0.341 ms | 187 B |
The cold build reparsed both sources. The warm build reported two cache hits and no invalidation or
reparse. This confirms that the incremental state and attestation path can avoid extraction. A
scaled manifest, invalidation, extraction, assembly, and publication benchmark remains missing.
## Rendering and live graph baseline
Milestone 0 has a supported `generic_html` manual renderer and a generation-pinned live graph
viewer. It does not have `ManualRenderPlan`, `GraphViewPlan`, or a portable graph renderer.
The 1,000-node manual render takes 280.007 ms and emits 583,150 bytes. Render status takes
150.591 ms because it recompiles the complete manual in memory before comparing the expected hash.
Once one validated index generation is pinned, the live viewer shows the actual SQLite read cost:
| Operation | Median | Response |
|---|---:|---:|
| Snapshot pin including validation | 143.761 ms | — |
| Overview | 1.078 ms | 977 B |
| Search, limit 20 | 1.363 ms | 10,423 B |
| Neighborhood, depth 8 | 0.425 ms | 4,891 B |
| Convergence web, depth 8 | 0.396 ms | 5,811 B |
## Measured bottlenecks
1. Generic `Project.load()` walks the source set twice, parses and validates every source, rereads
captured files for mutation detection, hashes the generation, and queries the Git revision.
2. Exact index retrieval validates twice. At 1,000 files it performs about 2,000 Markdown
front-matter parses around one bounded SQLite query.
3. Context compilation validates three times and performs about 3,000 source parses.
4. The fast attestation path applies to incremental projects. `verify_rows=False` does not make a
generic-project check cheap.
5. Full index publication intentionally reloads sources to detect concurrent mutation.
6. Missing-index synchronization compounds failed validation, locked revalidation, build, and
final validation.
7. Render status recompiles the complete manual to derive its expected hash.
8. Large response construction becomes material before SQLite retrieval does.
The profiler corroborated these paths. In the scale fixture, pinned SQLite operations remain about
0.41.4 ms while ordinary exact retrieval remains about 286 ms. Later optimization should first
remove redundant full-project work and introduce stable request snapshots or cheap source
generations. The evidence does not justify replacing SQLite.
## Recorded gaps
Milestone 0 deliberately records these missing measurements and gates:
- Compiler stages are not separately timed.
- There is no scaled incremental adapter fixture.
- No threshold policy yet defines acceptable regressions.
- There is no zero-source-parse assertion for routine warm reads.
- Per-tool MCP response-size budgets are not individually frozen.
- Manual planning is not separated from rendering.
- Portable graph planning and rendering do not exist.
- Recovery timing is not maintained for every corruption and degraded-refresh path.
- End-to-end stdio MCP request latency is not maintained beyond startup.
- Only Python 3.14 was exercised in this environment.
- The package declares MIT metadata but has no tracked standalone `LICENSE`, `COPYING`, or
`NOTICE` file.
These are inputs to later milestones. They are not permission to expand Milestone 0 into a
compiler, renderer, storage, or packaging redesign.

View file

@ -0,0 +1,102 @@
# DocForge2 Milestone 1 baseline
Milestone 1 removes repeated whole-project work from routine warm reads while retaining complete
loading and deep validation as recovery and equivalence oracles. The maintained machine-readable
evidence is
[`benchmarks/milestone1-2026-07-29.json`](../benchmarks/milestone1-2026-07-29.json), captured from
clean commit `6253c45a5eca01efa8c73ea3dfe4d85c55878ada`.
## Environment and method
- Platform: x86-64 Linux 7.1.3 with glibc 2.43.
- Python: CPython 3.14.6.
- Fixture: 1,000 Markdown files, 1,000 nodes, and 999 dependency edges.
- Samples: ten measured invocations after validated warmups.
- Duration clock: `time.perf_counter_ns()`.
- Percentile: nearest rank, so p95 is the maximum with ten samples.
- Response size: UTF-8 bytes of compact, sorted JSON.
- Process memory: cumulative main-process `RUSAGE_SELF` high-water mark.
All canonical sources, caches, indexes, changesets, renders, and viewer state were created in a
disposable temporary directory. The run did not read WorldForge, ScrapeStation, legacy DocForge
indexes, or production MCP state.
The benchmark validates every warmup and measured result. It also fails when a routine warm
operation performs a project load, parses canonical source, rebuilds an adapter projection,
extracts adapter sources, builds an index, prepares a render, constructs rendered output, or hashes
complete rendered output.
## Maintained 1,000-node results
| Operation | Median | p95 | Gate | Response |
|---|---:|---:|---:|---:|
| Warm no-change synchronization | 9.192 ms | 9.324 ms | 100 ms | 1,570 B |
| Exact node | 17.887 ms | 18.577 ms | 50 ms | 1,487 B |
| Missing-node error | 18.170 ms | 18.320 ms | 50 ms | 1,130 B |
| Search, limit 20 | 19.518 ms | 19.884 ms | 100 ms | 11,164 B |
| Filter, limit 20 | 18.274 ms | 19.534 ms | 100 ms | 8,572 B |
| Backlinks, limit 20 | 18.039 ms | 18.455 ms | 100 ms | 1,210 B |
| Dependencies, depth 8 | 18.117 ms | 19.061 ms | 100 ms | 2,559 B |
| Impact, depth 8 | 18.059 ms | 18.716 ms | 100 ms | 2,553 B |
| Context, 32,000-token budget, page 20 | 25.395 ms | 25.867 ms | 250 ms | 13,152 B |
| Current render receipt status | 18.969 ms | 19.613 ms | 50 ms | 1,603 B |
| Stale render receipt status | 18.910 ms | 19.087 ms | 50 ms | 1,551 B |
| Missing render receipt status | 17.986 ms | 18.746 ms | 50 ms | 1,352 B |
| Corrupt render receipt status | 18.007 ms | 18.511 ms | 50 ms | 1,352 B |
| Current visualization status | 9.875 ms | 10.836 ms | 50 ms | 1,446 B |
| Stale visualization status | 10.050 ms | 10.472 ms | 50 ms | 1,438 B |
| Not-running visualization status | 0.258 ms | 0.288 ms | 50 ms | 1,008 B |
| Unavailable visualization status | 0.053 ms | 0.081 ms | 50 ms | 1,138 B |
Every measured p95 passed its maintained ceiling. Exact retrieval is 15.4 times faster than the
Milestone 0 median. Warm synchronization is 15.5 times faster. The paged context response is 17.2
times faster and 19.6 times smaller than the inherited full response.
The cumulative process peak was 940,116 KiB. This is not an operation-local steady-state value. It
includes fixture construction, all benchmark phases, and Python allocator high-water behavior. It
excludes the detached visualization worker. Milestone 0's isolated subprocess measurements remain
the better evidence for per-operation steady-state memory until a maintained operation-local memory
harness is added.
## Work-proof counters
Routine retrieval and status operations recorded:
- Zero complete project loads.
- Zero canonical files or bytes parsed.
- Zero adapter projection loads and source extractions.
- Zero index builds.
- Zero render preparations, output bytes constructed, or complete output bytes hashed.
- One index check and two cheap source-generation checks for each pinned retrieval.
- One viewer-manager request for each running visualization-status query.
No-change synchronization recorded one synchronization and no build. Receipt and visualization
status recorded no hidden synchronization. The counter contract is fixed, schema-validated, and
executed by the repository gate.
## Meaning of the result
The Milestone 0 evidence showed that SQLite queries were already fast after a generation was
pinned. Milestone 1 confirms that repeated source discovery, parsing, and validation were the
dominant cost. A versioned source-generation receipt, immutable SQLite read snapshot, and bounded
indexed operations remove that cost without changing graph authority or storage.
The evidence still does not justify replacing SQLite. Complete project loading, complete index
checking, full adapter projection, and deep render validation remain independent truth and recovery
oracles.
## Known limits
- The context compiler still materializes its bounded selected graph before transport pagination.
A streaming planner requires separate scale evidence.
- Generic stat identities are cheap publication proofs, not cryptographic integrity scans.
- Legacy non-incremental adapters may not provide a cheap generation identity.
- Incremental adapter manifest, invalidation, extraction, and assembly are not yet measured at
1,000-source scale.
- Pagination cursors detect corruption and stale generations. They are not authenticated
authorization tokens.
- An individually oversized context entry is returned as explicit hash-identified omission
evidence. Targeted retrieval is required for its content.
- The maintained process peak is cumulative and excludes detached worker memory.
- Manual and graph render plans do not exist until Milestone 3.

View file

@ -0,0 +1,99 @@
# DocForge2 Milestone 1 closeout
Milestone 1 establishes a fast, observable core without changing graph meaning, canonical
authority, the supported `docforge` identity, or the legacy adapter boundary.
## Completed contracts
- Generic projects publish a versioned source-generation receipt only after complete stable
verification.
- Routine reads validate file and membership-directory identities without parsing canonical
sources.
- Every indexed read uses one immutable read-only SQLite transaction pinned between source and
index identity checks.
- Dependency validation is linear in nodes and edges and uses an iterative deterministic cycle
check.
- Search, filtering, backlinks, dependency, impact, context, and changeset review results are
bounded independently of project size.
- Version-1 cursors bind the project, adapter, canonical generation, query, collection, and
position. Corrupt and stale cursors fail closed.
- Mutation tools preflight their minimum receipt and never report `result_too_large` after a
committed operation.
- Render status uses a publication receipt and performs no hidden render, source parse, index
rebuild, or repair.
- Visualization status separates lifecycle from source and index freshness and performs no hidden
SQLite validation or source parse.
- Request-local diagnostics expose fixed bounded stage timings and work counters without recording
source text, paths, node IDs, queries, or SQL.
Complete loading, deep index checking, deep render status, and full adapter projection remain the
recovery and equivalence oracles.
## Compatibility
The distribution, import package, three executable names, existing CLI commands, existing MCP tool
names, and required arguments remain supported. New limits, cursors, deep-status switches, and
diagnostics are additive.
A one-method `load_projection()` adapter remains supported. Incremental behavior remains optional.
The no-AST binding continues to reject Logic publication and every Logic retrieval surface,
including application refresh and live visualization.
The disposable SQLite index schema is version 3. Version 2 indexes rebuild automatically. No
canonical source or stored proposal is migrated to satisfy the new index.
## Verification
The complete repository gate passed at clean commit
`6253c45a5eca01efa8c73ea3dfe4d85c55878ada`:
- Ruff formatting and lint.
- HTML, rendered-manual HTML, CSS, and JavaScript lint.
- Pyright with zero diagnostics.
- Python compilation.
- Public-contract and no-AST checks.
- 138 tests and 77 subtests under warnings-as-errors.
- Lock and JavaScript dependency-tree checks.
- Wheel and source-distribution builds.
- Milestone 0 benchmark smoke.
- Milestone 1 counter and latency smoke.
The maintained clean 1,000-node benchmark passed every latency and work-counter threshold. Exact
retrieval measured 18.577 ms p95. Paged 32,000-token context measured 25.867 ms p95. Warm no-change
synchronization measured 9.324 ms p95. Render receipt status measured 19.613 ms p95.
Visualization status measured 10.836 ms p95.
Detailed evidence is in
[`MILESTONE_1_BASELINE.md`](MILESTONE_1_BASELINE.md) and
[`benchmarks/milestone1-2026-07-29.json`](../benchmarks/milestone1-2026-07-29.json).
## Measured decisions
SQLite remains the derived retrieval store. The benchmark demonstrates that whole-source
validation around SQLite, not SQLite retrieval itself, caused the inherited latency. No speculative
storage rewrite was made.
Receipt caches remain disposable and fail closed. A missing, corrupt, incompatible, foreign, or
changed receipt falls back to the complete oracle or reports an explicit unverified state according
to the operation's safety contract.
Pagination is transport state, not project authority. It adds no database and grants no
authorization.
## Remaining weaknesses
- Context selection is bounded but not yet streaming internally.
- Scaled incremental-adapter performance remains unmeasured.
- Generic cheap generation proof uses filesystem identity rather than content hashing on every
read.
- Legacy adapters without incremental state cannot always prove current identity cheaply.
- One oversized context entry requires targeted retrieval after an explicit omission.
- Operation-local and detached-worker memory need a maintained isolated harness.
- Tree-sitter and its JavaScript and C++ grammars remain mandatory package dependencies.
- `ManualRenderPlan`, `GraphViewPlan`, and independent renderer packages remain Milestone 3 work.
## Scope confirmation
Milestone 1 did not self-host DocForge2, change WorldForge or ScrapeStation, repoint a production
MCP integration, modify the legacy Forgejo repository, create a tag, or create a release.

View file

@ -0,0 +1,85 @@
# Milestone 2 baseline
## Scope and method
This baseline records the agent-retrieval and client-integration behavior added in Milestone 2.
It was captured on 2026-07-29 from clean candidate commit
`fb0df5e4a1c591c2a84788fd4814d98550f11863`.
The maintained command was:
```bash
.venv/bin/python tools/milestone2_benchmark.py \
--nodes 1000 \
--samples 10 \
--output /tmp/docforge-milestone2-final.json
```
The fixture contains 1,000 Markdown nodes and 999 edges in a direct fan-in around one focus node.
The configured MCP response limit is 200,000 characters. Durations use
`time.perf_counter_ns()` and nearest-rank p95. Peak memory uses an isolated child process and
`RUSAGE_SELF`. Every warmup and measured invocation is validated.
Environment:
- Linux 7.1.3-200.nobara.fc44.x86_64.
- CPython 3.14.6.
- x86_64.
- Ten warm samples after one warmup.
- Isolated memory ceiling: 262,144 KiB.
The complete machine-readable result is
[`benchmarks/milestone2-2026-07-29.json`](../benchmarks/milestone2-2026-07-29.json).
## Results
| Operation | Median | p95 | Limit | Maximum response |
|---|---:|---:|---:|---:|
| Read bootstrap | 9.406 ms | 9.884 ms | 100 ms | 7,495 B |
| No-AST bootstrap | 9.153 ms | 9.379 ms | 100 ms | 8,796 B |
| Task diagnostic page | 73.326 ms | 93.115 ms | 500 ms | 7,190 B |
| Task complete traversal | 703.561 ms | 721.847 ms | 2,500 ms | 151,172 B/page |
| Generation diagnostic page | 40.961 ms | 41.650 ms | 100 ms | 66,516 B |
| Generation maximum page | 55.565 ms | 59.184 ms | 100 ms | 199,566 B |
| Generation complete traversal | 418.607 ms | 425.315 ms | 500 ms | 66,516 B/page |
| Codex configuration preview | 314.365 ms | 364.383 ms | 500 ms | 2,627 B |
| Claude configuration preview | 314.326 ms | 364.365 ms | 500 ms | 2,748 B |
| OpenClaw configuration preview | 314.401 ms | 364.532 ms | 500 ms | 2,869 B |
| Codex doctor | 0.421 ms | 0.556 ms | 100 ms | 3,595 B |
| Claude doctor | 0.364 ms | 0.446 ms | 100 ms | 3,669 B |
| OpenClaw doctor | 0.384 ms | 0.484 ms | 100 ms | 3,602 B |
Isolated peak RSS was 86,448 KiB.
Task traversal returned 108 evidence records and 892 explicit omissions across 11 pages. One
individually oversized focus record became a response-limit surrogate bound to the original record
hash. The remaining omissions were token-budget evidence. The benchmark verified every unique
subject, reconstructed the original collection hash, and matched the exact 1,000-node fixture.
Generation traversal returned all 1,000 changed-node details across 10 pages. It reconstructed the
stored retained-collection hash. The maximum generation page approached the response limit and
proved that optional diagnostics were dropped before the primary result.
## Structured-work gates
Configuration preview and doctor performed zero project loads, source parses, adapter projection
loads, adapter extraction, index checks, synchronization, index builds, render preparation,
rendered-byte construction or hashing, and viewer-manager requests.
Task-context pages performed exactly one index check and two cheap source-generation checks. They
performed none of the hidden work above. Generation-diff pages performed exactly two cheap
source-generation checks and no index check or hidden work.
## Measured limits and future notes
- Continuation is stateless and regenerates the task capsule for each page. The complete
11-page traversal remains within its gate, but later work can avoid repeated planning without
weakening generation binding.
- Configuration preview deliberately spends about 314 ms proving that the exact isolated
interpreter can import the MCP module. Discovery-only checks were rejected as unsafe.
- Claude configuration syntax is supported, but its timeout representation remains unverified.
Doctor therefore reports degraded rather than healthy.
- Doctor is a configuration inspector, not an MCP connection or SQLite integrity test.
- Legacy adapters without cheap source-generation identity report unknown for generation-diff
freshness.
- The results do not justify a storage rewrite. SQLite remains fast after one generation is pinned.

View file

@ -0,0 +1,68 @@
# Milestone 2 closeout
## Outcome
Milestone 2 is complete. One project-bound server can expose an explicit effective policy and
return compact, task-shaped, explainable context. Users can generate deterministic client
fragments and inspect their bindings without hidden runtime work.
Implemented contracts:
- Version-1 effective policy and capability-aware bootstrap.
- Version-1 retrieval plans and context capsules.
- Bounded task-context pagination with evidence gaps and explicit omissions.
- One disposable latest-generation transition receipt and paged read surface.
- Deterministic Codex, Claude, and OpenClaw standalone configuration fragments.
- Fixed-inventory read-only doctor results.
- Dedicated configuration and doctor JSON schemas.
- Repository-native Milestone 2 contract, smoke, scale, response-size, counter, and memory gates.
## Candidate evidence
The frozen implementation candidate is
`fb0df5e4a1c591c2a84788fd4814d98550f11863`.
The complete repository gate passed:
- Ruff formatting and lint.
- HTML, rendered-manual HTML, CSS, and JavaScript checks.
- Pyright with zero diagnostics.
- Warning-strict compilation and tests.
- 205 tests and 120 subtests.
- Lock and npm dependency-tree checks.
- Wheel and source-distribution builds.
- Milestone 0, 1, and 2 smoke benchmarks.
Three independent read-only adversarial audits covered client publication and policy binding,
doctor race and malformed-input behavior, and benchmark/contract evidence. Reproduced descriptor,
parent, target, filesystem, policy, secret-redaction, ambiguity, parser, response-size, and hidden
work defects were fixed and regression-tested before the candidate was frozen.
The clean ten-sample 1,000-node benchmark passed every threshold. Exact measurements and counter
ranges are recorded in
[`MILESTONE_2_BASELINE.md`](MILESTONE_2_BASELINE.md) and
[`benchmarks/milestone2-2026-07-29.json`](../benchmarks/milestone2-2026-07-29.json).
## Preserved boundaries
- The `docforge` package, imports, CLI executable, MCP executable, and existing tool names remain.
- Legacy one-method `load_projection()` adapters remain supported.
- The no-AST shorthand and legacy adapter-policy payload remain compatible.
- Project descriptor schema version 1 remains unchanged.
- No storage replacement was introduced.
- No legacy DocForge MCP or DocForge2 self-hosting was used.
- WorldForge and ScrapeStation were not touched.
- No production MCP integration was repointed.
- The legacy Forgejo repository and `legacy` remote were not changed.
- No tag, release, release announcement, or visibility change was created.
## Known follow-up work
The next active milestone may improve projection independence. It must not silently absorb these
separate future ideas:
- Avoid recomputing a complete task capsule for every continuation page.
- Add authenticated cursors only if a stronger threat model requires them.
- Verify Claude's native timeout representation.
- Add versioned adapter-owned launcher metadata before generating custom-adapter configurations.
- Keep doctor read-only; a live connection test must be an explicit separate operation.

View file

@ -0,0 +1,90 @@
# Milestone 3 baseline
## Scope and method
This baseline records the independent-projection behavior completed in Milestone 3. It was
captured on 2026-07-29 from clean candidate commit
`f5dccb5e1c312121f1af63780162f593d9363b98`.
The maintained command was:
```bash
.venv/bin/python tools/milestone3_benchmark.py \
--mode full \
--nodes 1000 \
--samples 10 \
--output benchmarks/milestone3-2026-07-29.json
```
The synthetic generic fixture contains 1,000 manual pages, 1,000 portable-graph nodes, and 999
edges. Durations use `time.perf_counter_ns()` and nearest-rank p95. In-process peak memory uses
`tracemalloc`; detached worker peak memory comes from the worker receipt and `RUSAGE_SELF`.
Every measured result is checked for deterministic semantic identity and bounded response size.
Environment:
- Linux 7.1.3-200.nobara.fc44.x86_64.
- CPython 3.14.6.
- x86_64.
- Ten samples except the one-time production cold render and fragment-cache population.
- In-process and detached-worker memory ceiling: 268,435,456 bytes.
- Detached artifact-transfer ceiling: 20,000,000 bytes.
The complete machine-readable result is
[`benchmarks/milestone3-2026-07-29.json`](../benchmarks/milestone3-2026-07-29.json).
## Results
| Operation | Median | p95 | Limit | Maximum response |
|---|---:|---:|---:|---:|
| Manual full plan/package/render | 801.948 ms | 810.490 ms | 15,000 ms | 664 B |
| Manual detached worker | 591.317 ms | 599.394 ms | 20,000 ms | 667 B |
| Fragment-assisted equivalence | 638.999 ms | 666.119 ms | 15,000 ms | 664 B |
| Production incremental cold | 2,827.152 ms | 2,827.152 ms | 20,000 ms | 668 B |
| Production incremental warm | 2,143.388 ms | 2,206.540 ms | 20,000 ms | 669 B |
| Production forced full | 944.135 ms | 978.870 ms | 20,000 ms | 668 B |
| Fragment cache miss sweep | 106.203 ms | 106.888 ms | 5,000 ms | 145 B |
| Fragment cache hit sweep | 498.911 ms | 514.324 ms | 5,000 ms | 145 B |
| Portable graph full plan/package/render | 303.736 ms | 323.690 ms | 10,000 ms | 657 B |
| Portable graph detached worker | 265.448 ms | 268.428 ms | 20,000 ms | 660 B |
| Manual receipt-only status | 62.881 ms | 111.381 ms | 500 ms | 1,542 B |
| Portable graph receipt-only status | 58.467 ms | 59.331 ms | 500 ms | 879 B |
The largest traced in-process peak was 35,160,716 bytes. The direct manual worker track peaked at
88,580,096 bytes and the portable graph worker at 89,583,616 bytes. The production manual paths,
including cold, warm, forced-full, and mutation variants, peaked at 104,771,584 bytes. Every child
peak was validated from its projection receipt against the 268,435,456-byte gate.
The manual artifact was 583,149 bytes. The portable graph artifact was 718,383 bytes. The manual
plan was 1,006,393 bytes and its ordinary package was 1,007,297 bytes. The graph plan was 398,158
bytes and its package was 398,715 bytes.
## Equivalence and no-work gates
The benchmark proved exact output equivalence for:
- Manual in-process and detached rendering.
- Manual full and fragment-assisted rendering.
- Production cold, warm, and forced-full rendering.
- Production add, change, delete, and reorder variants.
- Portable graph in-process and detached rendering.
Manual and portable-graph status each performed zero project loads, source parses, adapter
projection loads, adapter extraction, index checks, synchronization, index builds, render
preparation, output construction, output hashing, and viewer-manager requests. Each status path
performed only two cheap source-generation checks and verified committed receipt or manifest
evidence.
## Measured limits and future notes
- Fragment reuse is a correctness, isolation, and recovery boundary in this milestone. At 1,000
pages, production warm fragment validation is slower than the forced-full path. Later
optimization must start from this measurement and preserve byte equivalence.
- Full rendering remains the oracle and recovery path. Invalid, corrupt, oversized, stale, or
mismatched fragment records fall back without changing canonical facts.
- The benchmark main-process `ru_maxrss` value was 102,692 KiB. It is cumulative across all
main-process operations and is recorded only as diagnostic context. Detached child peaks are
measured separately. Per-operation traced peaks and every detached receipt peak own the memory
gates.
- The results do not justify a storage rewrite, render farm, remote renderer, or separate render
MCP.

View file

@ -0,0 +1,89 @@
# Milestone 3 closeout
## Outcome
Milestone 3 is complete. Manual compilation, portable graph rendering, and the live viewer are
separate generation-pinned consumers of the validated graph. They cannot become canonical or
retrieval authority.
Implemented contracts:
- Version-1 `ManualRenderPlan`, `GraphViewPlan`, projection package, and projection receipt.
- Strict canonical JSON identities and packaged Draft 2020-12 schemas.
- Independent manual and portable-graph renderer import boundaries.
- One isolated, fixed, one-request detached worker protocol with bounded request, response,
artifact, timeout, environment, and renderer inventory.
- Content-addressed portable graph artifacts, renderer receipts, generation/view manifests,
receipt-only status, repair, and degraded committed-publication evidence.
- Disposable semantic fragment records with bounded cache inventory, corruption recovery, and
full-render equivalence.
- Version-2 independent projection policy while preserving version-1 effective-policy behavior.
- Generation-pinned live source reads and a separate read-only viewer-manager lifecycle.
- Automated axe-tag and keyboard gates for the manual, portable graph, and live viewer.
- Repository-native contract, smoke, scale, response-size, memory, and equivalence gates.
## Candidate evidence
The frozen implementation candidate is
`f5dccb5e1c312121f1af63780162f593d9363b98`.
The complete repository gate passed:
- Ruff formatting and lint.
- HTML, rendered-manual HTML, portable-graph HTML, CSS, and JavaScript checks.
- Pyright with zero diagnostics.
- Warning-strict compilation and tests.
- 281 tests and 272 subtests.
- Three Playwright and axe accessibility flows. The alpha manual is checked with WCAG 2.0/2.1
A/AA axe tags; portable and live graph flows add WCAG 2.2 A/AA tags and keyboard interaction.
- Lock and npm dependency-tree checks.
- Wheel and source-distribution builds.
- Milestone 0, 1, 2, and 3 smoke benchmarks.
The maintained projection contract subset passed 142 tests and 236 subtests. A 10,000-node deep
chain and one 10,000-node strongly connected component prove that manual cycle planning has no
recursion-depth failure.
An isolated wheel installation passed CLI and MCP startup, a real detached manual render, and the
closed malformed-worker-request contract. Six Milestone 3 commits and the complete candidate tree
passed Gitleaks 8.30.1 with no findings.
Three independent adversarial review tracks covered manual isolation and fragment integrity,
portable publication and policy binding, and worker/accessibility/benchmark gates. Reproduced
project import, hostile environment, unbounded stdout, fragment forgery, cache growth, aggregate
overflow, coordinated policy drift, render-limit compatibility, deep-graph, and module-startup
defects were fixed and regression-tested before closeout.
The clean ten-sample 1,000-node benchmark passed every threshold. Exact measurements, equivalence
results, memory peaks, and response sizes are recorded in
[`MILESTONE_3_BASELINE.md`](MILESTONE_3_BASELINE.md) and
[`benchmarks/milestone3-2026-07-29.json`](../benchmarks/milestone3-2026-07-29.json).
## Preserved boundaries
- The `docforge` distribution, package, CLI, MCP executable, and existing tool names remain.
- The frozen alpha manual remains exactly 2,043 bytes with its legacy output hash and render
identity.
- Legacy one-method `load_projection()` adapters remain supported.
- Effective policy version 1, no-AST behavior, and existing client bindings remain compatible.
- Project descriptor schema version 1 and SQLite index schema version 3 remain unchanged.
- Configured `max_render_bytes` values above the detached transfer ceiling still load; a small
actual artifact renders normally. Actual detached transfer remains capped at 20,000,000 bytes.
- No storage replacement or self-hosting dependency was introduced.
- WorldForge and ScrapeStation were not touched.
- No production MCP integration was repointed.
- The legacy Forgejo repository and `legacy` remote were not changed.
- No tag, release, release announcement, or visibility change was created.
## Known follow-up work
Milestone 4 remains directional and is not active. Its adapter SDK and product-documentation work
must not silently absorb these separate future ideas:
- Optimize production fragment reuse only from measured profiles while preserving the forced-full
oracle.
- Add authenticated cursors only if a stronger threat model requires them.
- Verify Claude's native timeout representation.
- Add versioned adapter-owned launcher metadata before generating custom-adapter configurations.
- Keep remote render services, shared render farms, third-party renderers, storage replacement,
and self-hosting deferred until their own evidence justifies them.

View file

@ -0,0 +1,95 @@
# Milestone 4 baseline
## Scope and method
This baseline records the adapter SDK, Python reference adapter, incremental equivalence, and
recovery behavior completed in Milestone 4. It was captured on 2026-07-29 from clean executable
candidate `95271dcf2e48045b9d3aed9b9ea09c7fc155692c`.
The maintained command was:
```bash
.venv/bin/python tools/milestone4_benchmark.py \
--mode full \
--output benchmarks/milestone4-2026-07-29.json
```
The synthetic project contains 334 Python files. Each file contributes one module, one function,
and one argument node, for 1,002 primary nodes. Imports form a deterministic chain. Durations use
`time.perf_counter_ns()` and nearest-rank p95. Each ordinary result is serialized as compact sorted
JSON for its response-size gate. Per-operation memory uses `tracemalloc`; cumulative process
high-water uses `RUSAGE_SELF`.
Environment:
- Linux 7.1.3-200.nobara.fc44.x86_64 with glibc 2.43.
- CPython 3.14.6.
- x86_64.
- Three warm samples; cold, equivalence, and recovery operations run once.
- Per-operation traced-memory ceiling: 268,435,456 bytes.
- Process high-water ceiling: 536,870,912 bytes.
- Response ceiling: 524,288 bytes.
The complete machine-readable result is
[`benchmarks/milestone4-2026-07-29.json`](../benchmarks/milestone4-2026-07-29.json).
Its SHA-256 is `b6a871dde730a119fc0a138c47bd25f2c533ed3171a9a3d098075aef08173600`.
The stable evidence payload SHA-256 is
`4a8db461a311b5df4abd5aa00063e9a347d8b9dba19b30e35684f561d5271549`.
## Results
| Operation | Median | p95 | Limit | Traced peak | Response |
|---|---:|---:|---:|---:|---:|
| Cold incremental build | 1,599.760 ms | 1,599.760 ms | 20,000 ms | 68,573,540 B | 14,580 B |
| Warm incremental build | 1,244.427 ms | 1,265.387 ms | 20,000 ms | 69,997,166 B | 14,578 B |
| Complete/incremental equivalence | 1,897.919 ms | 1,897.919 ms | 30,000 ms | 64,305,919 B | 230 B |
| Corrupt extraction-cache recovery | 1,695.406 ms | 1,695.406 ms | 30,000 ms | 69,776,923 B | 14,580 B |
| Corrupt index recovery | 1,707.879 ms | 1,707.879 ms | 30,000 ms | 68,561,924 B | 14,779 B |
Process high-water was 78,798,848 bytes. The regression limits intentionally leave multiple times
the measured headroom; they are tripwires, not performance promises.
## Deterministic graph and Logic evidence
The candidate produced:
- 1,002 primary nodes with hash
`f7705dedf8dd388857a20d11f459dc74de797bcd12d3ed36e7f3aa75d67c328f`.
- 1,001 primary edges with hash
`00cc90998b6783afc8c9d1fd900409e5e3ba352868c4ba30e9bb2cbd59c52d35`.
- 334 Logic projections containing 2,338 Logic nodes and 2,338 Logic edges, with hash
`c86a3ae74777c2cec3a82c83e6e5bcca0196772ccea63cb13340fa9141593b0c`.
- Complete assembly hash
`2888182ab765fbffe3ba873c1613345640e8e6d89be74ddfca7452c0a5056345`.
- Source hash
`30ef23aa41061de2d4a7c995fe109d7a41518d9ee5493d805dd256581b47dae2`.
The independently loaded complete assembly and the incremental assembly matched exactly across
project identity, revision, source, nodes, edges, and Logic.
## Work and recovery gates
The warm build recorded all 334 sources as cache hits, zero reparsed sources, zero `ast.parse`
calls, and zero `extract_source` calls. Corrupting the extraction cache forced all sources through
extraction and reproduced the same graph and Logic hashes. Corrupting SQLite rebuilt the index
entirely from cache hits with zero parsing or extraction and reproduced those hashes.
This zero-parser evidence applies to the Python benchmark. Focused JavaScript and TypeScript tests
prove parser-free manifests. The C++ reference manifest uses Tree-sitter for bounded quoted-include
discovery and does not make a zero-warm-parser claim.
## Fresh-wheel and reference evidence
The offline adoption proof built and installed the base wheel without Tree-sitter packages,
built and checked a real Python reference project, started the isolated reference MCP server, and
performed bootstrap, search, and exact retrieval over its 21 read tools. Selecting C++ without its
extra failed with `optional_dependency_missing` and `docforge[cpp]` remediation.
Focused fixtures produced:
- Python: 14 nodes, 13 edges, 5 Logic projections.
- JavaScript: 14 nodes, 14 edges, 5 Logic projections.
- TypeScript: 13 nodes, 14 edges, 4 Logic projections.
- C++: 17 nodes, 16 edges, 5 Logic projections.
These fixture counts verify implementations; they are not language-wide completeness claims.

View file

@ -0,0 +1,83 @@
# Milestone 4 closeout
## Outcome
Milestone 4 is complete. New projects can adopt a public adapter SDK or one of four narrow
repository reference integrations, attach a fixed read-only MCP server, and follow maintained
product documentation without reading core implementation.
Implemented contracts:
- Stable `docforge.adapter_sdk` authoring imports.
- Independent complete primary-graph-plus-Logic oracle and exact incremental equivalence.
- Bounded adapter assemblies and version-1 extraction caches.
- Base Python, optional JavaScript, optional TypeScript, and optional C++ reference integrations.
- Closed `.docforge/reference-adapter.toml` and fixed `docforge.reference_mcp` read-only binding.
- Immutable, project-bound, launchable `AdapterLauncherV1` declarations and generated Codex,
Claude, and OpenClaw fragments for custom adapters.
- Live implementation-derived CLI and MCP reference tables with race-safe publication and drift
checking.
- Strict documentation graph, link, anchor, H1, reachability, required-page, generated-notice, and
documented-reference-config checks.
- Offline fresh-wheel adoption and maintained 1,002-node scale/recovery gates.
## Candidate evidence
The frozen executable candidate is
`95271dcf2e48045b9d3aed9b9ea09c7fc155692c`.
Its complete executable gate passed:
- Ruff formatting and Python lint.
- HTML, rendered-manual HTML, portable-graph HTML, CSS, and JavaScript checks.
- Pyright with zero diagnostics.
- Warning-strict compilation.
- 142 contract tests and 268 subtests.
- 347 complete tests and 402 subtests.
- Three Playwright and axe accessibility flows for the manual, portable graph, and live viewer.
- Lock and npm dependency-tree checks.
- Wheel and source-distribution builds.
- Offline fresh-wheel adoption.
- Milestone 0, 1, 2, 3, and 4 smoke benchmarks.
The clean full benchmark passed exact graph-plus-Logic equivalence, warm zero Python parser and
extraction work, corrupt-cache recovery, corrupt-index recovery, response, memory, and latency
gates. Exact results are in [the Milestone 4 baseline](MILESTONE_4_BASELINE.md) and
[`benchmarks/milestone4-2026-07-29.json`](../benchmarks/milestone4-2026-07-29.json).
Gitleaks 8.30.1 scanned the Milestone 4 commit range and candidate tree with no findings. The SSH
remote syntax prevents Gitleaks from constructing finding hyperlinks; it does not affect scanning.
## Reference scope
- Python uses the standard-library AST and publishes syntax plus local imports.
- JavaScript and TypeScript use distinct optional Tree-sitter grammars and publish syntax plus
project-local static relative imports and re-exports.
- C++ uses a confined `compile_commands.json` as inert translation-unit inventory and publishes
syntax plus directly resolvable project-local quoted includes.
- The references do not claim resolved calls, inheritance, types, symbol references, compiler
include semantics, macro semantics, runtime behavior, or semantic ownership.
The C++ reference never executes a compiler or compilation-database command. It is not a Clang
semantic adapter.
## Preserved boundaries
- The `docforge` distribution, Python package, CLI, MCP executable, and tool names remain.
- Generic projects and one-method `load_projection()` adapters remain supported.
- Descriptor schema version 1, index schema version 3, effective policy version 1, and no-AST
behavior remain.
- Heavy language frontends are optional. Base generic and Python operation installs no
Tree-sitter distribution.
- Reference MCP is read-only. Proposal and application remain explicit project-owned gates.
- No language adapter was separately published.
- No WorldForge, ScrapeStation, legacy-repository, production-binding, storage, or self-hosting
change was made.
- No tag or Forgejo release was created for Milestone 4.
## Later work
Milestone 5 owns stabilization and the first DocForge2 release. Release identity, compatibility
matrix, migration and recovery proofs, comparative real-task evidence, versioning, tagging, and
publication must be validated there. Remote adapters, render farms, third-party renderers,
cross-project graphs, storage replacement, and self-hosting remain deferred without measured need.

View file

@ -0,0 +1,189 @@
# Milestone 5 baseline
## Status and method
This baseline records the frozen DocForge `1.4.0` release lineage and the evidence used to publish
annotated tag `v1.4.0`.
The executable implementation freeze is
`d2bb95fe6190e659cf66ba57c78be53b63b53240`. Fresh-clone legacy-tag verification was added in
`97f3b6b1ae303c972387508557a3d54ea621a702`, and the maintained aggregate recovery proofs were
completed in
`2b98059b44f4d46b4d4cce776f163e893c647c76`. Documentation-bearing candidate
`49e1a87c138cdc63fb5abb85fc6eb2cf9f4a9d73` passed the complete clean local release gate. The
annotated tag points to its final documentation-only descendant after the exact fresh-clone gate.
The clean executable candidate used:
- Linux 7.1.3-200.nobara.fc44.x86_64 with glibc 2.43.
- CPython 3.14.6 on x86_64.
- The repository lockfile and offline wheel inputs.
- Maintained repository-native gates rather than a self-hosted DocForge development loop.
No WorldForge, ScrapeStation, legacy repository, production binding, or production MCP
configuration was changed.
## Quality and compatibility gates
The clean executable candidate passed:
- Ruff formatting and lint, web lint, strict Pyright with zero diagnostics, and compilation.
- 142 contract tests plus 272 subtests.
- 371 complete tests plus 419 subtests.
- Three Playwright and axe accessibility flows covering the manual, portable graph, and live
viewer.
- Lockfile, dependency-tree, wheel, source-distribution, generated-reference, and documentation
checks.
- 116 compatibility tests plus 263 subtests.
- 29 concurrency tests plus 2 subtests.
- Offline fresh-wheel adoption, exact version identity, reproducible artifact, and secret-scan
gates.
- The complete maintained Milestone 0 through Milestone 4 benchmark sequence.
The later maintained recovery proof adds four tests without changing executable product code. Its
aggregate recovery gate passes 72 tests plus 62 subtests. The final clean clone reruns the complete
totals from the documentation-bearing commit before tagging.
Compatibility remains additive:
- The `docforge` distribution and imports, CLI, generic MCP executable, MCP tool names, generic
project descriptor, result envelopes, legacy one-method adapter, and no-AST behavior remain.
- Effective policy version 1, projection policy version 2, descriptor schema version 1, and index
schema version 3 remain the current authorities.
- Complete and incremental maintained adapters produce exact primary-graph and Logic equivalence.
## Version and artifact identity
One source file owns version `1.4.0`. The following executable surfaces reported that exact
version:
- `docforge`
- `docforge-mcp`
- `docforge-viewer-manager`
- `python -m docforge.reference_mcp`
The package metadata also reported `docforge 1.4.0`, Python 3.12 or newer, the MIT license
expression, the public Forgejo repository, and its issues URL. Generated client bindings include
the version in their validated, hash-bound identity.
Two independent builds from the executable candidate and the same source-date epoch produced
identical artifacts:
| Artifact | Bytes | Executable-candidate SHA-256 |
|---|---:|---|
| Wheel | 315,811 | `c3b9bfa320d00d827154e0f858a6b970459ae0eba6ab14e1cd4f6ba470e33b6a` |
| Source distribution | 685,891 | `bb5f9333e5fa2365cf0f8ac12d7920122315b4835c9f81b7aa457eb05f1702f5` |
Both contained the license and version authority. These identify the executable freeze only.
Documentation changes alter the final release artifacts, so final checksums must be generated from
the exact tagged documentation-bearing commit.
The release channel is Forgejo only. PyPI is excluded because the `docforge` name is occupied by an
unrelated project.
## Exact migration evidence
The migration gate archives and executes the real annotated `v1.0.0` lineage:
- Tag object: `2d7d306a37da89f1c860c7f0be161c45386acf61`.
- Peeled commit: `593c173b453236a6872d0a4e88e7a51a67a21cde`.
- Canonical byte hash before and after:
`9fde91b6b08669177d690cdf9f91b162120baee1f7e24b05fec67f56617f286a`.
- Snapshot hash before and after:
`45bef8b0e1a4dac976e096dcf8f9048e1268211cd7f7638cdf99951428ec0500`.
- Active proposal hash:
`2c055dfae45443b4a4d9d4087ef70959e7293beae8d27acb14ffffe52eca7111`.
- Active proposal file hash:
`4658d494b43bd7c6cc3e3f5933a2c878c817b52bc4566e9e8429e7b8e43ca007`.
The version-1 disposable index rebuilt from schema 1 to schema 3. Canonical bytes, graph meaning,
and the active proposal remained exact. The current CLI is a 28-command superset of the version-1
20-command surface. The current generic MCP surface is a 36-tool superset of the version-1
25-tool surface.
The proof also preserves a real inherited version-1 inconsistency: package metadata reports
`1.0.0`, while its Python module and server report `0.15.0`. Migration evidence records that fact;
it does not rewrite history to make the old identities agree.
## Recovery and concurrency evidence
The maintained gates prove:
- Concurrent canonical-source changes fail closed without publishing or serving mixed
generations.
- Exact-hash canonical application uses project-owned serialization and compare-and-swap
publication, rejects stale proposals, and preserves raced source data.
- Extraction caches and SQLite indexes rebuild from canonical sources or complete adapter
projections.
- Corrupt index attestations, manual render receipts, generation-diff receipts, and portable-graph
manifests recover through their normal synchronize, rebuild, or render entry points.
- Recovery preserves exact canonical bytes, the canonical collection hash, the complete snapshot
hash, and primary-graph-plus-Logic index identity.
- Derived publication uses durable atomic replacement and directory synchronization.
The release does not claim that a multi-file canonical application is process-death atomic. A
malicious same-UID process deliberately modifying the private mode-0700 transaction directory is
also outside the application race contract. Those limits do not weaken the maintained
concurrent-canonical-source compare-and-swap proof.
## Representative real-task evidence
The real-package track uses installed, lock-pinned `markdown-it-py 4.2.0`:
- 66 Python files.
- 225,945 source bytes.
- Source-tree SHA-256:
`bd57c9f332fcf6507282ec2023e6804fce0cf844631696336ee17cbe46e63aad`.
- No network, production binding, installed-source mutation, or self-hosting.
- Python reference-adapter local-import projection for graph-assisted work.
- Independent standard-library AST import inspection for source-only work.
Both workflows returned the exact reviewed answer for all three tasks:
| Task | Graph inspected | Source inspected | Graph median | Source median |
|---|---:|---:|---:|---:|
| Renderer direct dependencies | 349 B | 10,628 B | 0.008 ms | 1.341 ms |
| Three-level HTML-block impact | 419 B | 225,945 B | 0.009 ms | 32.108 ms |
| CLI-to-core-state path | 212 B | 225,945 B | 0.013 ms | 32.142 ms |
Per-task latency excludes the separately reported one-time graph preparation. Graph-assisted
answers inspected substantially fewer agent-visible bytes and carried generation-bound
provenance. The source-only answers were smaller on the final response-size metric. The comparison
therefore records the measured tradeoff rather than claiming that every metric favors the graph.
The real-package semantic evidence SHA-256 is
`c8c19b2900fa47abdaff94f5b13fd9ca537241089edc97b30d32d9db8809a2c1`.
The combined real and generated-track evidence SHA-256 is
`45952883b47449eb4fd862b51854aa2001749928e0d37c5a6e58c2d3292fbee4`.
## Machine-readable evidence
The maintained evidence generators emit canonical compact JSON and can write it atomically with
their `--output` option:
```bash
.venv/bin/python tools/milestone5_migration.py --output /tmp/docforge-m5-migration.json
.venv/bin/python tools/milestone5_task_evidence.py \
--mode full --output /tmp/docforge-m5-task-evidence.json
.venv/bin/python tools/check_release_identity.py \
--mode full --require-clean --tag-state absent \
--output /tmp/docforge-m5-release-identity.json
.venv/bin/python tools/milestone5_fresh_clone.py \
--commit "$(git rev-parse HEAD)" \
--output /tmp/docforge-m5-fresh-clone.json
```
Migration, task, release identity, artifact hashes, and the final anonymous-clone result therefore
remain reproducible machine evidence rather than prose-only claims. The final Forgejo release
attaches the release-identity JSON beside the wheel and source distribution.
## Final release evidence
The documentation-bearing candidate passed 378 tests plus 422 subtests and all three accessibility
flows. The final documentation-only descendant is synchronized across `main`, `dev`,
`origin/main`, and `origin/dev`, then passes the maintained anonymous fresh-clone rehearsal at that
exact commit. Annotated tag `v1.4.0` and the Forgejo release identify that commit.
Final artifact checksums are generated from the tagged source and recorded in the attached
machine-readable release-identity evidence. The release includes the wheel and source
distribution. It does not publish to PyPI.

View file

@ -0,0 +1,97 @@
# Milestone 5 closeout
## Outcome
DocForge `1.4.0` completes the first DocForge2 successor release. Compatibility, migration,
determinism, concurrency, recovery, policy, projection isolation, accessibility, performance,
adoption, artifact reproducibility, and representative real-task evidence are maintained and
passing.
The documentation-bearing candidate passed its complete local release gate. Its final
documentation-only descendant passes the anonymous exact-commit clone gate and is identified by
the annotated `v1.4.0` tag and public Forgejo release.
## Candidate lineage
- Merged Milestone 4 baseline:
`6d06195950d33bcd2d712f8819bbfb3d6652ad03`.
- Frozen executable implementation:
`d2bb95fe6190e659cf66ba57c78be53b63b53240`.
- Fresh-clone legacy-tag verification:
`97f3b6b1ae303c972387508557a3d54ea621a702`.
- Maintained aggregate recovery proof:
`2b98059b44f4d46b4d4cce776f163e893c647c76`.
- Release version and tag: `1.4.0` and `v1.4.0`.
- Publication channel: public Forgejo release only.
The tag resolves to the final documentation-bearing descendant of this lineage, never to the
earlier executable-only commit.
## Closed release evidence
The clean executable release gate passed:
- Formatting, Python and web lint, strict types, compilation, lock, dependency, package-build,
generated-reference, and documentation checks.
- 142 contract tests plus 272 subtests.
- 378 complete tests plus 422 subtests.
- Three interactive accessibility flows.
- 116 compatibility tests plus 263 subtests.
- 29 concurrency tests plus 2 subtests.
- Offline fresh-wheel adoption.
- Reproducible wheel and source-distribution builds with exact version and MIT license identity.
- Gitleaks scans of reachable history and the candidate directory with no findings.
- Full maintained Milestone 0, 1, 2, 3, and 4 benchmarks.
The proof-only recovery commit adds four maintained tests. The aggregate recovery gate passes 72
tests plus 62 subtests.
Detailed migration identities, artifact evidence, recovery boundaries, and task measurements are
in the [Milestone 5 baseline](MILESTONE_5_BASELINE.md).
## Compatibility and migration result
Version `1.4.0` preserves the established distribution, imports, CLI, MCP, schema-1 descriptor,
generic-project, one-method adapter, exact-hash changeset, rendering, result-envelope, and no-AST
surfaces. New adapter, retrieval, projection, and release capabilities are additive.
The real annotated `v1.0.0` archive migrates without canonical or proposal changes. Its schema-1
index rebuilds as schema 3, and its graph identity remains exact. The current CLI and MCP
registrations are supersets of the version-1 surfaces. The proof reports the inherited version-1
metadata/runtime mismatch instead of hiding it.
## Representative task result
The pinned real-package comparison uses `markdown-it-py 4.2.0`, 66 Python files, and 225,945 bytes.
Graph-assisted and source-only workflows both return exact reviewed answers for direct
dependencies, bounded reverse impact, and a dependency path.
Graph-assisted medians were 0.008, 0.009, and 0.013 ms after one-time preparation, versus 1.341,
32.108, and 32.142 ms for source-only inspection. Graph-assisted work also reduced inspected bytes
from 10,628 to 349, from 225,945 to 419, and from 225,945 to 212. Source-only final responses were
smaller, so no universal response-size advantage is claimed.
## Preserved boundaries
- Canonical project files remain authoritative. Indexes, caches, receipts, render output, graph
output, client fragments, and viewer state remain disposable.
- Full rebuild remains the recovery and equivalence oracle.
- Derived publication is durable and atomic. Multi-file canonical application does not claim
process-death atomicity.
- `--no-ast` remains a restrictive binding policy, not a parser detector or filesystem sandbox.
- Reference adapters publish narrow static evidence and do not claim resolved calls, types,
inheritance, runtime behavior, compiler semantics, or semantic ownership.
- Reference MCP remains read-only.
- No WorldForge, ScrapeStation, legacy repository, production binding, storage, or self-hosting
change belongs to this release.
- No PyPI publication belongs to this release.
## Publication
`main`, `dev`, `origin/main`, and `origin/dev` resolve to the exact documentation-bearing release
commit. An anonymous HTTPS clone of that commit verifies the frozen annotated `v1.0.0` migration
tag and repeats the complete release gate before `v1.4.0` is created.
The public Forgejo release attaches the reproducible wheel, source distribution, and
machine-readable release-identity evidence containing their exact SHA-256 checksums. No PyPI
publication, production binding change, or legacy-repository mutation is part of this release.

View file

@ -1,6 +1,276 @@
# DocForge setup moved to the user manual
# New-project quickstart
The complete installation, project setup, visualization, CLI, MCP, application, adapter, and
troubleshooting reference now lives in the [DocForge user manual](USER_MANUAL.md).
This guide takes a new installation from an empty project binding to a validated generic manual or
one of DocForge's fixed reference source graphs. For the complete operating reference, see the
[user manual](USER_MANUAL.md).
This file remains only so existing bookmarks and links continue to resolve.
## 1. Install DocForge
DocForge requires Python 3.12 or newer. A base installation includes the generic project service,
the public adapter SDK, the Python reference integration, CLI, MCP server, renderers, and viewer
assets. It does not install Tree-sitter.
From a source checkout:
```bash
git clone <repository-url> /absolute/path/DocForge
cd /absolute/path/DocForge
uv sync --group dev
DOCFORGE=/absolute/path/DocForge/.venv/bin/docforge
DOCFORGE_MCP=/absolute/path/DocForge/.venv/bin/docforge-mcp
DOCFORGE_PYTHON=/absolute/path/DocForge/.venv/bin/python
```
For an isolated consumer environment, install the checkout or a built wheel with `uv pip install`.
Add only the language extras that project needs:
```bash
uv venv /absolute/path/docforge-env --python 3.12
uv pip install --python /absolute/path/docforge-env/bin/python /absolute/path/DocForge
uv pip install --python /absolute/path/docforge-env/bin/python \
"/absolute/path/DocForge[javascript]"
uv pip install --python /absolute/path/docforge-env/bin/python \
"/absolute/path/DocForge[typescript]"
uv pip install --python /absolute/path/docforge-env/bin/python \
"/absolute/path/DocForge[cpp]"
```
The `languages` extra installs all three optional grammar families. JavaScript and TypeScript are
separate extras because they use distinct grammars. The C++ extra supplies a syntax grammar, not a
compiler or Clang semantic frontend.
Confirm the installation:
```bash
"$DOCFORGE" --help
"$DOCFORGE_MCP" --help
```
## 2. Choose a project route
Use a generic project when Markdown or TOML documentation is canonical. Use a fixed reference
adapter when you want a bounded source inventory and syntax-level graph for Python, JavaScript,
TypeScript, or C++. Use a project-owned adapter when the production contract must supply richer
semantics.
- Generic manual: continue with [Create a generic project](#create-a-generic-project).
- Fixed source example: continue with [Use a reference adapter](#use-a-reference-adapter).
- Production language frontend: follow the [Adapter authoring guide](ADAPTER_AUTHORING_GUIDE.md).
## Create a generic project
Set the absolute project root and assess it without writing:
```bash
PROJECT=/absolute/path/MyProject
"$DOCFORGE" --project-root "$PROJECT" onboard
```
The assessment reports detected languages, build evidence, documentation candidates, current
configuration, and available capabilities. Detection never invents a source graph.
Create a generic starter explicitly:
```bash
"$DOCFORGE" --project-root "$PROJECT" onboard \
--scaffold \
--project-id my-project \
--title "My Project"
```
Scaffolding is create-only and refuses existing target files. It creates:
- `.docforge/project.toml`, the generic project descriptor;
- `docs/docforge/content/architecture-overview.md`, one canonical node;
- `.docforge/templates/manual.html`, one built-in-renderer template;
- derived index, receipt, and rendered output below `.docforge`.
If a language was detected, source graph status remains `adapter_required`. The generic starter
does not claim source semantics.
Validate and inspect it:
```bash
"$DOCFORGE" --project-root "$PROJECT" validate
"$DOCFORGE" --project-root "$PROJECT" reindex
"$DOCFORGE" --project-root "$PROJECT" search architecture
"$DOCFORGE" --project-root "$PROJECT" show architecture.overview
"$DOCFORGE" --project-root "$PROJECT" render-status
```
The descriptor is explained field by field in [Project descriptor](PROJECT_DESCRIPTOR.md). Add
canonical nodes only after choosing their authority, stable IDs, families, statuses, and allowed
relationships; [Core concepts and authority](CORE_CONCEPTS_AND_AUTHORITY.md) defines those terms.
### Start a generic MCP binding
Start read-only first:
```bash
"$DOCFORGE_MCP" \
--project-root "$PROJECT" \
--capability-mode read
```
Call `docforge_bootstrap` before other tools. It reports the exact project identity, generation,
effective policy, projection policy, available capabilities, and recommended first read.
To enable proposals, the descriptor must declare the writer and the process must select it:
```bash
"$DOCFORGE_MCP" \
--project-root "$PROJECT" \
--capability-mode proposal \
--proposal-writer project-editor
```
Canonical application is a separate startup gate. Do not add it to a read-only client:
```bash
"$DOCFORGE_MCP" \
--project-root "$PROJECT" \
--capability-mode application \
--proposal-writer project-editor \
--canonical-applier project-editor
```
Application accepts one exact reviewed changeset hash. It does not commit, push, build, deploy, or
publish the project.
### Generate a generic client fragment
Preview is side-effect free:
```bash
"$DOCFORGE" configure codex --project "$PROJECT"
"$DOCFORGE" configure claude --project "$PROJECT"
"$DOCFORGE" configure openclaw --project "$PROJECT"
```
Add `--output /absolute/path/new-fragment` to create one new private standalone file. Generation
does not merge with or replace a different existing file. Diagnose an installed binding with:
```bash
"$DOCFORGE" doctor --client codex --project "$PROJECT"
```
Doctor is a bounded configuration inspector, not a connection test. See [Agent
integration](AGENT_INTEGRATION.md) for client-specific layouts and limitations.
## Use a reference adapter
Reference adapters read one fixed descriptor:
`.docforge/reference-adapter.toml`. The path is not selectable.
Create a Python project configuration:
```toml reference-adapter
schema_version = 1
project_id = "my-python-project"
title = "My Python Project"
language = "python"
source_roots = ["src"]
```
JavaScript uses `language = "javascript"` and the `javascript` extra. TypeScript uses
`language = "typescript"` and the `typescript` extra.
C++ additionally requires a confined compilation database:
```toml reference-adapter
schema_version = 1
project_id = "my-cpp-project"
title = "My C++ Project"
language = "cpp"
source_roots = ["src", "include"]
compilation_database = "compile_commands.json"
```
The C++ integration reads `compile_commands.json` only as bounded translation-unit inventory and
fingerprint evidence. It never executes a recorded command or compiler.
Run the fixed read-only server through the same installed Python interpreter:
```bash
"$DOCFORGE_PYTHON" -I -m docforge.reference_mcp \
--project-root "$PROJECT" \
--capability-mode read
```
The binding chooses one in-package provider from the descriptor language. It accepts no provider
module, command, argument list, working directory, environment, discovery rule, proposal writer,
or canonical applier. Its MCP surface is the read subset documented in the [generated command
reference](COMMAND_REFERENCE.md).
### Generate a reference-adapter client fragment
Generic `docforge configure` intentionally refuses custom adapters. Construct the adapter project
and immutable launcher, then call the custom-adapter generator:
```python
from pathlib import Path
from docforge.adapter_launcher import AdapterLauncherV1
from docforge.client_config import generate_adapter_client_configuration
from docforge.reference_mcp import REFERENCE_MCP_MODULE, create_reference_project
root = Path("/absolute/path/MyProject").resolve(strict=True)
project = create_reference_project(root)
launcher = AdapterLauncherV1.for_project(project, module=REFERENCE_MCP_MODULE)
preview = generate_adapter_client_configuration(
project,
launcher,
"codex",
capability_mode="read",
)
print(preview["artifact"]["content"])
```
Use `client="claude"` or `client="openclaw"` for those formats. Pass an absolute `output` path only
when creating a new standalone private fragment. The generated launch is bound to the selected
project, descriptor hash, adapter identity, installed module, isolated interpreter, policy, and
source availability evidence.
Read [Reference adapters](REFERENCE_ADAPTERS.md) before relying on the graph. The Python example
publishes local imports, the JavaScript and TypeScript examples publish project-local static
relative imports and re-exports, and the C++ example publishes directly resolvable project-local
quoted includes. None is a complete semantic compiler frontend.
## 3. Add visualization only when needed
Install the per-user viewer manager once:
```bash
/absolute/path/DocForge/.venv/bin/docforge-viewer-manager install-user-service
```
Then start a project-bound snapshot:
```bash
"$DOCFORGE" --project-root "$PROJECT" visualize
"$DOCFORGE" --project-root "$PROJECT" visualization-status
```
The listener is loopback-only and tokenized. The viewer is a derived snapshot, not canonical
authority and not a continuously monitored filesystem view. Read [Rendering and
visualization](RENDERING_AND_VISUALIZATION.md) for manual, portable, and live-viewer differences.
## 4. Verify the maintained checkout
Contributors can run:
```bash
make command-reference-check
make docs-check
make adoption-m4
make benchmark-m4-smoke
```
Run `make gate` before a release candidate. `benchmark-m4-smoke` is routine coverage;
`benchmark-m4` is the maintained full adapter workload.
Continue with [Project onboarding](PROJECT_ONBOARDING.md) for production integration,
[Policy precedence](POLICY_PRECEDENCE.md) before widening a binding, and [Security](SECURITY.md)
before exposing any MCP process.

203
docs/POLICY_PRECEDENCE.md Normal file
View file

@ -0,0 +1,203 @@
# Policy precedence
DocForge composes an immutable effective policy for each project-bound process. It resolves
restrictions in a fixed order:
```text
core safety
> explicit binding
> no-AST shorthand
> resource availability
```
A lower layer can make a requested operation unavailable; it cannot override a higher-layer
prohibition. Bootstrap and generated client evidence report the composed result and hashes so a
client does not need to infer policy from command-line arguments.
## 1. Core safety
Core safety is unconditional. No capability mode exposes:
- arbitrary file access or project switching;
- arbitrary renderer, module, command, shell, argument, working-directory, or environment
selection;
- Git mutation;
- project builds or compiler execution;
- deployment or publication.
Canonical application is limited to DocForge's validated serializer boundary. Adapter launch is
limited to the immutable launcher contract. Rendering is limited to declared views and fixed
built-in workers. An indexed instruction cannot alter any of these rules.
## 2. Explicit binding
A process is bound at startup to one project root, descriptor, adapter, capability mode, writer and
applier identities when present, no-AST selection, diagnostics selection, and projection modes.
The binding does not change during the process lifetime.
Capability modes are:
- `read`: register only the read surface;
- `proposal`: add proposal tools when a valid selected writer is available;
- `application`: require a startup-bound canonical applier and expose exact-hash application;
- `operator`: reserved; it currently adds no tools.
Mode describes the maximum registered surface. Actual authority can be narrower. A descriptor must
declare the selected writer, including allowed families and operation types. Application defaults
to a matching configured writer, changeset creator, and canonical-applier identity. A project-owned
server may explicitly authorize its applier to accept changesets from additional configured writers.
A mode name cannot create a missing descriptor grant or extend that accepted-writer allowlist.
Generic generated client fragments default to read mode. Other construction paths preserve their
documented compatible factory defaults. Treat `docforge_bootstrap.session_contract` and its actual
capabilities as authoritative for a running server.
## 3. No-AST shorthand
`--no-ast` is a restrictive compatibility shorthand. It composes:
- adapter evolution `preserve`;
- AST analysis `forbidden`;
- Logic indexing `off`;
- `docforge_get_logic` blocked;
- prohibitions on AST, Tree-sitter, compiler-AST, and function-Logic upgrades.
Non-AST source fingerprinting and incremental caching remain allowed. Existing one-method adapters
continue to use `load_projection()`.
The shorthand does not inspect how an existing adapter was implemented, sandbox its filesystem
reads, or transform an AST-based adapter into a no-AST adapter. Do not run the Python,
JavaScript/TypeScript, or C++ syntax reference integrations and then describe the binding as a
proved no-AST integration. See [Legacy and no-AST operation](LEGACY_AND_NO_AST.md).
## 4. Resource availability
Even an allowed policy cannot create a missing resource:
- application mode requires a startup-bound canonical applier;
- manual `explicit` or `auto` requires declared manual render configuration;
- manual `auto` also requires canonical application in the current operation or server;
- portable graph `explicit` requires declared portable graph configuration;
- live viewer `on-demand` requires the viewer runtime;
- an optional reference language requires its installed extra;
- a project-owned launcher module must be installed, isolated, project-owned, and unchanged.
Unavailable requested modes fail before hidden work. Missing optional language grammars return an
actionable install target such as `docforge[typescript]` or `docforge[cpp]`; DocForge does not
silently downgrade to a different frontend.
## Effective process policy
The version-1 effective policy reports:
- capability mode and whether it came from a factory default or explicit selection;
- adapter evolution, AST analysis, and Logic indexing;
- automatic synchronization and validated integrity;
- the compatible manual-render projection;
- profiling state;
- blocked tools and prohibitions;
- the exact precedence list.
This version-1 projection preserves compatibility. It is not the complete version-2 rendering
policy; portable graph and live-viewer choices are reported separately.
## Independent projection policy
Manual rendering, portable graph publication, and live visualization use a separate immutable
version-2 policy:
```text
manual: auto | explicit | disabled
portable_graph: explicit | disabled
live_viewer: on-demand | disabled
```
When a selector is omitted, composition uses availability-aware defaults:
- manual is `auto` only when manual configuration and canonical application are both available;
otherwise it is `explicit` when configured, or `disabled`;
- portable graph is `explicit` when configured, otherwise `disabled`;
- live viewer is `on-demand` when the runtime is available, otherwise `disabled`.
An explicit non-disabled selection for a missing resource returns
`projection_policy_unavailable`. An active operation prohibited by the selected policy returns
`projection_policy_forbids_operation` before planning, rendering, or viewer startup.
Status remains intentionally narrower than active work. Manual and portable receipt-only status
are available when their active operations are disabled. Viewer status and explicit stop remain
available when viewer start is disabled.
Ordinary standalone CLI rendering uses manual `explicit`. Manual `auto` belongs to a canonical
application operation that owns automatic regeneration.
## Diagnostics do not grant authority
`--diagnostics` enables bounded request-local stage timings and compiler-work counters. It does not
enable tools, broaden paths, retain project content, or displace a primary MCP result that already
needs the response budget.
## Descriptor policy and process policy
The project descriptor is canonical project configuration. The process policy is a runtime
restriction. They compose by intersection:
```text
operation is available
only if core permits it
and the startup binding registers it
and no-AST permits it
and required resources exist
and the descriptor grants the requested project authority
and current graph/hash preconditions validate
```
Changing a descriptor does not retarget a running project-owned process. Descriptor, adapter
implementation, or launcher drift requires a fresh process.
## Common decisions
For an agent that only reads documentation:
```bash
docforge-mcp \
--project-root /absolute/path/MyProject \
--capability-mode read
```
For an agent that may prepare reviewable proposals:
```bash
docforge-mcp \
--project-root /absolute/path/MyProject \
--capability-mode proposal \
--proposal-writer project-editor
```
For a tightly controlled application process:
```bash
docforge-mcp \
--project-root /absolute/path/MyProject \
--capability-mode application \
--proposal-writer project-editor \
--canonical-applier project-editor
```
For a read binding with every active output projection disabled:
```bash
docforge-mcp \
--project-root /absolute/path/MyProject \
--capability-mode read \
--manual-render-policy disabled \
--portable-graph-policy disabled \
--live-viewer-policy disabled
```
Prefer the narrowest binding that completes the workflow. Call `docforge_bootstrap` first and use
the returned effective policy, projection policy, actual capabilities, and prohibitions rather
than assumptions based on client configuration.
See the [MCP contract](MCP_CONTRACT.md), [Security](SECURITY.md), [Project
descriptor](PROJECT_DESCRIPTOR.md), and [Rendering and
visualization](RENDERING_AND_VISUALIZATION.md).

279
docs/PROJECT_DESCRIPTOR.md Normal file
View file

@ -0,0 +1,279 @@
# Project descriptor
A generic DocForge project is selected by one fixed file:
`.docforge/project.toml` beneath an explicit project root. The descriptor is schema version 1.
DocForge validates both the JSON-schema shape in `schemas/project.schema.json` and runtime
invariants that schema alone cannot prove.
Reference integrations use a different fixed descriptor,
`.docforge/reference-adapter.toml`; see [Reference adapters](REFERENCE_ADAPTERS.md).
Project-owned adapters construct the same runtime `ProjectDescriptor` contract through the public
adapter SDK.
## Complete generic example
```toml
schema_version = 1
project_id = "my-project"
title = "My Project"
adapter = "generic"
[sources]
content_roots = ["docs/docforge/content"]
authority_files = []
[derived]
cache_root = ".docforge/cache"
index = ".docforge/cache/index.sqlite3"
[changesets]
root = ".docforge/changesets"
[[changesets.writers]]
id = "project-editor"
families = ["architecture", "operations", "system"]
operations = ["create", "update", "move", "delete"]
[render]
template_root = ".docforge/templates"
preview_root = ".docforge/previews"
[[render.views]]
id = "manual"
renderer = "generic_html"
template = "manual.html"
output = ".docforge/rendered/manual.html"
title = "My Project Manual"
families = ["architecture", "operations", "system"]
[graph_render]
output_root = ".docforge/portable-graph"
[[graph_render.views]]
id = "architecture"
renderer = "portable_graph_html"
output = "architecture.html"
title = "Architecture"
root = "architecture.overview"
initial_mode = "web"
depth = 3
max_nodes = 250
max_edges = 1000
max_work = 100000
families = ["architecture", "system"]
relations = ["depends_on", "owns", "calls", "reads", "writes", "tested_by", "relates_to"]
authorities = []
statuses = ["current", "active", "verified"]
tags = []
include_logic = false
[graph]
allowed_relations = [
"calls",
"depends_on",
"owns",
"reads",
"relates_to",
"tested_by",
"writes",
]
[limits]
max_source_bytes = 500000
max_nodes = 10000
max_query_chars = 500
max_results = 100
max_traversal_depth = 6
max_context_tokens = 12000
max_tool_output_chars = 200000
max_changesets = 100
max_changeset_operations = 100
max_changeset_bytes = 1000000
max_render_views = 20
max_template_bytes = 1000000
max_render_bytes = 1000000
[[profiles]]
id = "development"
families = ["architecture", "operations", "system"]
statuses = ["current", "active", "verified"]
required_nodes = ["architecture.overview"]
token_budget = 8000
dependency_depth = 3
```
Rendering sections are optional. Profiles and limits may also be omitted; runtime defaults then
apply. The required top-level fields are `schema_version`, `project_id`, `title`, `adapter`,
`sources`, `derived`, `changesets`, and `graph`.
## Identity fields
`schema_version` must be `1`.
`project_id` is the stable machine identity. It is lowercase and may contain digits, dots,
underscores, and hyphens after its first character. Do not derive it from a mutable display title.
`title` is the human-readable project name.
`adapter` is `generic` for this file format. Project-owned adapters publish a validated
`adapter_id@adapter_version` identity through their runtime descriptor; changing adapter identity
invalidates incompatible derived state.
The descriptor's byte content contributes to a descriptor hash. Client fragments, launchers, index
evidence, and policy results use that hash to detect drift.
## Canonical sources
`sources.content_roots` lists the project-relative directories containing generic Markdown and TOML
nodes. Each path must resolve beneath the project root. Canonical content roots may not overlap the
derived cache.
`sources.authority_files` lists additional project-relative regular files whose content belongs to
the canonical project identity. They are not automatically parsed as nodes.
A Markdown node contains one TOML metadata block followed by Markdown content:
```markdown
+++
schema_version = 1
id = "architecture.overview"
title = "Architecture overview"
family = "architecture"
authority = "authoritative"
status = "current"
tags = ["architecture"]
summary = "Defines the top-level architecture and ownership."
+++
# Architecture overview
Describe systems, ownership, runtime flow, failure behavior, and proof.
```
Every node ID is project-wide and stable. Every relationship target must resolve. A TOML source may
contain multiple `[[nodes]]` records; proposal-enabled multi-node files need stable
`source_anchor` values where creation or movement requires an exact record boundary.
## Derived state
`derived.cache_root` owns disposable indexes, attestations, extraction caches, render receipts,
projection artifacts, and viewer registry state.
`derived.index` must be inside `derived.cache_root`. Canonical content and cache paths must not
overlap.
Derived state is not a backup. If it is deleted or rejected as corrupt, DocForge rebuilds it from
validated canonical sources.
## Changesets and writers
`changesets.root` is the confined proposal store. It must not overlap canonical content or the
derived cache.
Each `changesets.writers` entry declares:
- a stable writer `id`;
- the node `families` that writer may change;
- allowed `operations`: `create`, `update`, `move`, and/or `delete`.
The descriptor grant is necessary but not sufficient. A process must also select that writer at
startup, and canonical application requires a separately bound matching applier. Capability mode
does not broaden the descriptor grant. See [Policy precedence](POLICY_PRECEDENCE.md).
## Allowed relationships
`graph.allowed_relations` is the exact project vocabulary accepted on edges. It must be nonempty.
Relationship names are stable IDs. DocForge rejects relationships that are not declared and gives
`depends_on` additional cycle validation.
The core does not reinterpret a custom relationship just because its spelling resembles a known
term. Task-context retrieval classifies only the documented versioned aliases and preserves
unknown allowed relationships as `unclassified`.
## Context profiles
Each `profiles` entry defines one bounded context compilation:
- `id` selects the profile;
- `families` and `statuses` filter eligible nodes;
- `required_nodes` names stable nodes that must be present;
- `token_budget` limits compiled content;
- `dependency_depth` bounds relationship expansion.
Profiles choose derived retrieval scope. They do not change node authority or writer permissions.
## Limits
Positive limits bound input, graph, retrieval, proposal, and rendering work. Current defaults are:
- `max_source_bytes = 1000000`
- `max_nodes = 10000`
- `max_query_chars = 500`
- `max_results = 100`
- `max_traversal_depth = 8`
- `max_context_tokens = 32000`
- `max_tool_output_chars = 200000`
- `max_changesets = 1000`
- `max_changeset_operations = 100`
- `max_changeset_bytes = 1000000`
- `max_render_views = 100`
- `max_template_bytes = 1000000`
- `max_render_bytes = 1000000`
Smaller project limits are useful policy. They cannot widen fixed internal worker, package,
response, or cache ceilings. In particular, detached renderer transfer has its own fixed boundary
even if a compatibility descriptor retains a larger `max_render_bytes`.
## Manual rendering
The optional `render` section declares:
- one confined `template_root`;
- one isolated `preview_root`;
- one or more stable views.
Each view uses the built-in `generic_html` renderer, a template beneath `template_root`, one
declared output path, a title, and a family filter. Paths may not overlap canonical content,
authority files, changesets, cache, templates, or previews in unsafe ways.
Templates are inert UTF-8 files with a fixed token vocabulary. They cannot select executable
renderers or commands. See [Rendering and visualization](RENDERING_AND_VISUALIZATION.md).
## Portable graph rendering
The optional `graph_render` section declares an output root and one or more
`portable_graph_html` views. Each view selects exactly one:
- `root`, an exact stable node ID; or
- `query`, a bounded metadata-only lexical seed.
It may then restrict families, relations, authorities, statuses, and tags, plus depth, node, edge,
and work limits. `initial_mode` is `nodes`, `flow`, or `web`. Portable graph contract version 1
requires `include_logic = false`.
The declared `output` is relative to `graph_render.output_root`.
## Paths and confinement
Descriptor paths are project-relative. Absolute paths and parent traversal are rejected. Runtime
validation also rejects symlink escapes, unexpected file types, unsafe overlap, changing path
identity during sensitive reads or publication, and derived outputs outside their declared roots.
The explicit CLI `--project-root` is the only project selector. The MCP process binds it at startup
and exposes no project-switching tool.
## Validate changes safely
After editing the descriptor:
```bash
docforge --project-root /absolute/path/MyProject validate
docforge --project-root /absolute/path/MyProject reindex
docforge --project-root /absolute/path/MyProject check
```
Descriptor or adapter implementation drift makes a project-owned running process fail closed; start
a fresh process after changing those boundaries.
For first-time creation, prefer the create-only [New-project
quickstart](NEW_PROJECT_QUICKSTART.md). For full invariants, read the [Core contract](CONTRACT.md).

350
docs/PROJECT_ONBOARDING.md Normal file
View file

@ -0,0 +1,350 @@
# Project onboarding
DocForge onboarding has two separate outcomes:
1. A generic manual can be configured, indexed, rendered, visualized, and exposed through the MCP.
2. A source graph additionally requires one validated language frontend per source language.
DocForge ships narrow fixed references for Python, JavaScript, TypeScript, and C++. They are useful
for syntax-scoped projects and adoption proof, but production semantic requirements may still
require a project-owned compiler or language-service adapter. Review
[Reference Adapters](REFERENCE_ADAPTERS.md) before selecting a frontend.
The onboarding command never claims that source semantics exist merely because it found source
files. It reports each detected language as `adapter_required` until a project integration supplies
and proves that frontend.
## Start with a read-only assessment
```bash
docforge --project-root /absolute/path/MyProject onboard
```
The assessment:
- detects common source languages and build-system evidence;
- excludes version-control, dependency, generated, cache, and build directories;
- inventories likely documentation;
- reports whether the project is already configured;
- states which capabilities are ready and which still need an adapter;
- does not create or modify files.
Limit detection to one or more known profiles when automatic discovery is not appropriate:
```bash
docforge --project-root /absolute/path/MyProject onboard --language rust
docforge --project-root /absolute/path/MyProject onboard --language java
docforge --project-root /absolute/path/MyProject onboard \
--language cpp \
--language typescript
```
Current profile IDs are `c`, `cpp`, `csharp`, `go`, `java`, `javascript`, `kotlin`, `lua`, `php`,
`python`, `ruby`, `rust`, `scala`, `swift`, and `typescript`. A profile recognizes project
evidence. It is not itself a parser.
## Scaffold a generic manual
After reviewing the assessment:
```bash
docforge --project-root /absolute/path/MyProject onboard \
--scaffold \
--project-id my-project \
--title "My Project"
```
Scaffolding creates:
- `.docforge/project.toml`;
- `.docforge/templates/manual.html`;
- `docs/docforge/content/project-overview.md`;
- the derived SQLite index;
- the rendered starter manual.
The command refuses to replace any existing target. Canonical files are written before the
descriptor, and a failed write removes files created by that attempt. The configured manual is
immediately usable through the generic CLI, viewer, and MCP.
The starter overview records detected languages and states that the source graph is unavailable
until a language frontend passes the adapter proof. That limitation is deliberate.
## Configure a fixed reference adapter
The generic onboarding scaffold and fixed reference configuration are separate project routes.
For a syntax-scoped reference project, create `.docforge/reference-adapter.toml`:
```toml
schema_version = 1
project_id = "my-python-project"
title = "My Python project"
language = "python"
source_roots = ["src"]
```
Then start the fixed read-only binding:
```bash
python -I -m docforge.reference_mcp \
--project-root /absolute/path/MyProject \
--capability-mode read
```
The configuration selects only a fixed in-repository provider and cannot name a command or custom
module. JavaScript, TypeScript, and C++ require their respective optional extras; C++ also requires
`compilation_database = "compile_commands.json"`. The complete configuration and supported-fact
contract are in [Reference Adapters](REFERENCE_ADAPTERS.md).
## Complete onboarding checklist
### 1. Repository assessment
- [ ] Resolve one explicit project root.
- [ ] Detect version-control and worktree boundaries.
- [ ] Detect source languages and build systems.
- [ ] Find existing manuals, design notes, API references, plans, and proof records.
- [ ] Exclude vendored, generated, dependency, cache, and build trees.
- [ ] Estimate source, documentation, and expected graph size.
- [ ] Report missing tools without changing the repository.
- [ ] Review the assessment before scaffolding.
Done when authored source is distinguishable from disposable and external files.
### 2. Identity and authority
- [ ] Assign a stable project ID and title.
- [ ] Declare canonical content roots.
- [ ] Declare authority files.
- [ ] Declare derived cache, changeset, preview, template, and render roots.
- [ ] Define documentation families and allowed relationships.
- [ ] Define proposal writers and operations.
- [ ] Decide which views are public, internal, or restricted.
- [ ] Keep source mutation disabled unless separately designed and authorized.
Done when every durable documentation fact has one authoritative source and every derived output
can be deleted without losing that fact.
### 3. Manual foundation
- [ ] Scaffold or adapt `.docforge/project.toml`.
- [ ] Create at least one authoritative overview node.
- [ ] Assign stable node IDs, families, authorities, statuses, tags, and summaries.
- [ ] Import existing documents without silently changing their meaning.
- [ ] Separate current implementation, approved plans, proposals, and history.
- [ ] Define bounded context profiles for common development tasks.
- [ ] Validate, index, render, and visualize the manual.
Done when every rendered passage can be traced to a canonical source.
### 4. Language frontend selection
For every source language:
- [ ] Select or implement one frontend.
- [ ] Record its frontend and extractor versions.
- [ ] Define source discovery from authoritative build information.
- [ ] Define stable symbol identities.
- [ ] Define ownership for shared or generated declarations.
- [ ] Define supported node kinds and relationships.
- [ ] Define dependency discovery.
- [ ] State unsupported semantic facts explicitly.
Decide whether the project needs production semantic evidence or the narrower syntax-only
reference scope. The Python reference publishes only project-local imports. JavaScript and
TypeScript publish only project-local static relative imports and re-exports. The C++ reference
publishes only directly resolvable project-local quoted includes and does not run a compiler. None
of those references resolves calls, inheritance, types, symbols, runtime behavior, or semantic
ownership.
All frontends emit the same DocForge contracts:
- `AdapterManifest` inventories fingerprinted extraction units and dependencies.
- `AdapterSourceProjection` owns nodes, relationships, and optional function Logic for one unit.
- `AdapterProjection` provides the deterministic complete rebuild.
- `AdapterAssembly` optionally resolves overlapping raw evidence into the single published graph.
Language metadata may differ. Graph publication, indexing, querying, visualization, and MCP
behavior do not.
Done when repeated extraction produces the same stable identities without inferred or guessed
facts.
When compiler or language tooling repeats shared declarations across extraction units, use the
optional assembly contract. Cache the raw source contributions through DocForge, then
deterministically select or merge ownership from the complete contribution set. Do not hide a
second extraction cache inside the project adapter.
Before implementing a frontend, read the
[Language Adapter Authoring Guide](ADAPTER_AUTHORING_GUIDE.md). It defines the complete extraction,
identity, ownership, normalization, incremental-equivalence, troubleshooting, and proof route that
this checklist summarizes.
### 5. Build-system evidence
#### C and C++
- [ ] Use an authoritative compilation database.
- [ ] Preserve target flags, definitions, language standards, and include paths.
- [ ] Resolve headers shared by multiple translation units.
- [ ] Assign shared symbols to one deterministic source contribution.
- [ ] Record compiler-derived project include dependencies.
These are production semantic-adapter expectations. The built-in C++ reference uses
`compile_commands.json` only as bounded translation-unit inventory and fingerprint evidence. It
parses commands and arguments as inert data, executes no command or compiler, and does not claim
compiler include semantics, symbol ownership, or a Clang-derived graph.
#### Rust
- [ ] Read the Cargo workspace and package graph.
- [ ] Respect packages, targets, features, and conditional compilation.
- [ ] Model crates, modules, traits, implementations, functions, and supported macros.
- [ ] Treat expanded macro output as derived evidence.
- [ ] Record the exact toolchain and extraction backend.
#### Java
- [ ] Read Gradle, Maven, or explicit source-root configuration.
- [ ] Respect modules, source sets, language level, and classpath.
- [ ] Model packages, classes, interfaces, records, methods, fields, and supported annotations.
- [ ] Separate authored source from generated and annotation-processor output.
- [ ] Record inheritance and interface implementation.
Other languages follow the same rule: the language frontend translates authoritative build and
source evidence into the common adapter contract.
Done when a clean machine can reproduce the same source inventory from declared configuration.
### 6. Complete reference projection
- [ ] Extract the complete supported source tree.
- [ ] Generate stable source and symbol nodes.
- [ ] Generate only evidence-backed relationships.
- [ ] Generate optional function-scoped Logic separately from the primary graph.
- [ ] Reject duplicate node or Logic ownership.
- [ ] Reject missing relationship endpoints.
- [ ] Reject unsafe source paths.
- [ ] Record project identity, source hash, counts, and duration.
- [ ] Repeat the build and compare exact output.
Done when two unchanged complete builds are identical.
### 7. Incremental compilation
- [ ] Fingerprint each extraction unit.
- [ ] Record extractor versions.
- [ ] Record direct source dependencies.
- [ ] Invalidate reverse dependents.
- [ ] Remove deleted-source contributions.
- [ ] Treat missing or corrupt caches as cache misses.
- [ ] Publish cache and graph generations atomically.
- [ ] Keep the complete projection as the equivalence oracle.
Required proof:
- [ ] cold build;
- [ ] unchanged warm build;
- [ ] implementation-file change;
- [ ] shared-header or shared-module change;
- [ ] added, renamed, and deleted source;
- [ ] build-feature or compiler-setting change;
- [ ] corrupt cache;
- [ ] interrupted extraction;
- [ ] complete-versus-incremental equivalence.
Done when incremental extraction produces exactly the complete projection.
### 8. Source and manual integration
- [ ] Link documented systems to their implementation.
- [ ] Link API reference nodes to extracted symbols.
- [ ] Link roadmap work to affected systems.
- [ ] Link relevant tests and proof artifacts.
- [ ] Report implemented but undocumented systems.
- [ ] Report documented systems without implementation.
- [ ] Keep uncertain links as proposals.
- [ ] Keep source and manual projections independently rebuildable.
Done when a developer can navigate from a decision to implementation and back without guessing
from filenames.
### 9. Views and MCP
- [ ] Configure the generic graph browser.
- [ ] Configure manual, source, API, roadmap, and proof views as needed.
- [ ] Verify search, filters, backlinks, dependencies, impact, and Logic.
- [ ] Generate the exact project-bound MCP command.
- [ ] Select read, proposal, and application capabilities explicitly.
- [ ] Register and reload the client.
- [ ] Call `docforge_bootstrap`.
- [ ] Verify project ID, root fingerprint, adapter version, revision, source hash, and index health.
- [ ] Verify the MCP cannot switch projects or weaken project authority.
Done when a new session can identify and retrieve the correct project without being told its file
layout.
The fixed reference server is read-only. Generic and custom-adapter client generation, including
the immutable custom launcher boundary, is documented in
[Agent Integration](AGENT_INTEGRATION.md).
### 10. Operating guide and maintenance
- [ ] Record the authority and progressive-reading order.
- [ ] Explain exact lookup, search, context, backlinks, and impact analysis.
- [ ] Explain proposal, review, approval, and application.
- [ ] Explain cache invalidation and recovery.
- [ ] Explain frontend and adapter version changes.
- [ ] Run complete/incremental equivalence in continuous integration.
- [ ] Add contract tests for newly supported language features.
- [ ] Never hand-resolve generated-output conflicts.
- [ ] Never convert an inferred relationship into canonical truth silently.
Done when a developer unfamiliar with the repository can use DocForge without loading the entire
manual or inventing another documentation workflow.
### 11. Release-candidate documentation cadence
- [ ] Read the relevant canonical nodes during intake.
- [ ] Record the expected documentation impact in the working plan.
- [ ] Keep canonical sources and DocForge proposals unchanged during implementation and focused
test loops.
- [ ] Freeze one release candidate after implementation stops changing.
- [ ] Run the complete project gate, deployment preflight, candidate deployment, live checks, and
release-identity checks before proposing documentation updates.
- [ ] Return to implementation when candidate validation fails.
- [ ] Register one atomic changeset that covers every affected canonical node after the candidate
is green.
- [ ] Inspect the exact diff and previews, then apply only the reviewed changeset hash.
- [ ] Run documentation-only validation and render checks after application.
- [ ] Permit at most one narrow evidence-only correction for facts that could not exist before
deployment.
- [ ] Commit, tag, and publish the final revision only after implementation and canonical
documentation agree.
Done when documentation describes the verified release candidate instead of intermediate attempts,
and the project normally performs one canonical documentation write per release slice.
## CLI and MCP boundary
Initial assessment and scaffolding belong to the CLI because an MCP server cannot be registered
until the project exists. The MCP begins at `docforge_bootstrap`, after its process has been fixed
to one configured project root.
DocForge does not let an MCP call install dependencies, run project builds, modify Git, deploy, or
publish. A project integration may use its own normal development workflow for those actions.
## Reference and production frontend boundaries
The public authoring namespace is `docforge.adapter_sdk`. Project-owned and separately distributed
production frontends should build on that contract without putting language-specific rules into
the graph, index, viewer, or MCP core.
The in-repository Python, JavaScript, TypeScript, and C++ adapters are reference implementations.
Their syntax-scoped behavior is useful without becoming a claim that every project in those
languages has complete semantic coverage. A compiler-backed production C++ adapter may resolve
build flags, calls, types, inheritance, include semantics, and ownership when it can prove those
facts. The reference C++ adapter does none of that and is not a Clang semantic adapter.
New Rust, Java, or other frontends should use their authoritative build and language tooling and
must pass the graph-plus-Logic complete/incremental contract in the
[Language Adapter Authoring Guide](ADAPTER_AUTHORING_GUIDE.md).

View file

@ -0,0 +1,125 @@
# Recovery and performance
DocForge keeps canonical project evidence separate from disposable indexes, caches, receipts, and
rendered projections. Recovery rebuilds derived state from current authority; it does not rewrite
canonical content to make a cache look valid.
For the underlying boundaries, see [core authority](CORE_CONCEPTS_AND_AUTHORITY.md) and the
[security model](SECURITY.md).
## Recovery order
Use the narrowest verified recovery:
1. Run `docforge check` or the corresponding MCP status operation.
2. Run `docforge sync` to repair a missing or stale disposable index when safe.
3. Run `docforge reindex` for an explicit complete rebuild.
4. Restart a long-running MCP or viewer binding after descriptor or adapter implementation drift.
5. Regenerate a derived client fragment, command reference, preview, manual, or portable graph
from current project evidence.
Never delete or rewrite canonical sources, active changesets, or reviewed hashes as cache cleanup.
## Extraction-cache recovery
Incremental adapter contributions are stored in a version-1 disposable extraction cache. Reads are
no-follow, regular-file-only, identity-checked, bounded to 64,000,000 bytes and 10,000 source
records, and fail to a cache miss on malformed or incompatible data. Publication is atomic.
After a miss, current manifest sources are extracted again and the complete assembly is validated.
The active SQLite generation remains authoritative for reads until a new verified index is
published. An extraction cache may therefore be safely ahead of the last index; the two files do
not pretend to be one transaction.
## Index and receipt recovery
A validated SQLite index is a generation-pinned derived snapshot. Missing, corrupt, unattested, or
stale indexes rebuild from the complete project or adapter oracle. Generation receipts and render
receipts are post-publication evidence. Failure to write a receipt after a committed artifact is
reported as degraded committed success, not as permission to repeat a mutation.
The live viewer pins one validated index identity. Index replacement makes the running snapshot
stale and causes a later visualize request to start a fresh worker.
Milestone 5 maintains exact-oracle recovery proofs for four independently disposable artifacts:
- A corrupt index attestation is rejected, then `synchronize()` recreates the exact attestation
after complete verification while preserving canonical bytes, snapshot hash, and index identity.
- A corrupt manual-render receipt reports `unverified/receipt_corrupt`; explicit rendering restores
the exact output bytes and semantic receipt, and normal and deep status return `current`.
- A corrupt generation-diff receipt reports `unverified/corrupt_receipt`; a complete index build
recreates the exact current-graph baseline with `baseline/no_meaningful_transition`.
- A corrupt portable-graph manifest reports missing publication evidence; explicit rendering
restores the exact artifact bytes and semantic manifest, and status returns `current`.
These proofs are maintained by `tests/test_milestone5_recovery.py`. They do not promote receipts,
attestations, or manifests to canonical authority.
## Proposal and application recovery
Hash or base conflicts are not cache failures. Retrieve the current changeset and diff, then
review the new exact hash. Rebase is allowed only when every touched node, relationship, source,
permission, and graph invariant still matches. A content conflict requires a new proposal.
Generic canonical create, update, and delete publication compares exact target identity at the
commit boundary. Concurrent target mutation fails closed. In-process failures roll back only when
the exact displaced state remains provable; otherwise DocForge preserves foreign data and returns
`application_recovery_required` with retained evidence.
Canonical application is not journaled across several files. Each file publication is atomic, but
process or host death between publications can leave a partial canonical application. Inspect the
named canonical targets, the active proposal, and `.docforge/application/transaction-*` before
deciding whether to restore or create a new proposal. Do not retry the old approved hash merely
because a process ended.
If semantic application committed but private transaction cleanup degraded, the result remains
`applied`. The proposal is closed and its compact lifecycle receipt records
`application_recovery.status = "cleanup_required"`, retained paths, and remediation. Preserve and
inspect those files. Remove only artifacts proven to be DocForge-owned. If a canonical serializer
fails its round-trip check before success, use its reported rollback state. See
[migrating from v1](MIGRATING_FROM_V1.md) for rollback planning.
## Milestone 4 scale evidence
The maintained Python reference benchmark creates 334 Python source files and produces:
- 1,002 primary nodes and 1,001 primary edges.
- 334 Logic projections with 2,338 Logic nodes and 2,338 Logic edges.
- Exact complete/incremental primary graph and Logic equality.
- Exact output after corrupt extraction-cache recovery and corrupt-index recovery.
- Zero `ast.parse` calls and zero `extract_source` calls during a warm build.
On the frozen Milestone 4 candidate, operation p95 values were 1.266 to 1.898 seconds. Per-operation
traced peaks were about 61 to 67 MiB, and process high-water was 78,798,848 bytes. These are
regression measurements from one machine, not universal latency promises. The machine-readable
record is `../benchmarks/milestone4-2026-07-29.json`; detailed method and hashes are in
[the Milestone 4 baseline](MILESTONE_4_BASELINE.md).
The zero-parser claim is deliberately Python-only. Focused JavaScript and TypeScript tests prove
their manifests avoid Tree-sitter. The C++ reference manifest currently uses Tree-sitter while
discovering bounded quoted includes, so a warm C++ cache hit is not evidence of zero parser work.
## Maintained gates
Run:
```bash
make gate
make adoption-m4
make benchmark-m4-full
make docs-check
make compatibility-m5
make migration-m5
make concurrency-m5
make recovery-m5
make task-evidence-m5
make release-gate
make fresh-clone-m5
```
The main gate includes smoke benchmarks. Full milestone evidence is recorded separately from a
clean candidate so smoke or dirty-tree results cannot become release claims. `release-gate`
aggregates the complete quality, compatibility, migration, concurrency, recovery, task-evidence,
fresh-wheel, identity, reproducible-artifact, secret-scan, and full benchmark proofs.
`fresh-clone-m5` repeats that aggregate gate from anonymous HTTPS at one exact published commit
after verifying the frozen annotated `v1.0.0` migration tag.

149
docs/REFERENCE_ADAPTERS.md Normal file
View file

@ -0,0 +1,149 @@
# Reference adapters
DocForge includes fixed, syntax-scoped reference adapters for Python, JavaScript, TypeScript, and
C++. They demonstrate the public adapter contract, deterministic complete and incremental
publication, function Logic, cache recovery, and a runnable read-only MCP binding. They are not
compiler or language-service replacements.
Use a reference adapter when its deliberately narrow graph is sufficient or when proving a fresh
DocForge integration. Use the [Language Adapter Authoring Guide](ADAPTER_AUTHORING_GUIDE.md) for a
production adapter that needs resolved symbols, calls, inheritance, types, build semantics, or
semantic ownership.
## Installation
The Python reference adapter uses the standard library and is available in the base wheel. The
other frontends are separate optional extras:
```bash
python -m pip install docforge
python -m pip install "docforge[javascript]"
python -m pip install "docforge[typescript]"
python -m pip install "docforge[cpp]"
```
Install `docforge[languages]` only when one environment intentionally needs all three Tree-sitter
frontends. JavaScript and TypeScript use distinct grammar packages and distinct extras.
## Fixed project configuration
The runnable reference binding reads exactly
`.docforge/reference-adapter.toml` below the selected project root. The configuration is data only:
it cannot name a provider, Python module, command, arguments, working directory, environment, or
discovery rule.
Reference adapter configuration `.docforge/reference-adapter.toml`:
```toml
schema_version = 1
project_id = "example-python"
title = "Example Python project"
language = "python"
source_roots = ["src"]
```
`language` is one of `python`, `javascript`, `typescript`, or `cpp`. Source roots must be sorted,
unique, non-overlapping project-relative directories outside `.docforge`. The configuration,
roots, and inventoried source files must be regular confined paths without symlink traversal. The
configuration is limited to 65,536 bytes, at most 64 source roots, and 65,536 examined inventory
entries.
C++ additionally requires one explicit project-confined compilation database:
```toml
schema_version = 1
project_id = "example-cpp"
title = "Example C++ project"
language = "cpp"
source_roots = ["include", "src"]
compilation_database = "compile_commands.json"
```
`compilation_database` is forbidden for the other three languages.
## Published scope
All four adapters publish deterministic file, module or translation-unit, class or struct, and
function nodes where their syntax supports those categories. They publish lexical `contains`
relationships, a narrow set of project-local `depends_on` relationships, and function-scoped
Logic. Calls that appear inside Logic are syntax steps, not resolved symbol relationships.
### Python
Python uses `ast` from the standard library for source extraction and Logic. Manifest construction
does not build a Python AST; it tokenizes imports only far enough to publish dependencies that
resolve to another module in the declared source inventory.
It does not import or execute project code. It does not resolve dynamic imports, calls,
inheritance, imported symbols, types, overloads, re-exports, decorators, metaclasses, descriptors,
or runtime-generated behavior.
### JavaScript and TypeScript
JavaScript uses the optional `tree-sitter-javascript` grammar. TypeScript uses the distinct
optional `tree-sitter-typescript` grammar. Their focused incremental tests prove that manifest
construction and an unchanged warm build do not invoke the Tree-sitter extraction parser.
Only static relative imports and re-exports that resolve to another inventoried source file become
dependencies. Dynamic `import()`, `require()`, bare package specifiers, aliases, `tsconfig` paths,
loader hooks, types, interfaces, overloads, calls, inheritance, symbols, and runtime behavior are
not resolved. The adapters never import, compile, transpile, or execute project code.
### C++
C++ treats `compile_commands.json` as the authoritative bounded translation-unit inventory and as
fingerprint evidence. Commands, arguments, directories, and output fields are parsed as inert
data. The adapter executes no compiler, build tool, command, project binary, or project code.
The reference C++ manifest uses the optional `tree-sitter-cpp` grammar to parse inventoried sources
and discover quoted includes. It publishes a dependency only when that quoted include resolves
directly to a real project-local header in the declared roots. Angle-bracket includes, compiler
include paths, frameworks, generated headers, compiler-provided headers, conditional compilation,
and macro expansion are omitted.
This is not a Clang semantic adapter. It does not claim resolved calls, types, templates, aliases,
concepts, references, inheritance, overload ownership, out-of-line semantic ownership, macro
semantics, or compiler include semantics. A warm C++ extraction-cache hit is therefore not evidence
that manifest construction performed zero parser work.
## Complete and incremental proof
Each reference adapter implements:
- `load_projection()` for the complete primary graph compatibility oracle;
- `load_complete_assembly()` for the complete graph-plus-Logic oracle;
- `load_manifest()` and `extract_source()` for incremental extraction; and
- deterministic assembly of the complete current contribution set.
The maintained fixtures prove complete determinism, exact complete/incremental graph and Logic
parity, reverse-dependency invalidation, additions and deletions, corrupt extraction-cache
recovery, confinement, and exact unsupported-fact inventories. See
[Incremental Adapter Indexing](INCREMENTAL_INDEXING.md) for the cache contract and
[Legacy Adapters and No-AST Policy](LEGACY_AND_NO_AST.md) for compatibility and policy limits.
Current fixture evidence is:
| Language | Primary nodes | Relationships | Logic projections |
|---|---:|---:|---:|
| Python | 14 | 13 | 5 |
| JavaScript | 14 | 14 | 5 |
| TypeScript | 13 | 14 | 4 |
| C++ | 17 | 16 | 5 |
These are regression-fixture shapes, not promises for arbitrary repositories.
## Read-only reference server
Start the fixed server with:
```bash
python -I -m docforge.reference_mcp \
--project-root /absolute/path/to/project \
--capability-mode read
```
The server selects one of the four in-repository providers solely from the validated fixed
configuration. Its cache stays below `.docforge/cache/reference-adapter/<language>`. It registers
the 21-tool read surface and no proposal or application tools. See
[Agent Integration](AGENT_INTEGRATION.md) for generated client fragments and
[MCP Boundary](MCP_CONTRACT.md) for the exact tool contract.

View file

@ -0,0 +1,233 @@
# Rendering and visualization
DocForge has three independent ways to present one validated graph generation:
```text
validated generation
├── manual plan → immutable package → detached HTML renderer → declared manual output
├── graph plan → immutable package → detached graph renderer → portable static artifact
└── pinned index → managed loopback viewer → interactive Nodes, Flow, Web, and lazy Logic
```
They share validated facts but not authority, publication, or lifecycle. A manual render does not
publish a portable graph. A portable graph does not start the live viewer. None is canonical
project content.
## Manual rendering
A generic descriptor may declare a manual template root, isolated preview root, and stable views:
```toml
[render]
template_root = ".docforge/templates"
preview_root = ".docforge/previews"
[[render.views]]
id = "manual"
renderer = "generic_html"
template = "manual.html"
output = ".docforge/rendered/manual.html"
title = "My Project Manual"
families = ["architecture", "operations", "system"]
```
The renderer name is fixed to `generic_html`. Templates are confined UTF-8 assets with a fixed
token vocabulary. They cannot name commands, Python modules, executable renderers, or arbitrary
publication paths. Raw HTML in canonical Markdown is disabled by the pinned CommonMark path.
Plan and render identity covers the canonical source hash, optional changeset hash, selected nodes
and relationships, view configuration, template hash, renderer contract, and parser version.
Use:
```bash
docforge --project-root "$PROJECT" render-status
docforge --project-root "$PROJECT" render-status manual --deep
docforge --project-root "$PROJECT" render manual
docforge --project-root "$PROJECT" preview CHANGESET_ID manual
```
Normal status verifies bounded source, configuration, template, output, renderer, and receipt
identities without reconstructing output. `--deep` explicitly runs the side-effect-free full-render
oracle. A changeset preview writes only to the isolated preview root.
Manual `auto` means that a successful canonical application owns regeneration of every declared
manual view. It does not mean background rendering, and ordinary standalone CLI render cannot
select `auto`; use `explicit`.
## Portable graph rendering
Portable graph configuration is separate:
```toml
[graph_render]
output_root = ".docforge/portable-graph"
[[graph_render.views]]
id = "architecture"
renderer = "portable_graph_html"
output = "architecture.html"
title = "Architecture"
root = "architecture.overview"
initial_mode = "web"
depth = 3
max_nodes = 250
max_edges = 1000
max_work = 100000
families = ["architecture", "system"]
relations = ["depends_on", "owns", "calls", "reads", "writes", "tested_by", "relates_to"]
authorities = []
statuses = ["current", "active", "verified"]
tags = []
include_logic = false
```
A view selects exactly one exact `root` node or bounded metadata-only lexical `query`. Exact
filters and node, edge, depth, and work limits close the selection. The initial mode is `nodes`,
`flow`, or `web`. Portable graph contract version 1 excludes function-scoped Logic.
Use:
```bash
docforge --project-root "$PROJECT" graph-plan architecture
docforge --project-root "$PROJECT" graph-render architecture
docforge --project-root "$PROJECT" graph-render-status architecture
```
`graph-plan` validates and returns the generation-pinned plan without publication. `graph-render`
is an explicit local CLI publication action. It commits a content-addressed artifact, renderer
receipt, and bounded generation/view manifest in that order; the manifest is the publication
commit. `graph-render-status` checks only bounded committed evidence and never plans or renders.
MCP may plan and inspect portable graph evidence, but it does not publish the portable artifact.
Use the explicit project-local CLI command for publication.
Portable output is a complete static artifact. JavaScript is progressive enhancement, not a
requirement for the graph facts to be present.
## Live viewer
The live viewer is a managed, read-only browser over one generation-pinned validated SQLite file.
Install the per-user manager once:
```bash
docforge-viewer-manager install-user-service
```
Then operate a project viewer:
```bash
docforge --project-root "$PROJECT" visualize
docforge --project-root "$PROJECT" visualize --node architecture.overview
docforge --project-root "$PROJECT" visualize --query persistence
docforge --project-root "$PROJECT" visualization-status
docforge --project-root "$PROJECT" visualization-stop
```
`visualize` accepts only an optional stable node ID or lexical query, bounded traversal depth, and
the local `--no-open` presentation choice. It does not accept a project root override, database
path, SQL, template, command, renderer, bind address, or module.
The HTTP listener binds to `127.0.0.1` on an operating-system-selected port. A random token is part
of every accepted path. The server supports only `GET` and `HEAD`, sets no-store and restrictive
browser security headers, and has no write, project-selection, arbitrary-query, or static
filesystem endpoint.
The viewer is a snapshot. Index replacement or alteration makes that snapshot fail closed; start a
new visualization to use a new validated generation. Source inspection reads only project-confined
source evidence bound to the pinned generation.
## Nodes, Flow, Web, and Logic
The interactive viewer offers four complementary projections:
- **Nodes** shows a bounded relation-neutral incoming and outgoing neighborhood.
- **Flow** presents semantic contributors toward the focus. Prerequisite-style stored
relationships may be reversed for presentation without changing stored direction.
- **Web** expands the convergence picture with contributors, callers, containers, members, and
contextual relationships.
- **Logic** loads a function-scoped control-flow projection only when requested.
Logic supports the explicit control paths published by Python, JavaScript, TypeScript, and C++
integrations. It stays outside the primary graph, portable graph version 1, search, and
generation-diff receipts.
Hiding a node is presentation-only. In Flow and Web, ancestors without another path to the focus
are removed. In Logic, an omitted-path bridge preserves downstream readability. Restore reverses
the presentation change; neither action mutates the index.
## Immutable plan and worker boundary
Manual and graph plans are versioned, canonical JSON with deterministic ordering and fixed
structural and byte limits. They contain selected graph facts and bounded content, but no live
project object, SQLite handle, absolute project or index path, arbitrary query, command,
executable path, or caller-selected renderer module.
A projection package binds one plan to inert assets, fixed component versions, a closed built-in
renderer identity, and an exact artifact inventory. The detached worker:
- runs one fixed private module through isolated Python;
- uses a trusted working directory and sanitized environment;
- accepts one canonical newline-terminated JSON request;
- returns one bounded canonical JSON response;
- has a fixed renderer allowlist and timeout;
- cannot select graph facts, read project state, choose output paths, or mutate canonical files.
The package contract is bounded, and actual artifact transfer has a fixed 20,000,000-byte ceiling.
A larger descriptor `max_render_bytes` compatibility value does not widen that worker boundary.
## Manual fragment reuse
Manual fragments are disposable semantic cache records, not publication authority. A cold record
is accepted only after byte-exact comparison against a full detached render. On a warm hit, the
worker independently recomputes the expected fragment before reuse.
Corrupt, forged, stale, incompatible, individually oversized, or aggregate-oversized records fall
back to the complete full-render oracle. The current cache is bounded to 10,000 records and
64,000,000 bytes.
Fragment reuse is a correctness and recovery boundary. Do not promise a speedup without current
measurements.
## Projection policy
The version-2 projection policy is independent of capability mode:
```text
manual: auto | explicit | disabled
portable_graph: explicit | disabled
live_viewer: on-demand | disabled
```
Set global CLI flags before the subcommand:
```bash
docforge --project-root "$PROJECT" --manual-render-policy disabled render manual
docforge --project-root "$PROJECT" --portable-graph-policy disabled graph-plan architecture
docforge --project-root "$PROJECT" --live-viewer-policy disabled visualize
```
An active operation blocked by policy fails before hidden work. Receipt-only manual and portable
status remain available. Viewer status and explicit stop remain available when viewer startup is
disabled.
A non-disabled selection also requires its resource: declared manual configuration, declared
portable graph configuration, or viewer runtime. Manual `auto` additionally requires canonical
application. Read [Policy precedence](POLICY_PRECEDENCE.md) for exact composition.
## Failure and recovery
Render input drift detected before replacement fails without publishing a current receipt for stale
output. Portable graph publication uses content-addressed evidence and a final manifest commit.
Status never repairs implicitly.
Recovery is explicit:
- rerun a manual render from validated canonical state;
- use deep manual status for the full, side-effect-free equivalence oracle;
- rerun portable graph publication, or repair only from validated content-addressed evidence;
- stop and restart a stale live viewer against a current validated index.
Do not recover a derived output by editing its receipt or treating it as canonical. See [Recovery
and performance](RECOVERY_AND_PERFORMANCE.md), [Security](SECURITY.md), and the [Viewer
manager](VIEWER_MANAGER.md).

126
docs/SECURITY.md Normal file
View file

@ -0,0 +1,126 @@
# Security model
DocForge is a project-bound knowledge compiler. Its security boundary is an explicit project root,
closed configuration, bounded data, and exact identities. It is not a general process or
filesystem sandbox.
Start with [core authority](CORE_CONCEPTS_AND_AUTHORITY.md), then use
[policy precedence](POLICY_PRECEDENCE.md) to decide which capabilities a server should expose.
## Project and path confinement
Descriptors, reference-adapter configurations, canonical sources, authority files, templates,
changesets, caches, indexes, previews, and declared outputs are resolved against one project root.
DocForge rejects absolute paths where only project-relative paths are allowed, parent traversal,
symbolic-link escapes, unsafe file types, and protected-root overlap. Important reads use
no-follow file descriptors and compare file identity before and after reading.
Confinement protects DocForge operations. It does not stop another process with repository access
from changing files. Long-running bindings revalidate descriptor and adapter implementation
identity and require a restart after drift.
Generic canonical application stages backups and replacements below
`.docforge/application/transaction-*` in mode-0700 directories. That private namespace confines
ordinary path access and prevents access by other users. Deliberate arbitrary tampering by another
process running as the same operating-system user is outside this boundary. DocForge still
identity-checks private files before using or removing them, but mode `0700` is not isolation from
the same UID.
## Untrusted project content
Documentation, source text, templates, adapter metadata, compiler-database entries, and changeset
content are data. They cannot redefine policy or instruct DocForge to execute a command. Generic
manual rendering disables raw HTML, accepts a fixed template vocabulary, and rejects script-like
content. Portable graph and manual workers accept validated inert packages and fixed built-in
renderer identities.
The C++ reference adapter reads `compile_commands.json` only as bounded translation-unit inventory
and fingerprint evidence. It never executes the recorded command, compiler, response file, or
project program.
## Adapter launcher boundary
`AdapterLauncherV1` contains one Python module name and project identity. It contains no command,
shell string, arbitrary argument list, working directory, environment, discovery rule, or callable
selector.
Custom modules must be installed top-level modules. An isolated `python -I` probe resolves the
module without importing it and requires its regular-file origin to remain inside the bound
project. The sole trusted dotted exception is the packaged `docforge.reference_mcp` module. Client
fragments use the exact validated interpreter and canonical fixed arguments with an empty
environment.
This proves that the declared module is resolvable and project-bound. It does not make arbitrary
module code safe. Project owners remain responsible for the implementation they install.
## Mutation boundary
Normal MCP and the fixed reference server are read-only. Proposal tools exist only when a
startup-bound writer is authorized by the descriptor. Canonical application exists only when a
matching applier is explicitly configured.
Every proposal append, rebase, abandonment, and application is hash-bound. Application requires
the exact changeset hash that was reviewed. Source identity, content hashes, permissions,
conflicts, graph validity, and serializer round trips are checked before success. DocForge never
turns prose approval into a fuzzy merge.
The generic applier compares exact canonical file identity immediately before each publication.
Create uses no-clobber publication. Update and delete use atomic exchange and no-replace moves.
Concurrent canonical-target mutation therefore fails closed, rolls back when the exact displaced
state is still provable, or retains recovery evidence without overwriting foreign data.
This compare-and-swap protection is not a process-death journal. One file publication is atomic,
and an in-process failure runs exact rollback, but an application spanning several canonical files
does not promise crash atomicity if the process or host dies between publications. Operators must
inspect canonical state and retained transaction evidence before retrying after such an
interruption.
## Derived state and publication
SQLite indexes, source-generation receipts, extraction caches, render fragments, previews, and
portable artifacts are disposable. Corrupt, stale, foreign, oversized, or mismatched derived
state is rejected or rebuilt from current project evidence.
Derived publication stages complete bounded output, flushes file and directory state, and commits
with atomic replacement or no-clobber compare-and-swap. It is crash-safe: an interruption leaves
the prior verified artifact, the complete new artifact, or explicit degraded post-commit evidence,
not a mixed publication. Generated command-reference publication also serializes cooperating
writers. A raced target is restored or retained for recovery instead of being silently discarded.
Projection and canonical-application lifecycles record when semantic content committed but later
private cleanup or receipt verification degraded. Canonical success closes the applied proposal
and persists compact `application_recovery` metadata with retained paths and remediation. A
completed mutation is never reported as an ordinary retryable failure.
## Limits and denial-of-service resistance
Inputs, results, traversal, context, changesets, renders, worker protocols, manifests, adapter
assemblies, and extraction caches have explicit count and byte limits. Incremental extraction
caches are capped at 10,000 sources and 64,000,000 bytes. Adapter primary nodes use the project
`max_nodes` limit; edges and Logic have deterministic multipliers over that limit.
Limits reduce accidental and adversarial amplification. An in-process adapter can still allocate
memory before returning data, so only trusted project-owned adapter code should run in the server
process.
## Secrets and network behavior
DocForge does not copy the parent environment into generated client fragments or detached
projection workers. Doctor checks never return environment values. The live viewer binds to
loopback, uses an unguessable URL token, supports read-only methods, and serves no arbitrary
filesystem tree.
Project secrets must not be placed in canonical documentation, adapter configuration, compiler
databases, templates, or changesets. Repository release gates include secret scanning, but that
scan is not a substitute for credential hygiene.
## No-AST boundary
`--no-ast` is a binding policy that preserves the selected adapter and prohibits Logic publication
and retrieval. It is not a parser detector, filesystem sandbox, or promise that unrelated
processes cannot parse source. See [legacy and no-AST operation](LEGACY_AND_NO_AST.md).
## Reporting and recovery
Do not bypass a confinement, identity, policy, hash, or limit error. Preserve the failing evidence,
stop the affected binding, and follow [recovery and performance](RECOVERY_AND_PERFORMANCE.md).

View file

@ -5,6 +5,18 @@ people and AI agents can search, inspect, visualize, and change through reviewab
Canonical project files remain authoritative. The SQLite graph, previews, rendered manuals, and
viewer processes are derived and can be rebuilt.
This manual describes the DocForge 2.0.0 release. The tagged `v1.0.0` baseline was the first stable
product release. Version 2.0.0 preserves its project-scoped graph, CLI and MCP query
surfaces, hash-approved proposal application, generic and project-owned adapters, declared
rendering, and Nodes/Flow/Web model while adding the maintained incremental, projection, adapter
SDK, recovery, and release proofs documented below.
Later incremental-compiler capabilities are additive. A Release 1 adapter with only
`load_projection()` remains valid and follows the same complete-rebuild path. No existing project
descriptor, canonical document, changeset, or adapter must be rewritten. Source-scoped caching and
lazy logic projections activate only for adapters that explicitly implement the optional
incremental methods while retaining the full loader as a fallback.
## Features
- Project-bound Markdown and TOML documentation graphs with stable node IDs.
@ -12,12 +24,21 @@ viewer processes are derived and can be rebuilt.
- Disposable SQLite indexing with lexical search, filters, backlinks, dependencies, and impact.
- Bounded context profiles for AI agents, including source paths and content hashes.
- Isolated, optimistic changesets with create, update, move, delete, validation, diffs, and previews.
- Relationship-only changeset operations that do not rewrite node content.
- Hash-bound canonical application through both CLI and an explicitly enabled MCP tool.
- Opt-in incremental adapter extraction with reverse-dependency invalidation.
- Lazy function-scoped logic projections that do not densify the primary graph.
- Declared HTML render views. Arbitrary templates, render commands, and output paths are rejected.
- Separate versioned manual and portable graph plans, immutable packages, detached built-in
renderers, and validated receipts.
- Content-addressed portable Nodes/Flow/Web artifacts with receipt-only status and repair.
- Independent manual, portable-graph, and live-viewer policy.
- A loopback-only graph browser with Nodes, semantic Flow, and convergence Web views,
relationship keys, source inspection, branch-aware node hiding, panel resizing, zooming, and
managed idle shutdown.
- A generic Markdown/TOML adapter plus contracts for deterministic project-owned adapters.
- A public adapter SDK, optional Python/JavaScript/TypeScript/C++ reference integrations, and one
fixed read-only reference MCP binding.
DocForge does not run shell commands from documentation, mutate Git, build an application, deploy,
publish, choose a project globally, or cross project boundaries.
@ -34,7 +55,11 @@ Canonical files own facts:
canonical Markdown/TOML or adapter sources
↓ validate
disposable SQLite graph
↓ query / visualize / compile context
├── query / compile context
├── ManualRenderPlanV1 → detached manual renderer → declared manual
├── GraphViewPlanV1 → detached graph renderer → portable Nodes/Flow/Web artifact
└── pinned index → managed live Nodes/Flow/Web/Logic viewer
people and agents
↓ propose
isolated changeset + preview
@ -53,6 +78,12 @@ The generic adapter can serialize its Markdown and TOML nodes directly. A custom
provide its own canonical applier because only that project knows how a graph node maps back to its
source format.
Generic canonical application compare-and-swaps each target against its exact expected identity.
A concurrent create, update, or delete fails closed, rolls back when the exact displaced state is
still provable, or preserves recovery evidence without overwriting foreign data. Per-file
publication is atomic, but an application spanning several canonical files has no process-death
journal and does not claim crash atomicity across the group.
## Setup
### Requirements
@ -64,8 +95,8 @@ source format.
Clone and verify DocForge:
```bash
git clone forgejo@repo.andraxion.net:administrator/DocForge.git /absolute/path/DocForge
cd /absolute/path/DocForge
git clone forgejo@repo.andraxion.net:administrator/DocForge2.git /absolute/path/DocForge2
cd /absolute/path/DocForge2
uv sync --group dev
npm ci
@ -79,6 +110,92 @@ uv run pytest -q
Use the executables under `/absolute/path/DocForge/.venv/bin/` when DocForge is not installed into
the active shell environment.
For an installed distribution, choose only the language extras the project needs:
```bash
python -m pip install docforge
python -m pip install 'docforge[javascript]'
python -m pip install 'docforge[typescript]'
python -m pip install 'docforge[cpp]'
```
The base wheel contains the Python reference adapter and no Tree-sitter distribution. JavaScript,
TypeScript, and C++ require their matching optional extras. `docforge[languages]` installs all
three optional frontend groups.
Verify the four executable surfaces from the exact installed environment:
```bash
python -m docforge.cli --version
python -m docforge.mcp_server --version
python -m docforge.reference_mcp --version
python -m docforge.viewer_manager --version
```
For version 2.0.0 these report `docforge 2.0.0`, `docforge-mcp 2.0.0`,
`python -m docforge.reference_mcp 2.0.0`, and `docforge-viewer-manager 2.0.0`. Package metadata,
Python imports, generated generic and adapter configurations, and these commands share the same
version authority.
### Configure a reference source project
Reference adapters are a narrow alternative to the generic documentation descriptor. Create
`.docforge/reference-adapter.toml`:
```toml reference-adapter
schema_version = 1
project_id = "my-python-project"
title = "My Python Project"
language = "python"
source_roots = ["src"]
```
Start the fixed read-only server:
```bash
python -I -m docforge.reference_mcp \
--project-root /absolute/path/MyProject \
--capability-mode read
```
For C++, set `language = "cpp"` and add a project-relative
`compilation_database = "compile_commands.json"`. The database is inert bounded inventory; the
reference adapter does not execute its commands or compiler.
The reference integrations publish syntax and local static relationships only. They do not claim
resolved calls, types, inheritance, macro behavior, compiler include semantics, runtime behavior,
or semantic ownership. See [reference adapters](REFERENCE_ADAPTERS.md) for exact evidence and
limitations.
### Assess and onboard an unconfigured project
Run a read-only assessment before writing configuration:
```bash
docforge --project-root /absolute/path/MyProject onboard
```
The result reports detected languages, build evidence, likely documentation, existing
configuration, and capability status. Detection does not claim that a language frontend exists.
Limit the assessment with one or more `--language` options when needed.
Create, index, and render a generic starter manual explicitly:
```bash
docforge --project-root /absolute/path/MyProject onboard \
--language rust \
--scaffold \
--project-id my-project \
--title "My Project"
```
Scaffolding refuses to replace existing target files. It leaves source-graph status at
`adapter_required` until a project integration implements and proves the adapter contract.
See [Project onboarding](PROJECT_ONBOARDING.md) for the complete language-neutral checklist.
Use the [Language Adapter Authoring Guide](ADAPTER_AUTHORING_GUIDE.md) when implementing that
frontend. It covers stable identities, overlapping compiler evidence, deterministic ownership,
normalization, incremental equivalence, failure recovery, and the required proof matrix.
### Configure a generic project
Create `/absolute/path/MyProject/.docforge/project.toml`:
@ -117,6 +234,27 @@ output = "Docs/Rendered/Manual.html"
title = "My Project Manual"
families = ["architecture", "system", "operations", "roadmap"]
[graph_render]
output_root = ".docforge/portable-graph"
[[graph_render.views]]
id = "architecture"
renderer = "portable_graph_html"
output = "architecture.html"
title = "Architecture"
root = "architecture.overview"
initial_mode = "web"
depth = 3
max_nodes = 250
max_edges = 1000
max_work = 100000
families = ["architecture", "system"]
relations = ["depends_on", "owns", "calls", "reads", "writes", "tested_by", "relates_to"]
authorities = []
statuses = ["current", "active", "verified"]
tags = []
include_logic = false
[graph]
allowed_relations = ["depends_on", "owns", "calls", "reads", "writes", "tested_by", "relates_to"]
@ -212,6 +350,17 @@ docforge --project-root "$PROJECT" visualize --query persistence
The command opens the default browser. Add `--no-open` when a script only needs the returned JSON
URL. Use `visualization-status` and `visualization-stop` to inspect or stop the project viewer.
Status separates the worker lifecycle from snapshot freshness. A worker may remain `running` while
`snapshot_state` is `stale`; it will not be reused by the next `visualize` call. `freshness.index`
checks the exact pinned index publication with file identity only. `freshness.source` compares the
cheap project generation when the project can prove one. Unavailable proof is `unknown`, never
silently `current`. Status does not load project content, open SQLite, rebuild the index, or renew
browser activity.
The freshness protocol requires viewer manager version 2. After upgrading an already running
installation, rerun `docforge-viewer-manager install-user-service` or restart the foreground
manager before requesting status.
## Visualization usage
- Left-click a node for its compact descriptor.
@ -283,6 +432,59 @@ Adjacent traversal is deliberately bounded. After DocForge includes a direct mem
dependency owned by the focus, it continues toward that branch rather than fanning back out
through unrelated siblings. Depth and edge limits provide a second guard against an unbounded web.
### Logic: possible control paths
**Logic** answers: “What decisions and actions can occur inside this function or method?”
Logic appears when the focused node owns a function-scoped `LogicProjection`. It loads that
projection on demand instead of adding statements and conditions to the primary architecture
graph. The view presents:
- **Entry** and **Exit** terminals.
- **Decision** cards for `if`, `elif`, compound booleans, loop conditions, `match` cases, and
assertions.
- **Action** cards for executable statement blocks and calls.
- **Control** cards for loops, `break`, and `continue`.
- **Convergence** cards where alternate paths rejoin, including decision, case, loop-exit, and
exception convergence.
- **Terminal** cards for returns and raised exceptions.
Edges use explicit labels and independent colors for `TRUE`, `FALSE`, `NEXT`, `CASE`, `LOOP`,
`EXCEPTION`, `RETURN`, `RAISE`, `BREAK`, and `CONTINUE`. Long predicates wrap on the card. The full
expression and source anchor remain available through inspection and source navigation.
The built-in analyzers cover Python, JavaScript, and C++. Python uses the standard-library AST.
JavaScript and C++ use pinned Tree-sitter grammars behind the same language-neutral
`LogicProjection` contract. Tree-sitter handles concrete syntax; DocForge keeps a thin
language-specific control-flow profile for constructs such as conditions, loops, cases,
exceptions, returns, and short-circuit operators. Adding a language therefore requires a grammar
and a semantic profile, not a new visualization or database design.
Logic is static analysis. It shows paths the indexed source permits, not the branch that ran for a
particular request or the runtime value of a boolean. Dynamic dispatch, reflection, generated
behavior, and values returned by other processes may require runtime tracing to resolve.
### Finding the right node
The left panel combines independent filters rather than forcing users to scan the complete node
list:
- **Text** searches indexed titles, summaries, and content.
- **Family** selects the project-defined family.
- **Node type** selects callables or an exact indexed kind such as function, method, class, route,
test, module, or document.
- **Language** selects an indexed language tag such as Python, JavaScript, or C++.
- **Capability** selects nodes with source navigation or an available Logic projection.
Quick presets select common combinations for Logic-ready nodes, Python callables, tests, routes,
and documentation. Filters compose, so `JavaScript` plus `Logic available` lists only JavaScript
functions that can open Logic. Result cards show the readable leaf name, kind, language, path, and
source anchor. Full identities remain in the tooltip and inspector.
Selecting any canvas node highlights its directly connected nodes and the exact edges between
them. Other nodes and edges remain visible at reduced opacity. This local trace works in Nodes,
Flow, Web, and Logic without changing the root or querying a different graph.
### Reading graph cards
The canvas presents nodes as compact semantic cards rather than anonymous circles:
@ -319,6 +521,9 @@ index, or future graph queries. The focus cannot be hidden; focus another node f
- In **Nodes**, hiding removes only the selected node and its incident edges.
- In **Flow** and **Web**, hiding removes the selected node, then prunes every upstream ancestor
whose only remaining route to the focus passed through it.
- In **Logic**, hiding removes the selected control-flow step and inserts an `omitted` bridge
between its visible predecessors and successors. This preserves the readable path without
pretending the hidden code disappeared from the indexed source.
- Descendant nodes between the hidden node and the focus remain visible.
- Ancestors with another valid path to the focus remain visible through that alternate path.
- The status line reports how many nodes were hidden or isolated.
@ -340,6 +545,10 @@ Every command emits deterministic JSON:
docforge --project-root /absolute/path/MyProject <command>
```
The sections below group common workflows. The implementation-derived list of all 28 current
commands, exact invocations, 36 generic MCP tools, arguments, and input-schema hashes is the
[generated command reference](COMMAND_REFERENCE.md).
### Project and index commands
```text
@ -347,6 +556,7 @@ info
validate
build
reindex
sync
check
validate-index
```
@ -355,6 +565,7 @@ validate-index
- `validate` validates current canonical sources without requiring an index.
- `build` rebuilds the disposable index.
- `reindex` rebuilds and checks the index in one operation.
- `sync` checks the index and rebuilds it only when it is missing, stale, or invalid.
- `check` and `validate-index` verify that the existing index matches current sources.
### Query commands
@ -363,24 +574,43 @@ validate-index
show NODE_ID
search QUERY [--limit N]
filter [--family X] [--authority X] [--status X] [--tag X] [--limit N]
backlinks NODE_ID [--relation RELATION]
dependencies NODE_ID [--depth N]
impact NODE_ID [--depth N]
context PROFILE [--budget N]
backlinks NODE_ID [--relation RELATION] [--limit N]
dependencies NODE_ID [--depth N] [--limit N]
impact NODE_ID [--depth N] [--limit N]
context PROFILE [--budget N] [--limit N] [--cursor OPAQUE]
generation-diff [--limit N] [--cursor OPAQUE]
```
`generation-diff` returns the latest verified primary-graph transition. It is not a history query.
Current results carry a version-1 page, a `receipt_header` bound to the complete stored receipt by
`stored_receipt_hash`, and one top-level pagination cursor. Missing, unsafe, stale, corrupt, or
unprovable disposable evidence is reported as a non-repairing receipt status. The command never
builds or repairs the index.
### Render and proposal commands
```text
render-status [VIEW_ID]
render-status [VIEW_ID] [--deep]
render VIEW_ID
graph-plan VIEW_ID
graph-render VIEW_ID
graph-render-status [VIEW_ID]
preview CHANGESET_ID VIEW_ID
apply CHANGESET_ID --changeset-hash SHA256 --applier WRITER_ID
```
`graph-plan` validates and returns one declared `GraphViewPlanV1` without publishing. A portable
view must select exactly one stable `root` or metadata-only lexical `query`. It may use only Nodes,
Flow, or Web as `initial_mode`; portable version 1 excludes function-scoped Logic.
`graph-render` explicitly publishes the declared static artifact, content-addressed renderer
evidence, and generation/view manifest. `graph-render-status` verifies only bounded committed
evidence and never plans or renders. Portable publication is a local CLI action.
The CLI apply command supports the generic adapter. It verifies that the configured writer owns the
changeset, applies the exact reviewed hash, rebuilds the index, checks it, and regenerates every
declared render. It does not commit or push the result.
declared manual render when manual policy is `auto`. It does not publish portable graphs, commit,
or push the result.
### Viewer commands
@ -390,6 +620,154 @@ visualization-status
visualization-stop
```
### Client configuration and doctor
Preview one deterministic standalone client fragment:
```bash
docforge configure codex --project /absolute/path/MyProject
docforge configure claude --project /absolute/path/MyProject
docforge configure openclaw --project /absolute/path/MyProject
```
Preview is the default. Add `--output /absolute/path/fragment` to create a new private fragment in
an existing real directory. Publication is create-only. DocForge accepts an identical existing
private single-link file as unchanged, but it never merges, replaces, broadens permissions, or
follows a symlink. Descriptor, parent, and target identities are revalidated across the
publication commit.
The generated command uses the exact current Python interpreter with isolated module startup.
Generation first proves that this interpreter can import `docforge.mcp_server`. The result binds
the project root, effective policy, arguments, artifact bytes, and all hashes. It copies no ambient
environment values.
Select authority explicitly:
```bash
docforge configure codex \
--project /absolute/path/MyProject \
--capability-mode proposal \
--proposal-writer project-editor
docforge configure codex \
--project /absolute/path/MyProject \
--capability-mode application \
--proposal-writer project-editor \
--canonical-applier project-editor
```
Read mode is the default. Proposal and application modes fail closed unless the descriptor
declares the named writer, and application requires the same writer/applier identity. Add
`--no-ast` to preserve the no-AST binding. Generic CLI generation refuses project-owned adapters
because it cannot safely reconstruct their composition.
A custom adapter owner supplies the already constructed project and immutable launcher through the
Python API:
```python
from docforge.adapter_launcher import AdapterLauncherV1
from docforge.client_config import generate_adapter_client_configuration
launcher = AdapterLauncherV1.for_project(
project,
module="my_project_docforge",
)
fragment = generate_adapter_client_configuration(
project,
launcher,
"codex",
capability_mode="read",
)
```
The top-level module must be installed for the exact isolated Python environment and resolve to a
regular file inside the project root. The fixed `docforge.reference_mcp` module is the only trusted
dotted exception. Generation probes resolution without importing the custom module, binds current
source availability and policy, and emits no arbitrary command, arguments, working directory, or
environment.
Select projection behavior independently:
```bash
docforge configure codex \
--project /absolute/path/MyProject \
--manual-render-policy explicit \
--portable-graph-policy disabled \
--live-viewer-policy on-demand
```
The generated version-1 configuration result carries an additive version-2 `projection_policy`,
its hash, projection availability, and the exact descriptor hash. Omitted default selectors are
validated against that descriptor rather than trusted as self-reported output.
Inspect one configured client binding:
```bash
docforge doctor --client codex --project /absolute/path/MyProject
docforge doctor --client codex \
--project /absolute/path/MyProject \
--config /absolute/path/config.toml \
--server-name my-project-docforge
```
Doctor returns `healthy`, `degraded`, or `unhealthy` with exit codes 0, 1, or 2. Its fixed
version-1 inventory checks project and descriptor binding, the client driver and entry, executable
and arguments, project root, effective policy, no-AST state, timeouts, environment-key names,
tool-filter representation, and stat-only index presence.
Doctor is intentionally not a connection test. It never loads canonical sources, opens SQLite,
starts MCP, executes the configured command, synchronizes, builds, renders, starts a viewer, or
writes configuration. Claude timeout representation and client filtering that cannot be proved
locally remain explicit warnings.
## Independent projection behavior
Manual and portable graph renderers consume immutable, path-free packages. A package binds one
generation-pinned plan, inert assets, fixed component versions, a built-in renderer identity, and
an exact artifact inventory. The detached child cannot select nodes, open the project or index,
choose a publication path, execute project code, or mutate canonical facts.
Child startup is fixed to isolated Python, a private module entrypoint, a trusted working
directory, and a sanitized environment. One request and response use canonical newline-terminated
JSON. The request, response, receipt, execution time, and disk-spooled stdout are bounded. Actual
artifact transfer is capped at 20,000,000 bytes even when the descriptor retains a larger
`max_render_bytes` compatibility value.
Manual fragment records are disposable semantic cache entries. On a cold miss, DocForge performs a
trusted full detached render, extracts candidate page fragments, and compares fragment-assisted
output byte-for-byte before publishing records. On a warm hit, the worker recomputes each expected
page fragment before accepting cached bytes. Corrupt, forged, stale, individually oversized, or
aggregate-oversized records fall back to the full oracle. Fragment reuse is currently a correctness
and recovery boundary, not a promised speedup.
Projection policy version 2 is:
```text
manual: auto | explicit | disabled
portable_graph: explicit | disabled
live_viewer: on-demand | disabled
```
For ordinary CLI commands, place the corresponding global flag before the subcommand:
```bash
docforge --project-root "$PROJECT" --manual-render-policy disabled render manual
docforge --project-root "$PROJECT" --portable-graph-policy disabled graph-plan architecture
docforge --project-root "$PROJECT" --live-viewer-policy disabled visualize
```
An active operation blocked by policy returns `projection_policy_forbids_operation` before hidden
work. Manual and portable receipt-only status remain available. Viewer status and explicit stop
remain available when viewer start is disabled.
A non-disabled projection also requires its declared configuration or runtime. Manual `explicit`
requires manual render configuration. Manual `auto` additionally requires canonical application in
the current operation or server capability. Portable graph `explicit` requires portable graph
render configuration, and live viewer `on-demand` requires its runtime. An unavailable selection
returns `projection_policy_unavailable` before work begins. In particular, ordinary CLI `render`
operations cannot select manual `auto`; use `explicit`, or let a configured canonical `apply`
operation own automatic regeneration.
## MCP usage
Run one MCP server per project with absolute paths:
@ -402,6 +780,31 @@ docforge-mcp \
Omit `--proposal-writer` when the MCP client should not create or append proposals.
For `.docforge/reference-adapter.toml`, use the fixed `docforge.reference_mcp` command shown in
[setup](#configure-a-reference-source-project). It exposes exactly the 21 read tools and never
registers proposal or application tools.
Select the session's declared surface explicitly when useful:
```bash
docforge-mcp \
--project-root /absolute/path/MyProject \
--capability-mode read \
--manual-render-policy explicit \
--portable-graph-policy explicit \
--live-viewer-policy on-demand
```
Supported modes are `read`, `proposal`, `application`, and `operator`. Existing startup defaults
remain compatible. Capability mode describes the registered surface; bootstrap separately reports
whether a configured writer or applier actually grants mutation access. Application mode refuses
startup without a canonical applier. Operator mode is reserved and currently adds no tools.
Add `--diagnostics` when profiling a development or benchmark session. Each MCP response then
includes bounded stage timings and compiler-work counters. The same flag is available on
`docforge`. Diagnostics are disabled by default, record no project content or paths, and never
displace a primary MCP result that already needs the configured output budget.
To expose canonical application, add a separate explicit startup gate:
```bash
@ -412,7 +815,20 @@ docforge-mcp \
```
Without `--canonical-applier`, `docforge_apply_changeset` is not registered. The flag is an
identity, not a command. The changeset creator, configured writer, and canonical applier must agree.
identity, not a command. Generic CLI and MCP application require the changeset creator, configured
writer, and canonical applier to agree.
A project-owned adapter server can separately pass `accepted_proposal_writers` to
`create_project_server`. This explicit allowlist lets its startup-bound applier accept an exact
reviewed changeset from another configured contributor identity. The default remains the applier
identity only. Accepted contributors retain their original proposal permissions and do not receive
canonical application authority.
Call `docforge_bootstrap` first. Its version-1 `session_contract` contains the fixed binding,
current graph generation, effective policy, actual capabilities, render policies, prohibitions,
and a recommended first operation. The result also carries the independently composed version-2
`projection_policy` and hash. Workflow guidance does not recommend registration or application
when those startup capabilities are unavailable.
Example MCP client configuration:
@ -436,29 +852,43 @@ Example MCP client configuration:
### Read tools
- `docforge_bootstrap`
- `docforge_sync`
- `docforge_project_info`
- `docforge_get_contract`
- `docforge_get_node`
- `docforge_get_logic`
- `docforge_search`
- `docforge_filter_nodes`
- `docforge_backlinks`
- `docforge_dependencies`
- `docforge_impact`
- `docforge_get_context`
- `docforge_get_task_context`
- `docforge_validate_project`
- `docforge_render_status`
- `docforge_graph_plan`
- `docforge_graph_render_status`
- `docforge_visualize`
- `docforge_visualization_status`
- `docforge_stop_visualization`
- `docforge_get_generation_diff`
MCP graph plan and status are read-only. MCP does not expose portable graph publication; use the
explicit local `graph-render` CLI command.
### Proposal tools
- `docforge_create_changeset`
- `docforge_register_changes`
- `docforge_list_changesets`
- `docforge_get_changeset`
- `docforge_rebase_changeset`
- `docforge_abandon_changeset`
- `docforge_propose_node_create`
- `docforge_propose_node_update`
- `docforge_propose_node_move`
- `docforge_propose_relationship_update`
- `docforge_propose_node_delete`
- `docforge_validate_changeset`
- `docforge_get_changeset_diff`
@ -472,33 +902,258 @@ The application call requires `changeset_id` and `expected_changeset_hash`. Alwa
inspect the final diff after the last proposal mutation. Apply that exact hash. A proposal mutation
creates a new hash, so an earlier approval cannot silently apply later content.
Recommended agent sequence:
Use `docforge_get_task_context` when an agent needs one bounded task-shaped intake instead of a
named profile. Choose `task_kind` from `change`, `implementation`, `failure`, `ownership`, `test`,
`operation`, or `release`. Supply `focus_node_id` when the stable node is known. Without it,
DocForge performs a bounded lexical focus search and refuses a tied best match instead of silently
choosing one.
1. Read the contract and relevant nodes.
2. Create a changeset.
3. Add structured operations using the hash returned by each previous mutation.
4. Validate the changeset.
5. Inspect its structured diff and preview.
6. Obtain human approval for the final changeset hash when required by the client workflow.
7. Call `docforge_apply_changeset` with that exact hash.
8. Report changed canonical files and derived refresh results.
The returned version-1 capsule includes:
- The exact project, adapter, source generation, effective policy, request, and retrieval-plan
hashes.
- Ordered focus and related evidence with source paths, content hashes, graph paths, and all
qualifying relationship reasons observed during the bounded traversal.
- Explicit evidence gaps and omissions, including whether a check completed.
- Provenance limitations for facts that the current graph does not carry, such as extractor
identity, observation time, and source provenance for relationships.
Project descriptors still own the valid relation vocabulary. The planner recognizes a fixed alias
map for structure, implementation, dependency, execution, data, evidence, and context. Any other
valid project relation is returned unchanged as `unclassified`; it is never assigned guessed task
semantics.
A relationship inside `relationship_path` describes the direction traveled from the preceding
node. A relationship inside `relationship_reasons` describes direction from the evidence item
itself. This keeps stored source and target identity exact while making each evidence explanation
locally readable.
Task context never exceeds 1,000 evidence items, 100,000 examined candidate edges, or 10,000 task
query characters, even when a project configures broader general limits. An edge-work or
unclassified-relation ceiling appears as an explicit omission rather than an unbounded response.
Use `docforge_get_generation_diff` after synchronization or a completed implementation slice to
inspect the one latest verified primary-graph transition. The version-1 receipt reports exact
added, removed, and changed node counts plus added and removed edge counts. Retained node details
identify changed fields and before/after hashes and source paths. Edge details retain the exact raw
relation triple. The receipt stores no source text, rendered content, Logic identities, or
historical sequence.
The first successful publication is an explicit baseline and does not claim every current node was
added. A corrupt, foreign, unsafe, or unavailable predecessor produces an unavailable comparison
rather than fabricated removals. A same-generation reindex preserves the latest meaningful
transition. Each later real transition atomically replaces the single disposable receipt.
Generation-diff reads use only the bounded receipt, stable file identities, and an adapter's cheap
source-generation proof. They do not open SQLite, load a complete adapter projection, parse source,
synchronize, build, or repair. Legacy adapters without cheap identity report `unknown`. Missing,
corrupt, foreign, oversized, or concurrently changed receipts report an explicit receipt state and
do not trigger hidden recovery.
Recommended release-candidate sequence:
1. Call `docforge_bootstrap`. It synchronizes derived state and reports the exact fixed binding.
2. Read only the relevant canonical context, implementation, configuration, tests, and release
rules.
3. Record the expected documentation impact in the working plan. Do not create or apply a
changeset yet.
4. Implement and run focused checks iteratively. Canonical documentation remains read-only during
this loop.
5. Freeze one release candidate after implementation stops changing.
6. Run the complete project gate, deployment preflight, candidate deployment, live checks, data
integrity checks, and release-identity checks.
7. If candidate validation fails, return to implementation. Do not document the failed candidate.
8. Call `docforge_sync` once after the candidate is green.
9. Call `docforge_register_changes` once with the complete operation list for every affected
canonical node.
10. Inspect the structured diff and every required preview.
11. Obtain human approval for the final changeset hash when required by the client workflow.
12. Call `docforge_apply_changeset` with that exact hash.
13. Run documentation-only validation and render checks.
14. Call `docforge_bootstrap` to verify the new canonical and derived identity.
15. Commit, tag, and publish the final revision containing both the verified implementation and
canonical documentation.
This cadence separates documentation intake from documentation publication. It avoids repeatedly
rewriting the manual around intermediate implementation states. One second documentation write is
allowed only for a narrow evidence correction that could not exist before deployment. If a late
check exposes an implementation defect, abandon or rebase the pending proposal and return to the
implementation loop.
The older create-and-append tools remain supported for interactive proposal construction.
`docforge_register_changes` avoids intermediate empty changesets and caller-managed hash chaining.
For update, move, and delete operations it captures the synchronized current node hash when
`expected_content_hash` is omitted.
MCP mutations are preflighted against the configured response limit. Small mutations keep their
full response. Large successful mutations return a compact or minimum version-1 receipt with
`mutation_committed = true` and the exact current changeset hash. A preflight size failure has
`mutation_committed = false`; it is safe to correct the request or policy before retrying. A
committed mutation is never reported as `result_too_large`.
Active changeset listing includes draft and ready proposals. Stale work remains available through
an explicit `status="stale"` query for rebase decisions. Applied and abandoned proposals are
terminal history, remain available by status or history request, and no longer block new proposals
against the same canonical base.
Context and changeset reads use version-1 continuation receipts when their evidence exceeds one
page. Follow `pagination.next_cursor` with the same tool and semantic arguments until
`pagination.has_more` is false. Page size may change between calls. Treat the cursor as opaque.
It is bound to the project, adapter, source generation, query, exact changeset hash, and collection
identity. `stale_cursor` means evidence changed between pages; discard prior pages and restart the
read instead of mixing generations.
`docforge_get_context` paginates one ordered evidence stream: selected entries followed by explicit
omissions. An entry too large for one MCP response is represented by a bounded omission carrying
its node ID and detail hash, and the cursor advances. `docforge_list_changesets`,
`docforge_get_changeset`, `docforge_validate_changeset`, and `docforge_get_changeset_diff` accept
the same optional `limit` and `cursor` fields. Small results keep their familiar fields. Large
inspection pages may use hash summaries. A large diff may return `result_mode =
"canonical_json_chunk"`; concatenate the chunks in order and verify `payload_hash` before decoding
the reconstructed `operations` and `changes` object.
`docforge_get_task_context` uses the same opaque continuation discipline over capsule evidence
followed by capsule omissions. Keep the semantic task arguments unchanged while paging. Page size
may change. Every page retains the same plan, collection, and capsule hashes. A `stale_cursor`
means that the generation, policy, plan, or collection changed; discard earlier pages and restart.
`docforge_get_generation_diff` paginates only the details retained in the latest bounded receipt.
Its summary counts and full collection hash still cover permanently truncated details. The cursor
binds the exact receipt, target generation, retained and full collection hashes, receipt state, and
effective policy. A replacement receipt returns `stale_cursor`; restart from its first page.
Canonical application records its terminal receipt immediately after the project-owned serializer
verifies the new canonical state. A later index or render refresh failure is reported as degraded
derived state with remediation, not as permission to apply the same canonical change again.
Likewise, failure to remove a private transaction artifact after semantic commit returns
`applied`, closes the proposal, and persists compact `application_recovery` lifecycle metadata
with `cleanup_required`, retained paths, and remediation. Inspect and remove only files proven to
be DocForge-owned.
Every successful declared render publishes a bounded version-1 receipt below the disposable cache.
Normal `render-status` compares cheap source-generation, view-configuration, template-file, and
output-file identities. It does not parse canonical nodes, prepare Markdown, construct HTML, or
hash the complete output. Missing or corrupt receipts are `unverified`; changed sources, templates,
or outputs are `stale`. Use `render-status --deep` only when explicitly requesting the
side-effect-free full-render equivalence oracle.
Use `docforge_propose_relationship_update` when the intended change is only an edge addition or
removal. It uses the same underlying validated update contract, but rejects empty relationship
lists and makes it explicit that node content will remain unchanged.
Custom adapters may expose the application tool only when they supply a project-owned
`CanonicalApplier`. Core DocForge will not guess how adapter nodes map back to canonical sources.
## Incremental adapter compilation
Release 1 complete-projection adapters remain supported. Adapters with large source trees can
implement the optional source-scoped manifest and extraction contract. DocForge then fingerprints
sources, reuses unchanged facts, reparses changed sources and their reverse dependents, validates a
complete candidate graph, and publishes the index atomically.
DocForge detects this capability structurally. An adapter without both `load_manifest()` and
`extract_source()` remains on the Release 1 path. Its behavior and query results are unchanged, but
it does not receive incremental performance until it opts in.
Build results report cache hits, reparsed sources, invalidated sources, deleted sources, and total
sources. A full projection remains the fallback and equivalence oracle.
Manual proposals remain separate from compilation. Applying an approved changeset updates
canonical sources first. Incremental compilation then notices those changed source fingerprints;
it never treats an unapplied proposal as canonical.
Function-scoped `LogicProjection` data is cached alongside its owning source but remains separate
from the primary Nodes, Flow, and Web graph. The Logic tab and `docforge_get_logic` load one
function or method on demand without adding every condition and basic block to ordinary graph
traversal.
See [Incremental Adapter Indexing](INCREMENTAL_INDEXING.md) for the complete contract, cache
invalidation rules, manual-application lifecycle, and lazy Logic boundary.
### Preserving an older non-AST adapter
Use `--no-ast` on the MCP binding when the project owner wants the existing adapter preserved
without AST, Tree-sitter, compiler-AST, or function-Logic upgrades:
```bash
docforge-mcp --project-root /absolute/project --no-ast
```
For a project-owned server, pass `no_ast=True` to `create_project_server()` or
`create_read_only_server()`. Bootstrap and contract responses then expose
`mode=preserve-no-ast`. The Logic tool is blocked, and DocForge refuses to publish nonempty Logic
projections.
This policy does not disable the Release 1 `load_projection()` path. It also permits incremental
fingerprinting and caching when those mechanisms do not add AST analysis. The adapter can
therefore benefit from current synchronization, proposals, application, rendering, and graph tools
without a source-analysis rewrite.
The binding rejects a pre-existing index containing Logic before reads or live visualization. A
configured canonical application service also refreshes through the same no-AST index policy.
DocForge does not inspect arbitrary adapter source to prove which parsing library it uses, so
repository permissions and project instructions remain responsible for adapter implementation
changes outside this process boundary.
## Troubleshooting
### `optional_dependency_missing`
Install the exact extra named in the error into the same Python environment that starts DocForge:
```bash
python -m pip install 'docforge[javascript]'
python -m pip install 'docforge[typescript]'
python -m pip install 'docforge[cpp]'
```
Do not install every frontend merely to suppress the check. A missing optional parser is a closed,
actionable capability error and does not affect base generic or Python reference operation.
### `adapter_launcher_unavailable` or `invalid_adapter_launcher`
Use one installed top-level Python module whose resolved regular-file origin is inside the project
root, or use the fixed `docforge.reference_mcp` binding. Arbitrary dotted modules, packages,
stdlib modules, missing modules, commands, argument strings, working directories, and environment
injection are rejected. Test the exact generated fragment rather than editing its command by hand.
See [agent integration](AGENT_INTEGRATION.md) and the [security model](SECURITY.md).
### `adapter_restart_required`
The project-local adapter code, its declared descriptor, or another implementation file changed
after the project-bound MCP process started. DocForge rejects every further operation before
synchronization because the live Python objects still represent the prior implementation.
Restart the MCP server or start a fresh client session. Do not stage files merely to change the
adapter's source manifest, and do not attempt in-process module reloading. The error includes
bounded added, changed, and deleted path evidence to identify the changed implementation boundary.
### `stale_index` or `visualization_stale`
Canonical sources changed after the index or viewer snapshot was built.
Normal MCP operations automatically repair a missing, stale, or invalid disposable index under a
project lock. `docforge_sync` can be called explicitly to inspect whether synchronization was a
no-op or rebuild. The CLI equivalent is:
```bash
docforge --project-root "$PROJECT" reindex
docforge --project-root "$PROJECT" sync
docforge --project-root "$PROJECT" visualize
```
An existing graph browser intentionally stays pinned to its original index identity. Reopen it
after reindexing.
after synchronization or reindexing.
Every complete index build also writes a disposable whole-file SHA-256 attestation. A new MCP
process verifies the unchanged database against that receipt instead of reconstructing every graph
row. Missing or mismatched receipts fall back to complete verification and are recreated only after
the full check succeeds.
Milestone 5 maintains exact recovery for four corrupt derived artifacts. Synchronization restores
a corrupt index attestation after complete verification. Explicit `render` restores a corrupt
manual receipt to the exact output and receipt semantics. A complete `reindex` recreates a corrupt
generation-diff baseline against the exact current graph. Explicit `graph-render` recreates a
corrupt portable-graph manifest and exact artifact. Status operations diagnose these conditions
without hidden repair.
### `visualization_manager_unavailable`
@ -538,14 +1193,34 @@ new hash rather than retrying with the old approval.
- `content_conflict`: a target node no longer has the expected content hash.
- `proposal_conflict`: another active proposal from the same base touches the same node or source.
Do not force apply. Rebase the intended changes into a new changeset after inspecting current
canonical content.
Do not force apply. Call `docforge_rebase_changeset` with the exact current changeset hash. DocForge
will rebind it only when every touched fact is unchanged and the proposal still validates. A
content or relationship conflict remains fail-closed and requires a newly reviewed proposal.
### `application_mismatch`
The written sources did not reproduce the validated projection. DocForge rolls the generic
canonical files back. For a custom adapter, fix its serializer or node-to-source mapping before
retrying.
The written sources did not reproduce the validated projection. During an ordinary in-process
failure, DocForge rolls generic canonical files back when their exact publication identities are
still provable. If another process raced a target, DocForge preserves foreign and displaced data
and returns `application_recovery_required` rather than overwriting either. For a custom adapter,
fix its serializer or node-to-source mapping before retrying.
### `application_recovery_required` or `cleanup_required`
`application_recovery_required` means canonical publication or rollback encountered concurrent or
unprovable state. Preserve every retained file named in the error. Compare it with the canonical
target and resolve the project before creating a newly reviewed proposal. Do not retry the old
approved hash.
`cleanup_required` means semantic application already committed. The proposal is closed as
`applied`, and its lifecycle receipt names private transaction artifacts that could not be removed.
Inspect those files and remove only confirmed DocForge-owned artifacts. The canonical change must
not be applied again.
Generic application uses mode-0700 transaction directories, but DocForge is not a filesystem
sandbox. Deliberate arbitrary tampering by another process running as the same operating-system
user is outside that integrity boundary. A process or host death can also interrupt a multi-file
application because canonical application has no process-death journal.
### `path_escape`, `unsafe_template`, or missing source
@ -562,8 +1237,8 @@ ambiguous adapter evidence.
### Full inspector content does not fit
DocForge 0.15 uses a fixed header and footer with a scrollable inspector body. If an older page is
still open, stop and reopen the visualization so it loads the current `graph-browser@15` template.
DocForge 1.4 uses a fixed header and footer with a scrollable inspector body. If an older page is
still open, stop and reopen the visualization so it loads the current `graph-browser@17` template.
### Render output is stale
@ -572,8 +1247,29 @@ docforge --project-root "$PROJECT" render-status
docforge --project-root "$PROJECT" render VIEW_ID
```
Successful canonical apply regenerates all declared views automatically. A manual canonical edit
requires reindexing and rendering.
Successful canonical apply regenerates declared manual views only when manual policy is `auto`.
A manual canonical edit requires reindexing and explicit rendering. Portable graph publication
always remains a separate explicit CLI action.
Portable graph publication has separate status and policy:
```bash
docforge --project-root "$PROJECT" graph-render-status
docforge --project-root "$PROJECT" graph-render architecture
```
### `projection_policy_forbids_operation`
The process was deliberately started with the relevant manual, portable-graph, or live-viewer
operation disabled. Restart with an allowed selector after confirming that the integration should
receive that capability. Status and explicit stop operations remain available as described above.
### `projection_policy_unavailable`
The selected non-disabled projection has no matching project configuration or runtime. Add the
declared manual or portable graph render configuration, or make the live viewer runtime available,
before selecting that mode. Manual `auto` also requires an operation or MCP server with canonical
application enabled. Use manual `explicit` for a standalone CLI render.
### Descriptor changed after startup
@ -582,16 +1278,40 @@ the process so it binds the new descriptor deliberately.
## Development and verification
Run the complete release gate from the DocForge repository:
Run the ordinary repository gate:
```bash
npx pyright
npm run lint:web
uv run ruff check src tests tools
uv run ruff format --check src tests tools
uv run python -m compileall -q src tests tools
uv run pytest -q
make gate
```
Use `make benchmark` for the historical Milestone 0 baseline, `make benchmark-m1` for the
counter-gated warm-operation benchmark, `make benchmark-m2` for agent workflow gates, and
`make benchmark-m3-full` for the ten-sample 1,000-node projection, worker, fragment, status,
equivalence, response-size, and memory gates. `make benchmark-m4-full` runs the 1,002-node adapter
and recovery benchmark. `make adoption-m4` performs the offline fresh-wheel proof.
`make command-reference-check` rejects command-reference drift, and `make docs-check` validates
the maintained documentation graph. `make accessibility` runs the generated manual, portable
graph, and live viewer axe and keyboard flows.
Milestone 5 adds maintained compatibility, migration, concurrency, recovery, comparative-task,
release-identity, reproducible-artifact, secret-scan, and fresh-clone gates:
```bash
make compatibility-m5
make migration-m5
make concurrency-m5
make recovery-m5
make task-evidence-m5
make release-gate
make fresh-clone-m5
```
`release-gate` aggregates the full quality, browser, compatibility, migration, concurrency,
recovery, task-evidence, fresh-wheel, version, artifact, secret-scan, and benchmark suite.
`fresh-clone-m5` anonymously clones the exact published candidate over HTTPS, fetches and verifies
the frozen annotated `v1.0.0` migration tag, and repeats `release-gate`. Release operators use
`make release-pretag` before creating `v2.0.0` and `make release-posttag` after the annotated tag
points to the exact release commit.
Project-specific vocabulary, extraction rules, and serialization belong in the project adapter.
Generic core behavior must remain deterministic, project-bound, and recoverable.

View file

@ -21,4 +21,26 @@ export default [
"prefer-const": "error",
},
},
{
files: ["playwright.accessibility.config.mjs", "tests/accessibility.spec.mjs"],
...js.configs.recommended,
languageOptions: {
ecmaVersion: 2024,
sourceType: "module",
globals: {
...globals.browser,
...globals.node,
},
},
linterOptions: {
reportUnusedDisableDirectives: "error",
},
rules: {
...js.configs.recommended.rules,
eqeqeq: "error",
"no-implicit-coercion": "error",
"no-var": "error",
"prefer-const": "error",
},
},
];

88
package-lock.json generated
View file

@ -8,7 +8,9 @@
"name": "docforge-web-quality",
"version": "0.0.0",
"devDependencies": {
"@axe-core/playwright": "4.12.1",
"@eslint/js": "10.0.1",
"@playwright/test": "1.62.0",
"eslint": "10.8.0",
"globals": "17.7.0",
"html-validate": "11.5.6",
@ -18,6 +20,19 @@
"stylelint-csstree-validator": "4.0.0"
}
},
"node_modules/@axe-core/playwright": {
"version": "4.12.1",
"resolved": "https://registry.npmjs.org/@axe-core/playwright/-/playwright-4.12.1.tgz",
"integrity": "sha512-rMd7xriptqKpP+w5265i4Hdkv2X5kbu6uiBi/B2I7uf3hieRBM3qDCfaKPtxfiYb2mKXfF+yLODJwIx+Jv1GDw==",
"dev": true,
"license": "MPL-2.0",
"dependencies": {
"axe-core": "~4.12.1"
},
"peerDependencies": {
"playwright-core": ">= 1.0.0"
}
},
"node_modules/@babel/code-frame": {
"version": "7.29.7",
"resolved": "https://registry.npmjs.org/@babel/code-frame/-/code-frame-7.29.7.tgz",
@ -515,6 +530,22 @@
"node": ">= 8"
}
},
"node_modules/@playwright/test": {
"version": "1.62.0",
"resolved": "https://registry.npmjs.org/@playwright/test/-/test-1.62.0.tgz",
"integrity": "sha512-9zOJ6ZQRAena31MpOH9VSzIz8Ou3YJ/wtY/eQm5T2uhfhG7/U3COrMS8xOtUrZrp9OgdmzEnIYODye3nY1VqzA==",
"dev": true,
"license": "Apache-2.0",
"dependencies": {
"playwright": "1.62.0"
},
"bin": {
"playwright": "cli.js"
},
"engines": {
"node": ">=20"
}
},
"node_modules/@sindresorhus/merge-streams": {
"version": "4.0.0",
"resolved": "https://registry.npmjs.org/@sindresorhus/merge-streams/-/merge-streams-4.0.0.tgz",
@ -635,6 +666,16 @@
"node": ">=8"
}
},
"node_modules/axe-core": {
"version": "4.12.1",
"resolved": "https://registry.npmjs.org/axe-core/-/axe-core-4.12.1.tgz",
"integrity": "sha512-s7iGf5GaVMxEG0ENN9x+xTr7GFZCb1ZP/1uATUpCEK2X78nDB3RwbtFCo9pGAf9ru+VwoQ464DkaLEeRM08wJA==",
"dev": true,
"license": "MPL-2.0",
"engines": {
"node": ">=4"
}
},
"node_modules/balanced-match": {
"version": "4.0.4",
"resolved": "https://registry.npmjs.org/balanced-match/-/balanced-match-4.0.4.tgz",
@ -1943,6 +1984,53 @@
"url": "https://github.com/sponsors/jonschlinkert"
}
},
"node_modules/playwright": {
"version": "1.62.0",
"resolved": "https://registry.npmjs.org/playwright/-/playwright-1.62.0.tgz",
"integrity": "sha512-Z14dG305dgaLu6foB1TXQagFiW8JfSUIUaUuPaKQ6NtBPKF1P/qXcqfh6c6K/icPqdy37JmjbiBXf6JNg6Sylw==",
"dev": true,
"license": "Apache-2.0",
"dependencies": {
"playwright-core": "1.62.0"
},
"bin": {
"playwright": "cli.js"
},
"engines": {
"node": ">=20"
},
"optionalDependencies": {
"fsevents": "2.3.2"
}
},
"node_modules/playwright-core": {
"version": "1.62.0",
"resolved": "https://registry.npmjs.org/playwright-core/-/playwright-core-1.62.0.tgz",
"integrity": "sha512-nsNRyq0r2zsG8AcRHWknc9QRA5XCueC7gWMrs+Gx2tlZn9hcl8zudfh00lhJPY1DE7NmZ6bDsT9g2yey8mXljA==",
"dev": true,
"license": "Apache-2.0",
"bin": {
"playwright-core": "cli.js"
},
"engines": {
"node": ">=20"
}
},
"node_modules/playwright/node_modules/fsevents": {
"version": "2.3.2",
"resolved": "https://registry.npmjs.org/fsevents/-/fsevents-2.3.2.tgz",
"integrity": "sha512-xiqMQR4xAeHTuB9uWm+fFRcIOgKBMiOBP+eXiyT7jsgVCq1bkVygt00oASowB7EdtpOHaaPgKt812P9ab+DDKA==",
"dev": true,
"hasInstallScript": true,
"license": "MIT",
"optional": true,
"os": [
"darwin"
],
"engines": {
"node": "^8.16.0 || ^10.6.0 || >=11.0.0"
}
},
"node_modules/postcss": {
"version": "8.5.23",
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.23.tgz",

View file

@ -4,10 +4,14 @@
"private": true,
"packageManager": "npm@10.9.7",
"scripts": {
"lint:web": "uv run python tools/check_web_assets.py"
"install:accessibility-browser": "playwright install chromium",
"lint:web": "uv run python tools/check_web_assets.py && eslint --max-warnings=0 playwright.accessibility.config.mjs tests/accessibility.spec.mjs",
"test:accessibility": "npm run install:accessibility-browser && playwright test --config=playwright.accessibility.config.mjs"
},
"devDependencies": {
"@axe-core/playwright": "4.12.1",
"@eslint/js": "10.0.1",
"@playwright/test": "1.62.0",
"eslint": "10.8.0",
"globals": "17.7.0",
"html-validate": "11.5.6",

View file

@ -0,0 +1,24 @@
import { defineConfig } from "@playwright/test";
export default defineConfig({
testDir: "./tests",
testMatch: "accessibility.spec.mjs",
fullyParallel: false,
workers: 1,
retries: 0,
reporter: "line",
outputDir: "/tmp/docforge-playwright-accessibility",
timeout: 30_000,
expect: {
timeout: 5_000,
},
use: {
browserName: "chromium",
bypassCSP: true,
headless: true,
viewport: {
width: 1440,
height: 1000,
},
},
});

View file

@ -4,16 +4,51 @@ build-backend = "hatchling.build"
[project]
name = "docforge"
version = "1.0.0"
dynamic = ["version"]
description = "Project-scoped documentation indexing and context service"
readme = "README.md"
requires-python = ">=3.12"
license = { text = "MIT" }
license = "MIT"
authors = [{ name = "Worldforge contributors" }]
dependencies = ["markdown-it-py>=4.2,<5", "mcp>=1.28,<2"]
dependencies = [
"markdown-it-py>=4.2,<5",
"mcp>=1.28,<2",
]
[project.urls]
Repository = "https://repo.andraxion.net/administrator/DocForge2"
Issues = "https://repo.andraxion.net/administrator/DocForge2/issues"
[project.optional-dependencies]
javascript = [
"tree-sitter>=0.25,<0.26",
"tree-sitter-javascript>=0.25,<0.26",
]
typescript = [
"tree-sitter>=0.25,<0.26",
"tree-sitter-typescript>=0.23,<0.24",
]
cpp = [
"tree-sitter>=0.25,<0.26",
"tree-sitter-cpp>=0.23,<0.24",
]
languages = [
"tree-sitter>=0.25,<0.26",
"tree-sitter-cpp>=0.23,<0.24",
"tree-sitter-javascript>=0.25,<0.26",
"tree-sitter-typescript>=0.23,<0.24",
]
[dependency-groups]
dev = ["pytest>=9.1,<10", "ruff>=0.15,<1"]
dev = [
"jsonschema>=4.25,<5",
"pytest>=9.1,<10",
"ruff>=0.15,<1",
"tree-sitter>=0.25,<0.26",
"tree-sitter-cpp>=0.23,<0.24",
"tree-sitter-javascript>=0.25,<0.26",
"tree-sitter-typescript>=0.23,<0.24",
]
[project.scripts]
docforge = "docforge.cli:main"
@ -21,7 +56,13 @@ docforge-mcp = "docforge.mcp_server:main"
docforge-viewer-manager = "docforge.viewer_manager:main"
[tool.hatch.build.targets.wheel]
packages = ["src/docforge"]
packages = ["src/docforge", "src/docforge_renderers"]
[tool.hatch.build.targets.wheel.force-include]
schemas = "docforge/schemas"
[tool.hatch.version]
path = "src/docforge/_version.py"
[tool.ruff]
line-length = 100

View file

@ -0,0 +1,364 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/adapter-client-configuration-v1.json",
"title": "DocForge project-owned adapter client configuration",
"$defs": {
"sha256": {
"type": "string",
"pattern": "^[0-9a-f]{64}$"
},
"project": {
"type": "object",
"required": [
"project_id",
"project_root",
"project_root_fingerprint",
"adapter",
"descriptor_hash"
],
"properties": {
"project_id": {
"type": "string",
"pattern": "^[a-z0-9][a-z0-9._-]{1,127}$"
},
"project_root": {
"type": "string",
"minLength": 1,
"maxLength": 4096
},
"project_root_fingerprint": {
"type": "string",
"pattern": "^[0-9a-f]{16}$"
},
"adapter": {
"type": "string",
"pattern": "^[A-Za-z0-9][A-Za-z0-9_.-]{0,127}@[A-Za-z0-9][A-Za-z0-9_.+-]{0,63}$"
},
"descriptor_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
},
"source_availability": {
"type": "object",
"required": [
"schema_version",
"status",
"method",
"revision",
"source_hash"
],
"properties": {
"schema_version": { "const": 1 },
"status": { "const": "available" },
"method": {
"enum": ["incremental-state", "complete-projection"]
},
"revision": {
"type": "string",
"minLength": 1,
"maxLength": 4096
},
"source_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
},
"binding": {
"type": "object",
"required": [
"transport",
"capability_mode",
"adapter_policy",
"render_policy",
"command",
"args",
"environment",
"timeouts",
"launcher_hash"
],
"properties": {
"transport": { "const": "stdio" },
"capability_mode": {
"enum": ["read", "proposal", "application"]
},
"adapter_policy": {
"$ref": "https://docforge.local/schema/client-configuration-v1.json#/$defs/adapter_policy"
},
"render_policy": {
"type": "object",
"required": ["manual", "graph", "live_viewer"],
"properties": {
"manual": { "enum": ["auto", "explicit", "disabled"] },
"graph": { "const": "disabled" },
"live_viewer": { "const": "on-demand" }
},
"additionalProperties": false
},
"command": {
"type": "string",
"minLength": 1,
"maxLength": 4096
},
"args": {
"type": "array",
"minItems": 7,
"maxItems": 64,
"prefixItems": [
{ "const": "-I" },
{ "const": "-m" },
{
"type": "string",
"minLength": 1,
"maxLength": 255,
"pattern": "^[A-Za-z_][A-Za-z0-9_]*(\\.[A-Za-z_][A-Za-z0-9_]*){0,31}$",
"not": { "const": "docforge.mcp_server" }
},
{ "const": "--project-root" },
{
"type": "string",
"minLength": 1,
"maxLength": 4096
},
{ "const": "--capability-mode" },
{ "enum": ["read", "proposal", "application"] }
],
"items": {
"type": "string",
"minLength": 1,
"maxLength": 4096
}
},
"environment": {
"type": "object",
"maxProperties": 0
},
"timeouts": {
"type": "object",
"required": ["startup_seconds", "tool_seconds"],
"properties": {
"startup_seconds": {
"type": "integer",
"minimum": 1,
"maximum": 3600
},
"tool_seconds": {
"type": "integer",
"minimum": 1,
"maximum": 86400
}
},
"additionalProperties": false
},
"launcher_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
},
"projection_availability": {
"type": "object",
"required": [
"manual_configured",
"portable_graph_configured",
"application_enabled",
"live_viewer_available"
],
"properties": {
"manual_configured": { "type": "boolean" },
"portable_graph_configured": { "type": "boolean" },
"application_enabled": { "type": "boolean" },
"live_viewer_available": { "const": true }
},
"additionalProperties": false
}
},
"type": "object",
"required": [
"status",
"schema_version",
"docforge_version",
"operation",
"action",
"client",
"server_name",
"project",
"launcher",
"launcher_hash",
"source_availability",
"source_availability_hash",
"binding",
"effective_policy",
"projection_policy",
"projection_policy_hash",
"projection_availability",
"artifact",
"configuration_hash",
"warnings"
],
"properties": {
"status": { "const": "ok" },
"schema_version": { "const": 1 },
"docforge_version": {
"type": "string",
"minLength": 1,
"maxLength": 128
},
"operation": { "const": "adapter_client.configure" },
"action": { "enum": ["preview", "write"] },
"client": { "enum": ["codex", "claude", "openclaw"] },
"server_name": {
"type": "string",
"pattern": "^[a-z0-9][a-z0-9_-]{0,63}$"
},
"project": { "$ref": "#/$defs/project" },
"launcher": {
"$ref": "https://docforge.local/schema/adapter-launcher-v1.json"
},
"launcher_hash": { "$ref": "#/$defs/sha256" },
"source_availability": {
"$ref": "#/$defs/source_availability"
},
"source_availability_hash": { "$ref": "#/$defs/sha256" },
"binding": { "$ref": "#/$defs/binding" },
"effective_policy": {
"$ref": "https://docforge.local/schema/client-configuration-v1.json#/$defs/effective_policy"
},
"projection_policy": {
"$ref": "https://docforge.local/schema/client-configuration-v1.json#/$defs/projection_policy"
},
"projection_policy_hash": { "$ref": "#/$defs/sha256" },
"projection_availability": {
"$ref": "#/$defs/projection_availability"
},
"artifact": {
"$ref": "https://docforge.local/schema/client-configuration-v1.json#/properties/artifact"
},
"configuration_hash": { "$ref": "#/$defs/sha256" },
"warnings": {
"$ref": "https://docforge.local/schema/client-configuration-v1.json#/properties/warnings"
}
},
"allOf": [
{
"if": {
"properties": { "client": { "const": "codex" } },
"required": ["client"]
},
"then": {
"properties": {
"artifact": {
"properties": {
"format": { "const": "codex-toml-fragment-v1" }
}
}
}
}
},
{
"if": {
"properties": { "client": { "const": "claude" } },
"required": ["client"]
},
"then": {
"properties": {
"artifact": {
"properties": {
"format": { "const": "claude-json-fragment-v1" }
}
},
"warnings": {
"contains": {
"properties": {
"code": { "const": "timeout_format_unverified" }
},
"required": ["code"]
}
}
}
}
},
{
"if": {
"properties": { "client": { "const": "openclaw" } },
"required": ["client"]
},
"then": {
"properties": {
"artifact": {
"properties": {
"format": { "const": "openclaw-json-fragment-v1" }
}
}
}
}
},
{
"if": {
"properties": { "action": { "const": "preview" } },
"required": ["action"]
},
"then": {
"properties": {
"artifact": {
"properties": {
"output_path": { "type": "null" },
"write_state": { "const": "not_requested" },
"durability": { "const": "not_applicable" }
}
}
}
},
"else": {
"properties": {
"artifact": {
"properties": {
"write_state": { "enum": ["created", "unchanged"] }
}
}
}
}
},
{
"if": {
"properties": {
"binding": {
"properties": {
"adapter_policy": {
"properties": {
"mode": { "const": "preserve-no-ast" }
},
"required": ["mode"]
}
},
"required": ["adapter_policy"]
}
},
"required": ["binding"]
},
"then": {
"properties": {
"binding": {
"properties": {
"args": {
"contains": { "const": "--no-ast" },
"minContains": 1,
"maxContains": 1
}
}
}
}
},
"else": {
"properties": {
"binding": {
"properties": {
"args": {
"not": {
"contains": { "const": "--no-ast" }
}
}
}
}
}
}
}
],
"additionalProperties": false
}

View file

@ -0,0 +1,46 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/adapter-launcher-v1.json",
"title": "DocForge project-owned adapter launcher",
"type": "object",
"required": [
"schema_version",
"entry_point",
"project_id",
"project_root",
"adapter",
"descriptor_hash",
"module"
],
"properties": {
"schema_version": { "const": 1 },
"entry_point": { "const": "python-module" },
"project_id": {
"type": "string",
"pattern": "^[a-z0-9][a-z0-9._-]{1,127}$"
},
"project_root": {
"type": "string",
"minLength": 1,
"maxLength": 4096
},
"adapter": {
"type": "string",
"pattern": "^[A-Za-z0-9][A-Za-z0-9_.-]{0,127}@[A-Za-z0-9][A-Za-z0-9_.+-]{0,63}$"
},
"descriptor_hash": {
"type": "string",
"pattern": "^[0-9a-f]{64}$"
},
"module": {
"type": "string",
"minLength": 1,
"maxLength": 255,
"anyOf": [
{ "pattern": "^[A-Za-z_][A-Za-z0-9_]*$" },
{ "const": "docforge.reference_mcp" }
]
}
},
"additionalProperties": false
}

View file

@ -0,0 +1,934 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/client-configuration-v1.json",
"title": "DocForge deterministic client configuration plan",
"$defs": {
"sha256": {
"type": "string",
"pattern": "^[0-9a-f]{64}$"
},
"adapter_policy": {
"oneOf": [
{
"type": "object",
"required": [
"mode",
"ast_analysis",
"logic_projection",
"incremental_extraction",
"adapter_rewrite"
],
"properties": {
"mode": { "const": "standard" },
"ast_analysis": { "const": "allowed" },
"logic_projection": { "const": "allowed" },
"incremental_extraction": { "const": "allowed" },
"adapter_rewrite": { "const": "not_requested" }
},
"additionalProperties": false
},
{
"type": "object",
"required": [
"mode",
"ast_analysis",
"logic_projection",
"incremental_extraction",
"adapter_rewrite",
"blocked_tools",
"instruction"
],
"properties": {
"mode": { "const": "preserve-no-ast" },
"ast_analysis": { "const": "forbidden" },
"logic_projection": { "const": "forbidden" },
"incremental_extraction": { "const": "allowed" },
"adapter_rewrite": { "const": "forbidden" },
"blocked_tools": {
"const": ["docforge_get_logic"]
},
"instruction": {
"const": "Preserve the existing adapter extraction strategy. Do not add Python AST, Tree-sitter, compiler-AST, or function-Logic extraction. Non-AST incremental fingerprinting and caching remain allowed."
}
},
"additionalProperties": false
}
]
},
"effective_policy": {
"type": "object",
"required": [
"schema_version",
"capability_mode",
"capability_source",
"adapter_evolution",
"ast_analysis",
"logic_indexing",
"synchronization",
"integrity",
"manual_render",
"graph_render",
"live_viewer",
"profiling",
"blocked_tools",
"prohibitions",
"precedence"
],
"properties": {
"schema_version": { "const": 1 },
"capability_mode": {
"enum": ["read", "proposal", "application"]
},
"capability_source": { "const": "explicit" },
"adapter_evolution": { "enum": ["allowed", "preserve"] },
"ast_analysis": { "enum": ["allowed", "forbidden"] },
"logic_indexing": { "enum": ["full", "off"] },
"synchronization": { "const": "automatic" },
"integrity": { "const": "validated" },
"manual_render": { "enum": ["auto", "explicit", "disabled"] },
"graph_render": { "const": "disabled" },
"live_viewer": { "const": "on-demand" },
"profiling": { "const": "disabled" },
"blocked_tools": {
"type": "array",
"maxItems": 1,
"items": { "const": "docforge_get_logic" },
"uniqueItems": true
},
"prohibitions": {
"type": "array",
"minItems": 7,
"maxItems": 11,
"items": {
"enum": [
"arbitrary_file_access",
"arbitrary_renderer_execution",
"shell_execution",
"git_mutation",
"deployment",
"publication",
"project_switching",
"adapter_ast_upgrade",
"tree_sitter_upgrade",
"compiler_ast_upgrade",
"function_logic_extraction"
]
},
"uniqueItems": true
},
"precedence": {
"const": [
"core_safety",
"explicit_binding",
"no_ast_shorthand",
"resource_availability"
]
}
},
"additionalProperties": false
},
"projection_policy": {
"type": "object",
"required": ["schema_version", "manual", "portable_graph", "live_viewer"],
"properties": {
"schema_version": { "const": 2 },
"manual": { "enum": ["auto", "explicit", "disabled"] },
"portable_graph": { "enum": ["explicit", "disabled"] },
"live_viewer": { "enum": ["on-demand", "disabled"] }
},
"additionalProperties": false
},
"diagnostics": {
"type": "object",
"required": [
"schema_version",
"operation",
"outcome",
"elapsed_ns",
"stages",
"counters"
],
"properties": {
"schema_version": { "const": 1 },
"operation": { "const": "cli.configure" },
"outcome": { "const": "ok" },
"elapsed_ns": { "type": "integer", "minimum": 0 },
"stages": {
"type": "object",
"maxProperties": 14,
"propertyNames": {
"enum": [
"source.generation",
"source.parse",
"adapter.projection",
"adapter.extract",
"index.check",
"index.synchronize",
"index.build",
"index.read",
"render.status",
"render.prepare",
"render.output_hash",
"visualization.status",
"viewer.manager",
"mcp.runtime_validation"
]
},
"additionalProperties": {
"type": "object",
"required": ["calls", "elapsed_ns"],
"properties": {
"calls": { "type": "integer", "minimum": 1 },
"elapsed_ns": { "type": "integer", "minimum": 0 }
},
"additionalProperties": false
}
},
"counters": {
"type": "object",
"required": [
"project_loads",
"source_files_parsed",
"source_bytes_parsed",
"adapter_projection_loads",
"adapter_source_extractions",
"source_generation_checks",
"index_checks",
"index_synchronizations",
"index_builds",
"render_prepare_calls",
"render_output_bytes_built",
"render_output_bytes_hashed",
"viewer_manager_requests"
],
"additionalProperties": {
"type": "integer",
"minimum": 0
},
"maxProperties": 13
}
},
"additionalProperties": false
}
},
"type": "object",
"required": [
"status",
"schema_version",
"docforge_version",
"operation",
"action",
"client",
"server_name",
"project",
"binding",
"effective_policy",
"projection_policy",
"projection_policy_hash",
"projection_availability",
"artifact",
"configuration_hash",
"warnings"
],
"properties": {
"status": { "const": "ok" },
"schema_version": { "const": 1 },
"docforge_version": {
"type": "string",
"minLength": 1,
"maxLength": 128
},
"operation": { "const": "client.configure" },
"action": { "enum": ["preview", "write"] },
"client": { "enum": ["codex", "claude", "openclaw"] },
"server_name": {
"type": "string",
"pattern": "^[a-z0-9][a-z0-9_-]{0,63}$"
},
"project": {
"type": "object",
"required": [
"project_id",
"project_root",
"project_root_fingerprint",
"adapter",
"descriptor_hash"
],
"properties": {
"project_id": { "type": "string", "minLength": 1 },
"project_root": { "type": "string", "minLength": 1 },
"project_root_fingerprint": {
"type": "string",
"pattern": "^[0-9a-f]{16}$"
},
"adapter": { "type": "string", "minLength": 1 },
"descriptor_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
},
"binding": {
"type": "object",
"required": [
"transport",
"capability_mode",
"adapter_policy",
"render_policy",
"command",
"args",
"environment",
"timeouts"
],
"properties": {
"transport": { "const": "stdio" },
"capability_mode": {
"enum": ["read", "proposal", "application"]
},
"adapter_policy": { "$ref": "#/$defs/adapter_policy" },
"render_policy": {
"type": "object",
"required": ["manual", "graph", "live_viewer"],
"properties": {
"manual": { "enum": ["auto", "explicit", "disabled"] },
"graph": { "const": "disabled" },
"live_viewer": { "const": "on-demand" }
},
"additionalProperties": false
},
"command": { "type": "string", "minLength": 1 },
"args": {
"type": "array",
"minItems": 7,
"maxItems": 64,
"prefixItems": [
{ "const": "-I" },
{ "const": "-m" },
{ "const": "docforge.mcp_server" },
{ "const": "--project-root" },
{ "type": "string", "minLength": 1, "maxLength": 4096 },
{ "const": "--capability-mode" },
{ "enum": ["read", "proposal", "application"] }
],
"items": {
"type": "string",
"minLength": 1,
"maxLength": 4096
}
},
"environment": {
"type": "object",
"maxProperties": 0
},
"timeouts": {
"type": "object",
"required": ["startup_seconds", "tool_seconds"],
"properties": {
"startup_seconds": {
"type": "integer",
"minimum": 1,
"maximum": 3600
},
"tool_seconds": {
"type": "integer",
"minimum": 1,
"maximum": 86400
}
},
"additionalProperties": false
}
},
"additionalProperties": false
},
"effective_policy": { "$ref": "#/$defs/effective_policy" },
"projection_policy": { "$ref": "#/$defs/projection_policy" },
"projection_policy_hash": { "$ref": "#/$defs/sha256" },
"projection_availability": {
"type": "object",
"required": [
"manual_configured",
"portable_graph_configured",
"application_enabled",
"live_viewer_available"
],
"properties": {
"manual_configured": { "type": "boolean" },
"portable_graph_configured": { "type": "boolean" },
"application_enabled": { "type": "boolean" },
"live_viewer_available": { "const": true }
},
"additionalProperties": false
},
"artifact": {
"type": "object",
"required": [
"format",
"content",
"content_sha256",
"output_path",
"write_state",
"durability"
],
"properties": {
"format": {
"enum": [
"codex-toml-fragment-v1",
"claude-json-fragment-v1",
"openclaw-json-fragment-v1"
]
},
"content": {
"type": "string",
"minLength": 1,
"maxLength": 65536
},
"content_sha256": { "$ref": "#/$defs/sha256" },
"output_path": {
"type": ["string", "null"],
"minLength": 1
},
"write_state": {
"enum": ["not_requested", "created", "unchanged"]
},
"durability": {
"enum": ["not_applicable", "confirmed", "unconfirmed"]
}
},
"additionalProperties": false
},
"configuration_hash": { "$ref": "#/$defs/sha256" },
"warnings": {
"type": "array",
"maxItems": 8,
"items": {
"type": "object",
"required": ["code"],
"properties": {
"code": {
"enum": [
"timeout_format_unverified",
"publication_durability_unconfirmed",
"publication_location_unconfirmed",
"publication_binding_unconfirmed"
]
}
},
"additionalProperties": false
}
},
"diagnostics": { "$ref": "#/$defs/diagnostics" }
},
"allOf": [
{
"if": {
"properties": {
"warnings": {
"contains": {
"properties": {
"code": { "const": "publication_binding_unconfirmed" }
},
"required": ["code"]
}
}
},
"required": ["warnings"]
},
"then": {
"properties": {
"artifact": {
"properties": {
"output_path": { "type": "null" },
"write_state": { "const": "created" },
"durability": { "const": "unconfirmed" }
}
}
}
}
},
{
"if": {
"properties": { "client": { "const": "codex" } },
"required": ["client"]
},
"then": {
"properties": {
"artifact": {
"properties": {
"format": { "const": "codex-toml-fragment-v1" }
}
},
"warnings": {
"not": {
"contains": {
"properties": {
"code": { "const": "timeout_format_unverified" }
},
"required": ["code"]
}
}
}
}
}
},
{
"if": {
"properties": { "client": { "const": "openclaw" } },
"required": ["client"]
},
"then": {
"properties": {
"artifact": {
"properties": {
"format": { "const": "openclaw-json-fragment-v1" }
}
},
"warnings": {
"not": {
"contains": {
"properties": {
"code": { "const": "timeout_format_unverified" }
},
"required": ["code"]
}
}
}
}
}
},
{
"if": {
"properties": { "client": { "const": "claude" } },
"required": ["client"]
},
"then": {
"properties": {
"artifact": {
"properties": {
"format": { "const": "claude-json-fragment-v1" }
}
},
"warnings": {
"contains": {
"properties": {
"code": { "const": "timeout_format_unverified" }
},
"required": ["code"]
}
}
}
}
},
{
"if": {
"properties": { "action": { "const": "preview" } },
"required": ["action"]
},
"then": {
"properties": {
"artifact": {
"properties": {
"output_path": { "type": "null" },
"write_state": { "const": "not_requested" },
"durability": { "const": "not_applicable" }
}
}
}
},
"else": {
"properties": {
"artifact": {
"properties": {
"output_path": {
"type": ["string", "null"],
"minLength": 1
},
"write_state": { "enum": ["created", "unchanged"] }
},
"allOf": [
{
"if": {
"properties": { "write_state": { "const": "created" } },
"required": ["write_state"]
},
"then": {
"properties": {
"durability": { "enum": ["confirmed", "unconfirmed"] }
}
},
"else": {
"properties": {
"durability": { "const": "not_applicable" }
}
}
}
]
}
}
}
},
{
"if": {
"properties": {
"binding": {
"properties": { "capability_mode": { "const": "read" } },
"required": ["capability_mode"]
}
},
"required": ["binding"]
},
"then": {
"properties": {
"binding": {
"properties": {
"args": {
"prefixItems": [{}, {}, {}, {}, {}, {}, { "const": "read" }]
}
}
},
"effective_policy": {
"properties": { "capability_mode": { "const": "read" } }
}
}
}
},
{
"if": {
"properties": {
"binding": {
"properties": { "capability_mode": { "const": "proposal" } },
"required": ["capability_mode"]
}
},
"required": ["binding"]
},
"then": {
"properties": {
"binding": {
"properties": {
"args": {
"prefixItems": [{}, {}, {}, {}, {}, {}, { "const": "proposal" }]
}
}
},
"effective_policy": {
"properties": { "capability_mode": { "const": "proposal" } }
}
}
}
},
{
"if": {
"properties": {
"binding": {
"properties": { "capability_mode": { "const": "application" } },
"required": ["capability_mode"]
}
},
"required": ["binding"]
},
"then": {
"properties": {
"binding": {
"properties": {
"args": {
"prefixItems": [{}, {}, {}, {}, {}, {}, { "const": "application" }]
}
}
},
"effective_policy": {
"properties": { "capability_mode": { "const": "application" } }
}
}
}
},
{
"if": {
"properties": {
"binding": {
"properties": {
"adapter_policy": {
"properties": { "mode": { "const": "preserve-no-ast" } },
"required": ["mode"]
}
},
"required": ["adapter_policy"]
}
},
"required": ["binding"]
},
"then": {
"properties": {
"binding": {
"properties": {
"args": {
"contains": { "const": "--no-ast" },
"minContains": 1,
"maxContains": 1
}
}
},
"effective_policy": {
"properties": {
"adapter_evolution": { "const": "preserve" },
"ast_analysis": { "const": "forbidden" },
"logic_indexing": { "const": "off" },
"blocked_tools": { "const": ["docforge_get_logic"] },
"prohibitions": {
"const": [
"arbitrary_file_access",
"arbitrary_renderer_execution",
"shell_execution",
"git_mutation",
"deployment",
"publication",
"project_switching",
"adapter_ast_upgrade",
"tree_sitter_upgrade",
"compiler_ast_upgrade",
"function_logic_extraction"
]
}
}
}
}
},
"else": {
"properties": {
"binding": {
"properties": {
"args": {
"not": {
"contains": { "const": "--no-ast" }
}
}
}
},
"effective_policy": {
"properties": {
"adapter_evolution": { "const": "allowed" },
"ast_analysis": { "const": "allowed" },
"logic_indexing": { "const": "full" },
"blocked_tools": { "const": [] },
"prohibitions": {
"const": [
"arbitrary_file_access",
"arbitrary_renderer_execution",
"shell_execution",
"git_mutation",
"deployment",
"publication",
"project_switching"
]
}
}
}
}
}
},
{
"if": {
"properties": {
"binding": {
"properties": {
"render_policy": {
"properties": { "manual": { "const": "auto" } },
"required": ["manual"]
}
},
"required": ["render_policy"]
}
},
"required": ["binding"]
},
"then": {
"properties": {
"effective_policy": {
"properties": { "manual_render": { "const": "auto" } }
}
}
}
},
{
"if": {
"properties": {
"binding": {
"properties": {
"render_policy": {
"properties": { "manual": { "const": "explicit" } },
"required": ["manual"]
}
},
"required": ["render_policy"]
}
},
"required": ["binding"]
},
"then": {
"properties": {
"effective_policy": {
"properties": { "manual_render": { "const": "explicit" } }
}
}
}
},
{
"if": {
"properties": {
"binding": {
"properties": {
"render_policy": {
"properties": { "manual": { "const": "disabled" } },
"required": ["manual"]
}
},
"required": ["render_policy"]
}
},
"required": ["binding"]
},
"then": {
"properties": {
"effective_policy": {
"properties": { "manual_render": { "const": "disabled" } }
}
}
}
},
{
"if": {
"properties": {
"artifact": {
"properties": {
"durability": { "const": "unconfirmed" }
},
"required": ["durability"]
}
},
"required": ["artifact"]
},
"then": {
"properties": {
"artifact": {
"properties": {
"write_state": { "const": "created" }
}
},
"warnings": {
"anyOf": [
{
"contains": {
"properties": {
"code": { "const": "publication_durability_unconfirmed" }
},
"required": ["code"]
}
},
{
"contains": {
"properties": {
"code": { "const": "publication_location_unconfirmed" }
},
"required": ["code"]
}
},
{
"contains": {
"properties": {
"code": { "const": "publication_binding_unconfirmed" }
},
"required": ["code"]
}
}
]
}
}
}
},
{
"if": {
"properties": {
"warnings": {
"contains": {
"properties": {
"code": { "const": "publication_location_unconfirmed" }
},
"required": ["code"]
}
}
},
"required": ["warnings"]
},
"then": {
"properties": {
"artifact": {
"properties": {
"output_path": { "type": "null" },
"write_state": { "const": "created" },
"durability": { "const": "unconfirmed" }
}
}
}
}
},
{
"if": {
"properties": {
"action": { "const": "write" },
"artifact": {
"properties": { "output_path": { "type": "null" } },
"required": ["output_path"]
}
},
"required": ["action", "artifact"]
},
"then": {
"properties": {
"warnings": {
"anyOf": [
{
"contains": {
"properties": {
"code": { "const": "publication_location_unconfirmed" }
},
"required": ["code"]
}
},
{
"contains": {
"properties": {
"code": { "const": "publication_binding_unconfirmed" }
},
"required": ["code"]
}
}
]
}
}
}
},
{
"if": {
"properties": {
"warnings": {
"contains": {
"properties": {
"code": { "const": "publication_durability_unconfirmed" }
},
"required": ["code"]
}
}
},
"required": ["warnings"]
},
"then": {
"properties": {
"artifact": {
"properties": {
"write_state": { "const": "created" },
"durability": { "const": "unconfirmed" }
}
}
}
}
}
],
"additionalProperties": false
}

View file

@ -0,0 +1,435 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/context-capsule-v1.json",
"title": "DocForge task context capsule",
"$defs": {
"sha256": {
"type": "string",
"pattern": "^[0-9a-f]{64}$"
},
"pagination": {
"type": "object",
"required": [
"schema_version",
"kind",
"returned_count",
"limit",
"total_count",
"has_more",
"next_cursor"
],
"properties": {
"schema_version": { "const": 1 },
"kind": { "const": "task-context.items" },
"returned_count": { "type": "integer", "minimum": 0 },
"limit": { "type": "integer", "minimum": 1 },
"total_count": { "type": "integer", "minimum": 0 },
"has_more": { "type": "boolean" },
"next_cursor": {
"type": ["string", "null"],
"minLength": 1,
"maxLength": 8192
}
},
"additionalProperties": false
},
"category": {
"enum": [
"structure",
"implementation",
"dependency",
"execution",
"data",
"evidence",
"context",
"unclassified"
]
},
"step": {
"type": "object",
"required": [
"step_id",
"operation",
"relation_scope",
"relation_set_hash",
"direction",
"depth",
"limit",
"required",
"evidence_role"
],
"properties": {
"step_id": { "type": "string", "minLength": 1, "maxLength": 64 },
"operation": {
"enum": ["exact", "search", "outgoing", "incoming", "metadata"]
},
"relation_scope": { "enum": ["none", "project_allowed"] },
"relation_set_hash": {
"oneOf": [
{ "$ref": "#/$defs/sha256" },
{ "type": "null" }
]
},
"direction": { "enum": ["none", "outgoing", "incoming"] },
"depth": { "type": "integer", "minimum": 0 },
"limit": { "type": "integer", "minimum": 1 },
"required": { "type": "boolean" },
"evidence_role": { "type": "string", "minLength": 1, "maxLength": 64 }
},
"additionalProperties": false
},
"requirement": {
"type": "object",
"required": ["requirement_id", "check", "category", "required"],
"properties": {
"requirement_id": { "type": "string", "minLength": 1, "maxLength": 128 },
"check": { "const": "selected_relation_category" },
"category": { "$ref": "#/$defs/category" },
"required": { "const": true }
},
"additionalProperties": false
},
"plan": {
"type": "object",
"required": [
"schema_version",
"planner",
"task_kind",
"request_hash",
"effective_policy_hash",
"focus_node_id",
"limits",
"category_order",
"steps",
"requirements",
"plan_hash"
],
"properties": {
"schema_version": { "const": 1 },
"planner": {
"type": "object",
"required": ["id", "version"],
"properties": {
"id": { "const": "docforge.core.task-context" },
"version": { "const": 1 }
},
"additionalProperties": false
},
"task_kind": {
"enum": [
"change",
"implementation",
"failure",
"ownership",
"test",
"operation",
"release"
]
},
"request_hash": { "$ref": "#/$defs/sha256" },
"effective_policy_hash": { "$ref": "#/$defs/sha256" },
"focus_node_id": { "type": ["string", "null"], "maxLength": 256 },
"limits": {
"type": "object",
"required": [
"max_evidence",
"max_tokens",
"max_depth",
"max_candidate_edges"
],
"properties": {
"max_evidence": { "type": "integer", "minimum": 1, "maximum": 1000 },
"max_tokens": { "type": "integer", "minimum": 1 },
"max_depth": { "type": "integer", "minimum": 0 },
"max_candidate_edges": {
"type": "integer",
"minimum": 1,
"maximum": 100000
}
},
"additionalProperties": false
},
"category_order": {
"type": "array",
"minItems": 8,
"maxItems": 8,
"items": { "$ref": "#/$defs/category" },
"uniqueItems": true
},
"steps": {
"type": "array",
"minItems": 4,
"maxItems": 4,
"items": { "$ref": "#/$defs/step" }
},
"requirements": {
"type": "array",
"minItems": 1,
"maxItems": 8,
"items": { "$ref": "#/$defs/requirement" }
},
"plan_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
},
"relationship": {
"type": "object",
"required": [
"source_id",
"relation",
"target_id",
"direction",
"category",
"provenance"
],
"properties": {
"source_id": { "type": "string", "minLength": 1 },
"relation": { "type": "string", "minLength": 1 },
"target_id": { "type": "string", "minLength": 1 },
"direction": { "enum": ["outgoing", "incoming"] },
"category": { "$ref": "#/$defs/category" },
"provenance": {
"const": "validated_graph_edge_without_source_provenance"
}
},
"additionalProperties": false
},
"evidence": {
"type": "object",
"required": [
"evidence_hash",
"role",
"reason_code",
"node_id",
"title",
"family",
"authority",
"status",
"tags",
"summary",
"text",
"estimated_tokens",
"source",
"depth",
"relationship_path",
"relationship_reasons",
"provenance_limitations"
],
"properties": {
"evidence_hash": { "$ref": "#/$defs/sha256" },
"role": { "enum": ["focus", "related"] },
"reason_code": {
"enum": ["exact_focus", "lexical_focus", "relationship_path"]
},
"node_id": { "type": "string", "minLength": 1 },
"title": { "type": "string" },
"family": { "type": "string" },
"authority": { "type": "string" },
"status": { "type": "string" },
"tags": {
"type": "array",
"items": { "type": "string" }
},
"summary": { "type": "string" },
"text": { "type": "string" },
"estimated_tokens": { "type": "integer", "minimum": 1 },
"source": {
"type": "object",
"required": ["path", "anchor", "content_hash"],
"properties": {
"path": { "type": "string", "minLength": 1 },
"anchor": { "type": ["string", "null"] },
"content_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
},
"depth": { "type": "integer", "minimum": 0 },
"relationship_path": {
"type": "array",
"items": { "$ref": "#/$defs/relationship" }
},
"relationship_reasons": {
"type": "array",
"items": { "$ref": "#/$defs/relationship" },
"uniqueItems": true
},
"provenance_limitations": {
"const": [
"evidence_type_unavailable",
"extractor_identity_unavailable",
"relationship_provenance_unavailable",
"observation_time_unavailable"
]
}
},
"additionalProperties": false
},
"gap": {
"type": "object",
"required": [
"code",
"requirement_id",
"category",
"state",
"check_complete",
"detail"
],
"properties": {
"code": {
"enum": [
"focus_not_found",
"focus_ambiguous",
"category_not_declared",
"no_selected_evidence",
"evidence_incomplete",
"unclassified_relation"
]
},
"requirement_id": { "type": "string", "minLength": 1 },
"category": {
"oneOf": [
{ "$ref": "#/$defs/category" },
{ "type": "null" }
]
},
"state": { "enum": ["missing", "incomplete", "blocked", "limitation"] },
"check_complete": { "type": "boolean" },
"detail": { "type": "string", "minLength": 1, "maxLength": 1000 }
},
"additionalProperties": false
},
"omission": {
"type": "object",
"required": ["code", "subject", "detail_hash"],
"properties": {
"code": {
"enum": [
"result_limit",
"token_budget",
"response_limit",
"edge_examination_limit",
"unclassified_relation_limit"
]
},
"subject": { "type": "string", "minLength": 1, "maxLength": 256 },
"detail_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
}
},
"type": "object",
"required": [
"schema_version",
"state",
"task_kind",
"generation",
"plan",
"focus",
"evidence",
"gaps",
"omissions",
"summary",
"collection_hash",
"capsule_hash"
],
"properties": {
"schema_version": { "const": 1 },
"state": { "enum": ["complete", "incomplete", "blocked"] },
"task_kind": {
"enum": [
"change",
"implementation",
"failure",
"ownership",
"test",
"operation",
"release"
]
},
"generation": {
"type": "object",
"required": [
"project_id",
"project_root_fingerprint",
"adapter",
"revision",
"source_hash",
"index_schema_version"
],
"properties": {
"project_id": { "type": "string", "minLength": 1 },
"project_root_fingerprint": {
"type": "string",
"pattern": "^[0-9a-f]{16}$"
},
"adapter": { "type": "string", "minLength": 1 },
"revision": { "type": "string", "minLength": 1 },
"source_hash": { "$ref": "#/$defs/sha256" },
"index_schema_version": { "type": "integer", "minimum": 1 }
},
"additionalProperties": false
},
"plan": { "$ref": "#/$defs/plan" },
"focus": {
"type": "object",
"required": ["state", "node_id", "candidate_count"],
"properties": {
"state": { "enum": ["resolved", "not_found", "ambiguous"] },
"node_id": { "type": ["string", "null"] },
"candidate_count": { "type": "integer", "minimum": 0 }
},
"additionalProperties": false
},
"evidence": {
"type": "array",
"maxItems": 10000,
"items": { "$ref": "#/$defs/evidence" }
},
"gaps": {
"type": "array",
"maxItems": 10000,
"items": { "$ref": "#/$defs/gap" }
},
"omissions": {
"type": "array",
"maxItems": 10000,
"items": { "$ref": "#/$defs/omission" }
},
"summary": {
"type": "object",
"required": [
"evidence_count",
"gap_count",
"omission_count",
"selected_count",
"examined_edge_count",
"estimated_tokens",
"unclassified_relations"
],
"properties": {
"evidence_count": { "type": "integer", "minimum": 0 },
"gap_count": { "type": "integer", "minimum": 0 },
"omission_count": { "type": "integer", "minimum": 0 },
"selected_count": { "type": "integer", "minimum": 0 },
"examined_edge_count": { "type": "integer", "minimum": 0 },
"estimated_tokens": { "type": "integer", "minimum": 0 },
"unclassified_relations": {
"type": "array",
"maxItems": 10000,
"items": { "type": "string", "minLength": 1 },
"uniqueItems": true
},
"page_evidence_count": { "type": "integer", "minimum": 0 },
"page_omission_count": { "type": "integer", "minimum": 0 },
"page_item_count": { "type": "integer", "minimum": 0 }
},
"additionalProperties": false
},
"collection_hash": { "$ref": "#/$defs/sha256" },
"capsule_hash": { "$ref": "#/$defs/sha256" },
"page_state": { "enum": ["complete", "incomplete", "blocked"] },
"page_hash": { "$ref": "#/$defs/sha256" },
"pagination": { "$ref": "#/$defs/pagination" }
},
"additionalProperties": false
}

View file

@ -0,0 +1,466 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/doctor-result-v1.json",
"title": "DocForge bounded read-only integration doctor result",
"$defs": {
"diagnostics": {
"type": "object",
"required": [
"schema_version",
"operation",
"outcome",
"elapsed_ns",
"stages",
"counters"
],
"properties": {
"schema_version": { "const": 1 },
"operation": { "const": "cli.doctor" },
"outcome": { "const": "ok" },
"elapsed_ns": { "type": "integer", "minimum": 0 },
"stages": {
"type": "object",
"maxProperties": 14,
"propertyNames": {
"enum": [
"source.generation",
"source.parse",
"adapter.projection",
"adapter.extract",
"index.check",
"index.synchronize",
"index.build",
"index.read",
"render.status",
"render.prepare",
"render.output_hash",
"visualization.status",
"viewer.manager",
"mcp.runtime_validation"
]
},
"additionalProperties": {
"type": "object",
"required": ["calls", "elapsed_ns"],
"properties": {
"calls": { "type": "integer", "minimum": 1 },
"elapsed_ns": { "type": "integer", "minimum": 0 }
},
"additionalProperties": false
}
},
"counters": {
"type": "object",
"required": [
"project_loads",
"source_files_parsed",
"source_bytes_parsed",
"adapter_projection_loads",
"adapter_source_extractions",
"source_generation_checks",
"index_checks",
"index_synchronizations",
"index_builds",
"render_prepare_calls",
"render_output_bytes_built",
"render_output_bytes_hashed",
"viewer_manager_requests"
],
"additionalProperties": {
"type": "integer",
"minimum": 0
},
"maxProperties": 13
}
},
"additionalProperties": false
}
},
"type": "object",
"required": [
"status",
"schema_version",
"doctor_state",
"client",
"project",
"config",
"summary",
"guarantees",
"checks"
],
"properties": {
"status": { "const": "ok" },
"schema_version": { "const": 1 },
"doctor_state": { "enum": ["healthy", "degraded", "unhealthy"] },
"client": { "enum": ["codex", "claude", "openclaw"] },
"project": {
"type": "object",
"required": [
"project_id",
"project_root",
"project_root_fingerprint",
"adapter"
],
"properties": {
"project_id": { "type": "string", "minLength": 1, "maxLength": 128 },
"project_root": { "type": "string", "minLength": 1, "maxLength": 4096 },
"project_root_fingerprint": {
"type": "string",
"pattern": "^[0-9a-f]{16}$"
},
"adapter": { "type": "string", "minLength": 1, "maxLength": 256 }
},
"additionalProperties": false
},
"config": {
"type": "object",
"required": ["path", "server_name"],
"properties": {
"path": { "type": "string", "minLength": 1, "maxLength": 4096 },
"server_name": {
"type": ["string", "null"],
"maxLength": 256
}
},
"additionalProperties": false
},
"summary": {
"type": "object",
"required": ["passed", "warning", "failed", "skipped"],
"properties": {
"passed": { "type": "integer", "minimum": 0, "maximum": 14 },
"warning": { "type": "integer", "minimum": 0, "maximum": 14 },
"failed": { "type": "integer", "minimum": 0, "maximum": 14 },
"skipped": { "type": "integer", "minimum": 0, "maximum": 14 }
},
"additionalProperties": false
},
"guarantees": {
"type": "object",
"required": [
"read_only",
"project_loads",
"adapter_projection_loads",
"adapter_source_extractions",
"sqlite_opens",
"index_checks",
"index_synchronizations",
"index_builds",
"renders",
"viewer_operations",
"client_config_writes",
"configured_command_executions"
],
"properties": {
"read_only": { "const": true },
"project_loads": { "const": 0 },
"adapter_projection_loads": { "const": 0 },
"adapter_source_extractions": { "const": 0 },
"sqlite_opens": { "const": 0 },
"index_checks": { "const": 0 },
"index_synchronizations": { "const": 0 },
"index_builds": { "const": 0 },
"renders": { "const": 0 },
"viewer_operations": { "const": 0 },
"client_config_writes": { "const": 0 },
"configured_command_executions": { "const": 0 }
},
"additionalProperties": false
},
"checks": {
"type": "array",
"minItems": 14,
"maxItems": 14,
"items": {
"type": "object",
"required": ["check_id", "state", "code", "message", "details"],
"properties": {
"check_id": {
"enum": [
"project.binding",
"project.canonical_validation",
"client.driver",
"client.config",
"client.entry",
"server.executable",
"server.arguments",
"server.project_binding",
"policy.effective",
"policy.no_ast",
"client.timeouts",
"client.environment",
"client.tool_filter",
"derived.index"
]
},
"state": {
"enum": ["passed", "warning", "failed", "skipped"]
},
"code": {
"type": "string",
"minLength": 1,
"maxLength": 128
},
"message": {
"type": "string",
"minLength": 1,
"maxLength": 512
},
"details": {
"type": "object",
"maxProperties": 16,
"propertyNames": {
"type": "string",
"minLength": 1,
"maxLength": 128
},
"additionalProperties": {
"oneOf": [
{ "type": "string", "maxLength": 512 },
{ "type": "integer" },
{ "type": "boolean" },
{ "type": "null" },
{
"type": "array",
"maxItems": 16,
"items": {
"oneOf": [
{ "type": "string", "maxLength": 256 },
{ "type": "integer" },
{ "type": "boolean" },
{ "type": "null" }
]
}
}
]
}
}
},
"additionalProperties": false
}
},
"diagnostics": { "$ref": "#/$defs/diagnostics" }
},
"allOf": [
{
"properties": {
"checks": {
"contains": {
"properties": { "check_id": { "const": "project.binding" } },
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": {
"check_id": { "const": "project.canonical_validation" }
},
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": { "check_id": { "const": "client.driver" } },
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": { "check_id": { "const": "client.config" } },
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": { "check_id": { "const": "client.entry" } },
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": { "check_id": { "const": "server.executable" } },
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": { "check_id": { "const": "server.arguments" } },
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": {
"check_id": { "const": "server.project_binding" }
},
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": { "check_id": { "const": "policy.effective" } },
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": { "check_id": { "const": "policy.no_ast" } },
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": { "check_id": { "const": "client.timeouts" } },
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": { "check_id": { "const": "client.environment" } },
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": { "check_id": { "const": "client.tool_filter" } },
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"properties": {
"checks": {
"contains": {
"properties": { "check_id": { "const": "derived.index" } },
"required": ["check_id"]
},
"minContains": 1,
"maxContains": 1
}
}
},
{
"if": {
"properties": { "doctor_state": { "const": "healthy" } },
"required": ["doctor_state"]
},
"then": {
"properties": {
"summary": {
"properties": {
"warning": { "const": 0 },
"failed": { "const": 0 }
}
}
}
}
},
{
"if": {
"properties": { "doctor_state": { "const": "degraded" } },
"required": ["doctor_state"]
},
"then": {
"properties": {
"summary": {
"properties": {
"warning": { "minimum": 1 },
"failed": { "const": 0 }
}
}
}
}
},
{
"if": {
"properties": { "doctor_state": { "const": "unhealthy" } },
"required": ["doctor_state"]
},
"then": {
"properties": {
"summary": {
"properties": {
"failed": { "minimum": 1 }
}
}
}
}
}
],
"additionalProperties": false
}

View file

@ -0,0 +1,430 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/generation-diff-page-v1.json",
"title": "DocForge bounded latest-generation diff page",
"$defs": {
"sha256": {
"type": "string",
"pattern": "^[0-9a-f]{64}$"
},
"generation": {
"type": "object",
"required": [
"revision",
"source_hash",
"node_count",
"node_hash",
"edge_count",
"edge_hash",
"index_schema_version"
],
"properties": {
"revision": { "type": "string", "minLength": 1 },
"source_hash": { "$ref": "#/$defs/sha256" },
"node_count": { "type": "integer", "minimum": 0 },
"node_hash": { "$ref": "#/$defs/sha256" },
"edge_count": { "type": "integer", "minimum": 0 },
"edge_hash": { "$ref": "#/$defs/sha256" },
"index_schema_version": { "type": "integer", "minimum": 1 }
},
"additionalProperties": false
},
"summary": {
"type": "object",
"required": [
"nodes_added",
"nodes_removed",
"nodes_changed",
"edges_added",
"edges_removed",
"total_changes"
],
"properties": {
"nodes_added": { "type": "integer", "minimum": 0 },
"nodes_removed": { "type": "integer", "minimum": 0 },
"nodes_changed": { "type": "integer", "minimum": 0 },
"edges_added": { "type": "integer", "minimum": 0 },
"edges_removed": { "type": "integer", "minimum": 0 },
"total_changes": { "type": "integer", "minimum": 0 }
},
"additionalProperties": false
},
"node_item": {
"type": "object",
"required": [
"entity",
"change",
"node_id",
"before_content_hash",
"after_content_hash",
"before_source_path",
"after_source_path",
"before_node_hash",
"after_node_hash",
"changed_fields",
"item_hash"
],
"properties": {
"entity": { "const": "node" },
"change": { "enum": ["added", "removed", "changed"] },
"node_id": { "type": "string", "minLength": 1 },
"before_content_hash": {
"oneOf": [{ "$ref": "#/$defs/sha256" }, { "type": "null" }]
},
"after_content_hash": {
"oneOf": [{ "$ref": "#/$defs/sha256" }, { "type": "null" }]
},
"before_source_path": { "type": ["string", "null"] },
"after_source_path": { "type": ["string", "null"] },
"before_node_hash": {
"oneOf": [{ "$ref": "#/$defs/sha256" }, { "type": "null" }]
},
"after_node_hash": {
"oneOf": [{ "$ref": "#/$defs/sha256" }, { "type": "null" }]
},
"changed_fields": {
"type": "array",
"uniqueItems": true,
"items": {
"enum": [
"title",
"family",
"authority",
"status",
"tags",
"summary",
"content",
"source_path",
"source_anchor",
"content_hash"
]
}
},
"item_hash": { "$ref": "#/$defs/sha256" }
},
"allOf": [
{
"if": {
"properties": { "change": { "const": "added" } },
"required": ["change"]
},
"then": {
"properties": {
"before_content_hash": { "type": "null" },
"before_source_path": { "type": "null" },
"before_node_hash": { "type": "null" },
"after_content_hash": { "$ref": "#/$defs/sha256" },
"after_source_path": { "type": "string", "minLength": 1 },
"after_node_hash": { "$ref": "#/$defs/sha256" },
"changed_fields": { "maxItems": 0 }
}
}
},
{
"if": {
"properties": { "change": { "const": "removed" } },
"required": ["change"]
},
"then": {
"properties": {
"before_content_hash": { "$ref": "#/$defs/sha256" },
"before_source_path": { "type": "string", "minLength": 1 },
"before_node_hash": { "$ref": "#/$defs/sha256" },
"after_content_hash": { "type": "null" },
"after_source_path": { "type": "null" },
"after_node_hash": { "type": "null" },
"changed_fields": { "maxItems": 0 }
}
}
},
{
"if": {
"properties": { "change": { "const": "changed" } },
"required": ["change"]
},
"then": {
"properties": {
"before_content_hash": { "$ref": "#/$defs/sha256" },
"before_source_path": { "type": "string", "minLength": 1 },
"before_node_hash": { "$ref": "#/$defs/sha256" },
"after_content_hash": { "$ref": "#/$defs/sha256" },
"after_source_path": { "type": "string", "minLength": 1 },
"after_node_hash": { "$ref": "#/$defs/sha256" },
"changed_fields": { "minItems": 1 }
}
}
}
],
"additionalProperties": false
},
"edge_item": {
"type": "object",
"required": [
"entity",
"change",
"source_id",
"relation",
"target_id",
"item_hash"
],
"properties": {
"entity": { "const": "edge" },
"change": { "enum": ["added", "removed"] },
"source_id": { "type": "string", "minLength": 1 },
"relation": { "type": "string", "minLength": 1 },
"target_id": { "type": "string", "minLength": 1 },
"item_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
},
"receipt_header": {
"type": "object",
"required": [
"schema_version",
"diff_semantics_version",
"project_id",
"project_root_fingerprint",
"adapter",
"kind",
"reason",
"from_generation",
"to_generation",
"summary",
"full_item_count",
"retained_item_count",
"details_truncated",
"truncation_reason",
"full_collection_hash",
"retained_collection_hash",
"index_signature",
"stored_receipt_hash"
],
"properties": {
"schema_version": { "const": 1 },
"diff_semantics_version": { "const": 1 },
"project_id": { "type": "string", "minLength": 1 },
"project_root_fingerprint": {
"type": "string",
"pattern": "^[0-9a-f]{16}$"
},
"adapter": { "type": "string", "minLength": 1 },
"kind": { "enum": ["baseline", "transition"] },
"reason": {
"enum": [
null,
"no_predecessor",
"predecessor_unsafe",
"predecessor_unsupported_schema",
"predecessor_foreign",
"predecessor_policy_incompatible",
"predecessor_corrupt",
"predecessor_unattested",
"predecessor_changed",
"no_meaningful_transition"
]
},
"from_generation": {
"oneOf": [{ "$ref": "#/$defs/generation" }, { "type": "null" }]
},
"to_generation": { "$ref": "#/$defs/generation" },
"summary": { "$ref": "#/$defs/summary" },
"full_item_count": { "type": "integer", "minimum": 0 },
"retained_item_count": {
"type": "integer",
"minimum": 0,
"maximum": 1000
},
"details_truncated": { "type": "boolean" },
"truncation_reason": {
"enum": [null, "receipt_item_limit", "receipt_byte_limit"]
},
"full_collection_hash": { "$ref": "#/$defs/sha256" },
"retained_collection_hash": { "$ref": "#/$defs/sha256" },
"index_signature": {
"type": "object",
"required": ["device", "inode", "size", "mtime_ns", "ctime_ns"],
"properties": {
"device": { "type": "integer", "minimum": 0 },
"inode": { "type": "integer", "minimum": 0 },
"size": { "type": "integer", "minimum": 0 },
"mtime_ns": { "type": "integer", "minimum": 0 },
"ctime_ns": { "type": "integer", "minimum": 0 }
},
"additionalProperties": false
},
"stored_receipt_hash": { "$ref": "#/$defs/sha256" }
},
"allOf": [
{
"if": {
"properties": { "kind": { "const": "baseline" } },
"required": ["kind"]
},
"then": {
"properties": {
"from_generation": { "type": "null" },
"reason": { "not": { "type": "null" } },
"summary": {
"properties": {
"nodes_added": { "const": 0 },
"nodes_removed": { "const": 0 },
"nodes_changed": { "const": 0 },
"edges_added": { "const": 0 },
"edges_removed": { "const": 0 },
"total_changes": { "const": 0 }
}
},
"full_item_count": { "const": 0 },
"retained_item_count": { "const": 0 },
"details_truncated": { "const": false },
"truncation_reason": { "const": null },
"full_collection_hash": {
"const": "4f53cda18c2baa0c0354bb5f9a3ecbe5ed12ab4d8e11ba873c2f11161202b945"
},
"retained_collection_hash": {
"const": "4f53cda18c2baa0c0354bb5f9a3ecbe5ed12ab4d8e11ba873c2f11161202b945"
}
}
},
"else": {
"properties": {
"from_generation": { "$ref": "#/$defs/generation" },
"reason": { "type": "null" }
}
}
},
{
"if": {
"properties": { "truncation_reason": { "const": null } },
"required": ["truncation_reason"]
},
"then": {
"properties": {
"details_truncated": { "const": false },
"full_item_count": { "maximum": 1000 }
}
}
},
{
"if": {
"properties": {
"truncation_reason": { "const": "receipt_item_limit" }
},
"required": ["truncation_reason"]
},
"then": {
"properties": {
"details_truncated": { "const": true },
"full_item_count": { "minimum": 1001 },
"retained_item_count": { "const": 1000 }
}
}
},
{
"if": {
"properties": {
"truncation_reason": { "const": "receipt_byte_limit" }
},
"required": ["truncation_reason"]
},
"then": {
"properties": {
"details_truncated": { "const": true }
}
}
}
],
"additionalProperties": false
},
"page_body": {
"type": "object",
"required": [
"page_schema_version",
"receipt_header",
"items",
"omissions",
"page_hash"
],
"properties": {
"page_schema_version": { "const": 1 },
"receipt_header": { "$ref": "#/$defs/receipt_header" },
"items": {
"type": "array",
"maxItems": 1000,
"items": {
"oneOf": [
{ "$ref": "#/$defs/node_item" },
{ "$ref": "#/$defs/edge_item" }
]
}
},
"omissions": {
"type": "array",
"maxItems": 1,
"items": {
"type": "object",
"required": ["code", "item_hash"],
"properties": {
"code": { "const": "response_limit" },
"item_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
}
},
"page_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
},
"pagination": {
"type": "object",
"required": [
"schema_version",
"kind",
"returned_count",
"limit",
"total_count",
"has_more",
"next_cursor"
],
"properties": {
"schema_version": { "const": 1 },
"kind": { "const": "generation-diff.items" },
"returned_count": { "type": "integer", "minimum": 0 },
"limit": { "type": "integer", "minimum": 1, "maximum": 1000 },
"total_count": { "type": "integer", "minimum": 0, "maximum": 1000 },
"has_more": { "type": "boolean" },
"next_cursor": {
"type": ["string", "null"],
"minLength": 1,
"maxLength": 8192
}
},
"allOf": [
{
"if": {
"properties": { "has_more": { "const": true } },
"required": ["has_more"]
},
"then": {
"properties": {
"next_cursor": { "type": "string", "minLength": 1 }
}
},
"else": {
"properties": {
"next_cursor": { "type": "null" }
}
}
}
],
"additionalProperties": false
}
},
"type": "object",
"required": ["generation_diff", "pagination"],
"properties": {
"generation_diff": { "$ref": "#/$defs/page_body" },
"pagination": { "$ref": "#/$defs/pagination" }
},
"additionalProperties": false
}

View file

@ -0,0 +1,363 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/generation-diff-v1.json",
"title": "DocForge latest primary-graph generation diff receipt",
"$defs": {
"sha256": {
"type": "string",
"pattern": "^[0-9a-f]{64}$"
},
"generation": {
"type": "object",
"required": [
"revision",
"source_hash",
"node_count",
"node_hash",
"edge_count",
"edge_hash",
"index_schema_version"
],
"properties": {
"revision": { "type": "string", "minLength": 1 },
"source_hash": { "$ref": "#/$defs/sha256" },
"node_count": { "type": "integer", "minimum": 0 },
"node_hash": { "$ref": "#/$defs/sha256" },
"edge_count": { "type": "integer", "minimum": 0 },
"edge_hash": { "$ref": "#/$defs/sha256" },
"index_schema_version": { "type": "integer", "minimum": 1 }
},
"additionalProperties": false
},
"node_item": {
"type": "object",
"required": [
"entity",
"change",
"node_id",
"before_content_hash",
"after_content_hash",
"before_source_path",
"after_source_path",
"before_node_hash",
"after_node_hash",
"changed_fields",
"item_hash"
],
"properties": {
"entity": { "const": "node" },
"change": { "enum": ["added", "removed", "changed"] },
"node_id": { "type": "string", "minLength": 1 },
"before_content_hash": {
"oneOf": [
{ "$ref": "#/$defs/sha256" },
{ "type": "null" }
]
},
"after_content_hash": {
"oneOf": [
{ "$ref": "#/$defs/sha256" },
{ "type": "null" }
]
},
"before_source_path": { "type": ["string", "null"] },
"after_source_path": { "type": ["string", "null"] },
"before_node_hash": {
"oneOf": [
{ "$ref": "#/$defs/sha256" },
{ "type": "null" }
]
},
"after_node_hash": {
"oneOf": [
{ "$ref": "#/$defs/sha256" },
{ "type": "null" }
]
},
"changed_fields": {
"type": "array",
"items": {
"enum": [
"title",
"family",
"authority",
"status",
"tags",
"summary",
"content",
"source_path",
"source_anchor",
"content_hash"
]
},
"uniqueItems": true
},
"item_hash": { "$ref": "#/$defs/sha256" }
},
"allOf": [
{
"if": {
"properties": { "change": { "const": "added" } },
"required": ["change"]
},
"then": {
"properties": {
"before_content_hash": { "type": "null" },
"before_source_path": { "type": "null" },
"before_node_hash": { "type": "null" },
"after_content_hash": { "$ref": "#/$defs/sha256" },
"after_source_path": { "type": "string", "minLength": 1 },
"after_node_hash": { "$ref": "#/$defs/sha256" },
"changed_fields": { "maxItems": 0 }
}
}
},
{
"if": {
"properties": { "change": { "const": "removed" } },
"required": ["change"]
},
"then": {
"properties": {
"before_content_hash": { "$ref": "#/$defs/sha256" },
"before_source_path": { "type": "string", "minLength": 1 },
"before_node_hash": { "$ref": "#/$defs/sha256" },
"after_content_hash": { "type": "null" },
"after_source_path": { "type": "null" },
"after_node_hash": { "type": "null" },
"changed_fields": { "maxItems": 0 }
}
}
},
{
"if": {
"properties": { "change": { "const": "changed" } },
"required": ["change"]
},
"then": {
"properties": {
"before_content_hash": { "$ref": "#/$defs/sha256" },
"before_source_path": { "type": "string", "minLength": 1 },
"before_node_hash": { "$ref": "#/$defs/sha256" },
"after_content_hash": { "$ref": "#/$defs/sha256" },
"after_source_path": { "type": "string", "minLength": 1 },
"after_node_hash": { "$ref": "#/$defs/sha256" },
"changed_fields": { "minItems": 1 }
}
}
}
],
"additionalProperties": false
},
"edge_item": {
"type": "object",
"required": [
"entity",
"change",
"source_id",
"relation",
"target_id",
"item_hash"
],
"properties": {
"entity": { "const": "edge" },
"change": { "enum": ["added", "removed"] },
"source_id": { "type": "string", "minLength": 1 },
"relation": { "type": "string", "minLength": 1 },
"target_id": { "type": "string", "minLength": 1 },
"item_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
}
},
"type": "object",
"required": [
"schema_version",
"diff_semantics_version",
"project_id",
"project_root_fingerprint",
"adapter",
"kind",
"reason",
"from_generation",
"to_generation",
"summary",
"items",
"full_item_count",
"retained_item_count",
"details_truncated",
"truncation_reason",
"full_collection_hash",
"retained_collection_hash",
"index_signature",
"receipt_hash"
],
"properties": {
"schema_version": { "const": 1 },
"diff_semantics_version": { "const": 1 },
"project_id": { "type": "string", "minLength": 1 },
"project_root_fingerprint": {
"type": "string",
"pattern": "^[0-9a-f]{16}$"
},
"adapter": { "type": "string", "minLength": 1 },
"kind": { "enum": ["baseline", "transition"] },
"reason": {
"enum": [
null,
"no_predecessor",
"predecessor_unsafe",
"predecessor_unsupported_schema",
"predecessor_foreign",
"predecessor_policy_incompatible",
"predecessor_corrupt",
"predecessor_unattested",
"predecessor_changed",
"no_meaningful_transition"
]
},
"from_generation": {
"oneOf": [
{ "$ref": "#/$defs/generation" },
{ "type": "null" }
]
},
"to_generation": { "$ref": "#/$defs/generation" },
"summary": {
"type": "object",
"required": [
"nodes_added",
"nodes_removed",
"nodes_changed",
"edges_added",
"edges_removed",
"total_changes"
],
"properties": {
"nodes_added": { "type": "integer", "minimum": 0 },
"nodes_removed": { "type": "integer", "minimum": 0 },
"nodes_changed": { "type": "integer", "minimum": 0 },
"edges_added": { "type": "integer", "minimum": 0 },
"edges_removed": { "type": "integer", "minimum": 0 },
"total_changes": { "type": "integer", "minimum": 0 }
},
"additionalProperties": false
},
"items": {
"type": "array",
"maxItems": 1000,
"items": {
"oneOf": [
{ "$ref": "#/$defs/node_item" },
{ "$ref": "#/$defs/edge_item" }
]
}
},
"full_item_count": { "type": "integer", "minimum": 0 },
"retained_item_count": {
"type": "integer",
"minimum": 0,
"maximum": 1000
},
"details_truncated": { "type": "boolean" },
"truncation_reason": {
"enum": [null, "receipt_item_limit", "receipt_byte_limit"]
},
"full_collection_hash": { "$ref": "#/$defs/sha256" },
"retained_collection_hash": { "$ref": "#/$defs/sha256" },
"index_signature": {
"type": "object",
"required": ["device", "inode", "size", "mtime_ns", "ctime_ns"],
"properties": {
"device": { "type": "integer", "minimum": 0 },
"inode": { "type": "integer", "minimum": 0 },
"size": { "type": "integer", "minimum": 0 },
"mtime_ns": { "type": "integer", "minimum": 0 },
"ctime_ns": { "type": "integer", "minimum": 0 }
},
"additionalProperties": false
},
"receipt_hash": { "$ref": "#/$defs/sha256" }
},
"allOf": [
{
"if": {
"properties": { "kind": { "const": "baseline" } },
"required": ["kind"]
},
"then": {
"properties": {
"from_generation": { "type": "null" },
"reason": { "not": { "type": "null" } },
"summary": {
"properties": {
"nodes_added": { "const": 0 },
"nodes_removed": { "const": 0 },
"nodes_changed": { "const": 0 },
"edges_added": { "const": 0 },
"edges_removed": { "const": 0 },
"total_changes": { "const": 0 }
}
},
"full_item_count": { "const": 0 },
"retained_item_count": { "const": 0 },
"details_truncated": { "const": false },
"truncation_reason": { "const": null },
"full_collection_hash": {
"const": "4f53cda18c2baa0c0354bb5f9a3ecbe5ed12ab4d8e11ba873c2f11161202b945"
},
"retained_collection_hash": {
"const": "4f53cda18c2baa0c0354bb5f9a3ecbe5ed12ab4d8e11ba873c2f11161202b945"
}
}
},
"else": {
"properties": {
"from_generation": { "$ref": "#/$defs/generation" },
"reason": { "type": "null" }
}
}
},
{
"if": {
"properties": { "truncation_reason": { "const": null } },
"required": ["truncation_reason"]
},
"then": {
"properties": {
"details_truncated": { "const": false },
"full_item_count": { "maximum": 1000 }
}
}
},
{
"if": {
"properties": {
"truncation_reason": { "const": "receipt_item_limit" }
},
"required": ["truncation_reason"]
},
"then": {
"properties": {
"details_truncated": { "const": true },
"full_item_count": { "minimum": 1001 },
"retained_item_count": { "const": 1000 }
}
}
},
{
"if": {
"properties": {
"truncation_reason": { "const": "receipt_byte_limit" }
},
"required": ["truncation_reason"]
},
"then": {
"properties": {
"details_truncated": { "const": true }
}
}
}
],
"additionalProperties": false
}

View file

@ -0,0 +1,321 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/graph-view-plan-v1.json",
"title": "DocForge immutable portable graph view plan",
"$comment": "plan_id is the SHA-256 of canonical JSON without plan_id and is verified by the runtime contract validator. The version-1 nested graph vocabulary remains planner-owned.",
"$defs": {
"sha256": {
"type": "string",
"pattern": "^[0-9a-f]{64}$"
},
"project": {
"type": "object",
"required": [
"project_id",
"project_root_fingerprint",
"adapter",
"revision",
"source_hash"
],
"properties": {
"project_id": { "type": "string", "minLength": 1 },
"project_root_fingerprint": {
"type": "string",
"pattern": "^[0-9a-f]{16}$"
},
"adapter": { "type": "string", "minLength": 1 },
"revision": { "type": "string", "minLength": 1 },
"source_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
},
"filter": {
"type": "array",
"maxItems": 64,
"items": {
"type": "string",
"minLength": 1,
"maxLength": 1024
},
"uniqueItems": true
},
"edge": {
"type": "object",
"required": ["source_id", "relation", "target_id"],
"properties": {
"source_id": { "type": "string", "minLength": 1 },
"relation": { "type": "string", "minLength": 1 },
"target_id": { "type": "string", "minLength": 1 }
},
"additionalProperties": false
},
"node": {
"type": "object",
"required": [
"node_id",
"title",
"family",
"authority",
"status",
"tags",
"summary",
"content_hash"
],
"properties": {
"node_id": { "type": "string", "minLength": 1 },
"title": { "type": "string", "minLength": 1 },
"family": { "type": "string", "minLength": 1 },
"authority": { "type": "string", "minLength": 1 },
"status": { "type": "string", "minLength": 1 },
"tags": { "$ref": "#/$defs/filter" },
"summary": { "type": "string" },
"content_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
},
"omission": {
"oneOf": [
{
"type": "object",
"required": ["code", "subject", "limit", "minimum_omitted"],
"properties": {
"code": { "const": "node_result_limit" },
"subject": { "const": "nodes" },
"limit": { "type": "integer", "minimum": 1, "maximum": 1000 },
"minimum_omitted": { "type": "integer", "minimum": 1 }
},
"additionalProperties": false
},
{
"type": "object",
"required": ["code", "subject", "limit", "minimum_omitted"],
"properties": {
"code": { "const": "edge_result_limit" },
"subject": { "const": "edges" },
"limit": { "type": "integer", "minimum": 0, "maximum": 4000 },
"minimum_omitted": { "type": "integer", "minimum": 1 }
},
"additionalProperties": false
},
{
"type": "object",
"required": [
"code",
"subject",
"limit",
"examined",
"minimum_omitted"
],
"properties": {
"code": { "const": "work_limit" },
"subject": { "const": "selection" },
"limit": { "type": "integer", "minimum": 1, "maximum": 1000000 },
"examined": { "type": "integer", "minimum": 0 },
"minimum_omitted": { "type": "integer", "minimum": 1 }
},
"additionalProperties": false
},
{
"type": "object",
"required": ["code", "subject", "minimum_omitted"],
"properties": {
"code": { "const": "logic_forbidden" },
"subject": { "const": "logic" },
"minimum_omitted": { "type": "integer", "minimum": 1 }
},
"additionalProperties": false
}
]
}
},
"type": "object",
"required": [
"schema_version",
"contract",
"plan_id",
"project",
"view",
"bounds",
"policy",
"graph",
"omissions",
"diagnostics"
],
"properties": {
"schema_version": { "const": 1 },
"contract": { "const": "docforge.graph-view-plan" },
"plan_id": { "$ref": "#/$defs/sha256" },
"project": { "$ref": "#/$defs/project" },
"view": {
"type": "object",
"required": [
"view_id",
"title",
"initial_mode",
"scope",
"filters",
"detail_fields"
],
"properties": {
"view_id": {
"type": "string",
"minLength": 1,
"maxLength": 1024
},
"title": {
"type": "string",
"minLength": 1,
"maxLength": 1024
},
"initial_mode": { "enum": ["nodes", "flow", "web", "logic"] },
"scope": {
"oneOf": [
{
"type": "object",
"required": ["kind", "root_node_id", "depth"],
"properties": {
"kind": { "const": "exact_root" },
"root_node_id": {
"type": "string",
"minLength": 1,
"maxLength": 1024
},
"depth": { "type": "integer", "minimum": 1, "maximum": 32 }
},
"additionalProperties": false
},
{
"type": "object",
"required": ["kind", "query"],
"properties": {
"kind": { "const": "lexical" },
"query": {
"type": "string",
"minLength": 1,
"maxLength": 10000
}
},
"additionalProperties": false
}
]
},
"filters": {
"type": "object",
"required": [
"families",
"relations",
"authorities",
"statuses",
"tags"
],
"properties": {
"families": { "$ref": "#/$defs/filter" },
"relations": { "$ref": "#/$defs/filter" },
"authorities": { "$ref": "#/$defs/filter" },
"statuses": { "$ref": "#/$defs/filter" },
"tags": { "$ref": "#/$defs/filter" }
},
"additionalProperties": false
},
"detail_fields": {
"const": [
"node_id",
"title",
"family",
"authority",
"status",
"tags",
"summary",
"content_hash"
]
}
},
"additionalProperties": false
},
"bounds": {
"type": "object",
"required": ["depth", "max_nodes", "max_edges", "max_work"],
"properties": {
"depth": { "type": "integer", "minimum": 1, "maximum": 32 },
"max_nodes": { "type": "integer", "minimum": 1, "maximum": 1000 },
"max_edges": { "type": "integer", "minimum": 0, "maximum": 4000 },
"max_work": { "type": "integer", "minimum": 1, "maximum": 1000000 }
},
"additionalProperties": false
},
"policy": {
"type": "object",
"required": [
"visibility",
"source_paths",
"source_bodies",
"database_queries",
"executable_content",
"logic",
"logic_requested"
],
"properties": {
"visibility": { "const": "selected_graph_only" },
"source_paths": { "const": "excluded" },
"source_bodies": { "const": "excluded" },
"database_queries": { "const": "forbidden" },
"executable_content": { "const": "forbidden" },
"logic": { "enum": ["allowed", "forbidden"] },
"logic_requested": { "type": "boolean" }
},
"additionalProperties": false
},
"graph": {
"type": "object",
"required": ["root_node_id", "nodes", "edges", "logic_projections"],
"properties": {
"root_node_id": {
"type": ["string", "null"],
"minLength": 1,
"maxLength": 1024
},
"nodes": {
"type": "array",
"maxItems": 1000,
"items": { "$ref": "#/$defs/node" }
},
"edges": {
"type": "array",
"maxItems": 4000,
"items": { "$ref": "#/$defs/edge" }
},
"logic_projections": {
"type": "array",
"maxItems": 0
}
},
"additionalProperties": false
},
"omissions": {
"type": "array",
"maxItems": 4,
"items": { "$ref": "#/$defs/omission" }
},
"diagnostics": {
"type": "object",
"required": [
"selection",
"returned_nodes",
"returned_edges",
"examined_work_units",
"truncated",
"ordering"
],
"properties": {
"selection": { "enum": ["exact_root", "lexical"] },
"returned_nodes": { "type": "integer", "minimum": 0, "maximum": 1000 },
"returned_edges": { "type": "integer", "minimum": 0, "maximum": 4000 },
"examined_work_units": { "type": "integer", "minimum": 0 },
"truncated": { "type": "boolean" },
"ordering": { "const": "node_id;source_id,relation,target_id" }
},
"additionalProperties": false
}
},
"additionalProperties": false
}

View file

@ -0,0 +1,214 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/manual-render-plan-v1.json",
"title": "DocForge immutable manual render plan",
"$comment": "plan_id is the SHA-256 of canonical JSON without plan_id and is verified by the runtime contract validator.",
"$defs": {
"sha256": {
"type": "string",
"pattern": "^[0-9a-f]{64}$"
},
"project": {
"type": "object",
"required": [
"project_id",
"project_root_fingerprint",
"adapter",
"revision",
"source_hash"
],
"properties": {
"project_id": { "type": "string", "minLength": 1 },
"project_root_fingerprint": {
"type": "string",
"pattern": "^[0-9a-f]{16}$"
},
"adapter": { "type": "string", "minLength": 1 },
"revision": { "type": "string", "minLength": 1 },
"source_hash": { "$ref": "#/$defs/sha256" }
},
"additionalProperties": false
},
"edge": {
"type": "object",
"required": ["source_id", "relation", "target_id"],
"properties": {
"source_id": { "type": "string", "minLength": 1 },
"relation": { "type": "string", "minLength": 1 },
"target_id": { "type": "string", "minLength": 1 }
},
"additionalProperties": false
},
"page": {
"type": "object",
"required": [
"node_id",
"title",
"family",
"authority",
"status",
"tags",
"summary",
"content",
"content_hash",
"components",
"breadcrumbs",
"cross_references",
"backlinks"
],
"properties": {
"node_id": { "type": "string", "minLength": 1 },
"title": { "type": "string", "minLength": 1 },
"family": { "type": "string", "minLength": 1 },
"authority": { "type": "string", "minLength": 1 },
"status": { "type": "string", "minLength": 1 },
"tags": {
"type": "array",
"maxItems": 10000,
"items": { "type": "string", "minLength": 1 },
"uniqueItems": true
},
"summary": { "type": "string" },
"content": { "type": "string" },
"content_hash": { "$ref": "#/$defs/sha256" },
"components": {
"type": "array",
"maxItems": 32,
"items": { "type": "string", "minLength": 1 },
"uniqueItems": true
},
"breadcrumbs": {
"type": "array",
"maxItems": 10000,
"items": { "type": "string", "minLength": 1 }
},
"cross_references": {
"type": "array",
"maxItems": 10000,
"items": { "$ref": "#/$defs/edge" }
},
"backlinks": {
"type": "array",
"maxItems": 10000,
"items": { "$ref": "#/$defs/edge" }
}
},
"additionalProperties": false
},
"navigation_item": {
"type": "object",
"required": ["node_id", "title"],
"properties": {
"node_id": { "type": "string", "minLength": 1 },
"title": { "type": "string", "minLength": 1 }
},
"additionalProperties": false
},
"search_document": {
"type": "object",
"required": [
"node_id",
"title",
"summary",
"family",
"status",
"tags"
],
"properties": {
"node_id": { "type": "string", "minLength": 1 },
"title": { "type": "string", "minLength": 1 },
"summary": { "type": "string" },
"family": { "type": "string", "minLength": 1 },
"status": { "type": "string", "minLength": 1 },
"tags": {
"type": "array",
"maxItems": 10000,
"items": { "type": "string", "minLength": 1 },
"uniqueItems": true
}
},
"additionalProperties": false
}
},
"type": "object",
"required": [
"schema_version",
"contract",
"plan_id",
"project",
"view",
"changeset_hash",
"pages",
"navigation",
"search_documents",
"diagnostics"
],
"properties": {
"schema_version": { "const": 1 },
"contract": { "const": "docforge.manual-render-plan" },
"plan_id": { "$ref": "#/$defs/sha256" },
"project": { "$ref": "#/$defs/project" },
"view": {
"type": "object",
"required": ["view_id", "title", "families", "renderer"],
"properties": {
"view_id": { "type": "string", "minLength": 1 },
"title": { "type": "string", "minLength": 1 },
"families": {
"type": "array",
"maxItems": 10000,
"items": { "type": "string", "minLength": 1 },
"uniqueItems": true
},
"renderer": { "type": "string", "minLength": 1 }
},
"additionalProperties": false
},
"changeset_hash": {
"oneOf": [
{ "$ref": "#/$defs/sha256" },
{ "type": "null" }
]
},
"pages": {
"type": "array",
"maxItems": 10000,
"items": { "$ref": "#/$defs/page" }
},
"navigation": {
"type": "array",
"maxItems": 10000,
"items": { "$ref": "#/$defs/navigation_item" }
},
"search_documents": {
"type": "array",
"maxItems": 10000,
"items": { "$ref": "#/$defs/search_document" }
},
"diagnostics": {
"type": "object",
"required": ["orphans", "cycles"],
"properties": {
"orphans": {
"type": "array",
"maxItems": 10000,
"items": { "type": "string", "minLength": 1 },
"uniqueItems": true
},
"cycles": {
"type": "array",
"maxItems": 10000,
"items": {
"type": "array",
"minItems": 1,
"maxItems": 10000,
"items": { "type": "string", "minLength": 1 },
"uniqueItems": true
}
}
},
"additionalProperties": false
}
},
"additionalProperties": false
}

View file

@ -0,0 +1,62 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/policy-v1.json",
"title": "DocForge effective process policy",
"type": "object",
"required": [
"schema_version",
"capability_mode",
"capability_source",
"adapter_evolution",
"ast_analysis",
"logic_indexing",
"synchronization",
"integrity",
"manual_render",
"graph_render",
"live_viewer",
"profiling",
"blocked_tools",
"prohibitions",
"precedence"
],
"properties": {
"schema_version": { "const": 1 },
"capability_mode": {
"enum": ["read", "proposal", "application", "operator"]
},
"capability_source": {
"enum": ["factory_default", "explicit"]
},
"adapter_evolution": { "enum": ["allowed", "preserve"] },
"ast_analysis": { "enum": ["allowed", "forbidden"] },
"logic_indexing": { "enum": ["full", "off"] },
"synchronization": { "const": "automatic" },
"integrity": { "const": "validated" },
"manual_render": { "enum": ["auto", "explicit", "disabled"] },
"graph_render": { "const": "disabled" },
"live_viewer": { "const": "on-demand" },
"profiling": { "enum": ["enabled", "disabled"] },
"blocked_tools": {
"type": "array",
"maxItems": 64,
"items": { "type": "string", "minLength": 1 },
"uniqueItems": true
},
"prohibitions": {
"type": "array",
"maxItems": 64,
"items": { "type": "string", "minLength": 1 },
"uniqueItems": true
},
"precedence": {
"const": [
"core_safety",
"explicit_binding",
"no_ast_shorthand",
"resource_availability"
]
}
},
"additionalProperties": false
}

View file

@ -74,6 +74,82 @@
},
"additionalProperties": false
},
"graph_render": {
"type": "object",
"required": ["output_root", "views"],
"properties": {
"output_root": { "$ref": "#/$defs/relativePath" },
"views": {
"type": "array",
"minItems": 1,
"items": {
"type": "object",
"required": ["id", "renderer", "output", "title"],
"properties": {
"id": { "type": "string", "pattern": "^[a-z0-9][a-z0-9._-]{1,127}$" },
"renderer": { "const": "portable_graph_html" },
"output": {
"allOf": [
{ "$ref": "#/$defs/relativePath" },
{ "pattern": "\\.html$" }
]
},
"title": { "type": "string", "minLength": 1, "maxLength": 1024 },
"root": { "type": "string", "minLength": 1, "maxLength": 1024 },
"query": { "type": "string", "minLength": 1, "maxLength": 10000 },
"initial_mode": { "enum": ["nodes", "flow", "web"] },
"depth": { "type": "integer", "minimum": 1, "maximum": 32 },
"max_nodes": { "type": "integer", "minimum": 1, "maximum": 1000 },
"max_edges": { "type": "integer", "minimum": 0, "maximum": 4000 },
"max_work": { "type": "integer", "minimum": 1, "maximum": 1000000 },
"families": {
"type": "array",
"maxItems": 64,
"uniqueItems": true,
"items": { "type": "string", "minLength": 1, "maxLength": 1024 }
},
"relations": {
"type": "array",
"maxItems": 64,
"uniqueItems": true,
"items": { "type": "string", "minLength": 1, "maxLength": 1024 }
},
"authorities": {
"type": "array",
"maxItems": 64,
"uniqueItems": true,
"items": { "type": "string", "minLength": 1, "maxLength": 1024 }
},
"statuses": {
"type": "array",
"maxItems": 64,
"uniqueItems": true,
"items": { "type": "string", "minLength": 1, "maxLength": 1024 }
},
"tags": {
"type": "array",
"maxItems": 64,
"uniqueItems": true,
"items": { "type": "string", "minLength": 1, "maxLength": 1024 }
},
"include_logic": { "const": false }
},
"oneOf": [
{
"required": ["root"],
"not": { "required": ["query"] }
},
{
"required": ["query"],
"not": { "required": ["root"] }
}
],
"additionalProperties": false
}
}
},
"additionalProperties": false
},
"graph": {
"type": "object",
"required": ["allowed_relations"],

View file

@ -0,0 +1,190 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/projection-package-v1.json",
"title": "DocForge immutable renderer projection package",
"$comment": "package_id and embedded plan identity equality are verified by the runtime contract validator.",
"$defs": {
"sha256": {
"type": "string",
"pattern": "^[0-9a-f]{64}$"
},
"manual_plan": {
"type": "object",
"required": [
"schema_version",
"contract",
"plan_id",
"project",
"view",
"changeset_hash",
"pages",
"navigation",
"search_documents",
"diagnostics"
],
"properties": {
"schema_version": { "const": 1 },
"contract": { "const": "docforge.manual-render-plan" },
"plan_id": { "$ref": "#/$defs/sha256" },
"project": { "type": "object" },
"view": { "type": "object" },
"changeset_hash": {
"oneOf": [
{ "$ref": "#/$defs/sha256" },
{ "type": "null" }
]
},
"pages": { "type": "array", "maxItems": 10000 },
"navigation": { "type": "array", "maxItems": 10000 },
"search_documents": { "type": "array", "maxItems": 10000 },
"diagnostics": { "type": "object" }
},
"additionalProperties": false
},
"graph_plan": {
"type": "object",
"required": [
"schema_version",
"contract",
"plan_id",
"project",
"view",
"bounds",
"policy",
"graph",
"omissions",
"diagnostics"
],
"properties": {
"schema_version": { "const": 1 },
"contract": { "const": "docforge.graph-view-plan" },
"plan_id": { "$ref": "#/$defs/sha256" },
"project": { "type": "object" },
"view": { "type": "object" },
"bounds": { "type": "object" },
"policy": { "type": "object" },
"graph": { "type": "object" },
"omissions": { "type": "array", "maxItems": 10000 },
"diagnostics": { "type": "object" }
},
"additionalProperties": false
},
"renderer": {
"type": "object",
"required": ["renderer_id", "renderer_version"],
"properties": {
"renderer_id": { "type": "string", "minLength": 1 },
"renderer_version": { "type": "string", "minLength": 1 }
},
"additionalProperties": false
},
"component": {
"type": "object",
"required": ["component_id"],
"properties": {
"component_id": { "type": "string", "minLength": 1 }
},
"additionalProperties": false
},
"asset": {
"type": "object",
"required": ["asset_id", "media_type", "sha256", "text"],
"properties": {
"asset_id": {
"type": "string",
"minLength": 1,
"pattern": "^[^/]+$"
},
"media_type": { "type": "string", "minLength": 1 },
"sha256": { "$ref": "#/$defs/sha256" },
"text": { "type": "string" }
},
"additionalProperties": false
}
},
"type": "object",
"required": [
"schema_version",
"contract",
"package_id",
"kind",
"plan_id",
"plan",
"renderer",
"components",
"assets",
"output_policy"
],
"properties": {
"schema_version": { "const": 1 },
"contract": { "const": "docforge.projection-package" },
"package_id": { "$ref": "#/$defs/sha256" },
"kind": { "enum": ["manual", "graph"] },
"plan_id": { "$ref": "#/$defs/sha256" },
"plan": {
"oneOf": [
{ "$ref": "#/$defs/manual_plan" },
{ "$ref": "#/$defs/graph_plan" }
]
},
"renderer": { "$ref": "#/$defs/renderer" },
"components": {
"type": "array",
"maxItems": 32,
"items": { "$ref": "#/$defs/component" },
"uniqueItems": true
},
"assets": {
"type": "array",
"maxItems": 32,
"items": { "$ref": "#/$defs/asset" }
},
"output_policy": {
"type": "object",
"required": ["artifact_ids", "max_total_bytes"],
"properties": {
"artifact_ids": {
"type": "array",
"minItems": 1,
"maxItems": 32,
"items": {
"type": "string",
"minLength": 1,
"pattern": "^[^/]+$"
},
"uniqueItems": true
},
"max_total_bytes": {
"type": "integer",
"minimum": 1
}
},
"additionalProperties": false
}
},
"allOf": [
{
"if": {
"properties": { "kind": { "const": "manual" } },
"required": ["kind"]
},
"then": {
"properties": {
"plan": { "$ref": "#/$defs/manual_plan" }
}
}
},
{
"if": {
"properties": { "kind": { "const": "graph" } },
"required": ["kind"]
},
"then": {
"properties": {
"plan": { "$ref": "#/$defs/graph_plan" }
}
}
}
],
"additionalProperties": false
}

View file

@ -0,0 +1,19 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/projection-policy-v2.json",
"title": "DocForge independent projection policy",
"type": "object",
"required": [
"schema_version",
"manual",
"portable_graph",
"live_viewer"
],
"properties": {
"schema_version": { "const": 2 },
"manual": { "enum": ["auto", "explicit", "disabled"] },
"portable_graph": { "enum": ["explicit", "disabled"] },
"live_viewer": { "enum": ["on-demand", "disabled"] }
},
"additionalProperties": false
}

View file

@ -0,0 +1,89 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/projection-receipt-v1.json",
"title": "DocForge projection renderer receipt",
"$comment": "receipt_id is the SHA-256 of canonical JSON without receipt_id and is verified by the runtime contract validator.",
"$defs": {
"sha256": {
"type": "string",
"pattern": "^[0-9a-f]{64}$"
},
"renderer": {
"type": "object",
"required": ["renderer_id", "renderer_version"],
"properties": {
"renderer_id": { "type": "string", "minLength": 1 },
"renderer_version": { "type": "string", "minLength": 1 }
},
"additionalProperties": false
},
"artifact": {
"type": "object",
"required": ["artifact_id", "media_type", "sha256", "bytes"],
"properties": {
"artifact_id": {
"type": "string",
"minLength": 1,
"pattern": "^[^/]+$"
},
"media_type": { "type": "string", "minLength": 1 },
"sha256": { "$ref": "#/$defs/sha256" },
"bytes": { "type": "integer", "minimum": 0 }
},
"additionalProperties": false
}
},
"type": "object",
"required": [
"schema_version",
"contract",
"receipt_id",
"kind",
"package_id",
"plan_id",
"renderer",
"artifacts",
"diagnostics",
"timing",
"peak_memory_bytes"
],
"properties": {
"schema_version": { "const": 1 },
"contract": { "const": "docforge.projection-receipt" },
"receipt_id": { "$ref": "#/$defs/sha256" },
"kind": { "enum": ["manual", "graph"] },
"package_id": { "$ref": "#/$defs/sha256" },
"plan_id": { "$ref": "#/$defs/sha256" },
"renderer": { "$ref": "#/$defs/renderer" },
"artifacts": {
"type": "array",
"maxItems": 32,
"items": { "$ref": "#/$defs/artifact" }
},
"diagnostics": {
"type": "object",
"required": ["warnings"],
"properties": {
"warnings": {
"type": "array",
"maxItems": 10000,
"items": { "type": "string" }
}
},
"additionalProperties": false
},
"timing": {
"type": "object",
"required": ["elapsed_ns"],
"properties": {
"elapsed_ns": { "type": "integer", "minimum": 0 }
},
"additionalProperties": false
},
"peak_memory_bytes": {
"type": ["integer", "null"],
"minimum": 0
}
},
"additionalProperties": false
}

View file

@ -0,0 +1,78 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://andraxion.net/docforge/schemas/reference-adapter.schema.json",
"title": "DocForge Reference Adapter Configuration",
"type": "object",
"additionalProperties": false,
"required": [
"schema_version",
"project_id",
"title",
"language",
"source_roots"
],
"properties": {
"schema_version": {
"const": 1
},
"project_id": {
"type": "string",
"pattern": "^[a-z0-9][a-z0-9._-]{1,127}$"
},
"title": {
"type": "string",
"minLength": 1,
"maxLength": 256
},
"language": {
"enum": [
"python",
"javascript",
"typescript",
"cpp"
]
},
"source_roots": {
"type": "array",
"minItems": 1,
"maxItems": 64,
"uniqueItems": true,
"items": {
"type": "string",
"minLength": 1,
"pattern": "^(?!/)(?!.*(?:^|/)\\.\\.(?:/|$)).+$"
}
},
"compilation_database": {
"type": "string",
"minLength": 1,
"pattern": "^(?!/)(?!.*(?:^|/)\\.\\.(?:/|$)).+$"
}
},
"allOf": [
{
"if": {
"properties": {
"language": {
"const": "cpp"
}
},
"required": [
"language"
]
},
"then": {
"required": [
"compilation_database"
]
},
"else": {
"not": {
"required": [
"compilation_database"
]
}
}
}
]
}

View file

@ -2,6 +2,187 @@
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://docforge.local/schema/result-v1.json",
"title": "DocForge result envelope",
"$defs": {
"diagnostics": {
"type": "object",
"required": [
"schema_version",
"operation",
"outcome",
"elapsed_ns",
"stages",
"counters"
],
"properties": {
"schema_version": { "const": 1 },
"operation": {
"enum": [
"test",
"benchmark.m1",
"benchmark.m2",
"benchmark.m3",
"mcp.invoke",
"mcp.bootstrap",
"mcp.sync",
"mcp.project_info",
"mcp.contract",
"mcp.get_node",
"mcp.get_logic",
"mcp.search",
"mcp.filter",
"mcp.backlinks",
"mcp.dependencies",
"mcp.impact",
"mcp.context",
"mcp.task_context",
"mcp.generation_diff",
"mcp.validate_project",
"mcp.render_status",
"mcp.graph_plan",
"mcp.graph_render_status",
"mcp.visualize",
"mcp.visualization_status",
"mcp.stop_visualization",
"mcp.changeset",
"mcp.mutation",
"cli.onboard",
"cli.info",
"cli.validate",
"cli.build",
"cli.reindex",
"cli.sync",
"cli.check",
"cli.validate-index",
"cli.show",
"cli.search",
"cli.filter",
"cli.backlinks",
"cli.dependencies",
"cli.impact",
"cli.context",
"cli.generation-diff",
"cli.graph-plan",
"cli.graph-render",
"cli.graph-render-status",
"cli.configure",
"cli.doctor",
"cli.render",
"cli.render-status",
"cli.preview",
"cli.apply",
"cli.visualize",
"cli.visualization-status",
"cli.visualization-stop"
]
},
"outcome": { "enum": ["ok", "error"] },
"elapsed_ns": { "type": "integer", "minimum": 0 },
"stages": {
"type": "object",
"maxProperties": 14,
"propertyNames": {
"enum": [
"source.generation",
"source.parse",
"adapter.projection",
"adapter.extract",
"index.check",
"index.synchronize",
"index.build",
"index.read",
"render.status",
"render.prepare",
"render.output_hash",
"visualization.status",
"viewer.manager",
"mcp.runtime_validation"
]
},
"additionalProperties": {
"type": "object",
"required": ["calls", "elapsed_ns"],
"properties": {
"calls": { "type": "integer", "minimum": 1 },
"elapsed_ns": { "type": "integer", "minimum": 0 }
},
"additionalProperties": false
}
},
"counters": {
"type": "object",
"required": [
"project_loads",
"source_files_parsed",
"source_bytes_parsed",
"adapter_projection_loads",
"adapter_source_extractions",
"source_generation_checks",
"index_checks",
"index_synchronizations",
"index_builds",
"render_prepare_calls",
"render_output_bytes_built",
"render_output_bytes_hashed",
"viewer_manager_requests"
],
"properties": {
"project_loads": { "type": "integer", "minimum": 0 },
"source_files_parsed": { "type": "integer", "minimum": 0 },
"source_bytes_parsed": { "type": "integer", "minimum": 0 },
"adapter_projection_loads": { "type": "integer", "minimum": 0 },
"adapter_source_extractions": { "type": "integer", "minimum": 0 },
"source_generation_checks": { "type": "integer", "minimum": 0 },
"index_checks": { "type": "integer", "minimum": 0 },
"index_synchronizations": { "type": "integer", "minimum": 0 },
"index_builds": { "type": "integer", "minimum": 0 },
"render_prepare_calls": { "type": "integer", "minimum": 0 },
"render_output_bytes_built": { "type": "integer", "minimum": 0 },
"render_output_bytes_hashed": { "type": "integer", "minimum": 0 },
"viewer_manager_requests": { "type": "integer", "minimum": 0 }
},
"additionalProperties": false
}
},
"additionalProperties": false
},
"pagination": {
"type": "object",
"required": [
"schema_version",
"kind",
"returned_count",
"limit",
"total_count",
"has_more",
"next_cursor"
],
"properties": {
"schema_version": { "const": 1 },
"kind": {
"enum": [
"context.items",
"task-context.items",
"generation-diff.items",
"changeset.list",
"changeset.inspect",
"changeset.validate",
"changeset.diff",
"changeset.diff-chunks"
]
},
"returned_count": { "type": "integer", "minimum": 0 },
"limit": { "type": "integer", "minimum": 1 },
"total_count": { "type": "integer", "minimum": 0 },
"has_more": { "type": "boolean" },
"next_cursor": {
"type": ["string", "null"],
"minLength": 1,
"maxLength": 8192
}
},
"additionalProperties": false
}
},
"oneOf": [
{
"type": "object",
@ -10,27 +191,60 @@
"status": { "const": "ok" },
"project_id": { "type": "string" },
"revision": { "type": "string" },
"source_hash": { "type": "string", "pattern": "^[0-9a-f]{64}$" },
"adapter": { "type": "string" }
}
"source_hash": {
"type": ["string", "null"],
"pattern": "^[0-9a-f]{64}$"
},
"adapter": { "type": "string" },
"pagination": { "$ref": "#/$defs/pagination" },
"diagnostics": { "$ref": "#/$defs/diagnostics" }
},
"additionalProperties": true
},
{
"type": "object",
"required": ["status", "error"],
"properties": {
"status": { "const": "error" },
"project_id": { "type": "string" },
"project_root_fingerprint": {
"type": "string",
"pattern": "^[0-9a-f]{16}$"
},
"adapter": { "type": "string" },
"server_version": { "type": "string" },
"revision": { "type": "string" },
"source_hash": {
"type": ["string", "null"],
"pattern": "^[0-9a-f]{64}$"
},
"content_warning": { "type": "string" },
"staleness": { "enum": ["current", "stale", "unknown"] },
"synchronization": { "type": "object" },
"diagnostics": { "$ref": "#/$defs/diagnostics" },
"error": {
"type": "object",
"required": ["code", "message", "details"],
"properties": {
"code": { "type": "string", "minLength": 1 },
"message": { "type": "string", "minLength": 1 },
"details": { "type": "object" }
"details": { "type": "object" },
"remediation": {
"type": "object",
"required": ["retryable"],
"properties": {
"retryable": { "type": "boolean" },
"action": { "type": "string", "minLength": 1 },
"tool": { "type": "string", "minLength": 1 },
"arguments": { "type": "object" }
},
"additionalProperties": false
}
},
"additionalProperties": false
}
},
"additionalProperties": false
"additionalProperties": true
}
]
}

View file

@ -1,5 +1,6 @@
"""Project-scoped documentation retrieval, proposals, and gated application."""
from ._version import __version__
from .application import CanonicalApplicationService, CanonicalApplier, GenericCanonicalApplier
from .errors import DocForgeError
from .project import Project
@ -10,5 +11,5 @@ __all__ = [
"DocForgeError",
"GenericCanonicalApplier",
"Project",
"__version__",
]
__version__ = "0.15.0"

367
src/docforge/_fs_safety.py Normal file
View file

@ -0,0 +1,367 @@
"""Internal directory binding helpers for disposable publication paths."""
from __future__ import annotations
import ctypes
import errno
import os
import secrets
import stat
from collections.abc import Callable
from contextlib import suppress
from pathlib import Path
from typing import Protocol, cast
from .errors import DocForgeError
RENAME_NOREPLACE = 1
RENAME_EXCHANGE = 2
class _RenameAt2(Protocol):
argtypes: list[object]
restype: object
def __call__(
self,
old_directory_fd: int,
old_name: bytes,
new_directory_fd: int,
new_name: bytes,
flags: int,
/,
) -> int: ...
def _rename_at2(
old_directory_fd: int,
old_name: str,
new_directory_fd: int,
new_name: str,
flags: int,
) -> int:
library = ctypes.CDLL(None, use_errno=True)
try:
rename_at2 = cast(_RenameAt2, library.renameat2)
except AttributeError as error:
raise DocForgeError(
"atomic_exchange_unavailable",
"Atomic exchange is unavailable on this platform",
) from error
rename_at2.argtypes = [
ctypes.c_int,
ctypes.c_char_p,
ctypes.c_int,
ctypes.c_char_p,
ctypes.c_uint,
]
rename_at2.restype = ctypes.c_int
ctypes.set_errno(0)
result = rename_at2(
old_directory_fd,
os.fsencode(old_name),
new_directory_fd,
os.fsencode(new_name),
flags,
)
if result == 0:
return 0
error_number = ctypes.get_errno()
if error_number in {errno.ENOSYS, errno.EINVAL, errno.EOPNOTSUPP}:
raise DocForgeError(
"atomic_exchange_unavailable",
"Atomic exchange is unavailable on this filesystem",
)
return error_number
def rename_exchange_between_at(
first_directory_fd: int,
first: str,
second_directory_fd: int,
second: str,
) -> None:
"""Atomically exchange names between two bound directories on one filesystem."""
error_number = _rename_at2(
first_directory_fd,
first,
second_directory_fd,
second,
RENAME_EXCHANGE,
)
if error_number == 0:
return
raise DocForgeError(
"publication_failure",
"Could not exchange atomic publication paths",
error_number=error_number,
) from OSError(error_number, os.strerror(error_number))
def rename_exchange_at(directory_fd: int, first: str, second: str) -> None:
"""Atomically exchange two names inside one already bound directory."""
rename_exchange_between_at(directory_fd, first, directory_fd, second)
def rename_noreplace_between_at(
source_directory_fd: int,
source: str,
target_directory_fd: int,
target: str,
) -> bool:
"""Atomically move one name without replacing a target that appeared."""
error_number = _rename_at2(
source_directory_fd,
source,
target_directory_fd,
target,
RENAME_NOREPLACE,
)
if error_number == 0:
return True
if error_number == errno.EEXIST:
return False
raise DocForgeError(
"publication_failure",
"Could not move an atomic publication path without replacement",
error_number=error_number,
) from OSError(error_number, os.strerror(error_number))
def open_bound_directory(path: Path) -> int:
"""Open one real directory and bind its current inode for later operations."""
try:
path_status = path.lstat()
if (
stat.S_ISLNK(path_status.st_mode)
or not stat.S_ISDIR(path_status.st_mode)
or path.resolve(strict=True) != path
):
raise DocForgeError(
"path_escape",
"Derived cache root is not a safe real directory",
)
directory_fd = os.open(path, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW)
except FileNotFoundError as error:
raise DocForgeError(
"missing_index",
"Derived cache root does not exist",
) from error
except OSError as error:
raise DocForgeError(
"path_escape",
"Derived cache root cannot be opened safely",
) from error
try:
opened_status = os.fstat(directory_fd)
if opened_status.st_dev != path_status.st_dev or opened_status.st_ino != path_status.st_ino:
raise DocForgeError(
"path_escape",
"Derived cache root changed while opening",
)
except Exception:
os.close(directory_fd)
raise
return directory_fd
def require_bound_directory(path: Path, directory_fd: int) -> None:
"""Require a path to still name the exact opened real directory."""
try:
path_status = path.lstat()
opened_status = os.fstat(directory_fd)
if (
stat.S_ISLNK(path_status.st_mode)
or not stat.S_ISDIR(path_status.st_mode)
or path.resolve(strict=True) != path
or opened_status.st_dev != path_status.st_dev
or opened_status.st_ino != path_status.st_ino
):
raise DocForgeError(
"path_escape",
"Derived cache root changed during publication",
)
except FileNotFoundError as error:
raise DocForgeError(
"path_escape",
"Derived cache root disappeared during publication",
) from error
def open_confined_directory(root: Path, path: Path, *, create: bool) -> int:
"""Open a descendant directory through stable no-follow directory descriptors."""
try:
unsafe = (
root.is_symlink()
or root.resolve(strict=True) != root
or not path.is_relative_to(root)
or path == root
)
except OSError as error:
raise DocForgeError("path_escape", "Project root cannot be resolved safely") from error
if unsafe:
raise DocForgeError("path_escape", "Derived output directory is not confined")
relative = path.relative_to(root)
try:
descriptor = os.open(root, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW)
except OSError as error:
raise DocForgeError("path_escape", "Project root cannot be opened safely") from error
try:
for part in relative.parts:
if part in {"", ".", ".."}:
raise DocForgeError("path_escape", "Derived output directory is not confined")
if create:
try:
os.mkdir(part, mode=0o700, dir_fd=descriptor)
except FileExistsError:
pass
except OSError as error:
raise DocForgeError(
"publication_failure",
"Derived output directory could not be created",
) from error
try:
next_descriptor = os.open(
part,
os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW,
dir_fd=descriptor,
)
except OSError as error:
raise DocForgeError(
"path_escape",
"Derived output directory is missing or unsafe",
) from error
os.close(descriptor)
descriptor = next_descriptor
require_bound_directory(path, descriptor)
return descriptor
except Exception:
os.close(descriptor)
raise
def safe_file_identity_at(
directory: Path,
directory_fd: int,
name: str,
) -> dict[str, object] | None:
"""Return one no-follow regular-file identity relative to a bound directory."""
del directory
if not name or "/" in name or name in {".", ".."}:
raise DocForgeError("path_escape", "Derived artifact name is unsafe")
try:
current = os.stat(name, dir_fd=directory_fd, follow_symlinks=False)
except FileNotFoundError:
return None
except OSError as error:
raise DocForgeError("path_escape", "Derived artifact cannot be inspected") from error
if not stat.S_ISREG(current.st_mode):
raise DocForgeError("path_escape", "Derived artifact is not a safe regular file")
return {
"path": name,
"device": current.st_dev,
"inode": current.st_ino,
"mode": current.st_mode,
"size": current.st_size,
"mtime_ns": current.st_mtime_ns,
"ctime_ns": current.st_ctime_ns,
}
def read_bounded_file_at(
directory_fd: int,
name: str,
maximum_bytes: int,
) -> bytes | None:
"""Read one regular file through a bound directory without following links."""
try:
descriptor = os.open(name, os.O_RDONLY | os.O_NOFOLLOW, dir_fd=directory_fd)
except FileNotFoundError:
return None
except OSError as error:
raise DocForgeError("path_escape", "Derived artifact cannot be opened safely") from error
with os.fdopen(descriptor, "rb") as handle:
current = os.fstat(handle.fileno())
if not stat.S_ISREG(current.st_mode) or current.st_size > maximum_bytes:
raise DocForgeError("invalid_projection", "Derived artifact is invalid or oversized")
content = handle.read(maximum_bytes + 1)
if len(content) > maximum_bytes:
raise DocForgeError("invalid_projection", "Derived artifact is oversized")
return content
def atomic_replace_bytes_at(
path: Path,
directory_fd: int,
name: str,
content: bytes,
*,
verify: Callable[[], None],
) -> dict[str, object]:
"""Durably replace one file inside an already bound directory."""
if not name or "/" in name or name in {".", ".."}:
raise DocForgeError("path_escape", "Derived artifact name is unsafe")
existing = safe_file_identity_at(path, directory_fd, name)
del existing
temporary = f".docforge-projection-{secrets.token_hex(12)}"
descriptor: int | None = None
committed = False
try:
descriptor = os.open(
temporary,
os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW,
0o600,
dir_fd=directory_fd,
)
with os.fdopen(descriptor, "wb") as handle:
descriptor = None
handle.write(content)
handle.flush()
os.fsync(handle.fileno())
verify()
require_bound_directory(path, directory_fd)
os.replace(
temporary,
name,
src_dir_fd=directory_fd,
dst_dir_fd=directory_fd,
)
committed = True
os.fsync(directory_fd)
identity = safe_file_identity_at(path, directory_fd, name)
if identity is None:
raise DocForgeError(
"publication_failure",
"Derived artifact disappeared after publication",
mutation_committed=True,
)
return identity
except DocForgeError as error:
if committed:
raise DocForgeError(
"publication_failure",
"Derived artifact was replaced but final publication verification failed",
mutation_committed=True,
cause=error.code,
) from error
raise
except OSError as error:
raise DocForgeError(
"publication_failure",
"Derived artifact publication failed",
mutation_committed=committed,
) from error
finally:
if descriptor is not None:
os.close(descriptor)
with suppress(OSError):
os.unlink(temporary, dir_fd=directory_fd)

View file

@ -0,0 +1,8 @@
"""Private module entry point for the detached projection worker."""
from __future__ import annotations
from .projection_worker import main
if __name__ == "__main__":
raise SystemExit(main())

3
src/docforge/_version.py Normal file
View file

@ -0,0 +1,3 @@
"""Single authoritative DocForge distribution and runtime version."""
__version__ = "2.0.0"

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,392 @@
"""Strict, data-only launch contracts for project-owned adapter MCP modules."""
from __future__ import annotations
import json
import re
import stat
import subprocess
import sys
from dataclasses import dataclass
from pathlib import Path
from typing import Literal, cast
from .changeset_contract import document_hash
from .config_validation import ID_PATTERN
from .errors import DocForgeError
from .models import IncrementalStateProject, ProjectService, RuntimeValidatedProject
ADAPTER_LAUNCHER_SCHEMA_VERSION = 1
_PYTHON_MODULE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*(?:\.[A-Za-z_][A-Za-z0-9_]*){0,31}")
_TOP_LEVEL_PYTHON_MODULE = re.compile(r"[A-Za-z_][A-Za-z0-9_]*")
_ADAPTER_IDENTITY = re.compile(r"[A-Za-z0-9][A-Za-z0-9_.-]{0,127}@[A-Za-z0-9][A-Za-z0-9_.+-]{0,63}")
_SHA256 = re.compile(r"[0-9a-f]{64}")
_TRUSTED_DOTTED_MODULE = "docforge.reference_mcp"
_MODULE_PROBE_TIMEOUT_SECONDS = 5
_MAX_MODULE_PROBE_BYTES = 8_192
_MODULE_PROBE = (
"import importlib.util,json,sys;"
"spec=importlib.util.find_spec(sys.argv[1]);"
"result=None if spec is None else {"
"'has_loader':spec.loader is not None,"
"'is_package':spec.submodule_search_locations is not None,"
"'origin':spec.origin};"
"print(json.dumps(result,sort_keys=True,separators=(',',':')))"
)
_run_module_probe = subprocess.run
__all__ = [
"ADAPTER_LAUNCHER_SCHEMA_VERSION",
"AdapterLauncherV1",
"AdapterSourceAvailabilityV1",
"adapter_source_availability",
"validate_adapter_launcher",
]
@dataclass(frozen=True)
class AdapterLauncherV1:
"""One immutable, project-bound Python module launch declaration.
The contract intentionally has no command, shell string, working directory,
environment, arbitrary arguments, discovery rule, or callable selector.
"""
schema_version: int
entry_point: Literal["python-module"]
project_id: str
project_root: Path
adapter: str
descriptor_hash: str
module: str
def __post_init__(self) -> None:
_validate_launcher_fields(self)
@classmethod
def for_project(
cls,
project: ProjectService,
*,
module: str,
) -> AdapterLauncherV1:
"""Bind a structural module entry point to one already constructed adapter project."""
descriptor = project.descriptor
if descriptor.adapter == "generic":
raise DocForgeError(
"adapter_launcher_unavailable",
"Project-owned launcher configuration requires a custom adapter",
)
launcher = cls(
schema_version=ADAPTER_LAUNCHER_SCHEMA_VERSION,
entry_point="python-module",
project_id=descriptor.project_id,
project_root=descriptor.root,
adapter=descriptor.adapter,
descriptor_hash=descriptor.descriptor_hash,
module=module,
)
validate_adapter_launcher(project, launcher)
return launcher
def as_dict(self) -> dict[str, object]:
return {
"schema_version": self.schema_version,
"entry_point": self.entry_point,
"project_id": self.project_id,
"project_root": str(self.project_root),
"adapter": self.adapter,
"descriptor_hash": self.descriptor_hash,
"module": self.module,
}
@property
def launcher_hash(self) -> str:
return document_hash(self.as_dict())
@dataclass(frozen=True)
class AdapterSourceAvailabilityV1:
"""One bounded source identity observed without executing the launcher module."""
schema_version: int
status: Literal["available"]
method: Literal["incremental-state", "complete-projection"]
revision: str
source_hash: str
def __post_init__(self) -> None:
revision = _runtime_value(self.revision)
if (
self.schema_version != 1
or self.status != "available"
or self.method not in {"incremental-state", "complete-projection"}
or not isinstance(revision, str)
or not revision
or _SHA256.fullmatch(self.source_hash) is None
):
raise DocForgeError(
"invalid_adapter_launcher",
"Adapter source availability evidence is invalid",
)
def as_dict(self) -> dict[str, object]:
return {
"schema_version": self.schema_version,
"status": self.status,
"method": self.method,
"revision": self.revision,
"source_hash": self.source_hash,
}
@property
def availability_hash(self) -> str:
return document_hash(self.as_dict())
def validate_adapter_launcher(
project: ProjectService,
launcher: AdapterLauncherV1,
) -> None:
"""Require one launcher to match the exact current adapter project binding."""
_validate_launcher_fields(launcher)
descriptor = project.descriptor
_validate_project_root(descriptor.root)
if descriptor.adapter == "generic":
raise DocForgeError(
"adapter_launcher_unavailable",
"Project-owned launcher configuration requires a custom adapter",
)
if (
launcher.project_id != descriptor.project_id
or launcher.project_root != descriptor.root
or launcher.adapter != descriptor.adapter
or launcher.descriptor_hash != descriptor.descriptor_hash
):
raise DocForgeError(
"adapter_launcher_mismatch",
"Adapter launcher does not match the selected project binding",
project_id=descriptor.project_id,
adapter=descriptor.adapter,
)
if isinstance(project, RuntimeValidatedProject):
project.validate_runtime()
def adapter_source_availability(
project: ProjectService,
launcher: AdapterLauncherV1,
) -> AdapterSourceAvailabilityV1:
"""Capture current source identity without importing or executing the launcher module."""
validate_adapter_launcher(project, launcher)
state = project.incremental_state() if isinstance(project, IncrementalStateProject) else None
if state is not None:
method: Literal["incremental-state", "complete-projection"] = "incremental-state"
revision = state.revision
source_hash = state.source_hash
else:
snapshot = project.load()
descriptor = snapshot.descriptor
if (
descriptor.project_id != launcher.project_id
or descriptor.root != launcher.project_root
or descriptor.adapter != launcher.adapter
or descriptor.descriptor_hash != launcher.descriptor_hash
):
raise DocForgeError(
"adapter_launcher_mismatch",
"Loaded adapter projection drifted from its launcher binding",
)
method = "complete-projection"
revision = snapshot.revision
source_hash = snapshot.source_hash
validate_adapter_launcher(project, launcher)
return AdapterSourceAvailabilityV1(
schema_version=1,
status="available",
method=method,
revision=revision,
source_hash=source_hash,
)
def _validate_launcher_fields(launcher: AdapterLauncherV1) -> None:
if type(launcher.schema_version) is not int or launcher.schema_version != 1:
raise DocForgeError(
"invalid_adapter_launcher",
"Adapter launcher schema version is unsupported",
supported=1,
)
if launcher.entry_point != "python-module":
raise DocForgeError(
"invalid_adapter_launcher",
"Adapter launcher entry point must be one isolated Python module",
)
project_id = _runtime_value(launcher.project_id)
project_root = _runtime_value(launcher.project_root)
adapter = _runtime_value(launcher.adapter)
descriptor_hash = _runtime_value(launcher.descriptor_hash)
module = _runtime_value(launcher.module)
if not isinstance(project_id, str) or ID_PATTERN.fullmatch(project_id) is None:
raise DocForgeError("invalid_adapter_launcher", "Adapter launcher project ID is invalid")
if not isinstance(project_root, Path):
raise DocForgeError("invalid_adapter_launcher", "Adapter launcher project root is invalid")
_validate_project_root(project_root)
if not isinstance(adapter, str) or _ADAPTER_IDENTITY.fullmatch(adapter) is None:
raise DocForgeError("invalid_adapter_launcher", "Adapter launcher identity is invalid")
if not isinstance(descriptor_hash, str) or _SHA256.fullmatch(descriptor_hash) is None:
raise DocForgeError("invalid_adapter_launcher", "Adapter descriptor hash is invalid")
if (
not isinstance(module, str)
or len(module) > 255
or _PYTHON_MODULE.fullmatch(module) is None
or module == "docforge.mcp_server"
or (module != _TRUSTED_DOTTED_MODULE and _TOP_LEVEL_PYTHON_MODULE.fullmatch(module) is None)
):
raise DocForgeError(
"invalid_adapter_launcher",
(
"Adapter launcher module must be one installed project-owned top-level "
"Python module or the fixed DocForge reference binding"
),
)
_validate_isolated_module(project_root, module)
def _validate_isolated_module(project_root: Path, module: str) -> None:
"""Prove ``python -I -m`` can resolve one module without importing project code."""
executable = Path(sys.executable)
try:
executable_status = executable.stat()
except OSError as error:
raise DocForgeError(
"adapter_launcher_unavailable",
"The isolated Python executable is unavailable",
) from error
if not stat.S_ISREG(executable_status.st_mode):
raise DocForgeError(
"adapter_launcher_unavailable",
"The isolated Python executable is not a regular file",
)
try:
completed = _run_module_probe(
[str(executable), "-I", "-c", _MODULE_PROBE, module],
capture_output=True,
text=True,
timeout=_MODULE_PROBE_TIMEOUT_SECONDS,
check=False,
)
except (OSError, subprocess.TimeoutExpired) as error:
raise DocForgeError(
"adapter_launcher_unavailable",
"The isolated adapter launcher probe could not complete",
module=module,
) from error
if (
completed.returncode != 0
or len(completed.stdout.encode("utf-8")) > _MAX_MODULE_PROBE_BYTES
or len(completed.stderr.encode("utf-8")) > _MAX_MODULE_PROBE_BYTES
):
raise DocForgeError(
"adapter_launcher_unavailable",
"The isolated adapter launcher module could not be resolved",
module=module,
)
try:
result: object = json.loads(completed.stdout)
except (UnicodeError, json.JSONDecodeError) as error:
raise DocForgeError(
"adapter_launcher_unavailable",
"The isolated adapter launcher probe returned invalid evidence",
module=module,
) from error
if result is None:
raise DocForgeError(
"adapter_launcher_unavailable",
"The isolated adapter launcher module is not installed",
module=module,
)
if not isinstance(result, dict):
raise DocForgeError(
"adapter_launcher_unavailable",
"The isolated adapter launcher probe returned invalid evidence",
module=module,
)
evidence = cast(dict[object, object], result)
if (
set(evidence) != {"has_loader", "is_package", "origin"}
or evidence["has_loader"] is not True
or evidence["is_package"] is not False
or not isinstance(evidence["origin"], str)
):
raise DocForgeError(
"invalid_adapter_launcher",
"The isolated adapter launcher must resolve to one executable module file",
module=module,
)
origin = Path(evidence["origin"])
try:
origin_status = origin.lstat()
resolved_origin = origin.resolve(strict=True)
except OSError as error:
raise DocForgeError(
"adapter_launcher_unavailable",
"The isolated adapter launcher module origin is unavailable",
module=module,
) from error
if (
not origin.is_absolute()
or stat.S_ISLNK(origin_status.st_mode)
or not stat.S_ISREG(origin_status.st_mode)
or resolved_origin != origin
or origin.suffix != ".py"
):
raise DocForgeError(
"invalid_adapter_launcher",
"The isolated adapter launcher must be one canonical Python module file",
module=module,
)
if module == _TRUSTED_DOTTED_MODULE:
expected = Path(__file__).with_name("reference_mcp.py").resolve(strict=True)
if origin != expected:
raise DocForgeError(
"adapter_launcher_mismatch",
"The fixed reference launcher resolved outside this DocForge installation",
module=module,
)
elif not origin.is_relative_to(project_root):
raise DocForgeError(
"invalid_adapter_launcher",
"The installed adapter launcher module is not owned by the selected project",
module=module,
)
def _validate_project_root(root: Path) -> None:
try:
status = root.lstat()
resolved = root.resolve(strict=True)
except OSError as error:
raise DocForgeError(
"invalid_adapter_launcher",
"Adapter launcher project root is unavailable",
) from error
if (
not root.is_absolute()
or stat.S_ISLNK(status.st_mode)
or not stat.S_ISDIR(status.st_mode)
or resolved != root
):
raise DocForgeError(
"invalid_adapter_launcher",
"Adapter launcher project root must be one canonical real directory",
)
def _runtime_value(value: object) -> object:
"""Keep runtime validation explicit even when static callers are typed."""
return value

View file

@ -0,0 +1,69 @@
"""Public typed surface for project adapter implementations and conformance checks."""
from .adapter_contract import (
AdapterAssembly,
AdapterConformanceReport,
AdapterEdge,
AdapterImplementation,
AdapterLoader,
AdapterManifest,
AdapterNode,
AdapterProject,
AdapterProjection,
AdapterProjectSettings,
AdapterSource,
AdapterSourceProjection,
CompleteAdapterAssemblyLoader,
IncrementalAdapterAssembler,
IncrementalAdapterLoader,
verify_adapter_conformance,
)
from .adapter_validation import (
validate_logic_projection,
validate_manifest,
validate_projection,
validate_source_projection,
)
from .models import (
Edge,
Limits,
LogicEdge,
LogicNode,
LogicProjection,
Node,
ProposalWriter,
RenderConfig,
RenderView,
)
__all__ = [
"AdapterAssembly",
"AdapterConformanceReport",
"AdapterEdge",
"AdapterImplementation",
"AdapterLoader",
"AdapterManifest",
"AdapterNode",
"AdapterProject",
"AdapterProjection",
"AdapterProjectSettings",
"AdapterSource",
"AdapterSourceProjection",
"CompleteAdapterAssemblyLoader",
"Edge",
"IncrementalAdapterAssembler",
"IncrementalAdapterLoader",
"Limits",
"LogicEdge",
"LogicNode",
"LogicProjection",
"Node",
"ProposalWriter",
"RenderConfig",
"RenderView",
"validate_logic_projection",
"validate_manifest",
"validate_projection",
"validate_source_projection",
"verify_adapter_conformance",
]

View file

@ -0,0 +1,334 @@
"""Validation and cache serialization for adapter-owned graph projections."""
from __future__ import annotations
from pathlib import Path
from typing import TYPE_CHECKING, cast
from .config_validation import AUTHORITIES, ID_PATTERN
from .errors import DocForgeError
from .models import Edge, LogicEdge, LogicNode, LogicProjection, Node
from .project import validate_graph
if TYPE_CHECKING:
from .adapter_contract import (
AdapterManifest,
AdapterProjection,
AdapterSource,
AdapterSourceProjection,
)
def validate_projection(projection: AdapterProjection) -> None:
"""Validate generic invariants without interpreting adapter metadata."""
root = projection.root.resolve(strict=True)
if not root.is_dir() or projection.root != root:
raise DocForgeError("invalid_adapter", "Adapter root must be a resolved directory")
for label, value in (
("project_id", projection.project_id),
("title", projection.title),
("adapter_id", projection.adapter_id),
("adapter_version", projection.adapter_version),
("revision", projection.revision),
("source_hash", projection.source_hash),
):
if not value.strip():
raise DocForgeError("invalid_adapter", f"Adapter {label} must not be empty")
if ID_PATTERN.fullmatch(projection.project_id) is None:
raise DocForgeError("invalid_adapter", "Adapter project ID is invalid")
validate_sha256(projection.source_hash, label="source hash")
ordered_nodes = tuple(sorted(projection.nodes, key=lambda item: item.node.node_id))
ordered_edges = tuple(
sorted(
projection.edges,
key=lambda item: (
item.edge.source_id,
item.edge.relation,
item.edge.target_id,
),
)
)
if projection.nodes != ordered_nodes or projection.edges != ordered_edges:
raise DocForgeError(
"invalid_adapter", "Adapter projection must be deterministically ordered"
)
for item in (*projection.nodes, *projection.edges):
keys = [key for key, _ in item.metadata]
if keys != sorted(keys) or len(keys) != len(set(keys)):
raise DocForgeError(
"invalid_adapter", "Adapter metadata keys must be unique and ordered"
)
for item in projection.nodes:
node = item.node
source = Path(node.source_path)
if ID_PATTERN.fullmatch(node.node_id) is None:
raise DocForgeError("invalid_adapter", "Adapter node ID is invalid", id=node.node_id)
if node.authority not in AUTHORITIES:
raise DocForgeError(
"invalid_adapter", "Adapter node authority is invalid", id=node.node_id
)
if (
not node.title.strip()
or not node.family.strip()
or not node.status.strip()
or not node.summary.strip()
or not node.content.strip()
):
raise DocForgeError(
"invalid_adapter", "Adapter node has empty required content", id=node.node_id
)
if source.is_absolute() or ".." in source.parts or not node.source_path:
raise DocForgeError(
"invalid_adapter", "Adapter node source path is unsafe", id=node.node_id
)
if len(node.tags) != len(set(node.tags)) or any(not tag for tag in node.tags):
raise DocForgeError("invalid_adapter", "Adapter node tags are invalid", id=node.node_id)
validate_sha256(node.content_hash, label="node content hash")
for item in projection.edges:
if ID_PATTERN.fullmatch(item.edge.relation) is None:
raise DocForgeError("invalid_adapter", "Adapter relationship type is invalid")
validate_graph(projection.core_nodes(), projection.core_edges())
def validate_manifest(manifest: AdapterManifest) -> None:
"""Validate a cheap incremental manifest without parsing project sources."""
root = manifest.root.resolve(strict=True)
if manifest.root != root or not root.is_dir():
raise DocForgeError("invalid_adapter", "Adapter root must be a resolved directory")
for label, value in (
("project_id", manifest.project_id),
("title", manifest.title),
("adapter_id", manifest.adapter_id),
("adapter_version", manifest.adapter_version),
("revision", manifest.revision),
("source_hash", manifest.source_hash),
):
if not value.strip():
raise DocForgeError("invalid_adapter", f"Adapter {label} must not be empty")
if ID_PATTERN.fullmatch(manifest.project_id) is None:
raise DocForgeError("invalid_adapter", "Adapter project ID is invalid")
validate_sha256(manifest.source_hash, label="source hash")
if manifest.estimated_nodes < 1:
raise DocForgeError("invalid_adapter", "Adapter estimated node count must be positive")
if (
manifest.families != tuple(sorted(set(manifest.families)))
or not manifest.families
or any(not family.strip() for family in manifest.families)
):
raise DocForgeError("invalid_adapter", "Adapter families must be unique and ordered")
if manifest.allowed_relations != tuple(sorted(set(manifest.allowed_relations))) or any(
ID_PATTERN.fullmatch(relation) is None for relation in manifest.allowed_relations
):
raise DocForgeError(
"invalid_adapter", "Adapter relationship types must be valid and ordered"
)
ordered = tuple(sorted(manifest.sources, key=lambda source: source.source_id))
if manifest.sources != ordered or len({source.source_id for source in ordered}) != len(ordered):
raise DocForgeError("invalid_adapter", "Adapter sources must be unique and ordered")
paths: set[str] = set()
source_ids = {source.source_id for source in ordered}
for source in ordered:
if ID_PATTERN.fullmatch(source.source_id) is None:
raise DocForgeError("invalid_adapter", "Adapter source ID is invalid")
path = Path(source.source_path)
if (
not source.source_path
or path.is_absolute()
or ".." in path.parts
or source.source_path in paths
):
raise DocForgeError("invalid_adapter", "Adapter source path is unsafe or repeated")
paths.add(source.source_path)
validate_sha256(source.fingerprint, label="source fingerprint")
if not source.extractor_version.strip():
raise DocForgeError("invalid_adapter", "Source extractor version must not be empty")
if (
source.dependencies != tuple(sorted(set(source.dependencies)))
or source.source_id in source.dependencies
or not set(source.dependencies).issubset(source_ids)
):
raise DocForgeError(
"invalid_adapter", "Adapter source dependencies are invalid or unordered"
)
def validate_source_projection(
source: AdapterSource, contribution: AdapterSourceProjection
) -> None:
"""Validate one extraction unit before it enters the reusable cache."""
if contribution.source_id != source.source_id or contribution.fingerprint != source.fingerprint:
raise DocForgeError(
"invalid_adapter", "Source projection identity does not match its manifest entry"
)
if contribution.nodes != tuple(
sorted(contribution.nodes, key=lambda item: item.node.node_id)
) or contribution.edges != tuple(
sorted(
contribution.edges,
key=lambda item: (
item.edge.source_id,
item.edge.relation,
item.edge.target_id,
),
)
):
raise DocForgeError(
"invalid_adapter", "Source projection must be deterministically ordered"
)
for item in (*contribution.nodes, *contribution.edges):
keys = [key for key, _ in item.metadata]
if keys != sorted(keys) or len(keys) != len(set(keys)):
raise DocForgeError(
"invalid_adapter", "Adapter metadata keys must be unique and ordered"
)
node_ids = {item.node.node_id for item in contribution.nodes}
if len(node_ids) != len(contribution.nodes):
raise DocForgeError("invalid_adapter", "Source projection contains duplicate nodes")
for logic in contribution.logic:
if logic.source_id != source.source_id or logic.owner_node_id not in node_ids:
raise DocForgeError(
"invalid_adapter", "Logic projection must be owned by a node in its source"
)
validate_logic_projection(logic)
def validate_logic_projection(projection: LogicProjection) -> None:
node_ids = [node.logic_id for node in projection.nodes]
if (
projection.nodes != tuple(sorted(projection.nodes, key=lambda node: node.logic_id))
or len(node_ids) != len(set(node_ids))
or any(ID_PATTERN.fullmatch(node_id) is None for node_id in node_ids)
):
raise DocForgeError("invalid_adapter", "Logic nodes must be valid, unique, and ordered")
ordered_edges = tuple(
sorted(
projection.edges,
key=lambda edge: (
edge.source_id,
edge.ordinal,
edge.relation,
edge.target_id,
edge.label or "",
),
)
)
if projection.edges != ordered_edges:
raise DocForgeError("invalid_adapter", "Logic edges must be deterministically ordered")
known = set(node_ids)
for edge in projection.edges:
if (
edge.source_id not in known
or edge.target_id not in known
or ID_PATTERN.fullmatch(edge.relation) is None
or edge.ordinal < 0
):
raise DocForgeError("invalid_adapter", "Logic edge is invalid")
def validate_sha256(value: str, *, label: str) -> None:
if len(value) != 64 or any(character not in "0123456789abcdef" for character in value):
raise DocForgeError("invalid_adapter", f"Adapter {label} must be lowercase SHA-256")
def source_payload(contribution: AdapterSourceProjection) -> dict[str, object]:
return {
"source_id": contribution.source_id,
"fingerprint": contribution.fingerprint,
"nodes": [item.as_dict() for item in contribution.nodes],
"edges": [item.as_dict() for item in contribution.edges],
"logic": [projection.as_dict() for projection in contribution.logic],
}
def source_projection(payload: dict[str, object]) -> AdapterSourceProjection:
from .adapter_contract import AdapterEdge, AdapterNode, AdapterSourceProjection
try:
raw_nodes = cast(list[dict[str, object]], payload["nodes"])
raw_edges = cast(list[dict[str, object]], payload["edges"])
raw_logic = cast(list[dict[str, object]], payload["logic"])
nodes = tuple(
AdapterNode(
node=node_from_dict(cast(dict[str, object], item["node"])),
metadata=tuple(
sorted(
(str(key), str(value))
for key, value in cast(dict[str, object], item["metadata"]).items()
)
),
)
for item in raw_nodes
)
edges = tuple(
AdapterEdge(
edge=Edge(**cast(dict[str, str], item["edge"])),
metadata=tuple(
sorted(
(str(key), str(value))
for key, value in cast(dict[str, object], item["metadata"]).items()
)
),
)
for item in raw_edges
)
logic = tuple(logic_from_dict(item) for item in raw_logic)
return AdapterSourceProjection(
source_id=str(payload["source_id"]),
fingerprint=str(payload["fingerprint"]),
nodes=nodes,
edges=edges,
logic=logic,
)
except (AttributeError, KeyError, TypeError, ValueError) as error:
raise DocForgeError(
"invalid_cache", "Incremental extraction cache contains invalid adapter data"
) from error
def node_from_dict(payload: dict[str, object]) -> Node:
return Node(
node_id=str(payload["node_id"]),
title=str(payload["title"]),
family=str(payload["family"]),
authority=str(payload["authority"]),
status=str(payload["status"]),
tags=tuple(str(value) for value in cast(list[object], payload["tags"])),
summary=str(payload["summary"]),
content=str(payload["content"]),
source_path=str(payload["source_path"]),
source_anchor=(
str(payload["source_anchor"]) if payload.get("source_anchor") is not None else None
),
content_hash=str(payload["content_hash"]),
)
def logic_from_dict(payload: dict[str, object]) -> LogicProjection:
return LogicProjection(
owner_node_id=str(payload["owner_node_id"]),
source_id=str(payload["source_id"]),
nodes=tuple(
LogicNode(
logic_id=str(item["logic_id"]),
kind=str(item["kind"]),
label=str(item["label"]),
source_anchor=(
str(item["source_anchor"]) if item.get("source_anchor") is not None else None
),
)
for item in cast(list[dict[str, object]], payload["nodes"])
),
edges=tuple(
LogicEdge(
source_id=str(item["source_id"]),
relation=str(item["relation"]),
target_id=str(item["target_id"]),
label=str(item["label"]) if item.get("label") is not None else None,
ordinal=int(cast(int, item["ordinal"])),
)
for item in cast(list[dict[str, object]], payload["edges"])
),
)

View file

@ -0,0 +1,25 @@
"""Repository-owned reference adapters built on the public adapter SDK."""
from .javascript import (
JAVASCRIPT_ADAPTER_VERSION,
JAVASCRIPT_EXTRACTOR_VERSION,
JavaScriptReferenceAdapter,
JavaScriptUnsupportedFact,
)
from .python import (
PYTHON_ADAPTER_VERSION,
PYTHON_EXTRACTOR_VERSION,
PythonReferenceAdapter,
PythonUnsupportedFact,
)
__all__ = [
"JAVASCRIPT_ADAPTER_VERSION",
"JAVASCRIPT_EXTRACTOR_VERSION",
"PYTHON_ADAPTER_VERSION",
"PYTHON_EXTRACTOR_VERSION",
"JavaScriptReferenceAdapter",
"JavaScriptUnsupportedFact",
"PythonReferenceAdapter",
"PythonUnsupportedFact",
]

1349
src/docforge/adapters/cpp.py Normal file

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,948 @@
"""Deterministic stdlib-AST reference adapter for explicitly confined Python roots.
The adapter parses Python source as untrusted data. It never imports or executes
project code. Its intentionally narrow dependency model publishes only imports
that resolve to another module in the same declared source inventory.
"""
from __future__ import annotations
import ast
import hashlib
import io
import os
import stat
import tokenize
from collections.abc import Iterable, Sequence
from dataclasses import dataclass
from pathlib import Path, PurePosixPath
from ..adapter_sdk import (
AdapterAssembly,
AdapterEdge,
AdapterManifest,
AdapterNode,
AdapterProjection,
AdapterSource,
AdapterSourceProjection,
Edge,
Node,
)
from ..errors import DocForgeError
from ..python_logic import PythonLogicOwner, analyze_python_source
PYTHON_ADAPTER_ID = "docforge.reference.python"
PYTHON_ADAPTER_VERSION = "1"
PYTHON_EXTRACTOR_VERSION = "stdlib-ast@1"
PYTHON_IDENTITY_VERSION = "python-reference-id@1"
PYTHON_SUPPORT_SCHEMA_VERSION = 1
_SUPPORTED_FACTS = (
"python_file",
"python_module",
"python_class",
"python_function",
"lexical_containment",
"local_import_dependency",
"function_logic",
)
@dataclass(frozen=True)
class PythonUnsupportedFact:
"""One semantic fact this syntax-only reference adapter does not claim."""
code: str
description: str
def as_dict(self) -> dict[str, str]:
return {"code": self.code, "description": self.description}
_UNSUPPORTED_FACTS = (
PythonUnsupportedFact(
"call_resolution",
"Calls are represented only inside function Logic and are not resolved to symbols.",
),
PythonUnsupportedFact(
"dynamic_import_resolution",
"Imports performed through runtime calls are not dependency evidence.",
),
PythonUnsupportedFact(
"inheritance_resolution",
"Class bases are syntax only and are not resolved to local or external types.",
),
PythonUnsupportedFact(
"runtime_generated_facts",
"Decorators, metaclasses, descriptors, and executed module code are never evaluated.",
),
PythonUnsupportedFact(
"symbol_reference_resolution",
"Imported names, variable references, types, overloads, and re-exports are not resolved.",
),
)
@dataclass(frozen=True)
class _SourceRecord:
source_id: str
source_path: str
module_name: str
fingerprint: str
raw: bytes
text: str
tree: ast.Module | None
dependencies: tuple[str, ...] = ()
@dataclass(frozen=True)
class _Definition:
kind: str
qualified_name: str
parent_node_id: str
node: ast.ClassDef | ast.FunctionDef | ast.AsyncFunctionDef
node_id: str
class PythonReferenceAdapter:
"""Reference Python adapter over explicit, non-overlapping source roots."""
def __init__(
self,
root: Path,
*,
source_roots: Iterable[str | Path],
project_id: str = "python-reference",
title: str = "Python reference project",
max_sources: int = 4_096,
max_source_bytes: int = 1_000_000,
max_logic_nodes_per_function: int = 2_000,
) -> None:
resolved_root = root.resolve(strict=True)
if not resolved_root.is_dir():
raise DocForgeError("invalid_adapter", "Python adapter root must be a directory")
if max_sources < 1 or max_source_bytes < 1 or max_logic_nodes_per_function < 2:
raise ValueError("Python adapter limits must be positive")
self.root = resolved_root
self.project_id = project_id
self.title = title
self.max_sources = max_sources
self.max_source_bytes = max_source_bytes
self.max_logic_nodes_per_function = max_logic_nodes_per_function
self.source_roots = self._normalize_source_roots(source_roots)
def support_report(self) -> dict[str, object]:
"""Return deterministic, machine-readable scope and limitation evidence."""
return {
"schema_version": PYTHON_SUPPORT_SCHEMA_VERSION,
"adapter_id": PYTHON_ADAPTER_ID,
"adapter_version": PYTHON_ADAPTER_VERSION,
"extractor_version": PYTHON_EXTRACTOR_VERSION,
"identity_version": PYTHON_IDENTITY_VERSION,
"frontend": "python-stdlib-ast",
"imports_project_code": False,
"executes_project_code": False,
"source_roots": list(self.source_roots),
"supported_facts": list(_SUPPORTED_FACTS),
"unsupported_facts": [fact.as_dict() for fact in _UNSUPPORTED_FACTS],
}
def unsupported_facts(self) -> tuple[PythonUnsupportedFact, ...]:
"""Return the adapter's fixed unsupported semantic fact inventory."""
return _UNSUPPORTED_FACTS
def load_manifest(self) -> AdapterManifest:
records = self._inventory(parse=False)
module_sources = {record.module_name: record.source_id for record in records}
if len(module_sources) != len(records):
raise DocForgeError(
"ambiguous_python_module",
"Declared Python source roots produce duplicate module names",
)
sources = tuple(
sorted(
(
AdapterSource(
source_id=record.source_id,
source_path=record.source_path,
fingerprint=record.fingerprint,
extractor_version=PYTHON_EXTRACTOR_VERSION,
dependencies=self._local_dependencies(record, module_sources),
)
for record in records
),
key=lambda source: source.source_id,
)
)
source_hash = self._source_hash(sources)
return AdapterManifest(
project_id=self.project_id,
title=self.title,
adapter_id=PYTHON_ADAPTER_ID,
adapter_version=PYTHON_ADAPTER_VERSION,
root=self.root,
revision=source_hash[:12],
source_hash=source_hash,
families=("code",),
allowed_relations=("contains", "depends_on"),
sources=sources,
estimated_nodes=max(1, len(sources) * 8),
)
def extract_source(self, source: AdapterSource) -> AdapterSourceProjection:
record = self._read_source(source.source_path)
if record.source_id != source.source_id or record.fingerprint != source.fingerprint:
raise DocForgeError(
"stale_adapter_source",
"Python source changed after its manifest was captured",
source=source.source_path,
)
return self._extract_record(record, source)
def assemble_projection(
self,
manifest: AdapterManifest,
contributions: tuple[AdapterSourceProjection, ...],
) -> AdapterAssembly:
expected = tuple(source.source_id for source in manifest.sources)
actual = tuple(sorted(contribution.source_id for contribution in contributions))
if actual != expected:
raise DocForgeError(
"invalid_adapter",
"Python assembly contributions do not match the current manifest",
)
nodes = tuple(
sorted(
(node for contribution in contributions for node in contribution.nodes),
key=lambda item: item.node.node_id,
)
)
edges = tuple(
sorted(
(edge for contribution in contributions for edge in contribution.edges),
key=lambda item: (
item.edge.source_id,
item.edge.relation,
item.edge.target_id,
),
)
)
logic = tuple(
sorted(
(projection for contribution in contributions for projection in contribution.logic),
key=lambda projection: projection.owner_node_id,
)
)
return AdapterAssembly(
projection=AdapterProjection(
project_id=manifest.project_id,
title=manifest.title,
adapter_id=manifest.adapter_id,
adapter_version=manifest.adapter_version,
root=manifest.root,
revision=manifest.revision,
source_hash=manifest.source_hash,
nodes=nodes,
edges=edges,
),
logic=logic,
)
def load_assembly(self) -> AdapterAssembly:
"""Load one cache-independent complete graph and Logic assembly."""
manifest = self.load_manifest()
contributions = tuple(self.extract_source(source) for source in manifest.sources)
stable = self.load_manifest()
if stable != manifest:
raise DocForgeError(
"source_changed",
"Python sources changed during complete adapter extraction",
)
return self.assemble_projection(manifest, contributions)
def load_complete_assembly(self) -> AdapterAssembly:
"""Implement the SDK's independent complete-assembly oracle."""
return self.load_assembly()
def load_projection(self) -> AdapterProjection:
"""Preserve the first-class legacy complete-projection contract."""
return self.load_assembly().projection
def _normalize_source_roots(self, source_roots: Iterable[str | Path]) -> tuple[str, ...]:
normalized: list[str] = []
for value in source_roots:
candidate = PurePosixPath(Path(value).as_posix())
if (
candidate.is_absolute()
or not candidate.parts
or ".." in candidate.parts
or str(candidate) in {"", "."}
):
raise DocForgeError(
"path_escape",
"Python source roots must be explicit project-relative directories",
)
relative = candidate.as_posix()
absolute = self.root.joinpath(*candidate.parts)
if absolute.is_symlink():
raise DocForgeError("path_escape", "Python source root cannot be a symbolic link")
try:
resolved = absolute.resolve(strict=True)
except OSError as error:
raise DocForgeError(
"invalid_adapter",
"Python source root does not exist",
source_root=relative,
) from error
if (
resolved != absolute
or not resolved.is_dir()
or not resolved.is_relative_to(self.root)
):
raise DocForgeError(
"path_escape",
"Python source root must be a confined real directory",
source_root=relative,
)
normalized.append(relative)
result = tuple(sorted(set(normalized)))
if not result:
raise DocForgeError("invalid_adapter", "At least one Python source root is required")
if len(result) != len(normalized):
raise DocForgeError("invalid_adapter", "Python source roots must be unique")
paths = [PurePosixPath(value) for value in result]
for index, left in enumerate(paths):
for right in paths[index + 1 :]:
if left in right.parents or right in left.parents:
raise DocForgeError(
"invalid_adapter",
"Python source roots must not overlap",
)
return result
def _inventory(self, *, parse: bool = True) -> tuple[_SourceRecord, ...]:
paths: list[str] = []
for source_root in self.source_roots:
absolute_root = self.root.joinpath(*PurePosixPath(source_root).parts)
if (
absolute_root.is_symlink()
or not absolute_root.is_dir()
or absolute_root.resolve(strict=True) != absolute_root
):
raise DocForgeError(
"path_escape",
"Python source root changed after adapter binding",
source_root=source_root,
)
for directory, directory_names, file_names in os.walk(
absolute_root,
topdown=True,
followlinks=False,
):
current = Path(directory)
for name in tuple(directory_names):
child = current / name
if child.is_symlink():
raise DocForgeError(
"path_escape",
"Python source inventory contains a symbolic-link directory",
source=child.relative_to(self.root).as_posix(),
)
directory_names[:] = sorted(
name for name in directory_names if name != "__pycache__"
)
for name in sorted(file_names):
if not name.endswith(".py"):
continue
child = current / name
relative = child.relative_to(self.root).as_posix()
if child.is_symlink():
raise DocForgeError(
"path_escape",
"Python source inventory contains a symbolic-link file",
source=relative,
)
paths.append(relative)
if len(paths) > self.max_sources:
raise DocForgeError(
"adapter_too_large",
"Python source inventory exceeds the configured limit",
maximum=self.max_sources,
)
records = tuple(self._read_source(path, parse=parse) for path in sorted(paths))
modules = [record.module_name for record in records]
if len(modules) != len(set(modules)):
duplicates = sorted(module for module in set(modules) if modules.count(module) > 1)
raise DocForgeError(
"ambiguous_python_module",
"Declared Python source roots produce duplicate module names",
modules=duplicates,
)
return records
def _read_source(self, relative: str, *, parse: bool = True) -> _SourceRecord:
path = PurePosixPath(relative)
if path.is_absolute() or ".." in path.parts or path.suffix != ".py":
raise DocForgeError("path_escape", "Python source path is unsafe")
source_root = self._source_root_for(path)
candidate = self.root.joinpath(*path.parts)
try:
before = candidate.lstat()
except OSError as error:
raise DocForgeError(
"stale_adapter_source",
"Python source is unavailable",
source=relative,
) from error
if (
stat.S_ISLNK(before.st_mode)
or not stat.S_ISREG(before.st_mode)
or candidate.resolve(strict=True) != candidate
or not candidate.is_relative_to(self.root)
):
raise DocForgeError(
"path_escape",
"Python source must be a confined regular file",
source=relative,
)
if before.st_size > self.max_source_bytes:
raise DocForgeError(
"source_too_large",
"Python source exceeds the configured adapter limit",
source=relative,
maximum=self.max_source_bytes,
)
descriptor = -1
try:
flags = os.O_RDONLY | getattr(os, "O_BINARY", 0)
flags |= getattr(os, "O_NOFOLLOW", 0)
descriptor = os.open(candidate, flags)
opened = os.fstat(descriptor)
if not stat.S_ISREG(opened.st_mode) or (opened.st_dev, opened.st_ino) != (
before.st_dev,
before.st_ino,
):
raise DocForgeError(
"path_escape",
"Python source identity changed before it was opened",
source=relative,
)
with os.fdopen(descriptor, "rb") as handle:
descriptor = -1
raw = handle.read(self.max_source_bytes + 1)
after_open = os.fstat(handle.fileno())
after = candidate.lstat()
except DocForgeError:
raise
except OSError as error:
raise DocForgeError(
"stale_adapter_source",
"Python source changed while being read",
source=relative,
) from error
finally:
if descriptor >= 0:
os.close(descriptor)
identity_before = (
before.st_dev,
before.st_ino,
before.st_size,
before.st_mtime_ns,
)
identity_after = (
after.st_dev,
after.st_ino,
after.st_size,
after.st_mtime_ns,
)
identity_opened = (
after_open.st_dev,
after_open.st_ino,
after_open.st_size,
after_open.st_mtime_ns,
)
if (
len(raw) > self.max_source_bytes
or identity_before != identity_opened
or identity_before != identity_after
or len(raw) != after.st_size
):
raise DocForgeError(
"stale_adapter_source",
"Python source changed while being read",
source=relative,
)
text = self._decode_source(raw, relative)
tree: ast.Module | None = None
if parse:
try:
tree = ast.parse(text, filename=relative, type_comments=True)
except SyntaxError as error:
raise DocForgeError(
"invalid_python_source",
"Python source cannot be parsed by the reference adapter",
source=relative,
line=error.lineno,
) from error
module_name = self._module_name(path, PurePosixPath(source_root))
return _SourceRecord(
source_id=self._source_id(relative),
source_path=relative,
module_name=module_name,
fingerprint=hashlib.sha256(raw).hexdigest(),
raw=raw,
text=text,
tree=tree,
)
def _source_root_for(self, path: PurePosixPath) -> str:
matches = [
source_root
for source_root in self.source_roots
if path == PurePosixPath(source_root) or PurePosixPath(source_root) in path.parents
]
if len(matches) != 1:
raise DocForgeError(
"path_escape",
"Python source is outside the declared source roots",
source=path.as_posix(),
)
return matches[0]
@staticmethod
def _decode_source(raw: bytes, source_path: str) -> str:
try:
encoding, _ = tokenize.detect_encoding(io.BytesIO(raw).readline)
return raw.decode(encoding)
except (LookupError, SyntaxError, UnicodeDecodeError) as error:
raise DocForgeError(
"invalid_python_source",
"Python source encoding is invalid",
source=source_path,
) from error
@staticmethod
def _module_name(path: PurePosixPath, source_root: PurePosixPath) -> str:
local = path.relative_to(source_root)
parts = list(local.with_suffix("").parts)
if parts and parts[-1] == "__init__":
parts.pop()
if not parts:
parts = list(source_root.parts)
return ".".join(parts)
@staticmethod
def _source_id(relative: str) -> str:
digest = hashlib.sha256(
f"{PYTHON_IDENTITY_VERSION}\0source\0{relative}".encode()
).hexdigest()[:24]
return f"python.source.{digest}"
@staticmethod
def _file_node_id(source_id: str) -> str:
return source_id.replace("python.source.", "python.file.", 1)
@staticmethod
def _module_node_id(source_id: str) -> str:
return source_id.replace("python.source.", "python.module.", 1)
@staticmethod
def _symbol_node_id(kind: str, source_id: str, qualified_name: str) -> str:
digest = hashlib.sha256(
(f"{PYTHON_IDENTITY_VERSION}\0{kind}\0{source_id}\0{qualified_name}").encode()
).hexdigest()[:24]
return f"python.{kind}.{digest}"
@staticmethod
def _source_hash(sources: Sequence[AdapterSource]) -> str:
digest = hashlib.sha256()
digest.update(PYTHON_ADAPTER_VERSION.encode())
digest.update(PYTHON_EXTRACTOR_VERSION.encode())
digest.update(PYTHON_IDENTITY_VERSION.encode())
for source in sources:
digest.update(b"\0source\0")
digest.update(source.source_id.encode())
digest.update(b"\0path\0")
digest.update(source.source_path.encode())
digest.update(b"\0fingerprint\0")
digest.update(source.fingerprint.encode())
for dependency in source.dependencies:
digest.update(b"\0dependency\0")
digest.update(dependency.encode())
return digest.hexdigest()
def _local_dependencies(
self,
record: _SourceRecord,
module_sources: dict[str, str],
) -> tuple[str, ...]:
dependencies: set[str] = set()
package = (
record.module_name
if record.source_path.endswith("/__init__.py")
else record.module_name.rpartition(".")[0]
)
for level, module, imported_names in self._lexical_imports(record):
candidates: list[str] = []
if not imported_names and level == 0 and module is not None:
candidates.append(module)
else:
base = self._import_from_base(package, level, module)
if base:
candidates.append(base)
candidates.extend(f"{base}.{name}" for name in imported_names if name != "*")
for module_name in candidates:
target = module_sources.get(module_name)
if target is not None and target != record.source_id:
dependencies.add(target)
return tuple(sorted(dependencies))
@staticmethod
def _lexical_imports(
record: _SourceRecord,
) -> tuple[tuple[int, str | None, tuple[str, ...]], ...]:
"""Inventory import dependencies without constructing a Python AST."""
try:
tokens = tuple(tokenize.generate_tokens(io.StringIO(record.text).readline))
except (IndentationError, tokenize.TokenError) as error:
line = error.args[1][0] if len(error.args) > 1 else None
raise DocForgeError(
"invalid_python_source",
"Python source cannot be tokenized by the reference adapter",
source=record.source_path,
line=line,
) from error
statements: list[list[tokenize.TokenInfo]] = []
current: list[tokenize.TokenInfo] = []
nesting = 0
ignored = {
tokenize.ENCODING,
tokenize.INDENT,
tokenize.DEDENT,
tokenize.NL,
tokenize.COMMENT,
}
for token in tokens:
if token.type in ignored:
continue
if token.type == tokenize.OP:
if token.string in "([{":
nesting += 1
elif token.string in ")]}":
nesting = max(0, nesting - 1)
elif token.string == ";" and nesting == 0:
if current:
statements.append(current)
current = []
continue
if token.type in {tokenize.NEWLINE, tokenize.ENDMARKER} and nesting == 0:
if current:
statements.append(current)
current = []
continue
current.append(token)
imports: list[tuple[int, str | None, tuple[str, ...]]] = []
for statement in statements:
for position, token in enumerate(statement):
if token.type != tokenize.NAME or token.string not in {"import", "from"}:
continue
if token.string == "import":
names = PythonReferenceAdapter._imported_names(statement[position + 1 :])
imports.extend((0, name, ()) for name in names)
break
import_position = next(
(
index
for index in range(position + 1, len(statement))
if statement[index].type == tokenize.NAME
and statement[index].string == "import"
),
None,
)
if import_position is None:
break
prefix = statement[position + 1 : import_position]
level = 0
for item in prefix:
if item.type == tokenize.NAME:
break
if item.type == tokenize.OP and set(item.string) == {"."}:
level += len(item.string)
module_parts = [item.string for item in prefix if item.type == tokenize.NAME]
module = ".".join(module_parts) or None
names = PythonReferenceAdapter._imported_names(statement[import_position + 1 :])
imports.append((level, module, names))
break
return tuple(imports)
@staticmethod
def _imported_names(tokens: Sequence[tokenize.TokenInfo]) -> tuple[str, ...]:
names: list[str] = []
current: list[str] = []
skip_alias = False
for token in tokens:
if token.type == tokenize.NAME and token.string == "as":
skip_alias = True
continue
if token.type == tokenize.OP and token.string == ",":
if current:
names.append(".".join(current))
current = []
skip_alias = False
continue
if token.type == tokenize.OP and token.string == "*":
if not skip_alias:
current.append("*")
continue
if token.type == tokenize.NAME and not skip_alias:
current.append(token.string)
if current:
names.append(".".join(current))
return tuple(name for name in names if name)
@staticmethod
def _import_from_base(package: str, level: int, module: str | None) -> str:
if level == 0:
return module or ""
package_parts = package.split(".") if package else []
keep = len(package_parts) - (level - 1)
if keep < 0:
return ""
prefix = package_parts[:keep]
if module:
prefix.extend(module.split("."))
return ".".join(prefix)
def _extract_record(
self,
record: _SourceRecord,
source: AdapterSource,
) -> AdapterSourceProjection:
tree = record.tree
if tree is None:
raise DocForgeError(
"invalid_adapter",
"Python extraction requires a parsed source record",
source=record.source_path,
)
file_id = self._file_node_id(source.source_id)
module_id = self._module_node_id(source.source_id)
nodes: list[AdapterNode] = [
self._node(
node_id=file_id,
title=record.source_path,
kind="file",
qualified_name=record.source_path,
content=record.text,
source_path=record.source_path,
anchor="L1",
),
self._node(
node_id=module_id,
title=record.module_name,
kind="module",
qualified_name=record.module_name,
content=(
ast.get_docstring(tree, clean=False) or f"Python module {record.module_name}."
),
source_path=record.source_path,
anchor="L1",
),
]
edges: list[AdapterEdge] = [self._edge(file_id, "contains", module_id, "syntax")]
definitions = self._definitions(record, module_id)
owners: list[PythonLogicOwner] = []
for definition in definitions:
content = ast.get_source_segment(record.text, definition.node)
if content is None or not content.strip():
content = f"Python {definition.kind} {definition.qualified_name}."
anchor = f"L{definition.node.lineno}"
nodes.append(
self._node(
node_id=definition.node_id,
title=definition.qualified_name,
kind=definition.kind,
qualified_name=f"{record.module_name}.{definition.qualified_name}",
content=content,
source_path=record.source_path,
anchor=anchor,
asynchronous=isinstance(definition.node, ast.AsyncFunctionDef),
)
)
edges.append(
self._edge(
definition.parent_node_id,
"contains",
definition.node_id,
"syntax",
anchor=anchor,
)
)
if definition.kind == "function":
owners.append(
PythonLogicOwner(
owner_node_id=definition.node_id,
qualified_name=definition.qualified_name,
line=definition.node.lineno,
)
)
for dependency in source.dependencies:
edges.append(
self._edge(
module_id,
"depends_on",
self._module_node_id(dependency),
"local_import",
)
)
logic = analyze_python_source(
record.text,
source_id=source.source_id,
owners=owners,
filename=record.source_path,
max_nodes_per_function=self.max_logic_nodes_per_function,
)
return AdapterSourceProjection(
source_id=source.source_id,
fingerprint=source.fingerprint,
nodes=tuple(sorted(nodes, key=lambda item: item.node.node_id)),
edges=tuple(
sorted(
edges,
key=lambda item: (
item.edge.source_id,
item.edge.relation,
item.edge.target_id,
),
)
),
logic=logic,
)
def _definitions(
self,
record: _SourceRecord,
module_node_id: str,
) -> tuple[_Definition, ...]:
tree = record.tree
if tree is None:
raise DocForgeError(
"invalid_adapter",
"Python definition extraction requires a parsed source record",
source=record.source_path,
)
definitions: list[_Definition] = []
qualified_names: set[str] = set()
adapter = self
class Collector(ast.NodeVisitor):
def __init__(self) -> None:
self.scope_names: list[str] = []
self.scope_node_ids: list[str] = [module_node_id]
def _definition(
self,
kind: str,
node: ast.ClassDef | ast.FunctionDef | ast.AsyncFunctionDef,
) -> None:
qualified_name = ".".join((*self.scope_names, node.name))
if qualified_name in qualified_names:
raise DocForgeError(
"ambiguous_python_symbol",
"Python source repeats a class or function identity",
source=record.source_path,
qualified_name=qualified_name,
)
qualified_names.add(qualified_name)
node_id = adapter._symbol_node_id(
kind,
record.source_id,
qualified_name,
)
definitions.append(
_Definition(
kind=kind,
qualified_name=qualified_name,
parent_node_id=self.scope_node_ids[-1],
node=node,
node_id=node_id,
)
)
self.scope_names.append(node.name)
self.scope_node_ids.append(node_id)
self.generic_visit(node)
self.scope_node_ids.pop()
self.scope_names.pop()
def visit_ClassDef(self, node: ast.ClassDef) -> None:
self._definition("class", node)
def visit_FunctionDef(self, node: ast.FunctionDef) -> None:
self._definition("function", node)
def visit_AsyncFunctionDef(self, node: ast.AsyncFunctionDef) -> None:
self._definition("function", node)
Collector().visit(tree)
return tuple(sorted(definitions, key=lambda item: item.node_id))
@staticmethod
def _node(
*,
node_id: str,
title: str,
kind: str,
qualified_name: str,
content: str,
source_path: str,
anchor: str,
asynchronous: bool = False,
) -> AdapterNode:
normalized = content.strip() or f"Python {kind} {qualified_name}."
tags = tuple(sorted({"python", kind, *(("async",) if asynchronous else ())}))
return AdapterNode(
node=Node(
node_id=node_id,
title=title,
family="code",
authority="derived",
status="active",
tags=tags,
summary=f"Python {kind} fact for {qualified_name}.",
content=normalized,
source_path=source_path,
source_anchor=anchor,
content_hash=hashlib.sha256(normalized.encode()).hexdigest(),
),
metadata=(
("extractor", PYTHON_EXTRACTOR_VERSION),
("identity", PYTHON_IDENTITY_VERSION),
("kind", kind),
("qualified_name", qualified_name),
),
)
@staticmethod
def _edge(
source_id: str,
relation: str,
target_id: str,
evidence: str,
*,
anchor: str | None = None,
) -> AdapterEdge:
metadata = [("evidence", evidence)]
if anchor is not None:
metadata.append(("source_anchor", anchor))
return AdapterEdge(
edge=Edge(source_id, relation, target_id),
metadata=tuple(metadata),
)

File diff suppressed because it is too large Load diff

View file

@ -26,7 +26,7 @@ header {
}
header h1 { margin: 0; font-size: 17px; }
.view-switch {
position: relative; display: grid; grid-template-columns: repeat(3, 58px);
position: relative; display: grid; grid-template-columns: repeat(4, 58px);
flex: 0 0 auto; padding: 3px; border: 1px solid var(--line); border-radius: 9px;
background: #08131f; isolation: isolate;
}
@ -38,10 +38,19 @@ header h1 { margin: 0; font-size: 17px; }
}
.view-switch[data-mode="flow"]::before { transform: translateX(58px); }
.view-switch[data-mode="web"]::before { transform: translateX(116px); }
.view-switch[data-mode="logic"]::before { transform: translateX(174px); }
.filter-presets { display: flex; flex-wrap: wrap; gap: 5px; }
.filter-presets button {
border: 1px solid var(--line); border-radius: 999px; padding: 4px 8px;
background: #0b1724; color: var(--muted); font-size: 10px; font-weight: 700;
}
.view-switch button {
min-height: 30px; border: 0; border-radius: 6px; padding: 4px 8px;
background: transparent; color: var(--muted); font-size: 12px; font-weight: 700;
}
.filter-presets button:hover, .filter-presets button:focus-visible {
border-color: var(--accent); color: var(--text); outline: none;
}
.view-switch button[aria-pressed="true"] { color: var(--text); }
.view-switch button:focus-visible { outline: 2px solid var(--accent); outline-offset: 1px; }
.stats { display: flex; flex: 0 0 auto; gap: 14px; color: var(--muted); }
@ -70,6 +79,9 @@ aside { min-height: 0; overflow: hidden; padding: 16px; background: var(--panel)
.panel-resizer.resizing::after { background: var(--accent); }
.panel-resizer:focus-visible { outline: 1px solid var(--accent); outline-offset: -1px; }
form { display: grid; flex: 0 0 auto; gap: 8px; }
form > label, .filter-grid label {
display: grid; gap: 6px; color: var(--muted); font-size: 11px; font-weight: 700;
}
input, select {
width: 100%; border: 1px solid var(--line); border-radius: 8px;
padding: 9px 10px; background: var(--panel-2); color: var(--text);
@ -78,6 +90,9 @@ input:focus-visible, select:focus-visible {
border-color: var(--accent); outline: 2px solid var(--accent); outline-offset: 1px;
}
.search-row { display: grid; grid-template-columns: 1fr auto; gap: 8px; }
.filter-grid {
display: grid; grid-template-columns: minmax(0, 1fr) minmax(0, 1fr); gap: 8px;
}
.button {
border: 1px solid #277fa0; border-radius: 8px; padding: 8px 12px;
background: #12384a; color: var(--text);
@ -161,8 +176,16 @@ input:focus-visible, select:focus-visible {
padding: 9px; background: var(--panel-2); color: var(--text);
}
.result:hover, .result:focus-visible { border-color: var(--accent); outline: none; }
.result strong, .result span { display: block; overflow: hidden; text-overflow: ellipsis; }
.result span { color: var(--muted); font-size: 12px; white-space: nowrap; }
.result strong, .result span { display: block; overflow-wrap: anywhere; }
.result > span { color: var(--muted); font-size: 11px; }
.result-badges {
display: flex !important; flex-wrap: wrap; gap: 4px; margin: 5px 0;
}
.result-badges small {
border: 1px solid #31526d; border-radius: 999px; padding: 1px 6px;
background: #0a1724; color: #b7c9da; font-size: 9px; font-weight: 750;
letter-spacing: .04em; text-transform: uppercase;
}
.canvas { position: relative; min-width: 0; min-height: 0; overflow: hidden; }
svg { width: 100%; height: 100%; background:
radial-gradient(circle at 52% 46%, rgba(26, 65, 89, .58) 0, rgba(10, 28, 45, .46) 34%,
@ -239,13 +262,24 @@ svg { width: 100%; height: 100%; background:
.relationship-key-empty { margin: 2px 0; color: var(--muted); font-size: 11px; }
.relationship-edge {
fill: none; stroke-opacity: .74; stroke-width: 1.7;
vector-effect: non-scaling-stroke;
vector-effect: non-scaling-stroke; transition: opacity .16s, stroke-width .16s, filter .16s;
}
.edge-label {
font-size: 9px; font-weight: 700; letter-spacing: .015em; pointer-events: none;
paint-order: stroke; stroke: #07101a; stroke-width: 4px; stroke-linejoin: round;
transition: opacity .16s;
}
.node { cursor: pointer; }
.relationship-edge.trace-connected {
stroke-opacity: 1; stroke-width: 3;
filter: drop-shadow(0 0 7px currentcolor);
}
.relationship-edge.trace-muted, .edge-label.trace-muted { opacity: .13; }
.edge-label.trace-connected { opacity: 1; font-size: 10px; }
.node { cursor: pointer; transition: opacity .16s, filter .16s; }
.node.trace-connected:not(.selected) {
filter: drop-shadow(0 0 8px rgba(165, 243, 252, .28));
}
.node.trace-muted { opacity: .24; }
.node:focus { outline: none; }
.node .node-surface {
stroke-width: 1.35; vector-effect: non-scaling-stroke;

View file

@ -15,6 +15,7 @@
<button id="view-nodes" type="button" aria-pressed="true">Nodes</button>
<button id="view-flow" type="button" aria-pressed="false">Flow</button>
<button id="view-web" type="button" aria-pressed="false">Web</button>
<button id="view-logic" type="button" aria-pressed="false">Logic</button>
</div>
<h1 id="project-title">DocForge graph</h1>
<div class="stats">
@ -34,6 +35,34 @@
</div>
<label for="family">Family</label>
<select id="family" name="family"><option value="">All families</option></select>
<div class="filter-grid">
<label>
Node type
<select id="kind" name="kind">
<option value="">All node types</option>
<option value="callable">Callable functions &amp; methods</option>
</select>
</label>
<label>
Language
<select id="language" name="language">
<option value="">All languages</option>
</select>
</label>
</div>
<label for="capability">Capability</label>
<select id="capability" name="capability">
<option value="">All capabilities</option>
<option value="logic">Logic available</option>
<option value="source">Source available</option>
</select>
<div class="filter-presets" role="group" aria-label="Quick node filters">
<button type="button" data-preset="logic">Logic-ready</button>
<button type="button" data-preset="python">Python callables</button>
<button type="button" data-preset="tests">Tests</button>
<button type="button" data-preset="routes">Routes</button>
<button type="button" data-preset="docs">Docs</button>
</div>
</form>
<div class="results-context">
<strong id="results-label">All nodes</strong>
@ -66,7 +95,7 @@
aria-label="Visible relationship color and symbol key"></ul>
</details>
<svg id="graph" viewBox="-600 -410 1200 820"
role="img" aria-label="Node neighborhood"></svg>
role="group" aria-label="Interactive node neighborhood"></svg>
<div class="empty" id="empty">Search for a node to inspect its neighborhood.</div>
<div class="connection-state" id="connection-state" role="alert" hidden>
<strong>Visualization disconnected</strong>

View file

@ -4,6 +4,7 @@ const state = {
overview: null,
graph: null,
root: null,
focusNode: null,
mode: "nodes",
depth: 1,
searchLimit: 1,
@ -22,6 +23,35 @@ const state = {
dialogDrag: null,
leaseTimer: null,
};
const nodeKindOptions = Object.freeze([
["function", "Function"],
["method", "Method"],
["class", "Class"],
["module", "Module"],
["package", "Package"],
["route", "Route"],
["command", "Command"],
["service", "Service"],
["plugin", "Plugin"],
["test", "Test"],
["table", "Table"],
["view", "View"],
["document", "Document"],
["manual", "Manual"],
["section", "Section"],
]);
const languageOptions = Object.freeze([
["python", "Python"],
["javascript", "JavaScript"],
["typescript", "TypeScript"],
["cpp", "C++"],
["c", "C"],
["csharp", "C#"],
["rust", "Rust"],
["java", "Java"],
["go", "Go"],
["sql", "SQL"],
]);
const relationStyles = Object.freeze({
contains: {
family: "Structure", color: "#60a5fa", dash: "", marker: "diamond-arrow",
@ -95,6 +125,50 @@ const relationStyles = Object.freeze({
family: "Context", color: "#94a3b8", dash: "5 5", marker: "open-arrow",
flow: null,
},
next: {
family: "Logic", color: "#8da2b8", dash: "", marker: "arrow",
flow: "forward",
},
when_true: {
family: "Logic", color: "#4ade80", dash: "", marker: "arrow",
flow: "forward",
},
when_false: {
family: "Logic", color: "#fb7185", dash: "5 3", marker: "arrow",
flow: "forward",
},
case: {
family: "Logic", color: "#c084fc", dash: "7 3", marker: "arrow",
flow: "forward",
},
loop: {
family: "Logic", color: "#2dd4bf", dash: "4 3", marker: "double-arrow",
flow: "forward",
},
exception: {
family: "Logic", color: "#f97316", dash: "3 3", marker: "open-arrow",
flow: "forward",
},
return: {
family: "Logic", color: "#38bdf8", dash: "", marker: "square-arrow",
flow: "forward",
},
raise: {
family: "Logic", color: "#f43f5e", dash: "", marker: "square-arrow",
flow: "forward",
},
break: {
family: "Logic", color: "#fbbf24", dash: "6 3", marker: "open-arrow",
flow: "forward",
},
continue: {
family: "Logic", color: "#22d3ee", dash: "6 3", marker: "open-arrow",
flow: "forward",
},
omitted: {
family: "Logic", color: "#64748b", dash: "2 5", marker: "open-arrow",
flow: "forward",
},
});
const contributionStyles = Object.freeze({
focus: {
@ -126,10 +200,33 @@ const contributionStyles = Object.freeze({
related: {
label: "Related", section: "Other connections", color: "#fb923c", fill: "#3b2719",
},
"logic-entry": {
label: "Entry", section: "Function boundary", color: "#67e8f9", fill: "#103745",
},
"logic-condition": {
label: "Decision", section: "Conditions & cases", color: "#facc15", fill: "#3b3112",
},
"logic-action": {
label: "Action", section: "Actions & calls", color: "#34d399", fill: "#15372e",
},
"logic-comment": {
label: "Comment", section: "Source commentary", color: "#7dd3fc", fill: "#173447",
},
"logic-control": {
label: "Control", section: "Loops & exception handling", color: "#c084fc", fill: "#302044",
},
"logic-convergence": {
label: "Convergence", section: "Control paths reunite", color: "#94a3b8", fill: "#252d39",
},
"logic-terminal": {
label: "Terminal", section: "Returns, raises & exits", color: "#fb7185", fill: "#41202a",
},
});
const contributionOrder = Object.freeze([
"focus", "composition", "behavior", "dependency", "execution",
"data", "evidence", "context", "related",
"logic-entry", "logic-condition", "logic-action", "logic-control",
"logic-comment", "logic-convergence", "logic-terminal",
]);
const compositionRelations = new Set(["contains", "defines", "defined_in"]);
const behaviorRelations = new Set(["inherits", "implemented_by"]);
@ -193,12 +290,17 @@ function humanize(value) {
}
function nodeDisplayName(node) {
const title = escapeText(node.title).trim() || escapeText(node.node_id);
if (Array.isArray(node.tags) && node.tags.includes("logic")) return title;
if (!title || /\s/.test(title)) return title;
const parts = title.split(/::|[./]/).filter(Boolean);
return parts.at(-1) || title;
}
function nodeKindLabel(node) {
const tags = new Set(Array.isArray(node.tags) ? node.tags.map(String) : []);
if (tags.has("logic")) {
const kind = [...tags].find((tag) => tag !== "logic");
return kind ? humanize(kind) : "Logic";
}
const kinds = [
"method", "function", "class", "module", "package", "property", "field",
"route", "command", "service", "plugin", "table", "column", "view",
@ -416,8 +518,31 @@ function selectNode(nodeId) {
group.classList.toggle("selected", selected);
group.setAttribute("aria-pressed", String(selected));
}
applyTraceHighlight(nodeId);
return true;
}
function applyTraceHighlight(nodeId) {
const graph = $("graph");
const connected = new Set([nodeId]);
for (const edge of graph.querySelectorAll(".relationship-edge")) {
const direct = edge.dataset.sourceId === nodeId || edge.dataset.targetId === nodeId;
edge.classList.toggle("trace-connected", direct);
edge.classList.toggle("trace-muted", !direct);
if (direct) {
connected.add(edge.dataset.sourceId);
connected.add(edge.dataset.targetId);
}
}
for (const label of graph.querySelectorAll(".edge-label")) {
const direct = label.dataset.sourceId === nodeId || label.dataset.targetId === nodeId;
label.classList.toggle("trace-connected", direct);
label.classList.toggle("trace-muted", !direct);
}
for (const group of graph.querySelectorAll(".node")) {
group.classList.toggle("trace-connected", connected.has(group.dataset.nodeId));
group.classList.toggle("trace-muted", !connected.has(group.dataset.nodeId));
}
}
function centerSelectedNode() {
const point = state.positions.get(state.selectedNode);
if (!point) return false;
@ -458,6 +583,7 @@ function restoreGraphStatus() {
nodes: "neighborhood",
flow: "semantic flow",
web: "convergence web",
logic: "control flow",
}[state.mode];
const pruned = state.prunedCount ? ` · ${state.prunedCount} hidden or isolated` : "";
setStatus(`${state.visibleNodeCount} nodes · ${state.visibleEdgeCount} edges in ${scope}${pruned}`);
@ -491,6 +617,39 @@ function renderOverview(data) {
option.textContent = `${item.value} (${item.count})`;
family.append(option);
}
const tagCounts = new Map((data.tags || []).map(
(item) => [String(item.value), Number(item.count)],
));
const kind = $("kind");
const callableCount = ["function", "method", "nested-function"]
.reduce((total, tag) => total + (tagCounts.get(tag) || 0), 0);
kind.options[1].textContent = `Callable functions & methods (${callableCount})`;
for (const [value, label] of nodeKindOptions) {
const count = tagCounts.get(value) || 0;
if (!count) continue;
const option = document.createElement("option");
option.value = value;
option.textContent = `${label} (${count})`;
kind.append(option);
}
const language = $("language");
for (const [value, label] of languageOptions) {
const count = tagCounts.get(value) || 0;
if (!count) continue;
const option = document.createElement("option");
option.value = value;
option.textContent = `${label} (${count})`;
language.append(option);
}
const capabilityCounts = new Map((data.capabilities || []).map(
(item) => [String(item.value), Number(item.count)],
));
for (const option of $("capability").options) {
const count = capabilityCounts.get(option.value);
if (option.value && count !== undefined) {
option.textContent = `${option.textContent} (${count})`;
}
}
}
function renderResults(items) {
const results = $("results");
@ -507,12 +666,26 @@ function renderResults(items) {
button.type = "button";
button.className = "result";
const title = document.createElement("strong");
title.textContent = item.title;
title.textContent = nodeDisplayName(item);
const badges = document.createElement("span");
badges.className = "result-badges";
const tags = new Set(item.tags || []);
const kind = nodeKindLabel(item);
const language = languageOptions.find(([value]) => tags.has(value))?.[1];
for (const label of [kind, language]) {
if (!label) continue;
const badge = document.createElement("small");
badge.textContent = label;
badges.append(badge);
}
const id = document.createElement("span");
id.textContent = item.node_id;
id.textContent = item.source_anchor
? `${item.source_path} · ${item.source_anchor}`
: item.source_path;
const family = document.createElement("span");
family.textContent = `${item.family} · ${item.source_path}`;
button.append(title, id, family);
family.textContent = item.family;
button.title = `${item.title}\n${item.node_id}`;
button.append(title, badges, id, family);
button.addEventListener("click", () => loadNode(item.node_id));
results.append(button);
}
@ -623,6 +796,40 @@ function buildWebGraph(data) {
]));
return {...data, nodes, edges, topology};
}
function buildLogicGraph(data) {
const nodeIds = new Set(data.nodes.map((node) => node.node_id));
const edges = data.edges.filter(
(edge) => nodeIds.has(edge.source_id) && nodeIds.has(edge.target_id),
);
const hops = new Map([[data.root, 0]]);
let frontier = [data.root];
while (frontier.length) {
const next = [];
for (const sourceId of frontier) {
for (const edge of edges) {
if (edge.source_id !== sourceId || hops.has(edge.target_id)) continue;
hops.set(edge.target_id, hops.get(sourceId) + 1);
next.push(edge.target_id);
}
}
frontier = next;
}
const nodes = data.nodes.filter((node) => hops.has(node.node_id));
return {
...data,
nodes,
edges: edges.filter(
(edge) => hops.has(edge.source_id) && hops.has(edge.target_id),
),
topology: new Map(nodes.map((node) => [
node.node_id,
{
hop: hops.get(node.node_id) ?? 0,
role: node.node_id === data.root ? "primary" : "child",
},
])),
};
}
function pruneConvergenceGraph(data, hiddenNodes) {
const candidates = new Set(
data.nodes
@ -656,7 +863,93 @@ function pruneConvergenceGraph(data, hiddenNodes) {
prunedCount: data.nodes.length - reachesFocus.size,
};
}
function pruneLogicGraph(data, hiddenNodes) {
const visible = new Set(
data.nodes
.filter((node) => node.node_id === data.root || !hiddenNodes.has(node.node_id))
.map((node) => node.node_id),
);
const outgoing = new Map(data.nodes.map((node) => [node.node_id, []]));
for (const edge of data.edges) outgoing.get(edge.source_id)?.push(edge);
const edges = [];
const keys = new Set();
const append = (edge) => {
const key = `${edge.source_id}\u0000${edge.relation}\u0000${edge.target_id}`;
if (keys.has(key)) return;
keys.add(key);
edges.push(edge);
};
for (const sourceId of visible) {
const stack = [...(outgoing.get(sourceId) || [])].map(
(edge) => ({edge, omitted: false, seen: new Set([sourceId])}),
);
while (stack.length) {
const current = stack.pop();
const targetId = current.edge.target_id;
if (current.seen.has(targetId)) continue;
const seen = new Set(current.seen);
seen.add(targetId);
if (visible.has(targetId)) {
append(current.omitted
? {
source_id: sourceId,
relation: "omitted",
target_id: targetId,
label: "HIDDEN PATH",
reversed: false,
}
: current.edge);
continue;
}
for (const nextEdge of outgoing.get(targetId) || []) {
stack.push({edge: nextEdge, omitted: true, seen});
}
}
}
const reachable = new Set([data.root]);
let frontier = [data.root];
while (frontier.length) {
const next = [];
for (const sourceId of frontier) {
for (const edge of edges) {
if (edge.source_id !== sourceId || reachable.has(edge.target_id)) continue;
reachable.add(edge.target_id);
next.push(edge.target_id);
}
}
frontier = next;
}
const nodes = data.nodes.filter((node) => reachable.has(node.node_id));
return {
...data,
nodes,
edges: edges.filter(
(edge) => reachable.has(edge.source_id) && reachable.has(edge.target_id),
),
topology: buildLogicGraph({
...data,
nodes,
edges,
}).topology,
prunedCount: data.nodes.length - nodes.length,
};
}
function nodeContributionCategory(nodeId, data, topology) {
const node = data.nodes.find((candidate) => candidate.node_id === nodeId);
if (Array.isArray(node?.tags) && node.tags.includes("logic")) {
const kind = escapeText(node.logic_kind
|| node.tags.find((tag) => tag !== "logic")).toLowerCase();
if (kind === "entry") return "logic-entry";
if (["condition", "case"].includes(kind)) return "logic-condition";
if (["action", "call"].includes(kind)) return "logic-action";
if (kind === "comment") return "logic-comment";
if (["loop", "try", "except", "finally", "break", "continue"].includes(kind)) {
return "logic-control";
}
if (["merge", "convergence"].includes(kind)) return "logic-convergence";
if (["return", "raise", "exit"].includes(kind)) return "logic-terminal";
return "logic-action";
}
if (nodeId === data.root) return "focus";
const nodeHop = topology.get(nodeId)?.hop ?? Number.POSITIVE_INFINITY;
const candidates = [];
@ -756,6 +1049,68 @@ function layoutFlow(nodes, rootId, topology, sizes = nodeSizeMap(nodes, rootId))
}
return positions;
}
function layoutLogic(nodes, rootId, topology, edges, sizes = nodeSizeMap(nodes, rootId)) {
const layers = new Map();
for (const node of nodes) {
const hop = topology.get(node.node_id).hop;
if (!layers.has(hop)) layers.set(hop, []);
layers.get(hop).push(node);
}
const positions = new Map();
const layerWidths = new Map([...layers].map(([hop, layer]) => [
hop,
Math.max(...layer.map((node) => sizes.get(node.node_id).width)),
]));
const layerX = new Map([[0, 0]]);
for (const hop of [...layers.keys()].sort((a, b) => a - b).filter((value) => value > 0)) {
const previous = layerX.get(hop - 1) || 0;
layerX.set(
hop,
previous + (layerWidths.get(hop - 1) || 188) / 2
+ (layerWidths.get(hop) || 188) / 2 + 190,
);
}
const relationRank = new Map([
["when_true", 0], ["case", 1], ["next", 2], ["when_false", 3],
["exception", 4], ["loop", 5],
]);
for (const [hop, layer] of [...layers.entries()].sort((a, b) => a[0] - b[0])) {
layer.sort((first, second) => {
const firstIncoming = edges.filter((edge) => edge.target_id === first.node_id);
const secondIncoming = edges.filter((edge) => edge.target_id === second.node_id);
const parentY = (incoming) => {
const points = incoming
.map((edge) => positions.get(edge.source_id)?.y)
.filter((value) => value !== undefined);
return points.length
? points.reduce((total, value) => total + value, 0) / points.length
: 0;
};
const relation = (incoming) => Math.min(
...incoming.map((edge) => relationRank.get(edge.relation) ?? 20),
20,
);
return parentY(firstIncoming) - parentY(secondIncoming)
|| relation(firstIncoming) - relation(secondIncoming)
|| first.node_id.localeCompare(second.node_id);
});
const verticalGap = 96;
const layerHeight = layer.reduce(
(total, node) => total + sizes.get(node.node_id).height + verticalGap,
-verticalGap,
);
let cursor = -layerHeight / 2;
for (const node of layer) {
const size = sizes.get(node.node_id);
positions.set(node.node_id, {
x: layerX.get(hop) || 0,
y: cursor + size.height / 2,
});
cursor += size.height + verticalGap;
}
}
return positions;
}
function darken(hex, amount) {
const value = Number.parseInt(hex.slice(1), 16);
const factor = 1 - Math.min(.5, Math.max(0, amount));
@ -798,6 +1153,7 @@ function renderNeighborhood(data, topology, categories) {
nodes: "Neighborhood",
flow: "Semantic flow",
web: "Convergence web",
logic: "Control flow",
}[state.mode];
renderNodeLegend(categories);
const container = $("neighborhood-sections");
@ -825,7 +1181,9 @@ function renderNeighborhood(data, topology, categories) {
button.type = "button";
button.className = "node-list-item";
button.style.setProperty("--item-color", palette.stroke);
button.title = `Focus ${node.title}`;
button.title = state.mode === "logic"
? `Inspect ${node.title}`
: `Focus ${node.title}`;
const swatch = document.createElement("i");
swatch.className = "node-swatch";
const copy = document.createElement("span");
@ -837,7 +1195,14 @@ function renderNeighborhood(data, topology, categories) {
meta.textContent = `${nodeKindLabel(node)} · ${style.label} · ${hopLabel}`;
copy.append(title, meta);
button.append(swatch, copy);
button.addEventListener("click", () => loadNode(node.node_id));
button.addEventListener("click", (event) => {
if (state.mode === "logic") {
selectNode(node.node_id);
showNodeCard(node.node_id, event);
} else {
loadNode(node.node_id);
}
});
list.append(button);
}
container.append(heading, list);
@ -873,6 +1238,30 @@ function edgeEndpoints(source, target, sourceSize, targetSize) {
y2: target.y - unitY * targetOffset,
};
}
function logicEdgeGeometry(points, lane) {
const {x1, y1, x2, y2} = points;
const deltaX = x2 - x1;
if (deltaX > 40) {
const bend = Math.max(70, deltaX * .42);
const controlY = lane * 14;
return {
path: `M ${x1} ${y1} C ${x1 + bend} ${y1 + controlY}, `
+ `${x2 - bend} ${y2 + controlY}, ${x2} ${y2}`,
label: {
x: (x1 + x2) / 2,
y: (y1 + y2) / 2 + controlY * .75 - 8,
},
};
}
const direction = lane % 2 === 0 ? -1 : 1;
const archY = Math.min(y1, y2) + direction * (130 + Math.abs(lane) * 22);
const reach = Math.max(90, Math.abs(deltaX) * .32);
return {
path: `M ${x1} ${y1} C ${x1 + reach} ${archY}, `
+ `${x2 - reach} ${archY}, ${x2} ${y2}`,
label: {x: (x1 + x2) / 2, y: archY - 8},
};
}
function topologyRoleLabel(category) {
const style = contributionStyles[category] || contributionStyles.related;
return category === "focus" ? `${state.mode} focus` : `${style.label} contributor`;
@ -906,7 +1295,9 @@ function renderGraph(data, preserveSelection = false) {
: data.root;
const completeView = state.mode === "flow"
? buildFlowGraph(data)
: state.mode === "web" ? buildWebGraph(data) : data;
: state.mode === "web"
? buildWebGraph(data)
: state.mode === "logic" ? buildLogicGraph(data) : data;
let view;
if (state.mode === "nodes") {
const visibleIds = new Set(
@ -923,6 +1314,8 @@ function renderGraph(data, preserveSelection = false) {
),
prunedCount: completeView.nodes.length - visibleIds.size,
};
} else if (state.mode === "logic") {
view = pruneLogicGraph(completeView, state.hiddenNodes);
} else {
view = pruneConvergenceGraph(completeView, state.hiddenNodes);
}
@ -945,9 +1338,11 @@ function renderGraph(data, preserveSelection = false) {
const topology = view.topology || analyzeTopology(view);
const categories = nodeCategoryMap(view, topology);
const sizes = nodeSizeMap(view.nodes, view.root);
const positions = state.mode !== "nodes"
? layoutFlow(view.nodes, view.root, topology, sizes)
: layoutNodes(view.nodes, view.root, topology, sizes);
const positions = state.mode === "logic"
? layoutLogic(view.nodes, view.root, topology, view.edges, sizes)
: state.mode !== "nodes"
? layoutFlow(view.nodes, view.root, topology, sizes)
: layoutNodes(view.nodes, view.root, topology, sizes);
state.positions = positions;
state.homeViewport = viewportForPositions(positions, sizes);
resetViewport();
@ -959,6 +1354,11 @@ function renderGraph(data, preserveSelection = false) {
}
const edgeLayer = svgElement("g");
const nodeLayer = svgElement("g");
const outgoing = new Map();
for (const edge of view.edges) {
if (!outgoing.has(edge.source_id)) outgoing.set(edge.source_id, []);
outgoing.get(edge.source_id).push(edge);
}
for (const edge of view.edges) {
const source = positions.get(edge.source_id);
const target = positions.get(edge.target_id);
@ -970,23 +1370,38 @@ function renderGraph(data, preserveSelection = false) {
sizes.get(edge.source_id),
sizes.get(edge.target_id),
);
const line = svgElement("line", {
...points,
const siblings = outgoing.get(edge.source_id);
const lane = siblings.indexOf(edge) - (siblings.length - 1) / 2;
const geometry = state.mode === "logic"
? logicEdgeGeometry(points, lane)
: {
path: `M ${points.x1} ${points.y1} L ${points.x2} ${points.y2}`,
label: {
x: (points.x1 + points.x2) / 2,
y: (points.y1 + points.y2) / 2 - 5,
},
};
const line = svgElement("path", {
d: geometry.path,
class: "relationship-edge",
stroke: style.color,
"marker-end": `url(#${relationMarkerId(edge.relation)})`,
"data-relation": edge.relation,
"data-source-id": edge.source_id,
"data-target-id": edge.target_id,
});
if (style.dash) line.setAttribute("stroke-dasharray", style.dash);
edgeLayer.append(line);
const label = svgElement("text", {
x: (points.x1 + points.x2) / 2,
y: (points.y1 + points.y2) / 2 - 5,
x: geometry.label.x,
y: geometry.label.y,
class: "edge-label",
fill: style.color,
"text-anchor": "middle",
"data-source-id": edge.source_id,
"data-target-id": edge.target_id,
});
label.textContent = relationLabel(edge.relation, edge.reversed);
label.textContent = edge.label || relationLabel(edge.relation, edge.reversed);
edgeLayer.append(label);
}
for (const node of view.nodes) {
@ -1077,6 +1492,7 @@ function renderGraph(data, preserveSelection = false) {
nodeLayer.append(group);
}
svg.append(definitions, edgeLayer, nodeLayer);
applyTraceHighlight(state.selectedNode);
}
function hideNode(nodeId) {
if (!state.graph || nodeId === state.root) {
@ -1093,6 +1509,10 @@ function hideNode(nodeId) {
: "";
setStatus(`Hidden ${nodeId}.${suffix} Restore hidden nodes from the graph controls.`);
}
function explorationTarget(nodeId) {
const node = state.graph?.nodes.find((candidate) => candidate.node_id === nodeId);
return escapeText(node?.logic_owner_id || nodeId);
}
function restoreHiddenNodes() {
const count = state.hiddenNodes.size;
state.hiddenNodes.clear();
@ -1294,16 +1714,25 @@ async function search() {
const params = new URLSearchParams({
q: $("search").value.trim(),
family: $("family").value,
kind: $("kind").value,
language: $("language").value,
capability: $("capability").value,
limit: String(state.searchLimit),
});
const filtersActive = [
$("search").value.trim(),
$("family").value,
$("kind").value,
$("language").value,
$("capability").value,
].some(Boolean);
try {
setStatus("Searching validated index…");
const data = await api(`search?${params}`);
renderResults(data.results || []);
setResultsContext(
$("search").value.trim() || $("family").value
? `${data.count} search results`
: "All nodes",
filtersActive ? `${data.count} filtered results` : "All nodes",
filtersActive,
);
setStatus(`${data.count} matching node${data.count === 1 ? "" : "s"}`);
} catch (error) {
@ -1320,8 +1749,12 @@ async function filterByDescriptor(category, value) {
closeNodeCard();
setStatus(`Filtering ${category} ${value}`);
const data = await api(`filter?${params}`);
$("search").value = "";
$("family").value = category === "family" ? value : "";
clearSearchFilters();
if (category === "family") $("family").value = value;
if (category === "tag") {
if (nodeKindOptions.some(([tag]) => tag === value)) $("kind").value = value;
if (languageOptions.some(([tag]) => tag === value)) $("language").value = value;
}
renderResults(data.results || []);
setResultsContext(`${category}: ${value} (${data.total})`, true);
const suffix = data.truncated ? ` · showing first ${data.count}` : "";
@ -1330,14 +1763,44 @@ async function filterByDescriptor(category, value) {
setStatus(error.message, true);
}
}
function clearSearchFilters() {
$("search").value = "";
$("family").value = "";
$("kind").value = "";
$("language").value = "";
$("capability").value = "";
}
function applyFilterPreset(preset) {
clearSearchFilters();
if (preset === "logic") {
$("kind").value = "callable";
$("capability").value = "logic";
} else if (preset === "python") {
$("kind").value = "callable";
$("language").value = "python";
} else if (preset === "tests") {
$("kind").value = "test";
} else if (preset === "routes") {
$("kind").value = "route";
} else if (preset === "docs") {
$("kind").value = $("kind").querySelector('option[value="document"]')
? "document"
: "manual";
}
search();
}
async function loadNode(nodeId) {
try {
const showingFlow = state.mode === "flow";
const showingWeb = state.mode === "web";
const showingLogic = state.mode === "logic";
state.focusNode = nodeId;
const action = showingFlow ? "Tracing semantic flow for"
: showingWeb ? "Building convergence web for" : "Loading";
: showingWeb ? "Building convergence web for"
: showingLogic ? "Tracing control flow for" : "Loading";
setStatus(`${action} ${nodeId}`);
const endpoint = showingFlow ? "lineage" : showingWeb ? "web" : "node";
const endpoint = showingFlow ? "lineage"
: showingWeb ? "web" : showingLogic ? "logic" : "node";
const params = new URLSearchParams(
showingFlow
? {id: nodeId, limit: "1000"}
@ -1347,16 +1810,20 @@ async function loadNode(nodeId) {
depth: String(state.depth),
limit: "1000",
}
: {id: nodeId, depth: String(state.depth), limit: "100"},
: showingLogic
? {id: nodeId}
: {id: nodeId, depth: String(state.depth), limit: "100"},
);
const data = await api(`${endpoint}?${params}`);
renderGraph(data);
const scope = showingFlow ? "semantic flow"
: showingWeb ? "convergence web" : "neighborhood";
: showingWeb ? "convergence web"
: showingLogic ? "control flow" : "neighborhood";
const suffix = data.truncated ? " · truncated at the safety limit" : "";
setStatus(
`${data.nodes.length} nodes · ${data.edges.length} edges in ${scope}${suffix}`,
);
const status = showingLogic && !data.available
? `No indexed logic is available for ${nodeId}; use the Logic-ready filter`
: `${data.nodes.length} nodes · ${data.edges.length} edges in ${scope}${suffix}`;
setStatus(status, showingLogic && !data.available);
history.replaceState(
null,
"",
@ -1367,13 +1834,14 @@ async function loadNode(nodeId) {
}
}
async function setViewMode(mode) {
if (!["nodes", "flow", "web"].includes(mode)) return;
if (!["nodes", "flow", "web", "logic"].includes(mode)) return;
state.mode = mode;
$("view-switch").dataset.mode = mode;
$("view-nodes").setAttribute("aria-pressed", String(mode === "nodes"));
$("view-flow").setAttribute("aria-pressed", String(mode === "flow"));
$("view-web").setAttribute("aria-pressed", String(mode === "web"));
if (state.root) await loadNode(state.root);
$("view-logic").setAttribute("aria-pressed", String(mode === "logic"));
if (state.focusNode) await loadNode(state.focusNode);
}
function clamp(value, minimum, maximum) {
return Math.min(maximum, Math.max(minimum, value));
@ -1465,15 +1933,20 @@ function endDialogDrag(event) {
state.dialogDrag = null;
}
$("search-form").addEventListener("submit", (event) => { event.preventDefault(); search(); });
$("family").addEventListener("change", search);
for (const id of ["family", "kind", "language", "capability"]) {
$(id).addEventListener("change", search);
}
for (const button of document.querySelectorAll("[data-preset]")) {
button.addEventListener("click", () => applyFilterPreset(button.dataset.preset));
}
$("clear-result-filter").addEventListener("click", () => {
$("search").value = "";
$("family").value = "";
clearSearchFilters();
search();
});
$("view-nodes").addEventListener("click", () => setViewMode("nodes"));
$("view-flow").addEventListener("click", () => setViewMode("flow"));
$("view-web").addEventListener("click", () => setViewMode("web"));
$("view-logic").addEventListener("click", () => setViewMode("logic"));
$("zoom-in").addEventListener("click", () => zoomAt(.8));
$("zoom-out").addEventListener("click", () => zoomAt(1.25));
$("reset-view").addEventListener("click", resetViewport);
@ -1491,7 +1964,7 @@ $("node-dialog").querySelector(".dialog-head").addEventListener("pointercancel",
$("explore-node").addEventListener("click", async () => {
const nodeId = state.inspectedNode;
closeNodeDialog();
if (nodeId) await loadNode(nodeId);
if (nodeId) await loadNode(explorationTarget(nodeId));
});
$("open-node-source").addEventListener("click", () => {
if (state.inspectedNode) openSource(state.inspectedNode);
@ -1508,7 +1981,7 @@ $("hide-card-node").addEventListener("click", () => {
$("explore-card-node").addEventListener("click", async () => {
const nodeId = state.cardNode;
closeNodeCard();
if (nodeId) await loadNode(nodeId);
if (nodeId) await loadNode(explorationTarget(nodeId));
});
$("node-dialog").addEventListener("click", (event) => {
if (event.target !== $("node-dialog")) return;
@ -1607,7 +2080,9 @@ applyViewport();
const params = new URLSearchParams(location.search);
state.depth = Math.max(1, Number(params.get("depth")) || 1);
const requestedView = params.get("view");
setViewMode(["flow", "web"].includes(requestedView) ? requestedView : "nodes");
setViewMode(
["flow", "web", "logic"].includes(requestedView) ? requestedView : "nodes",
);
const overview = await api("overview");
renderOverview(overview);
startViewerLease();

View file

@ -20,9 +20,12 @@ from .changeset_contract import (
)
from .errors import DocForgeError
from .models import Edge, Node, ProjectService, ProjectSnapshot, ProposalWriter
from .pagination import canonical_hash, decode_cursor, page_limit, page_receipt
from .project import project_root_fingerprint
from .proposal_projection import ProposalProjector
MAX_ABANDON_REASON_CHARS = 2_000
class ChangesetStore:
"""One project-bound proposal store with an optional immutable writer identity."""
@ -79,6 +82,73 @@ class ChangesetStore:
self._write(path, document)
return self._result(snapshot, document, valid=True)
def register(
self,
changeset_id: str,
operations: list[dict[str, Any]],
) -> dict[str, object]:
"""Create and validate one complete proposal in a single atomic write."""
writer = self._require_writer()
validate_id(changeset_id, "changeset_id")
if not operations:
raise DocForgeError(
"empty_changeset",
"Registered changes require at least one operation",
)
with self._lock():
path = self._path(changeset_id)
if path.exists():
raise DocForgeError(
"changeset_exists",
"Changeset ID already exists",
changeset_id=changeset_id,
)
existing = tuple(self._root().glob("*.json"))
if len(existing) >= self.project.descriptor.limits.max_changesets:
raise DocForgeError("changeset_limit", "Project changeset limit has been reached")
if len(operations) > self.project.descriptor.limits.max_changeset_operations:
raise DocForgeError(
"changeset_operation_limit",
"Changeset operation limit has been reached",
)
snapshot = self.project.load()
nodes = {node.node_id: node for node in snapshot.nodes}
normalized = [
normalize_operation(
self._complete_operation(operation, nodes),
sequence=sequence,
)
for sequence, operation in enumerate(operations, start=1)
]
if len({item["node_id"] for item in normalized}) != len(normalized):
raise DocForgeError(
"duplicate_operation",
"A changeset may touch a node only once",
)
document: dict[str, Any] = {
"schema_version": 1,
"changeset_id": changeset_id,
"project_id": snapshot.descriptor.project_id,
"root_fingerprint": project_root_fingerprint(snapshot.descriptor.root),
"base_revision": snapshot.revision,
"base_source_hash": snapshot.source_hash,
"creator": writer.writer_id,
"operations": normalized,
}
projected_nodes, projected_edges = self.projector.project(snapshot, document)
self._check_proposal_conflicts(document, snapshot)
self._write(path, document)
return self._result(
snapshot,
document,
valid=True,
lifecycle="ready",
ready_for_review=True,
projected_node_count=len(projected_nodes),
projected_edge_count=len(projected_edges),
)
def propose_create(
self,
*,
@ -158,6 +228,33 @@ class ChangesetStore:
},
)
def propose_relationship_update(
self,
*,
changeset_id: str,
expected_changeset_hash: str,
node_id: str,
expected_content_hash: str,
relationship_changes: list[dict[str, Any]],
rationale: str,
) -> dict[str, object]:
"""Queue relationship-only changes against one exact existing node."""
if not relationship_changes:
raise DocForgeError(
"invalid_operation", "Relationship-only updates require at least one change"
)
return self.propose_update(
changeset_id=changeset_id,
expected_changeset_hash=expected_changeset_hash,
node_id=node_id,
expected_content_hash=expected_content_hash,
metadata=None,
content=None,
relationship_changes=relationship_changes,
rationale=rationale,
)
def propose_delete(
self,
*,
@ -183,24 +280,53 @@ class ChangesetStore:
},
)
def validate(self, changeset_id: str) -> dict[str, object]:
def validate(
self,
changeset_id: str,
*,
limit: int | None = None,
cursor: str | None = None,
) -> dict[str, object]:
validate_id(changeset_id, "changeset_id")
with self._lock():
snapshot, document, nodes, edges = self._validate_locked(changeset_id)
return self._result(
result = self._result(
snapshot,
document,
valid=True,
projected_node_count=len(nodes),
projected_edge_count=len(edges),
)
return self._page_document_result(
result,
kind="changeset.validate",
limit=limit,
cursor=cursor,
)
def list_changesets(self) -> dict[str, object]:
def list_changesets(
self,
*,
include_history: bool = True,
status: str | None = None,
limit: int | None = None,
cursor: str | None = None,
) -> dict[str, object]:
with self._lock():
snapshot = self.project.load()
records: list[dict[str, object]] = []
for path in sorted(self._root().glob("*.json"), key=lambda item: item.name):
document = self._read(path)
base_state = self._base_state(document, snapshot)
lifecycle = self._lifecycle(document, base_state)
if status is not None and lifecycle["status"] != status:
continue
if (
status is None
and not include_history
and lifecycle["status"] in {"abandoned", "applied", "stale"}
):
continue
records.append(
{
"changeset_id": document["changeset_id"],
@ -208,24 +334,218 @@ class ChangesetStore:
"creator": document["creator"],
"base_revision": document["base_revision"],
"base_source_hash": document["base_source_hash"],
"base_state": self._base_state(document, snapshot),
"base_state": base_state,
"lifecycle": lifecycle,
"operation_count": len(document["operations"]),
}
)
return self._base_result(snapshot, count=len(records), changesets=records)
result = self._base_result(snapshot, count=len(records), changesets=records)
if limit is None and cursor is None:
return result
selected_limit = page_limit(
limit,
default=20,
maximum=self.project.descriptor.limits.max_results,
)
binding = {
"project_id": snapshot.descriptor.project_id,
"project_root_fingerprint": project_root_fingerprint(snapshot.descriptor.root),
"adapter": snapshot.descriptor.adapter,
"revision": snapshot.revision,
"source_hash": snapshot.source_hash,
"include_history": include_history,
"status": status,
"collection_hash": canonical_hash(records),
}
position = decode_cursor(
cursor,
kind="changeset.list",
binding=binding,
total_count=len(records),
)
page = records[position : position + selected_limit]
while page:
candidate = {
**result,
"count": len(page),
"total_count": len(records),
"changesets": page,
"pagination": page_receipt(
kind="changeset.list",
binding=binding,
position=position,
count=len(page),
limit=selected_limit,
total_count=len(records),
),
}
if self._encoded_length(candidate) <= self._safe_page_chars():
return candidate
page.pop()
if position < len(records):
compact_record = self._compact_list_record(records[position])
return {
**result,
"count": 1,
"total_count": len(records),
"changesets": [compact_record],
"result_mode": "changeset_summaries",
"pagination": page_receipt(
kind="changeset.list",
binding=binding,
position=position,
count=1,
limit=selected_limit,
total_count=len(records),
),
}
return {
**result,
"count": 0,
"total_count": len(records),
"changesets": [],
"pagination": page_receipt(
kind="changeset.list",
binding=binding,
position=position,
count=0,
limit=selected_limit,
total_count=len(records),
),
}
def inspect(self, changeset_id: str) -> dict[str, object]:
def inspect(
self,
changeset_id: str,
*,
limit: int | None = None,
cursor: str | None = None,
) -> dict[str, object]:
validate_id(changeset_id, "changeset_id")
with self._lock():
document = self._read(self._path(changeset_id))
snapshot = self.project.load()
result = self._result(
snapshot,
document,
base_state=self._base_state(document, snapshot),
lifecycle=self._lifecycle(
document,
self._base_state(document, snapshot),
),
)
return self._page_document_result(
result,
kind="changeset.inspect",
limit=limit,
cursor=cursor,
)
def rebase(
self,
changeset_id: str,
expected_changeset_hash: str,
) -> dict[str, object]:
"""Move a proposal to the current base when every touched fact is unchanged."""
validate_id(changeset_id, "changeset_id")
validate_hash(expected_changeset_hash, "expected_changeset_hash")
with self._lock():
path = self._path(changeset_id)
document = self._read(path)
actual_hash = document_hash(document)
if actual_hash != expected_changeset_hash:
raise DocForgeError(
"changeset_conflict",
"Changeset changed after the caller read it",
changeset_id=changeset_id,
expected=expected_changeset_hash,
actual=actual_hash,
)
snapshot = self.project.load()
self._require_mutable(document, snapshot)
if self._base_state(document, snapshot) == "current":
return self._result(
snapshot,
document,
valid=True,
rebased=False,
lifecycle=self._lifecycle(document, "current"),
)
candidate = {
**document,
"base_revision": snapshot.revision,
"base_source_hash": snapshot.source_hash,
}
nodes, edges = self.projector.project(snapshot, candidate)
self._check_proposal_conflicts(candidate, snapshot)
self._write(path, candidate)
return self._result(
snapshot,
candidate,
valid=True,
rebased=True,
lifecycle="ready",
projected_node_count=len(nodes),
projected_edge_count=len(edges),
)
def abandon(
self,
changeset_id: str,
expected_changeset_hash: str,
reason: str,
) -> dict[str, object]:
"""Mark one proposal as abandoned without deleting its audit record."""
validate_id(changeset_id, "changeset_id")
validate_hash(expected_changeset_hash, "expected_changeset_hash")
normalized_reason = reason.strip()
if not normalized_reason:
raise DocForgeError("invalid_operation", "Abandon reason must be non-empty")
if len(normalized_reason) > MAX_ABANDON_REASON_CHARS:
raise DocForgeError(
"changeset_too_large",
"Abandon reason exceeds its character limit",
maximum=MAX_ABANDON_REASON_CHARS,
)
with self._lock():
document = self._read(self._path(changeset_id))
actual_hash = document_hash(document)
if actual_hash != expected_changeset_hash:
raise DocForgeError(
"changeset_conflict",
"Changeset changed after the caller read it",
changeset_id=changeset_id,
expected=expected_changeset_hash,
actual=actual_hash,
)
snapshot = self.project.load()
self._require_mutable(document, snapshot)
receipt = self._write_state(
changeset_id,
{
"status": "abandoned",
"changeset_hash": actual_hash,
"reason": normalized_reason,
"revision": snapshot.revision,
"source_hash": snapshot.source_hash,
},
)
return self._result(
snapshot,
document,
base_state=self._base_state(document, snapshot),
lifecycle=receipt,
)
def diff(self, changeset_id: str) -> dict[str, object]:
def diff(
self,
changeset_id: str,
*,
limit: int | None = None,
cursor: str | None = None,
) -> dict[str, object]:
validate_id(changeset_id, "changeset_id")
with self._lock():
snapshot, document, _, _ = self._validate_locked(changeset_id)
@ -248,7 +568,14 @@ class ChangesetStore:
sorted(before_edges - edges),
)
)
return self._result(snapshot, document, valid=True, changes=changes)
result = self._result(snapshot, document, valid=True, changes=changes)
return self._page_document_result(
result,
kind="changeset.diff",
limit=limit,
cursor=cursor,
parallel_key="changes",
)
def projected_snapshot(self, changeset_id: str) -> tuple[ProjectSnapshot, str]:
"""Return a validated in-memory proposal projection for derived preview use."""
@ -271,6 +598,7 @@ class ChangesetStore:
changeset_id: str,
expected_changeset_hash: str,
applier_id: str,
accepted_creator_ids: frozenset[str] | None = None,
application: Callable[
[ProjectSnapshot, ProjectSnapshot, tuple[Mapping[str, object], ...]],
dict[str, object],
@ -286,7 +614,12 @@ class ChangesetStore:
"Canonical applier identity is not configured for this store",
)
with self._lock():
snapshot, document, nodes, edges = self._validate_locked(changeset_id)
document = self._read(self._path(changeset_id))
snapshot = self.project.load()
self._require_mutable(document, snapshot)
self._check_base(document, snapshot)
nodes, edges = self.projector.project(snapshot, document)
self._check_proposal_conflicts(document, snapshot)
actual_hash = document_hash(document)
if actual_hash != expected_changeset_hash:
raise DocForgeError(
@ -296,13 +629,17 @@ class ChangesetStore:
expected=expected_changeset_hash,
actual=actual_hash,
)
if document["creator"] != applier_id:
accepted_creators = (
frozenset({applier_id}) if accepted_creator_ids is None else accepted_creator_ids
)
if document["creator"] not in accepted_creators:
raise DocForgeError(
"changeset_owner_conflict",
"Canonical applier does not own this changeset",
"Canonical applier is not authorized to accept this changeset creator",
changeset_id=changeset_id,
owner=document["creator"],
applier=applier_id,
accepted_creators=sorted(accepted_creators),
)
projected = ProjectSnapshot(
descriptor=snapshot.descriptor,
@ -317,13 +654,39 @@ class ChangesetStore:
tuple(cast(Mapping[str, object], item) for item in document["operations"]),
)
current = self.project.load()
lifecycle_payload: dict[str, object] = {
"status": "applied",
"changeset_hash": actual_hash,
"revision": current.revision,
"source_hash": current.source_hash,
}
application_recovery = payload.get("application_recovery")
if isinstance(application_recovery, Mapping):
recovery_payload = cast(Mapping[str, object], application_recovery)
if recovery_payload.get("status") != "clean":
retained = recovery_payload.get("retained")
lifecycle_payload["application_recovery"] = {
"status": recovery_payload.get("status"),
"retained_count": (
len(cast(list[object], retained)) if isinstance(retained, list) else 0
),
"retained_root": ".docforge/application",
"remediation": recovery_payload.get("remediation"),
}
lifecycle = self._write_state(
changeset_id,
lifecycle_payload,
)
return self._result(
current,
document,
valid=True,
applied=True,
lifecycle=lifecycle,
applied_from_revision=snapshot.revision,
applied_from_source_hash=snapshot.source_hash,
proposal_creator=document["creator"],
applied_by=applier_id,
**payload,
)
@ -356,6 +719,8 @@ class ChangesetStore:
owner=document["creator"],
writer=writer.writer_id,
)
snapshot = self.project.load()
self._require_mutable(document, snapshot)
if (
len(document["operations"])
>= self.project.descriptor.limits.max_changeset_operations
@ -371,7 +736,6 @@ class ChangesetStore:
node_id=normalized["node_id"],
)
candidate = {**document, "operations": [*document["operations"], normalized]}
snapshot = self.project.load()
self._check_base(candidate, snapshot)
nodes, edges = self.projector.project(snapshot, candidate)
self._check_proposal_conflicts(candidate, snapshot)
@ -384,6 +748,51 @@ class ChangesetStore:
projected_edge_count=len(edges),
)
@staticmethod
def _complete_operation(
operation: dict[str, Any],
nodes: dict[str, Node],
) -> dict[str, Any]:
allowed = {
"operation",
"node_id",
"expected_content_hash",
"target_source",
"metadata",
"content",
"relationship_changes",
"rationale",
}
unknown = sorted(set(operation) - allowed)
if unknown:
raise DocForgeError(
"invalid_operation",
"Operation has unknown fields",
fields=unknown,
)
kind = operation.get("operation")
node_id = operation.get("node_id")
expected = operation.get("expected_content_hash")
if kind != "create" and expected is None and isinstance(node_id, str):
node = nodes.get(node_id)
if node is None:
raise DocForgeError(
"missing_node",
"No node has the requested stable ID",
node_id=node_id,
)
expected = node.content_hash
return {
"operation": kind,
"node_id": node_id,
"expected_content_hash": expected,
"target_source": operation.get("target_source"),
"metadata": operation.get("metadata"),
"content": operation.get("content"),
"relationship_changes": operation.get("relationship_changes", []),
"rationale": operation.get("rationale"),
}
def _validate_locked(
self, changeset_id: str
) -> tuple[ProjectSnapshot, dict[str, Any], dict[str, Node], set[tuple[str, str, str]]]:
@ -414,6 +823,63 @@ class ChangesetStore:
return "current"
return "stale"
def _lifecycle(
self,
document: dict[str, Any],
base_state: str,
) -> dict[str, object]:
state_path = self._state_root() / f"{document['changeset_id']}.json"
if state_path.is_file() and not state_path.is_symlink():
try:
if state_path.stat().st_size > self.project.descriptor.limits.max_changeset_bytes:
raise DocForgeError(
"changeset_too_large",
"Changeset lifecycle record exceeds the configured size limit",
changeset_id=document["changeset_id"],
)
parsed: object = json.loads(state_path.read_text(encoding="utf-8"))
except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error:
raise DocForgeError(
"invalid_changeset_state",
"Changeset lifecycle record is unreadable",
changeset_id=document["changeset_id"],
) from error
if not isinstance(parsed, dict):
raise DocForgeError(
"invalid_changeset_state",
"Changeset lifecycle record does not match its proposal",
changeset_id=document["changeset_id"],
)
payload = cast(dict[str, object], parsed)
if payload.get("changeset_hash") != document_hash(document) or payload.get(
"status"
) not in {"applied", "abandoned"}:
raise DocForgeError(
"invalid_changeset_state",
"Changeset lifecycle record does not match its proposal",
changeset_id=document["changeset_id"],
)
return payload
if base_state == "stale":
return {"status": "stale"}
if not document["operations"]:
return {"status": "draft"}
return {"status": "ready"}
def _require_mutable(
self,
document: dict[str, Any],
snapshot: ProjectSnapshot,
) -> None:
lifecycle = self._lifecycle(document, self._base_state(document, snapshot))
if lifecycle["status"] in {"applied", "abandoned"}:
raise DocForgeError(
"changeset_closed",
"Applied or abandoned changesets cannot be modified",
changeset_id=document["changeset_id"],
lifecycle=lifecycle["status"],
)
def _check_proposal_conflicts(
self, document: dict[str, Any], snapshot: ProjectSnapshot
) -> None:
@ -425,6 +891,12 @@ class ChangesetStore:
other = self._read(path)
if other["base_source_hash"] != document["base_source_hash"]:
continue
other_lifecycle = self._lifecycle(
other,
self._base_state(other, snapshot),
)
if other_lifecycle["status"] in {"applied", "abandoned"}:
continue
other_nodes, other_sources = self.projector.touches(other, snapshot)
shared_nodes = sorted(nodes & other_nodes)
shared_sources = sorted(sources & other_sources)
@ -461,6 +933,219 @@ class ChangesetStore:
**payload,
)
def _page_document_result(
self,
result: dict[str, object],
*,
kind: str,
limit: int | None,
cursor: str | None,
parallel_key: str | None = None,
) -> dict[str, object]:
"""Page bulky operation-aligned payloads while preserving direct full defaults."""
if limit is None and cursor is None:
return result
selected_limit = page_limit(
limit,
default=20,
maximum=self.project.descriptor.limits.max_results,
)
operations_value = result.get("operations")
if not isinstance(operations_value, list):
raise DocForgeError(
"invalid_pagination_source",
"Changeset result does not contain a deterministic operation list",
)
operations = cast(list[object], operations_value)
parallel: list[object] | None = None
if parallel_key is not None:
parallel_value = result.get(parallel_key)
if not isinstance(parallel_value, list):
raise DocForgeError(
"invalid_pagination_source",
"Changeset result does not contain an aligned detail list",
)
parallel = cast(list[object], parallel_value)
if len(parallel) != len(operations):
raise DocForgeError(
"invalid_pagination_source",
"Changeset detail list is not aligned with its operations",
)
binding = {
"project_id": result["project_id"],
"project_root_fingerprint": result["project_root_fingerprint"],
"revision": result["revision"],
"source_hash": result["source_hash"],
"changeset_id": result["changeset_id"],
"changeset_hash": result["changeset_hash"],
"adapter": result["adapter"],
"result_hash": canonical_hash(
{
"operations": operations,
**({parallel_key: parallel} if parallel_key is not None else {}),
}
),
}
if (
kind == "changeset.diff"
and parallel_key is not None
and parallel is not None
and self._encoded_length(result) > self._safe_page_chars()
):
return self._page_json_chunks(
result,
operations=operations,
changes=parallel,
binding=binding,
cursor=cursor,
)
position = decode_cursor(
cursor,
kind=kind,
binding=binding,
total_count=len(operations),
)
page = operations[position : position + selected_limit]
paged = {
**result,
"operations": page,
"returned_operation_count": len(page),
"pagination": page_receipt(
kind=kind,
binding=binding,
position=position,
count=len(page),
limit=selected_limit,
total_count=len(operations),
),
}
if parallel_key is not None and parallel is not None:
paged[parallel_key] = parallel[position : position + selected_limit]
if self._encoded_length(paged) > self._safe_page_chars():
paged["operations"] = [
self._operation_summary(item)
for item in operations[position : position + selected_limit]
]
paged["result_mode"] = "operation_summaries"
paged["detail_tool"] = "docforge_get_changeset_diff"
return paged
def _page_json_chunks(
self,
result: dict[str, object],
*,
operations: list[object],
changes: list[object],
binding: Mapping[str, object],
cursor: str | None,
) -> dict[str, object]:
payload = {"operations": operations, "changes": changes}
encoded = json.dumps(
payload,
sort_keys=True,
separators=(",", ":"),
ensure_ascii=True,
allow_nan=False,
)
chunk_chars = max(512, min(64_000, self._safe_page_chars() // 2))
chunks = [
encoded[offset : offset + chunk_chars] for offset in range(0, len(encoded), chunk_chars)
] or [""]
chunk_binding = {
**binding,
"payload_hash": canonical_hash(payload),
"chunk_chars": chunk_chars,
}
position = decode_cursor(
cursor,
kind="changeset.diff-chunks",
binding=chunk_binding,
total_count=len(chunks),
)
compact = {
key: value for key, value in result.items() if key not in {"operations", "changes"}
}
return {
**compact,
"result_mode": "canonical_json_chunk",
"payload": "changeset_diff",
"payload_hash": chunk_binding["payload_hash"],
"payload_characters": len(encoded),
"chunk": {
"index": position,
"characters": len(chunks[position]),
"content": chunks[position],
},
"pagination": page_receipt(
kind="changeset.diff-chunks",
binding=chunk_binding,
position=position,
count=1,
limit=1,
total_count=len(chunks),
),
}
@staticmethod
def _operation_summary(operation: object) -> dict[str, object]:
if not isinstance(operation, Mapping):
raise DocForgeError(
"invalid_pagination_source",
"Changeset operation is not a deterministic object",
)
payload = cast(Mapping[str, object], operation)
summary = {
key: payload.get(key)
for key in (
"sequence",
"operation",
"node_id",
"expected_content_hash",
"target_source",
)
}
for key in ("metadata", "content", "relationship_changes", "rationale"):
value = payload.get(key)
summary[f"{key}_hash"] = canonical_hash(value)
summary[f"{key}_characters"] = len(
json.dumps(
value,
sort_keys=True,
separators=(",", ":"),
ensure_ascii=True,
allow_nan=False,
)
)
return summary
@staticmethod
def _compact_list_record(record: dict[str, object]) -> dict[str, object]:
lifecycle = record.get("lifecycle")
if not isinstance(lifecycle, Mapping):
return record
lifecycle_payload = cast(Mapping[str, object], lifecycle)
reason = lifecycle_payload.get("reason")
if not isinstance(reason, str):
return record
return {
**record,
"lifecycle": {
**lifecycle_payload,
"reason": {
"characters": len(reason),
"sha256": canonical_hash(reason),
},
},
}
def _safe_page_chars(self) -> int:
return max(1_024, self.project.descriptor.limits.max_tool_output_chars - 2_048)
@staticmethod
def _encoded_length(value: Mapping[str, object]) -> int:
return len(json.dumps(value, sort_keys=True, separators=(",", ":")))
@staticmethod
def _base_result(snapshot: ProjectSnapshot, **payload: object) -> dict[str, object]:
return {
@ -551,6 +1236,11 @@ class ChangesetStore:
def _restore(path: Path, previous: bytes | None, root: Path) -> None:
if previous is None:
path.unlink(missing_ok=True)
directory_descriptor = os.open(root, os.O_RDONLY)
try:
os.fsync(directory_descriptor)
finally:
os.close(directory_descriptor)
return
restore_descriptor, restore_name = tempfile.mkstemp(prefix=".rollback-", dir=root)
restore = Path(restore_name)
@ -560,6 +1250,11 @@ class ChangesetStore:
handle.flush()
os.fsync(handle.fileno())
os.replace(restore, path)
directory_descriptor = os.open(root, os.O_RDONLY)
try:
os.fsync(directory_descriptor)
finally:
os.close(directory_descriptor)
except Exception:
restore.unlink(missing_ok=True)
raise
@ -578,6 +1273,49 @@ class ChangesetStore:
raise DocForgeError("path_escape", "Changeset root changed or resolves unexpectedly")
return root
def _state_root(self) -> Path:
root = self._root() / ".state"
root.mkdir(parents=True, exist_ok=True)
if (
not root.is_dir()
or root.is_symlink()
or not root.resolve().is_relative_to(self._root())
):
raise DocForgeError("path_escape", "Changeset state root is not safe")
return root
def _write_state(
self,
changeset_id: str,
payload: dict[str, object],
) -> dict[str, object]:
root = self._state_root()
path = root / f"{changeset_id}.json"
raw = json.dumps(payload, sort_keys=True, indent=2).encode("utf-8") + b"\n"
if len(raw) > self.project.descriptor.limits.max_changeset_bytes:
raise DocForgeError(
"changeset_too_large",
"Changeset lifecycle record exceeds the configured size limit",
changeset_id=changeset_id,
)
descriptor, temporary_name = tempfile.mkstemp(prefix=".state-", dir=root)
temporary = Path(temporary_name)
try:
with os.fdopen(descriptor, "wb") as handle:
handle.write(raw)
handle.flush()
os.fsync(handle.fileno())
os.replace(temporary, path)
directory_descriptor = os.open(root, os.O_RDONLY)
try:
os.fsync(directory_descriptor)
finally:
os.close(directory_descriptor)
except Exception:
temporary.unlink(missing_ok=True)
raise
return payload
@contextmanager
def _lock(self) -> Generator[None]:
root = self._root()

View file

@ -8,23 +8,90 @@ import sys
import webbrowser
from pathlib import Path
from ._version import __version__
from .application import CanonicalApplicationService, GenericCanonicalApplier
from .client_config import CLIENT_NAMES, generate_client_configuration
from .context import compile_context
from .doctor import run_doctor
from .errors import DocForgeError
from .graph_rendering import GraphRenderService
from .index import ProjectIndex
from .onboarding import assess_project, scaffold_project
from .project import Project, project_root_fingerprint
from .projection_policy import compose_projection_policy
from .rendering import RenderService
from .telemetry import request
from .viewer_manager import ViewerManagerClient
def _parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(prog="docforge")
parser.add_argument("--project-root", type=Path, required=True)
parser.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
parser.add_argument("--project-root", type=Path)
parser.add_argument(
"--diagnostics",
action="store_true",
help="Attach bounded request-local stage timings and counters",
)
parser.add_argument(
"--manual-render-policy",
choices=("auto", "explicit", "disabled"),
)
parser.add_argument(
"--portable-graph-policy",
choices=("explicit", "disabled"),
)
parser.add_argument(
"--live-viewer-policy",
choices=("on-demand", "disabled"),
)
commands = parser.add_subparsers(dest="command", required=True)
configure = commands.add_parser("configure")
configure.add_argument("client", choices=CLIENT_NAMES)
configure.add_argument("--project", type=Path, required=True)
configure.add_argument("--name")
configure.add_argument(
"--capability-mode",
choices=("read", "proposal", "application"),
default="read",
)
configure.add_argument("--proposal-writer")
configure.add_argument("--canonical-applier")
configure.add_argument("--no-ast", action="store_true")
configure.add_argument(
"--manual-render-policy",
choices=("auto", "explicit", "disabled"),
default=argparse.SUPPRESS,
)
configure.add_argument(
"--portable-graph-policy",
choices=("explicit", "disabled"),
default=argparse.SUPPRESS,
)
configure.add_argument(
"--live-viewer-policy",
choices=("on-demand", "disabled"),
default=argparse.SUPPRESS,
)
configure.add_argument("--startup-timeout", type=int, default=30)
configure.add_argument("--tool-timeout", type=int, default=300)
configure.add_argument("--output", type=Path)
doctor = commands.add_parser("doctor")
doctor.add_argument("--client", choices=CLIENT_NAMES, required=True)
doctor.add_argument("--project", type=Path)
doctor.add_argument("--config", type=Path)
doctor.add_argument("--server-name")
onboard = commands.add_parser("onboard")
onboard.add_argument("--language", action="append", default=[])
onboard.add_argument("--scaffold", action="store_true")
onboard.add_argument("--project-id")
onboard.add_argument("--title")
onboard.add_argument("--content-root", default="docs/docforge/content")
commands.add_parser("info")
commands.add_parser("validate")
commands.add_parser("build")
commands.add_parser("reindex")
commands.add_parser("sync")
commands.add_parser("check")
commands.add_parser("validate-index")
show = commands.add_parser("show")
@ -45,13 +112,26 @@ def _parser() -> argparse.ArgumentParser:
command.add_argument("--relation")
else:
command.add_argument("--depth", type=int, default=2)
command.add_argument("--limit", type=int)
context = commands.add_parser("context")
context.add_argument("profile")
context.add_argument("--budget", type=int)
context.add_argument("--limit", type=int)
context.add_argument("--cursor")
generation_diff = commands.add_parser("generation-diff")
generation_diff.add_argument("--limit", type=int)
generation_diff.add_argument("--cursor")
render = commands.add_parser("render")
render.add_argument("view_id")
render_status = commands.add_parser("render-status")
render_status.add_argument("view_id", nargs="?")
render_status.add_argument("--deep", action="store_true")
graph_plan = commands.add_parser("graph-plan")
graph_plan.add_argument("view_id")
graph_render = commands.add_parser("graph-render")
graph_render.add_argument("view_id")
graph_render_status = commands.add_parser("graph-render-status")
graph_render_status.add_argument("view_id", nargs="?")
preview = commands.add_parser("preview")
preview.add_argument("changeset_id")
preview.add_argument("view_id")
@ -71,8 +151,80 @@ def _parser() -> argparse.ArgumentParser:
def _run(arguments: argparse.Namespace) -> dict[str, object]:
if arguments.command == "configure":
project = Project.open(arguments.project)
return generate_client_configuration(
project,
arguments.client,
server_name=arguments.name,
capability_mode=arguments.capability_mode,
proposal_writer=arguments.proposal_writer,
canonical_applier=arguments.canonical_applier,
no_ast=arguments.no_ast,
manual_render_policy=arguments.manual_render_policy,
portable_graph_policy=arguments.portable_graph_policy,
live_viewer_policy=arguments.live_viewer_policy,
startup_timeout=arguments.startup_timeout,
tool_timeout=arguments.tool_timeout,
output=arguments.output,
)
if arguments.command == "doctor":
root = arguments.project or arguments.project_root or Path.cwd()
return run_doctor(
Project.open(root),
arguments.client,
config_path=arguments.config,
server_name=arguments.server_name,
)
if arguments.project_root is None:
raise DocForgeError(
"missing_project_root",
"This command requires --project-root",
)
if arguments.command == "onboard":
languages = tuple(arguments.language)
if arguments.scaffold:
scaffold = scaffold_project(
arguments.project_root,
requested_languages=languages,
project_id=arguments.project_id,
title=arguments.title,
content_root=arguments.content_root,
)
project = Project.open(arguments.project_root)
projection_policy = compose_projection_policy(
manual=arguments.manual_render_policy,
portable_graph=arguments.portable_graph_policy,
live_viewer=arguments.live_viewer_policy,
manual_configured=project.descriptor.render is not None,
portable_graph_configured=project.descriptor.graph_render is not None,
application_enabled=False,
)
build = ProjectIndex(project).build()
render = (
{
"status": "ok",
"state": "skipped",
"reason": "projection_policy_disabled",
}
if projection_policy.manual == "disabled"
else RenderService(
project,
manual_policy=projection_policy.manual,
).render("manual")
)
return {**scaffold, "build": build, "render": render}
return assess_project(arguments.project_root, requested_languages=languages)
project = Project.open(arguments.project_root)
index = ProjectIndex(project)
projection_policy = compose_projection_policy(
manual=arguments.manual_render_policy,
portable_graph=arguments.portable_graph_policy,
live_viewer=arguments.live_viewer_policy,
manual_configured=project.descriptor.render is not None,
portable_graph_configured=project.descriptor.graph_render is not None,
application_enabled=arguments.command == "apply",
)
if arguments.command == "info":
snapshot = project.load()
return {
@ -107,6 +259,8 @@ def _run(arguments: argparse.Namespace) -> dict[str, object]:
"reindexed": True,
"check": index.check(),
}
if arguments.command == "sync":
return index.synchronize()
if arguments.command == "check":
return index.check()
if arguments.command == "validate-index":
@ -124,27 +278,91 @@ def _run(arguments: argparse.Namespace) -> dict[str, object]:
limit=arguments.limit,
)
if arguments.command == "backlinks":
return index.backlinks(arguments.node_id, relation=arguments.relation)
return index.backlinks(
arguments.node_id,
relation=arguments.relation,
limit=arguments.limit,
)
if arguments.command == "dependencies":
return index.dependencies(arguments.node_id, depth=arguments.depth)
return index.dependencies(
arguments.node_id,
depth=arguments.depth,
limit=arguments.limit,
)
if arguments.command == "impact":
return index.impact(arguments.node_id, depth=arguments.depth)
return index.impact(
arguments.node_id,
depth=arguments.depth,
limit=arguments.limit,
)
if arguments.command == "context":
if arguments.limit is not None or arguments.cursor is not None:
from .mcp_server import DocForgeService
return DocForgeService(project).context(
arguments.profile,
arguments.budget,
limit=arguments.limit,
cursor=arguments.cursor,
)
return compile_context(index, arguments.profile, arguments.budget)
if arguments.command == "generation-diff":
from .mcp_server import DocForgeService
return DocForgeService(
project,
capability_mode_name="read",
).generation_diff(
limit=arguments.limit,
cursor=arguments.cursor,
)
if arguments.command == "render":
return RenderService(project).render(arguments.view_id)
return RenderService(
project,
manual_policy=projection_policy.manual,
).render(arguments.view_id)
if arguments.command == "render-status":
return RenderService(project).status(arguments.view_id)
rendering = RenderService(
project,
manual_policy=projection_policy.manual,
)
return (
rendering.deep_status(arguments.view_id)
if arguments.deep
else rendering.status(arguments.view_id)
)
if arguments.command == "graph-plan":
return GraphRenderService(
project,
portable_graph_policy=projection_policy.portable_graph,
).plan(arguments.view_id)
if arguments.command == "graph-render":
return GraphRenderService(
project,
portable_graph_policy=projection_policy.portable_graph,
).render(arguments.view_id)
if arguments.command == "graph-render-status":
return GraphRenderService(
project,
portable_graph_policy=projection_policy.portable_graph,
).status(arguments.view_id)
if arguments.command == "preview":
return RenderService(project).preview(arguments.changeset_id, arguments.view_id)
return RenderService(
project,
manual_policy=projection_policy.manual,
).preview(arguments.changeset_id, arguments.view_id)
if arguments.command == "apply":
return CanonicalApplicationService(
project,
applier_id=arguments.applier,
applier=GenericCanonicalApplier(project),
manual_policy=projection_policy.manual,
).apply(arguments.changeset_id, arguments.changeset_hash)
if arguments.command == "visualize":
visualization = ViewerManagerClient(index).start(
visualization = ViewerManagerClient(
index,
live_viewer_policy=projection_policy.live_viewer,
).start(
node_id=arguments.node,
query=arguments.query,
depth=arguments.depth,
@ -161,21 +379,36 @@ def _run(arguments: argparse.Namespace) -> dict[str, object]:
"visualization": visualization,
}
if arguments.command == "visualization-status":
return ViewerManagerClient(index).status()
return ViewerManagerClient(
index,
live_viewer_policy=projection_policy.live_viewer,
).status()
if arguments.command == "visualization-stop":
return ViewerManagerClient(index).stop()
return ViewerManagerClient(
index,
live_viewer_policy=projection_policy.live_viewer,
).stop()
raise DocForgeError("invalid_command", "Unknown command")
def main(argv: list[str] | None = None) -> int:
parser = _parser()
arguments = parser.parse_args(argv)
try:
result = _run(arguments)
code = 0
except DocForgeError as error:
result = {"status": "error", "error": error.as_dict()}
code = 2
with request(
f"cli.{arguments.command}",
enabled=arguments.diagnostics,
) as collector:
try:
result = _run(arguments)
doctor_state = result.get("doctor_state")
code = 2 if doctor_state == "unhealthy" else (1 if doctor_state == "degraded" else 0)
except DocForgeError as error:
result = {"status": "error", "error": error.as_dict()}
code = 2
if collector is not None:
result["diagnostics"] = collector.as_dict(
outcome="ok" if result.get("status") == "ok" else "error",
)
print(json.dumps(result, sort_keys=True, indent=2))
return code

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,242 @@
"""Deterministic command references derived from live CLI and MCP registrations."""
from __future__ import annotations
import argparse
import hashlib
import json
from collections.abc import Collection, Iterable, Mapping
from dataclasses import dataclass
from typing import Any, cast
from mcp.types import Tool
@dataclass(frozen=True)
class CliCommandReference:
"""One CLI command and the exact normalized usage emitted by argparse."""
name: str
invocation: str
@dataclass(frozen=True)
class McpToolReference:
"""One registered MCP tool with arguments derived from its input schema."""
surface: str
name: str
required_arguments: tuple[str, ...]
optional_arguments: tuple[str, ...]
description: str
input_schema_hash: str
def cli_command_references(
parser: argparse.ArgumentParser | None = None,
) -> tuple[CliCommandReference, ...]:
"""Read CLI reference rows from the parser used by ``docforge``."""
effective_parser = parser or _docforge_parser()
commands = _command_parsers(effective_parser)
return tuple(
CliCommandReference(
name=name,
invocation=_normalize_usage(command_parser.format_usage()),
)
for name, command_parser in sorted(commands.items())
)
def mcp_tool_references(
tools: Iterable[Tool],
*,
expected_names: Collection[str] | None = None,
) -> tuple[McpToolReference, ...]:
"""Read MCP reference rows from registered tools returned by ``list_tools``.
``expected_names`` makes documentation generation fail closed when the selected
server surface is incomplete or has drifted.
"""
surface_by_name = _mcp_surface_by_name()
references: list[McpToolReference] = []
observed: set[str] = set()
for tool in tools:
if tool.name in observed:
raise ValueError(f"MCP tool registration repeats {tool.name!r}")
observed.add(tool.name)
try:
surface = surface_by_name[tool.name]
except KeyError as error:
raise ValueError(
f"MCP tool {tool.name!r} has no declared capability surface"
) from error
schema = tool.inputSchema
properties = _schema_properties(schema)
required = _schema_required(schema, properties)
references.append(
McpToolReference(
surface=surface,
name=tool.name,
required_arguments=tuple(sorted(required)),
optional_arguments=tuple(sorted(set(properties) - required)),
description=_normalize_text(tool.description or ""),
input_schema_hash=_canonical_hash(schema),
)
)
if expected_names is not None:
expected = set(expected_names)
if observed != expected:
missing = sorted(expected - observed)
unexpected = sorted(observed - expected)
raise ValueError(
"Registered MCP tools do not match the requested reference surface: "
f"missing={missing!r}, unexpected={unexpected!r}"
)
surface_rank = {"read": 0, "proposal": 1, "application": 2}
return tuple(sorted(references, key=lambda item: (surface_rank[item.surface], item.name)))
def render_cli_reference_markdown(references: Iterable[CliCommandReference]) -> str:
"""Render a deterministic Markdown table for CLI commands."""
rows = sorted(references, key=lambda item: item.name)
lines = [
"| Command | Invocation |",
"|---|---|",
]
lines.extend(
f"| `{_escape_markdown(item.name)}` | `{_escape_markdown(item.invocation)}` |"
for item in rows
)
return "\n".join(lines) + "\n"
def render_mcp_reference_markdown(references: Iterable[McpToolReference]) -> str:
"""Render a deterministic Markdown table for registered MCP tools."""
surface_rank = {"read": 0, "proposal": 1, "application": 2}
rows = sorted(references, key=lambda item: (surface_rank[item.surface], item.name))
lines = [
"| Surface | Tool | Required arguments | Optional arguments | "
"Input schema SHA-256 | Description |",
"|---|---|---|---|---|---|",
]
lines.extend(
"| "
f"{_escape_markdown(item.surface)} | "
f"`{_escape_markdown(item.name)}` | "
f"{_argument_list(item.required_arguments)} | "
f"{_argument_list(item.optional_arguments)} | "
f"`{item.input_schema_hash}` | "
f"{_escape_markdown(item.description) or ''} |"
for item in rows
)
return "\n".join(lines) + "\n"
def render_command_reference_markdown(
cli_references: Iterable[CliCommandReference],
mcp_references: Iterable[McpToolReference],
) -> str:
"""Render the complete deterministic CLI and MCP command reference."""
return (
"# DocForge command reference\n\n"
"> Generated from the live CLI parser and MCP registrations. Do not edit this file "
"by hand.\n\n"
"Global CLI options are documented in `docforge --help` and are not repeated in each "
"command row. The MCP table is the complete generic application-enabled surface; "
"the fixed reference-adapter server exposes only its read rows.\n\n"
"## CLI commands\n\n"
f"{render_cli_reference_markdown(cli_references)}"
"\n## MCP tools\n\n"
f"{render_mcp_reference_markdown(mcp_references)}"
)
def _docforge_parser() -> argparse.ArgumentParser:
from .cli import _parser # pyright: ignore[reportPrivateUsage]
return _parser()
def _command_parsers(
parser: argparse.ArgumentParser,
) -> Mapping[str, argparse.ArgumentParser]:
actions = parser._actions # pyright: ignore[reportPrivateUsage]
subparsers = [
action
for action in actions
if action.dest == "command" and isinstance(action.choices, dict)
]
if len(subparsers) != 1:
raise ValueError("CLI parser must define exactly one command subparser")
return cast(Mapping[str, argparse.ArgumentParser], subparsers[0].choices)
def _mcp_surface_by_name() -> dict[str, str]:
from .mcp_server import APPLICATION_TOOLS, PROPOSAL_TOOLS, READ_TOOLS
groups = {
"read": READ_TOOLS,
"proposal": PROPOSAL_TOOLS,
"application": APPLICATION_TOOLS,
}
surface_by_name: dict[str, str] = {}
for surface, names in groups.items():
for name in names:
if name in surface_by_name:
raise ValueError(f"MCP tool surface declaration repeats {name!r}")
surface_by_name[name] = surface
return surface_by_name
def _normalize_usage(usage: str) -> str:
return _normalize_text(usage.removeprefix("usage: "))
def _normalize_text(value: str) -> str:
return " ".join(value.split())
def _schema_properties(schema: Mapping[str, Any]) -> dict[str, Any]:
raw_properties: object = schema.get("properties", {})
if not isinstance(raw_properties, dict):
raise ValueError("MCP tool input schema properties must be a string-keyed object")
properties: dict[str, Any] = {}
for name, value in cast(dict[object, object], raw_properties).items():
if not isinstance(name, str):
raise ValueError("MCP tool input schema properties must be a string-keyed object")
properties[name] = value
return properties
def _schema_required(schema: Mapping[str, Any], properties: Mapping[str, Any]) -> set[str]:
raw_required: object = schema.get("required", [])
if not isinstance(raw_required, list):
raise ValueError("MCP tool input schema required arguments must be strings")
required_names: set[str] = set()
for name in cast(list[object], raw_required):
if not isinstance(name, str):
raise ValueError("MCP tool input schema required arguments must be strings")
required_names.add(name)
if not required_names <= set(properties):
raise ValueError("MCP tool input schema requires an undeclared argument")
return required_names
def _canonical_hash(payload: Mapping[str, Any]) -> str:
encoded = json.dumps(payload, sort_keys=True, separators=(",", ":")).encode()
return hashlib.sha256(encoded).hexdigest()
def _argument_list(arguments: tuple[str, ...]) -> str:
if not arguments:
return ""
return ", ".join(f"`{_escape_markdown(argument)}`" for argument in arguments)
def _escape_markdown(value: str) -> str:
return value.replace("\\", "\\\\").replace("|", "\\|").replace("`", "\\`")

View file

@ -35,92 +35,83 @@ def _profile(snapshot: ProjectSnapshot, profile_id: str) -> ContextProfile:
def compile_context(
index: ProjectIndex, profile_id: str, budget: int | None = None
) -> dict[str, object]:
checked = index.check()
snapshot = index.project.load()
if snapshot.source_hash != checked["source_hash"] or snapshot.revision != checked["revision"]:
raise DocForgeError("source_changed", "Canonical source changed before context selection")
profile = _profile(snapshot, profile_id)
selected_budget = profile.token_budget if budget is None else budget
if (
isinstance(selected_budget, bool)
or selected_budget < 1
or selected_budget > snapshot.descriptor.limits.max_context_tokens
):
raise DocForgeError("invalid_budget", "Context budget is outside the configured range")
def select(snapshot: ProjectSnapshot) -> dict[str, object]:
profile = _profile(snapshot, profile_id)
selected_budget = profile.token_budget if budget is None else budget
if (
isinstance(selected_budget, bool)
or selected_budget < 1
or selected_budget > snapshot.descriptor.limits.max_context_tokens
):
raise DocForgeError("invalid_budget", "Context budget is outside the configured range")
node_by_id = {node.node_id: node for node in snapshot.nodes}
dependency_edges = {
node_id: tuple(
edge.target_id
for edge in snapshot.edges
if edge.source_id == node_id and edge.relation == "depends_on"
)
for node_id in node_by_id
}
reasons: dict[str, str] = {node_id: "required by profile" for node_id in profile.required_nodes}
queue = deque((node_id, 0) for node_id in profile.required_nodes)
while queue:
node_id, depth = queue.popleft()
if depth >= profile.dependency_depth:
continue
for dependency in dependency_edges[node_id]:
if dependency not in reasons:
reasons[dependency] = f"dependency of {node_id}"
queue.append((dependency, depth + 1))
node_by_id = {node.node_id: node for node in snapshot.nodes}
dependency_lists: dict[str, list[str]] = {node_id: [] for node_id in node_by_id}
for edge in snapshot.edges:
if edge.relation == "depends_on":
dependency_lists[edge.source_id].append(edge.target_id)
dependency_edges = {
node_id: tuple(targets) for node_id, targets in dependency_lists.items()
}
reasons: dict[str, str] = {
node_id: "required by profile" for node_id in profile.required_nodes
}
queue = deque((node_id, 0) for node_id in profile.required_nodes)
while queue:
node_id, depth = queue.popleft()
if depth >= profile.dependency_depth:
continue
for dependency in dependency_edges[node_id]:
if dependency not in reasons:
reasons[dependency] = f"dependency of {node_id}"
queue.append((dependency, depth + 1))
eligible = [
node
for node in snapshot.nodes
if (not profile.families or node.family in profile.families)
and (not profile.statuses or node.status in profile.statuses)
]
ordered_ids = [*profile.required_nodes]
ordered_ids.extend(sorted(set(reasons) - set(ordered_ids)))
ordered_ids.extend(node.node_id for node in eligible if node.node_id not in reasons)
eligible = [
node
for node in snapshot.nodes
if (not profile.families or node.family in profile.families)
and (not profile.statuses or node.status in profile.statuses)
]
ordered_ids = [*profile.required_nodes]
ordered_ids.extend(sorted(set(reasons) - set(ordered_ids)))
ordered_ids.extend(node.node_id for node in eligible if node.node_id not in reasons)
entries: list[ContextEntry] = []
omissions: list[dict[str, str]] = []
used_tokens = 0
required = set(profile.required_nodes)
for node_id in ordered_ids:
node = node_by_id[node_id]
text = _node_text(node)
tokens = _estimate_tokens(text)
if used_tokens + tokens > selected_budget:
if node_id in required:
raise DocForgeError(
"budget_too_small",
"Context budget cannot contain every required node",
entries: list[ContextEntry] = []
omissions: list[dict[str, str]] = []
used_tokens = 0
required = set(profile.required_nodes)
for node_id in ordered_ids:
node = node_by_id[node_id]
text = _node_text(node)
tokens = _estimate_tokens(text)
if used_tokens + tokens > selected_budget:
if node_id in required:
raise DocForgeError(
"budget_too_small",
"Context budget cannot contain every required node",
node_id=node_id,
required_tokens=used_tokens + tokens,
)
omissions.append({"node_id": node_id, "reason": "token budget"})
continue
entries.append(
ContextEntry(
node_id=node_id,
required_tokens=used_tokens + tokens,
reason=reasons.get(node_id, "eligible profile node"),
estimated_tokens=tokens,
source_path=node.source_path,
content_hash=node.content_hash,
text=text,
)
omissions.append({"node_id": node_id, "reason": "token budget"})
continue
entries.append(
ContextEntry(
node_id=node_id,
reason=reasons.get(node_id, "eligible profile node"),
estimated_tokens=tokens,
source_path=node.source_path,
content_hash=node.content_hash,
text=text,
)
)
used_tokens += tokens
used_tokens += tokens
after = index.check()
if after["source_hash"] != checked["source_hash"] or after["revision"] != checked["revision"]:
raise DocForgeError("source_changed", "Canonical source changed during context selection")
return {
"status": "ok",
"project_id": checked["project_id"],
"project_root_fingerprint": checked["project_root_fingerprint"],
"revision": checked["revision"],
"source_hash": checked["source_hash"],
"adapter": checked["adapter"],
"profile": profile.profile_id,
"budget": selected_budget,
"estimated_tokens": used_tokens,
"entries": [entry.as_dict() for entry in entries],
"omissions": omissions,
}
return {
"profile": profile.profile_id,
"budget": selected_budget,
"estimated_tokens": used_tokens,
"entries": [entry.as_dict() for entry in entries],
"omissions": omissions,
}
return index.read_project_snapshot(select)

1304
src/docforge/doctor.py Normal file

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,911 @@
"""Bounded latest-generation transition receipts for disposable graph indexes."""
from __future__ import annotations
import json
import os
import secrets
import stat
from collections.abc import Mapping
from contextlib import suppress
from dataclasses import dataclass
from pathlib import Path
from typing import Literal, cast
from ._fs_safety import open_bound_directory, require_bound_directory
from .errors import DocForgeError
from .models import Edge, Node, ProjectDescriptor, ProjectSnapshot
from .pagination import canonical_hash
GENERATION_DIFF_SCHEMA_VERSION = 1
GENERATION_DIFF_SEMANTICS_VERSION = 1
MAX_GENERATION_DIFF_ITEMS = 1_000
MAX_GENERATION_DIFF_BYTES = 1_048_576
GENERATION_DIFF_FILENAME = "generation-diff.json"
IndexSignature = tuple[int, int, int, int, int]
PredecessorReason = Literal[
"no_predecessor",
"predecessor_unsafe",
"predecessor_unsupported_schema",
"predecessor_foreign",
"predecessor_policy_incompatible",
"predecessor_corrupt",
"predecessor_unattested",
"predecessor_changed",
"no_meaningful_transition",
]
_GENERATION_KEYS = frozenset(
{
"revision",
"source_hash",
"node_count",
"node_hash",
"edge_count",
"edge_hash",
"index_schema_version",
}
)
_SUMMARY_KEYS = frozenset(
{
"nodes_added",
"nodes_removed",
"nodes_changed",
"edges_added",
"edges_removed",
"total_changes",
}
)
_SIGNATURE_KEYS = frozenset({"device", "inode", "size", "mtime_ns", "ctime_ns"})
_RECEIPT_KEYS = frozenset(
{
"schema_version",
"diff_semantics_version",
"project_id",
"project_root_fingerprint",
"adapter",
"kind",
"reason",
"from_generation",
"to_generation",
"summary",
"items",
"full_item_count",
"retained_item_count",
"details_truncated",
"truncation_reason",
"full_collection_hash",
"retained_collection_hash",
"index_signature",
"receipt_hash",
}
)
_BASELINE_REASONS = frozenset(
{
"no_predecessor",
"predecessor_unsafe",
"predecessor_unsupported_schema",
"predecessor_foreign",
"predecessor_policy_incompatible",
"predecessor_corrupt",
"predecessor_unattested",
"predecessor_changed",
"no_meaningful_transition",
}
)
_NODE_ITEM_KEYS = frozenset(
{
"entity",
"change",
"node_id",
"before_content_hash",
"after_content_hash",
"before_source_path",
"after_source_path",
"before_node_hash",
"after_node_hash",
"changed_fields",
"item_hash",
}
)
_NODE_CHANGED_FIELDS = frozenset(
{
"title",
"family",
"authority",
"status",
"tags",
"summary",
"content",
"source_path",
"source_anchor",
"content_hash",
}
)
_EDGE_ITEM_KEYS = frozenset(
{
"entity",
"change",
"source_id",
"relation",
"target_id",
"item_hash",
}
)
@dataclass(frozen=True)
class PublishedGraph:
"""One completely verified predecessor publication."""
generation: dict[str, object]
nodes: tuple[Node, ...]
edges: tuple[Edge, ...]
logic_hash: str
signature: IndexSignature
@dataclass(frozen=True)
class GenerationDiffDraft:
"""A receipt body awaiting the committed index file identity."""
fields: dict[str, object]
items: tuple[dict[str, object], ...]
def generation_diff_path(descriptor: ProjectDescriptor) -> Path:
"""Return the fixed project-confined latest-transition receipt path."""
return descriptor.cache_root / GENERATION_DIFF_FILENAME
def index_signature(path: Path) -> IndexSignature:
"""Return the exact identity of one safe regular index publication."""
try:
status = path.lstat()
except OSError as error:
raise DocForgeError("missing_index", "Derived index does not exist") from error
if stat.S_ISLNK(status.st_mode) or not stat.S_ISREG(status.st_mode):
raise DocForgeError("path_escape", "Derived index path is not a safe regular file")
return (
status.st_dev,
status.st_ino,
status.st_size,
status.st_mtime_ns,
status.st_ctime_ns,
)
def signature_payload(signature: IndexSignature) -> dict[str, int]:
"""Convert one stat identity to its versioned JSON representation."""
return {
"device": signature[0],
"inode": signature[1],
"size": signature[2],
"mtime_ns": signature[3],
"ctime_ns": signature[4],
}
def generation_identity(status: Mapping[str, object]) -> dict[str, object]:
"""Select the primary-graph identity stored in a public diff receipt."""
return {
"revision": status["revision"],
"source_hash": status["source_hash"],
"node_count": status["node_count"],
"node_hash": status["node_hash"],
"edge_count": status["edge_count"],
"edge_hash": status["edge_hash"],
"index_schema_version": status["index_schema_version"],
}
def prepare_generation_diff(
descriptor: ProjectDescriptor,
*,
predecessor: PublishedGraph | None,
predecessor_reason: PredecessorReason | None,
current_snapshot: ProjectSnapshot,
current_status: Mapping[str, object],
preserved_receipt: Mapping[str, object] | None = None,
) -> GenerationDiffDraft:
"""Prepare one deterministic latest transition without publishing it."""
current_generation = generation_identity(current_status)
if predecessor is None:
return _baseline_draft(
descriptor,
current_generation,
predecessor_reason or "no_predecessor",
)
if predecessor.generation == current_generation:
if preserved_receipt is not None and _receipt_targets(
preserved_receipt,
descriptor,
current_generation,
predecessor.signature,
):
fields = {
key: value
for key, value in preserved_receipt.items()
if key not in {"index_signature", "receipt_hash", "items"}
}
preserved_items = preserved_receipt.get("items")
if not isinstance(preserved_items, list):
raise DocForgeError(
"invalid_generation_diff",
"Preserved generation diff has no item collection",
)
items = tuple(
cast(dict[str, object], item)
for item in cast(list[object], preserved_items)
if isinstance(item, dict)
)
return GenerationDiffDraft(fields=fields, items=items)
return _baseline_draft(
descriptor,
current_generation,
"no_meaningful_transition",
)
items = _change_items(
predecessor.nodes,
predecessor.edges,
current_snapshot.nodes,
current_snapshot.edges,
)
summary = _summary(items)
return GenerationDiffDraft(
fields={
"schema_version": GENERATION_DIFF_SCHEMA_VERSION,
"diff_semantics_version": GENERATION_DIFF_SEMANTICS_VERSION,
"project_id": descriptor.project_id,
"project_root_fingerprint": _root_fingerprint(descriptor),
"adapter": descriptor.adapter,
"kind": "transition",
"reason": None,
"from_generation": predecessor.generation,
"to_generation": current_generation,
"summary": summary,
"full_item_count": len(items),
"full_collection_hash": canonical_hash([item["item_hash"] for item in items]),
},
items=items,
)
def finalize_generation_diff(
draft: GenerationDiffDraft,
*,
signature: IndexSignature,
) -> dict[str, object]:
"""Bind a draft to the committed index and enforce fixed receipt ceilings."""
full_count = cast(int, draft.fields["full_item_count"])
retained = list(draft.items[:MAX_GENERATION_DIFF_ITEMS])
item_limited = full_count > len(retained)
byte_limited = draft.fields.get("truncation_reason") == "receipt_byte_limit"
while True:
receipt = _final_receipt(
draft.fields,
retained,
signature=signature,
item_limited=item_limited,
byte_limited=byte_limited,
)
if len(_receipt_bytes(receipt)) <= MAX_GENERATION_DIFF_BYTES:
return receipt
if not retained:
raise DocForgeError(
"generation_diff_failure",
"Generation diff identity exceeds the fixed receipt byte limit",
)
retained.pop()
byte_limited = True
def publish_generation_diff(
descriptor: ProjectDescriptor,
receipt: Mapping[str, object],
) -> None:
"""Atomically replace the single latest-generation receipt."""
root = descriptor.cache_root
if generation_diff_path(descriptor).parent != root:
raise DocForgeError("path_escape", "Generation diff path is not confined")
raw = _receipt_bytes(receipt)
if len(raw) > MAX_GENERATION_DIFF_BYTES:
raise DocForgeError(
"generation_diff_failure",
"Generation diff receipt exceeds the fixed byte limit",
)
root_fd = open_bound_directory(root)
temporary_name = f".generation-diff-{secrets.token_hex(12)}"
temporary_created = False
try:
try:
status = os.stat(
GENERATION_DIFF_FILENAME,
dir_fd=root_fd,
follow_symlinks=False,
)
except FileNotFoundError:
status = None
if status is not None and (
stat.S_ISLNK(status.st_mode) or not stat.S_ISREG(status.st_mode)
):
raise DocForgeError(
"path_escape",
"Generation diff receipt path is not a safe regular file",
)
descriptor_fd = os.open(
temporary_name,
os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW,
0o600,
dir_fd=root_fd,
)
temporary_created = True
with os.fdopen(descriptor_fd, "wb") as handle:
handle.write(raw)
handle.flush()
os.fsync(handle.fileno())
require_bound_directory(root, root_fd)
os.replace(
temporary_name,
GENERATION_DIFF_FILENAME,
src_dir_fd=root_fd,
dst_dir_fd=root_fd,
)
temporary_created = False
os.fsync(root_fd)
require_bound_directory(root, root_fd)
except Exception:
if temporary_created:
with suppress(OSError):
os.unlink(temporary_name, dir_fd=root_fd)
raise
finally:
os.close(root_fd)
def load_generation_diff(
descriptor: ProjectDescriptor,
) -> tuple[dict[str, object] | None, str | None, IndexSignature | None]:
"""Read and validate one receipt without repairing any derived state."""
try:
root_fd = open_bound_directory(descriptor.cache_root)
except DocForgeError as error:
reason = "missing_receipt" if error.code == "missing_index" else "unsafe_receipt"
return None, reason, None
try:
try:
descriptor_fd = os.open(
GENERATION_DIFF_FILENAME,
os.O_RDONLY | os.O_NOFOLLOW,
dir_fd=root_fd,
)
except FileNotFoundError:
return None, "missing_receipt", None
except OSError:
return None, "unsafe_receipt", None
with os.fdopen(descriptor_fd, "rb") as handle:
status_before = os.fstat(handle.fileno())
if stat.S_ISLNK(status_before.st_mode) or not stat.S_ISREG(status_before.st_mode):
return None, "unsafe_receipt", None
before: IndexSignature = (
status_before.st_dev,
status_before.st_ino,
status_before.st_size,
status_before.st_mtime_ns,
status_before.st_ctime_ns,
)
if before[2] > MAX_GENERATION_DIFF_BYTES:
return None, "oversized_receipt", before
raw = handle.read(MAX_GENERATION_DIFF_BYTES + 1)
status_after = os.fstat(handle.fileno())
after: IndexSignature = (
status_after.st_dev,
status_after.st_ino,
status_after.st_size,
status_after.st_mtime_ns,
status_after.st_ctime_ns,
)
if len(raw) > MAX_GENERATION_DIFF_BYTES:
return None, "oversized_receipt", before
try:
path_status = os.stat(
GENERATION_DIFF_FILENAME,
dir_fd=root_fd,
follow_symlinks=False,
)
except OSError:
return None, "receipt_changed", before
path_signature: IndexSignature = (
path_status.st_dev,
path_status.st_ino,
path_status.st_size,
path_status.st_mtime_ns,
path_status.st_ctime_ns,
)
try:
require_bound_directory(descriptor.cache_root, root_fd)
except DocForgeError:
return None, "unsafe_receipt", before
finally:
os.close(root_fd)
if before != after or before != path_signature or len(raw) != before[2]:
return None, "receipt_changed", before
try:
parsed: object = json.loads(raw.decode("utf-8"))
except (UnicodeDecodeError, json.JSONDecodeError):
return None, "corrupt_receipt", before
if not isinstance(parsed, dict):
return None, "corrupt_receipt", before
receipt = cast(dict[str, object], parsed)
if not validate_generation_diff_receipt(receipt):
return None, "corrupt_receipt", before
if (
receipt.get("project_id") != descriptor.project_id
or receipt.get("project_root_fingerprint") != _root_fingerprint(descriptor)
or receipt.get("adapter") != descriptor.adapter
):
return None, "foreign_receipt", before
return receipt, None, before
def valid_generation_identity(value: object) -> bool:
"""Return whether a primary-graph generation has the exact version-1 shape."""
return _valid_generation(value)
def validate_generation_diff_receipt(
receipt: Mapping[str, object],
*,
descriptor: ProjectDescriptor | None = None,
) -> bool:
"""Strictly validate the complete version-1 receipt and its hashes."""
if frozenset(receipt) != _RECEIPT_KEYS:
return False
if (
receipt.get("schema_version") != GENERATION_DIFF_SCHEMA_VERSION
or receipt.get("diff_semantics_version") != GENERATION_DIFF_SEMANTICS_VERSION
or not _nonempty(receipt.get("project_id"))
or not _fingerprint(receipt.get("project_root_fingerprint"))
or not _nonempty(receipt.get("adapter"))
):
return False
if descriptor is not None and (
receipt["project_id"] != descriptor.project_id
or receipt["project_root_fingerprint"] != _root_fingerprint(descriptor)
or receipt["adapter"] != descriptor.adapter
):
return False
kind = receipt.get("kind")
reason = receipt.get("reason")
from_generation = receipt.get("from_generation")
if kind == "baseline":
if reason not in _BASELINE_REASONS or from_generation is not None:
return False
elif kind == "transition":
if (
reason is not None
or not _valid_generation(from_generation)
or from_generation == receipt.get("to_generation")
):
return False
else:
return False
if not _valid_generation(receipt.get("to_generation")):
return False
summary_value = receipt.get("summary")
items_value = receipt.get("items")
if not isinstance(summary_value, dict) or not isinstance(items_value, list):
return False
summary = cast(dict[str, object], summary_value)
items = cast(list[object], items_value)
if frozenset(summary) != _SUMMARY_KEYS or any(
type(value) is not int or value < 0 for value in summary.values()
):
return False
total = sum(
cast(int, summary[key])
for key in (
"nodes_added",
"nodes_removed",
"nodes_changed",
"edges_added",
"edges_removed",
)
)
if summary.get("total_changes") != total:
return False
if kind == "baseline" and (total != 0 or items):
return False
if len(items) > MAX_GENERATION_DIFF_ITEMS or not all(_valid_item(item) for item in items):
return False
typed_items = tuple(cast(dict[str, object], item) for item in items)
if list(typed_items) != sorted(typed_items, key=_item_sort_key):
return False
identities = tuple(_item_identity(item) for item in typed_items)
if len(identities) != len(set(identities)):
return False
retained_summary = _summary(typed_items)
if any(
retained_summary[key] > cast(int, summary[key])
for key in (
"nodes_added",
"nodes_removed",
"nodes_changed",
"edges_added",
"edges_removed",
)
):
return False
full_count = receipt.get("full_item_count")
retained_count = receipt.get("retained_item_count")
truncated = receipt.get("details_truncated")
truncation_reason = receipt.get("truncation_reason")
if (
type(full_count) is not int
or type(retained_count) is not int
or full_count != total
or retained_count != len(items)
or retained_count > full_count
or type(truncated) is not bool
or truncated != (retained_count < full_count)
):
return False
if truncation_reason is None:
if truncated or full_count > MAX_GENERATION_DIFF_ITEMS:
return False
elif truncation_reason == "receipt_item_limit":
if (
not truncated
or full_count <= MAX_GENERATION_DIFF_ITEMS
or retained_count != MAX_GENERATION_DIFF_ITEMS
):
return False
elif truncation_reason == "receipt_byte_limit":
if not truncated or retained_count >= min(full_count, MAX_GENERATION_DIFF_ITEMS):
return False
else:
return False
item_hashes = [cast(dict[str, object], item)["item_hash"] for item in items]
retained_hash = canonical_hash(item_hashes)
if receipt.get("retained_collection_hash") != retained_hash:
return False
full_hash = receipt.get("full_collection_hash")
if not _sha256(full_hash):
return False
if not truncated and full_hash != retained_hash:
return False
signature_value = receipt.get("index_signature")
if not isinstance(signature_value, dict):
return False
signature = cast(dict[str, object], signature_value)
if frozenset(signature) != _SIGNATURE_KEYS:
return False
if any(type(value) is not int or value < 0 for value in signature.values()):
return False
receipt_hash = receipt.get("receipt_hash")
if not _sha256(receipt_hash):
return False
unhashed = {key: value for key, value in receipt.items() if key != "receipt_hash"}
return receipt_hash == canonical_hash(unhashed)
def _baseline_draft(
descriptor: ProjectDescriptor,
current_generation: Mapping[str, object],
reason: PredecessorReason,
) -> GenerationDiffDraft:
empty_hash = canonical_hash([])
return GenerationDiffDraft(
fields={
"schema_version": GENERATION_DIFF_SCHEMA_VERSION,
"diff_semantics_version": GENERATION_DIFF_SEMANTICS_VERSION,
"project_id": descriptor.project_id,
"project_root_fingerprint": _root_fingerprint(descriptor),
"adapter": descriptor.adapter,
"kind": "baseline",
"reason": reason,
"from_generation": None,
"to_generation": dict(current_generation),
"summary": {
"nodes_added": 0,
"nodes_removed": 0,
"nodes_changed": 0,
"edges_added": 0,
"edges_removed": 0,
"total_changes": 0,
},
"full_item_count": 0,
"full_collection_hash": empty_hash,
},
items=(),
)
def _change_items(
before_nodes: tuple[Node, ...],
before_edges: tuple[Edge, ...],
after_nodes: tuple[Node, ...],
after_edges: tuple[Edge, ...],
) -> tuple[dict[str, object], ...]:
before_by_id = {node.node_id: node for node in before_nodes}
after_by_id = {node.node_id: node for node in after_nodes}
items: list[dict[str, object]] = []
for node_id in sorted(before_by_id.keys() | after_by_id.keys()):
before = before_by_id.get(node_id)
after = after_by_id.get(node_id)
if before == after:
continue
if before is None:
change = "added"
elif after is None:
change = "removed"
else:
change = "changed"
before_payload = before.as_dict() if before is not None else None
after_payload = after.as_dict() if after is not None else None
changed_fields = (
[]
if before_payload is None or after_payload is None
else sorted(key for key in before_payload if before_payload[key] != after_payload[key])
)
payload: dict[str, object] = {
"entity": "node",
"change": change,
"node_id": node_id,
"before_content_hash": None if before is None else before.content_hash,
"after_content_hash": None if after is None else after.content_hash,
"before_source_path": None if before is None else before.source_path,
"after_source_path": None if after is None else after.source_path,
"before_node_hash": (
None if before_payload is None else canonical_hash(before_payload)
),
"after_node_hash": (None if after_payload is None else canonical_hash(after_payload)),
"changed_fields": changed_fields,
}
payload["item_hash"] = canonical_hash(payload)
items.append(payload)
before_edge_set = {(edge.source_id, edge.relation, edge.target_id) for edge in before_edges}
after_edge_set = {(edge.source_id, edge.relation, edge.target_id) for edge in after_edges}
for change, values in (
("removed", sorted(before_edge_set - after_edge_set)),
("added", sorted(after_edge_set - before_edge_set)),
):
for source_id, relation, target_id in values:
payload = {
"entity": "edge",
"change": change,
"source_id": source_id,
"relation": relation,
"target_id": target_id,
}
payload["item_hash"] = canonical_hash(payload)
items.append(payload)
return tuple(sorted(items, key=_item_sort_key))
def _item_sort_key(item: Mapping[str, object]) -> tuple[str, str, str, str, str]:
return (
cast(str, item["entity"]),
cast(str, item.get("node_id", item.get("source_id", ""))),
cast(str, item.get("relation", "")),
cast(str, item.get("target_id", "")),
cast(str, item["change"]),
)
def _item_identity(item: Mapping[str, object]) -> tuple[str, ...]:
if item["entity"] == "node":
return ("node", cast(str, item["node_id"]))
return (
"edge",
cast(str, item["source_id"]),
cast(str, item["relation"]),
cast(str, item["target_id"]),
)
def _summary(items: tuple[dict[str, object], ...]) -> dict[str, int]:
result = {
"nodes_added": 0,
"nodes_removed": 0,
"nodes_changed": 0,
"edges_added": 0,
"edges_removed": 0,
"total_changes": len(items),
}
for item in items:
entity = cast(str, item["entity"])
change = cast(str, item["change"])
key = f"{entity}s_{change}"
result[key] += 1
return result
def _final_receipt(
fields: Mapping[str, object],
retained: list[dict[str, object]],
*,
signature: IndexSignature,
item_limited: bool,
byte_limited: bool,
) -> dict[str, object]:
full_count = cast(int, fields["full_item_count"])
truncated = len(retained) < full_count
if byte_limited:
reason: str | None = "receipt_byte_limit"
elif item_limited:
reason = "receipt_item_limit"
else:
reason = None
receipt = {
**fields,
"items": retained,
"retained_item_count": len(retained),
"details_truncated": truncated,
"truncation_reason": reason,
"retained_collection_hash": canonical_hash([item["item_hash"] for item in retained]),
"index_signature": signature_payload(signature),
}
receipt["receipt_hash"] = canonical_hash(receipt)
return receipt
def _receipt_targets(
receipt: Mapping[str, object],
descriptor: ProjectDescriptor,
generation: Mapping[str, object],
signature: IndexSignature,
) -> bool:
return (
validate_generation_diff_receipt(receipt, descriptor=descriptor)
and receipt.get("to_generation") == dict(generation)
and receipt.get("index_signature") == signature_payload(signature)
)
def _valid_generation(value: object) -> bool:
if not isinstance(value, dict):
return False
generation = cast(dict[str, object], value)
if frozenset(generation) != _GENERATION_KEYS:
return False
source_hash = generation.get("source_hash")
node_hash = generation.get("node_hash")
edge_hash = generation.get("edge_hash")
return (
_nonempty(generation.get("revision"))
and _sha256(source_hash)
and _sha256(node_hash)
and _sha256(edge_hash)
and type(generation.get("node_count")) is int
and cast(int, generation["node_count"]) >= 0
and type(generation.get("edge_count")) is int
and cast(int, generation["edge_count"]) >= 0
and type(generation.get("index_schema_version")) is int
and cast(int, generation["index_schema_version"]) >= 1
)
def _valid_item(value: object) -> bool:
if not isinstance(value, dict):
return False
payload = cast(dict[str, object], value)
entity = payload.get("entity")
if entity == "node":
if frozenset(payload) != _NODE_ITEM_KEYS:
return False
changed_fields_value = payload.get("changed_fields")
if not isinstance(changed_fields_value, list):
return False
changed_fields = cast(list[object], changed_fields_value)
if (
payload.get("change") not in {"added", "removed", "changed"}
or not _nonempty(payload.get("node_id"))
or any(
not _nonempty(field) or field not in _NODE_CHANGED_FIELDS
for field in changed_fields
)
or changed_fields != sorted(set(cast(list[str], changed_fields)))
):
return False
for key in (
"before_content_hash",
"after_content_hash",
"before_node_hash",
"after_node_hash",
):
candidate = payload.get(key)
if candidate is not None and not _sha256(candidate):
return False
for key in ("before_source_path", "after_source_path"):
candidate = payload.get(key)
if candidate is not None and not _nonempty(candidate):
return False
change = payload["change"]
before_values = (
payload["before_content_hash"],
payload["before_source_path"],
payload["before_node_hash"],
)
after_values = (
payload["after_content_hash"],
payload["after_source_path"],
payload["after_node_hash"],
)
if change == "added":
if (
any(value is not None for value in before_values)
or not all(value is not None for value in after_values)
or changed_fields
):
return False
elif change == "removed":
if (
not all(value is not None for value in before_values)
or any(value is not None for value in after_values)
or changed_fields
):
return False
elif (
not all(value is not None for value in (*before_values, *after_values))
or not changed_fields
or payload["before_node_hash"] == payload["after_node_hash"]
):
return False
elif entity == "edge":
if (
frozenset(payload) != _EDGE_ITEM_KEYS
or payload.get("change") not in {"added", "removed"}
or not _nonempty(payload.get("source_id"))
or not _nonempty(payload.get("relation"))
or not _nonempty(payload.get("target_id"))
):
return False
else:
return False
item_hash = payload.get("item_hash")
unhashed = {key: item for key, item in payload.items() if key != "item_hash"}
return _sha256(item_hash) and item_hash == canonical_hash(unhashed)
def _receipt_bytes(receipt: Mapping[str, object]) -> bytes:
return json.dumps(receipt, sort_keys=True, indent=2, ensure_ascii=True).encode("utf-8") + b"\n"
def _root_fingerprint(descriptor: ProjectDescriptor) -> str:
from .project import project_root_fingerprint
return project_root_fingerprint(descriptor.root)
def _nonempty(value: object) -> bool:
return isinstance(value, str) and bool(value)
def _sha256(value: object) -> bool:
return (
isinstance(value, str)
and len(value) == 64
and all(character in "0123456789abcdef" for character in value)
)
def _fingerprint(value: object) -> bool:
return (
isinstance(value, str)
and len(value) == 16
and all(character in "0123456789abcdef" for character in value)
)

View file

@ -0,0 +1,525 @@
"""Pure, bounded portable-graph planning over one immutable graph generation."""
from __future__ import annotations
import re
from dataclasses import dataclass
from typing import Literal, cast
from .errors import DocForgeError
from .models import Edge, Node, ProjectSnapshot
from .project import project_root_fingerprint
from .projection_contract import (
MAX_GRAPH_VIEW_DEPTH,
MAX_GRAPH_VIEW_EDGES,
MAX_GRAPH_VIEW_FILTERS,
MAX_GRAPH_VIEW_NODES,
MAX_GRAPH_VIEW_QUERY_CHARS,
MAX_GRAPH_VIEW_STRING_CHARS,
MAX_GRAPH_VIEW_WORK,
GraphViewPlanV1,
ProjectionPackageV1,
)
GraphViewMode = Literal["nodes", "flow", "web", "logic"]
_QUERY_TOKEN = re.compile(r"\w+", re.UNICODE)
_DETAIL_FIELDS = (
"node_id",
"title",
"family",
"authority",
"status",
"tags",
"summary",
"content_hash",
)
@dataclass(frozen=True)
class GraphViewRequestV1:
"""One closed, inert graph selection request."""
view_id: str
title: str
root_node_id: str | None = None
query: str | None = None
initial_mode: GraphViewMode = "nodes"
depth: int = 1
max_nodes: int = 100
max_edges: int = 400
max_work: int = 100_000
families: tuple[str, ...] = ()
relations: tuple[str, ...] = ()
authorities: tuple[str, ...] = ()
statuses: tuple[str, ...] = ()
tags: tuple[str, ...] = ()
include_logic: bool = False
@dataclass(frozen=True)
class _ValidatedRequest:
view_id: str
title: str
root_node_id: str | None
query: str | None
initial_mode: GraphViewMode
depth: int
max_nodes: int
max_edges: int
max_work: int
families: tuple[str, ...]
relations: tuple[str, ...]
authorities: tuple[str, ...]
statuses: tuple[str, ...]
tags: tuple[str, ...]
include_logic: bool
def _invalid(message: str, **details: object) -> DocForgeError:
return DocForgeError("invalid_graph_view_request", message, **details)
def _string(value: object, *, field: str, maximum: int = MAX_GRAPH_VIEW_STRING_CHARS) -> str:
if not isinstance(value, str) or not value.strip() or len(value) > maximum or "\0" in value:
raise _invalid("Graph view request string is invalid", field=field)
return value.strip()
def _filter_values(values: object, *, field: str) -> tuple[str, ...]:
if not isinstance(values, tuple):
raise _invalid("Graph view filter is invalid", field=field)
tuple_values = cast(tuple[object, ...], values)
if not all(isinstance(value, str) for value in tuple_values):
raise _invalid("Graph view filter is invalid", field=field)
if len(tuple_values) > MAX_GRAPH_VIEW_FILTERS:
raise _invalid("Graph view filter is invalid", field=field)
normalized = tuple(_string(value, field=field) for value in cast(tuple[str, ...], tuple_values))
if len(normalized) != len(set(normalized)):
raise _invalid("Graph view filter contains duplicates", field=field)
return tuple(sorted(normalized))
def _validated_request(request: GraphViewRequestV1) -> _ValidatedRequest:
view_id = _string(request.view_id, field="view_id")
title = _string(request.title, field="title")
if (request.root_node_id is None) == (request.query is None):
raise _invalid("Choose exactly one exact root or lexical query")
root_node_id = (
None
if request.root_node_id is None
else _string(request.root_node_id, field="root_node_id")
)
query = (
None
if request.query is None
else _string(
request.query,
field="query",
maximum=MAX_GRAPH_VIEW_QUERY_CHARS,
)
)
if request.initial_mode not in {"nodes", "flow", "web", "logic"}:
raise _invalid("Graph view initial mode is unsupported")
if type(request.depth) is not int or not 1 <= request.depth <= MAX_GRAPH_VIEW_DEPTH:
raise _invalid(
"Graph view depth is outside the fixed boundary",
maximum=MAX_GRAPH_VIEW_DEPTH,
)
for field, value, minimum, maximum in (
("max_nodes", request.max_nodes, 1, MAX_GRAPH_VIEW_NODES),
("max_edges", request.max_edges, 0, MAX_GRAPH_VIEW_EDGES),
("max_work", request.max_work, 1, MAX_GRAPH_VIEW_WORK),
):
if type(value) is not int or not minimum <= value <= maximum:
raise _invalid(
"Graph view bound is outside the fixed boundary",
field=field,
minimum=minimum,
maximum=maximum,
)
if type(request.include_logic) is not bool:
raise _invalid("Graph view Logic selection must be Boolean")
return _ValidatedRequest(
view_id=view_id,
title=title,
root_node_id=root_node_id,
query=query,
initial_mode=request.initial_mode,
depth=request.depth,
max_nodes=request.max_nodes,
max_edges=request.max_edges,
max_work=request.max_work,
families=_filter_values(request.families, field="families"),
relations=_filter_values(request.relations, field="relations"),
authorities=_filter_values(request.authorities, field="authorities"),
statuses=_filter_values(request.statuses, field="statuses"),
tags=_filter_values(request.tags, field="tags"),
include_logic=request.include_logic,
)
def _validated_graph(
snapshot: ProjectSnapshot,
) -> tuple[tuple[Node, ...], tuple[Edge, ...], dict[str, Node]]:
nodes = tuple(sorted(snapshot.nodes, key=lambda node: node.node_id))
node_by_id = {node.node_id: node for node in nodes}
if len(node_by_id) != len(nodes):
raise DocForgeError(
"invalid_projection",
"Graph view snapshot contains duplicate node identities",
)
edges = tuple(
sorted(
snapshot.edges,
key=lambda edge: (edge.source_id, edge.relation, edge.target_id),
)
)
edge_keys = {(edge.source_id, edge.relation, edge.target_id) for edge in edges}
if len(edge_keys) != len(edges) or any(
edge.source_id not in node_by_id or edge.target_id not in node_by_id for edge in edges
):
raise DocForgeError(
"invalid_projection",
"Graph view snapshot contains invalid relationships",
)
return nodes, edges, node_by_id
def _eligible(node: Node, request: _ValidatedRequest) -> bool:
return (
(not request.families or node.family in request.families)
and (not request.authorities or node.authority in request.authorities)
and (not request.statuses or node.status in request.statuses)
and (not request.tags or set(request.tags).issubset(node.tags))
)
def _relation_allowed(edge: Edge, request: _ValidatedRequest) -> bool:
return not request.relations or edge.relation in request.relations
def _lexical_text(node: Node) -> str:
return " ".join(
(
node.node_id,
node.title,
node.summary,
node.family,
node.authority,
node.status,
*node.tags,
)
).casefold()
def _lexical_nodes(
nodes: tuple[Node, ...],
request: _ValidatedRequest,
omissions: list[dict[str, object]],
) -> tuple[set[str], int, bool]:
query = request.query
maximum_nodes = request.max_nodes
maximum_work = request.max_work
assert query is not None
terms = tuple(dict.fromkeys(_QUERY_TOKEN.findall(query.casefold())))
if not terms:
raise _invalid("Lexical graph scope contains no searchable text")
selected: set[str] = set()
work = 0
work_limited = False
for node in nodes:
if work >= maximum_work:
work_limited = True
break
work += 1
if not _eligible(node, request) or not all(term in _lexical_text(node) for term in terms):
continue
if len(selected) >= maximum_nodes:
omissions.append(
{
"code": "node_result_limit",
"subject": "nodes",
"limit": maximum_nodes,
"minimum_omitted": 1,
}
)
break
selected.add(node.node_id)
return selected, work, work_limited
def _root_nodes(
edges: tuple[Edge, ...],
node_by_id: dict[str, Node],
request: _ValidatedRequest,
omissions: list[dict[str, object]],
) -> tuple[set[str], int, bool]:
root_node_id = request.root_node_id
maximum_nodes = request.max_nodes
maximum_work = request.max_work
depth = request.depth
assert root_node_id is not None
root = node_by_id.get(root_node_id)
if root is None:
raise DocForgeError(
"missing_node",
"No node has the requested stable ID",
node_id=root_node_id,
)
if not _eligible(root, request):
raise _invalid("Exact graph root is excluded by the closed node filters")
selected = {root_node_id}
frontier = {root_node_id}
work = 0
work_limited = False
node_limited = False
for _ in range(depth):
if not frontier:
break
next_frontier: set[str] = set()
for edge in edges:
if work >= maximum_work:
work_limited = True
break
work += 1
if not _relation_allowed(edge, request):
continue
candidate: str | None = None
if edge.source_id in frontier:
candidate = edge.target_id
elif edge.target_id in frontier:
candidate = edge.source_id
if candidate is None or candidate in selected:
continue
node = node_by_id[candidate]
if not _eligible(node, request):
continue
if len(selected) >= maximum_nodes:
node_limited = True
continue
selected.add(candidate)
next_frontier.add(candidate)
if work_limited:
break
frontier = next_frontier
if node_limited:
omissions.append(
{
"code": "node_result_limit",
"subject": "nodes",
"limit": maximum_nodes,
"minimum_omitted": 1,
}
)
return selected, work, work_limited
def _selected_edges(
edges: tuple[Edge, ...],
selected_ids: set[str],
request: _ValidatedRequest,
*,
initial_work: int,
omissions: list[dict[str, object]],
) -> tuple[list[Edge], int, bool]:
maximum_edges = request.max_edges
maximum_work = request.max_work
selected: list[Edge] = []
work = initial_work
work_limited = False
edge_limited = False
for edge in edges:
if work >= maximum_work:
work_limited = True
break
work += 1
if (
edge.source_id not in selected_ids
or edge.target_id not in selected_ids
or not _relation_allowed(edge, request)
):
continue
if len(selected) >= maximum_edges:
edge_limited = True
break
selected.append(edge)
if edge_limited:
omissions.append(
{
"code": "edge_result_limit",
"subject": "edges",
"limit": maximum_edges,
"minimum_omitted": 1,
}
)
return selected, work, work_limited
def _node_payload(node: Node) -> dict[str, object]:
return {
"node_id": node.node_id,
"title": node.title,
"family": node.family,
"authority": node.authority,
"status": node.status,
"tags": sorted(node.tags),
"summary": node.summary,
"content_hash": node.content_hash,
}
def _edge_payload(edge: Edge) -> dict[str, str]:
return {
"source_id": edge.source_id,
"relation": edge.relation,
"target_id": edge.target_id,
}
def build_graph_view_plan(
snapshot: ProjectSnapshot,
request: GraphViewRequestV1,
allow_logic: bool,
) -> GraphViewPlanV1:
"""Build one deterministic, path-free graph plan without rendering or storage access."""
if type(allow_logic) is not bool:
raise _invalid("Graph view Logic policy must be Boolean")
normalized = _validated_request(request)
nodes, edges, node_by_id = _validated_graph(snapshot)
omissions: list[dict[str, object]] = []
if normalized.root_node_id is not None:
selected_ids, work, work_limited = _root_nodes(
edges,
node_by_id,
normalized,
omissions,
)
scope: dict[str, object] = {
"kind": "exact_root",
"root_node_id": normalized.root_node_id,
"depth": normalized.depth,
}
else:
selected_ids, work, work_limited = _lexical_nodes(
nodes,
normalized,
omissions,
)
scope = {
"kind": "lexical",
"query": normalized.query,
}
selected_edges, work, edge_work_limited = _selected_edges(
edges,
selected_ids,
normalized,
initial_work=work,
omissions=omissions,
)
work_limited = work_limited or edge_work_limited
if work_limited:
omissions.append(
{
"code": "work_limit",
"subject": "selection",
"limit": normalized.max_work,
"examined": work,
"minimum_omitted": 1,
}
)
logic_requested = normalized.include_logic
if logic_requested and not allow_logic:
omissions.append(
{
"code": "logic_forbidden",
"subject": "logic",
"minimum_omitted": 1,
}
)
omissions.sort(key=lambda item: (str(item["code"]), str(item["subject"])))
selected_nodes = [node_by_id[node_id] for node_id in sorted(selected_ids)]
filters: dict[str, list[str]] = {
"families": list(normalized.families),
"relations": list(normalized.relations),
"authorities": list(normalized.authorities),
"statuses": list(normalized.statuses),
"tags": list(normalized.tags),
}
return GraphViewPlanV1.create(
{
"project": {
"project_id": snapshot.descriptor.project_id,
"project_root_fingerprint": project_root_fingerprint(snapshot.descriptor.root),
"adapter": snapshot.descriptor.adapter,
"revision": snapshot.revision,
"source_hash": snapshot.source_hash,
},
"view": {
"view_id": normalized.view_id,
"title": normalized.title,
"initial_mode": normalized.initial_mode,
"scope": scope,
"filters": filters,
"detail_fields": list(_DETAIL_FIELDS),
},
"bounds": {
"depth": normalized.depth,
"max_nodes": normalized.max_nodes,
"max_edges": normalized.max_edges,
"max_work": normalized.max_work,
},
"policy": {
"visibility": "selected_graph_only",
"source_paths": "excluded",
"source_bodies": "excluded",
"database_queries": "forbidden",
"executable_content": "forbidden",
"logic": "allowed" if allow_logic else "forbidden",
"logic_requested": logic_requested,
},
"graph": {
"root_node_id": normalized.root_node_id,
"nodes": [_node_payload(node) for node in selected_nodes],
"edges": [_edge_payload(edge) for edge in selected_edges],
"logic_projections": [],
},
"omissions": omissions,
"diagnostics": {
"selection": scope["kind"],
"returned_nodes": len(selected_nodes),
"returned_edges": len(selected_edges),
"examined_work_units": work,
"truncated": bool(omissions),
"ordering": "node_id;source_id,relation,target_id",
},
}
)
def build_graph_projection_package(
plan: GraphViewPlanV1,
*,
renderer_id: str,
renderer_version: str,
max_output_bytes: int,
) -> ProjectionPackageV1:
"""Bind a graph plan to the fixed portable renderer without adding runtime authority."""
return ProjectionPackageV1.create(
kind="graph",
plan=plan,
renderer={"renderer_id": renderer_id, "renderer_version": renderer_version},
components=[
{"component_id": "graph.portable-document@1"},
{"component_id": "graph.accessible-list@1"},
{"component_id": "graph.relationship-table@1"},
],
assets=[],
output_policy={
"artifact_ids": ["portable-graph.html"],
"max_total_bytes": max_output_bytes,
},
)

View file

@ -0,0 +1,285 @@
"""Strict parsing and confinement for optional portable graph artifacts."""
from __future__ import annotations
from pathlib import Path
from typing import Literal, cast
from .config_validation import (
ID_PATTERN,
confined_path,
positive_int,
require_string,
string_list,
)
from .errors import DocForgeError
from .models import GraphRenderConfig, GraphRenderView, Limits, RenderConfig
from .projection_contract import (
MAX_GRAPH_VIEW_DEPTH,
MAX_GRAPH_VIEW_EDGES,
MAX_GRAPH_VIEW_FILTERS,
MAX_GRAPH_VIEW_NODES,
MAX_GRAPH_VIEW_QUERY_CHARS,
MAX_GRAPH_VIEW_STRING_CHARS,
MAX_GRAPH_VIEW_WORK,
)
_CONFIG_KEYS = frozenset({"output_root", "views"})
_VIEW_KEYS = frozenset(
{
"id",
"renderer",
"output",
"title",
"root",
"query",
"initial_mode",
"depth",
"max_nodes",
"max_edges",
"max_work",
"families",
"relations",
"authorities",
"statuses",
"tags",
"include_logic",
}
)
_MODES = frozenset({"nodes", "flow", "web"})
def _overlaps(first: Path, second: Path) -> bool:
return first == second or first.is_relative_to(second) or second.is_relative_to(first)
def _optional_string(document: dict[str, object], key: str, source: Path) -> str | None:
if key not in document:
return None
return require_string(document, key, source)
def _bounded_string(
document: dict[str, object],
key: str,
source: Path,
*,
maximum: int,
) -> str:
value = require_string(document, key, source)
if len(value) > maximum:
raise DocForgeError("invalid_config", f"{key} exceeds its fixed character limit")
return value
def _bounded_strings(
value: object,
*,
key: str,
source: Path,
) -> tuple[str, ...]:
values = string_list(value, key=key, source=source)
if len(values) > MAX_GRAPH_VIEW_FILTERS or any(
len(item) > MAX_GRAPH_VIEW_STRING_CHARS for item in values
):
raise DocForgeError("invalid_config", f"{key} exceeds its fixed bounds")
return values
def load_graph_render_config(
root: Path,
document: object,
*,
descriptor_path: Path,
content_roots: tuple[Path, ...],
authority_files: tuple[Path, ...],
cache_root: Path,
index_path: Path,
changeset_root: Path,
manual_render: RenderConfig | None,
limits: Limits,
) -> GraphRenderConfig | None:
if document is None:
return None
if not isinstance(document, dict):
raise DocForgeError("invalid_config", "graph_render must be a table")
document = cast(dict[str, object], document)
unknown = sorted(set(document) - _CONFIG_KEYS)
if unknown:
raise DocForgeError("invalid_config", "graph_render has unknown fields", fields=unknown)
output_root = confined_path(
root,
document.get("output_root"),
field="graph_render.output_root",
must_exist=False,
)
protected = [*content_roots, cache_root, changeset_root]
if manual_render is not None:
protected.extend((manual_render.template_root, manual_render.preview_root))
protected.extend(view.output_path for view in manual_render.views)
if any(_overlaps(output_root, path) for path in protected):
raise DocForgeError(
"invalid_config",
"Portable graph output must not overlap canonical or other derived roots",
)
protected_files = (descriptor_path, index_path, *authority_files)
if any(path == output_root or path.is_relative_to(output_root) for path in protected_files):
raise DocForgeError(
"invalid_config",
"Portable graph output overlaps a protected project path",
)
view_values_value = document.get("views")
if not isinstance(view_values_value, list) or not view_values_value:
raise DocForgeError(
"invalid_config",
"graph_render.views must contain at least one view",
)
view_values = cast(list[object], view_values_value)
if len(view_values) > limits.max_render_views:
raise DocForgeError("invalid_config", "graph_render.views exceeds the configured limit")
views: list[GraphRenderView] = []
view_ids: set[str] = set()
outputs: set[Path] = set()
for value in view_values:
if not isinstance(value, dict):
raise DocForgeError("invalid_config", "Each portable graph view must be a table")
view = cast(dict[str, object], value)
unknown_view = sorted(set(view) - _VIEW_KEYS)
if unknown_view:
raise DocForgeError(
"invalid_config",
"Portable graph view has unknown fields",
fields=unknown_view,
)
view_id = require_string(view, "id", descriptor_path)
if ID_PATTERN.fullmatch(view_id) is None or view_id in view_ids:
raise DocForgeError(
"invalid_config",
"Portable graph view ID is invalid or duplicated",
id=view_id,
)
view_ids.add(view_id)
renderer = require_string(view, "renderer", descriptor_path)
if renderer != "portable_graph_html":
raise DocForgeError(
"unsupported_renderer",
"Portable graph view names an unsupported built-in renderer",
renderer=renderer,
)
output = confined_path(
output_root,
view.get("output"),
field="graph_render.view.output",
must_exist=False,
)
if output.suffix != ".html" or output in outputs:
raise DocForgeError(
"invalid_config",
"Portable graph outputs must be unique HTML files",
)
outputs.add(output)
root_node_id = _optional_string(view, "root", descriptor_path)
query = _optional_string(view, "query", descriptor_path)
if (root_node_id is None) == (query is None):
raise DocForgeError(
"invalid_config",
"Portable graph view requires exactly one root or query",
)
if root_node_id is not None and len(root_node_id) > MAX_GRAPH_VIEW_STRING_CHARS:
raise DocForgeError("invalid_config", "Portable graph root exceeds its fixed limit")
if query is not None and len(query) > MAX_GRAPH_VIEW_QUERY_CHARS:
raise DocForgeError("invalid_config", "Portable graph query exceeds its fixed limit")
initial_mode_value = view.get("initial_mode", "nodes")
if not isinstance(initial_mode_value, str):
raise DocForgeError(
"invalid_config",
"Portable graph initial mode is unsupported",
)
initial_mode = cast(
Literal["nodes", "flow", "web", "logic"],
initial_mode_value,
)
if initial_mode not in _MODES:
raise DocForgeError(
"invalid_config",
"Portable graph initial mode is unsupported",
)
depth = positive_int(view.get("depth", 1), "graph_render.view.depth")
max_nodes = positive_int(view.get("max_nodes", 100), "graph_render.view.max_nodes")
max_edges = positive_int(
view.get("max_edges", 400),
"graph_render.view.max_edges",
allow_zero=True,
)
max_work = positive_int(view.get("max_work", 100_000), "graph_render.view.max_work")
if (
depth > min(limits.max_traversal_depth, MAX_GRAPH_VIEW_DEPTH)
or max_nodes > min(limits.max_nodes, MAX_GRAPH_VIEW_NODES)
or max_edges > MAX_GRAPH_VIEW_EDGES
or max_work > MAX_GRAPH_VIEW_WORK
):
raise DocForgeError(
"invalid_config",
"Portable graph view exceeds project or fixed safety limits",
)
include_logic = view.get("include_logic", False)
if type(include_logic) is not bool:
raise DocForgeError(
"invalid_config",
"Portable graph include_logic must be Boolean",
)
if include_logic:
raise DocForgeError(
"unsupported_renderer",
"Portable graph renderer version 1 does not support Logic projections",
)
views.append(
GraphRenderView(
view_id=view_id,
renderer=renderer,
output_path=output,
title=_bounded_string(
view,
"title",
descriptor_path,
maximum=MAX_GRAPH_VIEW_STRING_CHARS,
),
root_node_id=root_node_id,
query=query,
initial_mode=initial_mode,
depth=depth,
max_nodes=max_nodes,
max_edges=max_edges,
max_work=max_work,
families=_bounded_strings(
view.get("families", []),
key="graph_render.view.families",
source=descriptor_path,
),
relations=_bounded_strings(
view.get("relations", []),
key="graph_render.view.relations",
source=descriptor_path,
),
authorities=_bounded_strings(
view.get("authorities", []),
key="graph_render.view.authorities",
source=descriptor_path,
),
statuses=_bounded_strings(
view.get("statuses", []),
key="graph_render.view.statuses",
source=descriptor_path,
),
tags=_bounded_strings(
view.get("tags", []),
key="graph_render.view.tags",
source=descriptor_path,
),
include_logic=include_logic,
)
)
return GraphRenderConfig(
output_root=output_root,
views=tuple(sorted(views, key=lambda item: item.view_id)),
)

View file

@ -0,0 +1,815 @@
"""Declared portable graph planning, publication, and receipt-only status."""
from __future__ import annotations
import fcntl
import json
import os
from collections.abc import Callable, Generator
from contextlib import contextmanager
from pathlib import Path
from typing import cast
from ._fs_safety import (
atomic_replace_bytes_at,
open_confined_directory,
read_bounded_file_at,
require_bound_directory,
safe_file_identity_at,
)
from .errors import DocForgeError
from .graph_projection import (
GraphViewRequestV1,
build_graph_projection_package,
build_graph_view_plan,
)
from .models import (
GenerationRecordingProject,
GraphRenderConfig,
GraphRenderView,
IncrementalStateProject,
ProjectService,
ProjectSnapshot,
ProjectState,
)
from .project import project_root_fingerprint
from .projection_contract import GraphViewPlanV1, ProjectionReceiptV1, projection_hash
from .projection_policy import (
PortableGraphProjectionMode,
validate_portable_graph_projection_mode,
)
from .projection_worker import render_projection_in_worker
GRAPH_RENDERER_ID = "portable_graph_html"
GRAPH_RENDERER_VERSION = "1"
GRAPH_PUBLICATION_MANIFEST_VERSION = 1
GRAPH_PUBLICATION_CONTRACT = "docforge.graph-publication"
MAX_GRAPH_PUBLICATION_BYTES = 256_000
class GraphRenderService:
"""Publish one declared artifact while keeping planning and rendering independent."""
def __init__(
self,
project: ProjectService,
*,
allow_logic: bool = False,
portable_graph_policy: PortableGraphProjectionMode = "explicit",
) -> None:
self.project = project
self.allow_logic = allow_logic
self.portable_graph_policy = validate_portable_graph_projection_mode(portable_graph_policy)
def _require_rendering(self, operation: str) -> None:
if self.portable_graph_policy == "disabled":
raise DocForgeError(
"projection_policy_forbids_operation",
"Portable graph projection policy disables rendering work",
projection="portable_graph",
mode=self.portable_graph_policy,
operation=operation,
)
def plan(self, view_id: str) -> dict[str, object]:
self._require_rendering("plan")
snapshot = self.project.load()
view = self._view(self._config(snapshot), view_id)
plan = self._plan(snapshot, view)
return {
"status": "ok",
**self._identity(snapshot),
"view_id": view.view_id,
"plan": plan.as_dict(),
}
def status(self, view_id: str | None = None) -> dict[str, object]:
config = self.project.descriptor.graph_render
current = self._current_state()
if config is None:
return self._status_result(
current,
configured=False,
state="not_configured",
outputs=[],
)
views = config.views if view_id is None else (self._view(config, view_id),)
first_outputs = [self._manifest_status(view, current) for view in views]
outputs = [self._manifest_status(view, current) for view in views]
if outputs != first_outputs:
for output in outputs:
if output["state"] == "current":
output["state"] = "stale"
output["reason"] = "publication_changed_during_status"
final = self._current_state()
if final != current:
for output in outputs:
if output["state"] == "current":
output["state"] = "stale"
output["reason"] = "source_changed_during_status"
identity = final if final is not None else current
return self._status_result(
identity,
configured=True,
state="current" if all(item["state"] == "current" for item in outputs) else "stale",
outputs=outputs,
)
def render(self, view_id: str) -> dict[str, object]:
self._require_rendering("render")
with self._lock():
current_status = self.status(view_id)
current_outputs = cast(list[dict[str, object]], current_status["outputs"])
if current_status["state"] == "current" and current_outputs:
return {
**current_status,
"publication": "unchanged",
"output": current_outputs[0],
}
snapshot = self.project.load()
view = self._view(self._config(snapshot), view_id)
plan = self._plan(snapshot, view)
package = build_graph_projection_package(
plan,
renderer_id=GRAPH_RENDERER_ID,
renderer_version=GRAPH_RENDERER_VERSION,
max_output_bytes=snapshot.descriptor.limits.max_render_bytes,
)
result = render_projection_in_worker(package)
if len(result.artifacts) != 1:
raise DocForgeError(
"invalid_projection",
"Portable graph renderer returned an unsupported artifact set",
)
artifact = result.artifacts[0]
def verify() -> None:
current = self.project.load()
if (
current.revision != snapshot.revision
or current.source_hash != snapshot.source_hash
):
raise DocForgeError(
"render_input_changed",
"Canonical input changed during portable graph rendering",
)
verify()
if isinstance(self.project, GenerationRecordingProject):
self.project.record_generation(snapshot)
artifact_evidence = artifact.evidence()
try:
store_identity = self._publish_artifact(
snapshot,
artifact_evidence["sha256"],
artifact.content,
verify=verify,
)
except DocForgeError as error:
if self._mutation_committed(error):
return self._degraded_publication(
snapshot,
view,
plan,
package.package_id,
result.receipt.as_dict(),
artifact_evidence,
stage="artifact_store",
error=error,
output_published=False,
)
raise
try:
output_identity = self._publish_output(
snapshot,
view,
artifact.content,
verify=verify,
)
except DocForgeError as error:
if self._mutation_committed(error):
return self._degraded_publication(
snapshot,
view,
plan,
package.package_id,
result.receipt.as_dict(),
artifact_evidence,
stage="output",
error=error,
output_published=True,
)
raise
manifest = self._manifest(
snapshot,
view,
plan,
package.package_id,
result.receipt.as_dict(),
artifact_evidence,
store_identity,
output_identity,
)
try:
self._publish_manifest(snapshot, view, manifest, verify=verify)
except DocForgeError as error:
return self._degraded_publication(
snapshot,
view,
plan,
package.package_id,
result.receipt.as_dict(),
artifact_evidence,
stage="manifest",
error=error,
output_published=True,
)
return {
"status": "ok",
**self._identity(snapshot),
"view_id": view.view_id,
"state": "current",
"publication": "published",
"plan_id": plan.plan_id,
"package_id": package.package_id,
"output": {
**artifact.evidence(),
"path": view.output_path.relative_to(snapshot.descriptor.root).as_posix(),
},
"receipt": result.receipt.as_dict(),
"manifest": {
"state": "current",
"publication_id": manifest["publication_id"],
},
}
@staticmethod
def _mutation_committed(error: DocForgeError) -> bool:
return error.details.get("mutation_committed") is True
def _degraded_publication(
self,
snapshot: ProjectSnapshot,
view: GraphRenderView,
plan: GraphViewPlanV1,
package_id: str,
receipt: dict[str, object],
artifact: dict[str, object],
*,
stage: str,
error: DocForgeError,
output_published: bool,
) -> dict[str, object]:
return {
"status": "ok",
**self._identity(snapshot),
"view_id": view.view_id,
"state": "degraded",
"publication": "published" if output_published else "partial",
"committed_stage": stage,
"plan_id": plan.plan_id,
"package_id": package_id,
"artifact": artifact,
"output": {
**artifact,
"path": view.output_path.relative_to(snapshot.descriptor.root).as_posix(),
"state": "unverified" if output_published else "not_published",
},
"receipt": receipt,
"manifest": {
"state": "failed",
"error": error.as_dict(),
},
}
def _manifest_status(
self,
view: GraphRenderView,
current: ProjectState | None,
) -> dict[str, object]:
manifest = self._read_manifest(view)
base = {
"view_id": view.view_id,
"renderer": GRAPH_RENDERER_ID,
"renderer_version": GRAPH_RENDERER_VERSION,
"path": view.output_path.relative_to(self.project.descriptor.root).as_posix(),
"verification": "manifest",
}
if manifest is None:
return {**base, "state": "missing", "reason": "manifest_missing"}
if not self._valid_manifest(view, manifest):
return {**base, "state": "unverified", "reason": "manifest_invalid"}
if current is None:
reason = (
"source_generation_changed"
if isinstance(self.project, GenerationRecordingProject)
else "source_generation_unavailable"
)
return {
**base,
"state": (
"stale"
if isinstance(self.project, GenerationRecordingProject)
else "unverified"
),
"reason": reason,
"plan_id": manifest.get("plan_id"),
"package_id": manifest.get("package_id"),
}
project = cast(dict[str, object], manifest["project"])
if project["revision"] != current.revision or project["source_hash"] != current.source_hash:
return {
**base,
"state": "stale",
"reason": "source_generation_changed",
"plan_id": manifest["plan_id"],
"package_id": manifest["package_id"],
}
artifact = cast(dict[str, object], manifest["artifact"])
store = cast(dict[str, object], manifest["store"])
artifact_root = self.project.descriptor.cache_root / "projection-artifacts"
if not artifact_root.exists():
return {
**base,
"state": "stale",
"reason": "artifact_store_missing",
"plan_id": manifest["plan_id"],
"package_id": manifest["package_id"],
}
if artifact_root.is_symlink() or not artifact_root.is_dir():
return {
**base,
"state": "unsafe",
"reason": "artifact_store_unsafe",
"plan_id": manifest["plan_id"],
"package_id": manifest["package_id"],
}
artifact_directory: int | None = None
try:
artifact_directory = open_confined_directory(
self.project.descriptor.root,
artifact_root,
create=False,
)
artifact_identity = safe_file_identity_at(
artifact_root,
artifact_directory,
f"{artifact['sha256']}.html",
)
except DocForgeError:
return {
**base,
"state": "unsafe",
"reason": "artifact_store_unsafe",
"plan_id": manifest["plan_id"],
"package_id": manifest["package_id"],
}
finally:
if artifact_directory is not None:
os.close(artifact_directory)
if artifact_identity is None:
return {
**base,
"state": "stale",
"reason": "artifact_store_missing",
"plan_id": manifest["plan_id"],
"package_id": manifest["package_id"],
}
if artifact_identity != store:
return {
**base,
"state": "stale",
"reason": "artifact_store_changed",
"plan_id": manifest["plan_id"],
"package_id": manifest["package_id"],
}
try:
directory = open_confined_directory(
self.project.descriptor.root,
view.output_path.parent,
create=False,
)
except DocForgeError:
return {**base, "state": "unsafe", "reason": "output_root_unsafe"}
try:
identity = safe_file_identity_at(
view.output_path.parent, directory, view.output_path.name
)
except DocForgeError:
return {**base, "state": "unsafe", "reason": "output_unsafe"}
finally:
os.close(directory)
expected = cast(dict[str, object], manifest["output"])
if identity != expected:
return {
**base,
"state": "stale",
"reason": "output_changed",
"plan_id": manifest["plan_id"],
"package_id": manifest["package_id"],
}
return {
**base,
"state": "current",
"reason": None,
"plan_id": manifest["plan_id"],
"package_id": manifest["package_id"],
"publication_id": manifest["publication_id"],
"artifact": manifest["artifact"],
}
def _read_manifest(self, view: GraphRenderView) -> dict[str, object] | None:
root = self._manifest_root()
if not root.is_dir() or root.is_symlink():
return None
try:
descriptor = open_confined_directory(
self.project.descriptor.root,
root,
create=False,
)
except DocForgeError:
return None
try:
raw = read_bounded_file_at(
descriptor,
f"{view.view_id}.json",
MAX_GRAPH_PUBLICATION_BYTES,
)
except DocForgeError:
return None
finally:
os.close(descriptor)
if raw is None:
return None
try:
value: object = json.loads(raw)
except (UnicodeDecodeError, json.JSONDecodeError):
return None
return cast(dict[str, object], value) if isinstance(value, dict) else None
def _valid_manifest(self, view: GraphRenderView, manifest: dict[str, object]) -> bool:
required = {
"schema_version",
"contract",
"publication_id",
"project",
"view_id",
"view_config_hash",
"plan_id",
"package_id",
"renderer",
"receipt",
"artifact",
"store",
"output",
}
try:
if (
set(manifest) != required
or manifest.get("schema_version") != GRAPH_PUBLICATION_MANIFEST_VERSION
or manifest.get("contract") != GRAPH_PUBLICATION_CONTRACT
or manifest.get("view_id") != view.view_id
or manifest.get("view_config_hash") != self._view_hash(view)
or not self._hash(manifest.get("plan_id"))
or not self._hash(manifest.get("package_id"))
):
return False
project = manifest.get("project")
descriptor = self.project.descriptor
if not isinstance(project, dict):
return False
project_document = cast(dict[str, object], project)
if (
set(project_document)
!= {
"project_id",
"project_root_fingerprint",
"adapter",
"revision",
"source_hash",
}
or project_document.get("project_id") != descriptor.project_id
or project_document.get("project_root_fingerprint")
!= project_root_fingerprint(descriptor.root)
or project_document.get("adapter") != descriptor.adapter
or not isinstance(project_document.get("revision"), str)
or not project_document["revision"]
or not self._hash(project_document.get("source_hash"))
):
return False
renderer = manifest.get("renderer")
if renderer != {
"renderer_id": GRAPH_RENDERER_ID,
"renderer_version": GRAPH_RENDERER_VERSION,
}:
return False
receipt_value = manifest.get("receipt")
if not isinstance(receipt_value, dict):
return False
receipt = ProjectionReceiptV1.from_dict(
dict(cast(dict[str, object], receipt_value))
).as_dict()
artifacts = receipt.get("artifacts")
if not isinstance(artifacts, list):
return False
artifact_values = cast(list[object], artifacts)
if (
receipt.get("kind") != "graph"
or receipt.get("plan_id") != manifest["plan_id"]
or receipt.get("package_id") != manifest["package_id"]
or receipt.get("renderer") != renderer
or len(artifact_values) != 1
or manifest.get("artifact") != artifact_values[0]
):
return False
artifact = artifact_values[0]
if not isinstance(artifact, dict):
return False
artifact_document = cast(dict[str, object], artifact)
if (
artifact_document.get("artifact_id") != "portable-graph.html"
or artifact_document.get("media_type") != "text/html; charset=utf-8"
):
return False
artifact_hash = artifact_document.get("sha256")
artifact_bytes = artifact_document.get("bytes")
store = manifest.get("store")
output = manifest.get("output")
if (
not self._file_identity(store, expected_name=f"{artifact_hash}.html")
or not self._file_identity(output, expected_name=view.output_path.name)
or type(artifact_bytes) is not int
or cast(dict[str, object], store)["size"] != artifact_bytes
or cast(dict[str, object], output)["size"] != artifact_bytes
):
return False
body = dict(manifest)
publication_id = body.pop("publication_id", None)
return self._hash(publication_id) and publication_id == projection_hash(body)
except (DocForgeError, KeyError, TypeError, ValueError):
return False
@staticmethod
def _hash(value: object) -> bool:
return (
isinstance(value, str)
and len(value) == 64
and all(character in "0123456789abcdef" for character in value)
)
@staticmethod
def _file_identity(value: object, *, expected_name: str) -> bool:
if not isinstance(value, dict):
return False
document = cast(dict[str, object], value)
required = {"path", "device", "inode", "mode", "size", "mtime_ns", "ctime_ns"}
return (
set(document) == required
and document.get("path") == expected_name
and all(
type(document.get(field)) is int and cast(int, document[field]) >= 0
for field in required - {"path"}
)
)
def _manifest(
self,
snapshot: ProjectSnapshot,
view: GraphRenderView,
plan: GraphViewPlanV1,
package_id: str,
receipt: dict[str, object],
artifact: dict[str, object],
store: dict[str, object],
output: dict[str, object],
) -> dict[str, object]:
body: dict[str, object] = {
"schema_version": GRAPH_PUBLICATION_MANIFEST_VERSION,
"contract": GRAPH_PUBLICATION_CONTRACT,
"project": self._identity(snapshot),
"view_id": view.view_id,
"view_config_hash": self._view_hash(view),
"plan_id": plan.plan_id,
"package_id": package_id,
"renderer": {
"renderer_id": GRAPH_RENDERER_ID,
"renderer_version": GRAPH_RENDERER_VERSION,
},
"receipt": receipt,
"artifact": artifact,
"store": store,
"output": output,
}
return {**body, "publication_id": projection_hash(body)}
def _publish_artifact(
self,
snapshot: ProjectSnapshot,
artifact_hash: object,
content: bytes,
*,
verify: Callable[[], None],
) -> dict[str, object]:
if not isinstance(artifact_hash, str):
raise DocForgeError("invalid_projection", "Artifact hash is invalid")
root = snapshot.descriptor.cache_root / "projection-artifacts"
descriptor = open_confined_directory(snapshot.descriptor.root, root, create=True)
name = f"{artifact_hash}.html"
try:
try:
existing = read_bounded_file_at(descriptor, name, len(content))
except DocForgeError as error:
if error.code != "invalid_projection":
raise
existing = None
if existing == content:
identity = safe_file_identity_at(root, descriptor, name)
assert identity is not None
return identity
return atomic_replace_bytes_at(root, descriptor, name, content, verify=verify)
finally:
os.close(descriptor)
def _publish_output(
self,
snapshot: ProjectSnapshot,
view: GraphRenderView,
content: bytes,
*,
verify: Callable[[], None],
) -> dict[str, object]:
root = view.output_path.parent
descriptor = open_confined_directory(snapshot.descriptor.root, root, create=True)
try:
try:
existing = read_bounded_file_at(
descriptor,
view.output_path.name,
len(content),
)
except DocForgeError as error:
if error.code != "invalid_projection":
raise
existing = None
if existing == content:
identity = safe_file_identity_at(root, descriptor, view.output_path.name)
assert identity is not None
return identity
return atomic_replace_bytes_at(
root,
descriptor,
view.output_path.name,
content,
verify=verify,
)
finally:
os.close(descriptor)
def _publish_manifest(
self,
snapshot: ProjectSnapshot,
view: GraphRenderView,
manifest: dict[str, object],
*,
verify: Callable[[], None],
) -> None:
raw = json.dumps(manifest, sort_keys=True, indent=2).encode() + b"\n"
if len(raw) > MAX_GRAPH_PUBLICATION_BYTES:
raise DocForgeError(
"projection_too_large",
"Portable graph publication manifest exceeds its fixed limit",
)
root = self._manifest_root()
descriptor = open_confined_directory(snapshot.descriptor.root, root, create=True)
try:
atomic_replace_bytes_at(
root,
descriptor,
f"{view.view_id}.json",
raw,
verify=verify,
)
finally:
os.close(descriptor)
@contextmanager
def _lock(self) -> Generator[None]:
root = self.project.descriptor.cache_root
descriptor = open_confined_directory(self.project.descriptor.root, root, create=True)
lock_descriptor: int | None = None
try:
lock_descriptor = os.open(
"graph-render.lock",
os.O_RDWR | os.O_CREAT | os.O_NOFOLLOW,
0o600,
dir_fd=descriptor,
)
fcntl.flock(lock_descriptor, fcntl.LOCK_EX)
require_bound_directory(root, descriptor)
yield
except OSError as error:
raise DocForgeError(
"publication_failure",
"Portable graph render lock is unavailable",
) from error
finally:
if lock_descriptor is not None:
os.close(lock_descriptor)
os.close(descriptor)
def _plan(self, snapshot: ProjectSnapshot, view: GraphRenderView) -> GraphViewPlanV1:
return build_graph_view_plan(
snapshot,
GraphViewRequestV1(
view_id=view.view_id,
title=view.title,
root_node_id=view.root_node_id,
query=view.query,
initial_mode=view.initial_mode,
depth=view.depth,
max_nodes=view.max_nodes,
max_edges=view.max_edges,
max_work=view.max_work,
families=view.families,
relations=view.relations,
authorities=view.authorities,
statuses=view.statuses,
tags=view.tags,
include_logic=view.include_logic,
),
self.allow_logic,
)
def _current_state(self) -> ProjectState | None:
if isinstance(self.project, IncrementalStateProject):
return self.project.incremental_state()
return None
def _config(self, snapshot: ProjectSnapshot) -> GraphRenderConfig:
config = snapshot.descriptor.graph_render
if config is None:
raise DocForgeError(
"graph_render_not_configured",
"Project has no portable graph render configuration",
)
return config
@staticmethod
def _view(config: GraphRenderConfig, view_id: str) -> GraphRenderView:
for view in config.views:
if view.view_id == view_id:
return view
raise DocForgeError(
"unknown_graph_render_view",
"Portable graph view is not declared",
view_id=view_id,
)
def _manifest_root(self) -> Path:
return self.project.descriptor.cache_root / "projection-publications" / "graph"
@staticmethod
def _view_hash(view: GraphRenderView) -> str:
return projection_hash(
{
"view_id": view.view_id,
"renderer": view.renderer,
"title": view.title,
"root_node_id": view.root_node_id,
"query": view.query,
"initial_mode": view.initial_mode,
"depth": view.depth,
"max_nodes": view.max_nodes,
"max_edges": view.max_edges,
"max_work": view.max_work,
"families": list(view.families),
"relations": list(view.relations),
"authorities": list(view.authorities),
"statuses": list(view.statuses),
"tags": list(view.tags),
"include_logic": view.include_logic,
}
)
def _status_result(self, current: ProjectState | None, **payload: object) -> dict[str, object]:
descriptor = self.project.descriptor
return {
"status": "ok",
"project_id": descriptor.project_id,
"project_root_fingerprint": project_root_fingerprint(descriptor.root),
"adapter": descriptor.adapter,
"revision": current.revision if current is not None else "unknown",
"source_hash": current.source_hash if current is not None else None,
**payload,
}
@staticmethod
def _identity(snapshot: ProjectSnapshot) -> dict[str, object]:
return {
"project_id": snapshot.descriptor.project_id,
"project_root_fingerprint": project_root_fingerprint(snapshot.descriptor.root),
"adapter": snapshot.descriptor.adapter,
"revision": snapshot.revision,
"source_hash": snapshot.source_hash,
}

228
src/docforge/incremental.py Normal file
View file

@ -0,0 +1,228 @@
"""Versioned, confined extraction-cache primitives for incremental adapters."""
from __future__ import annotations
import json
import os
import stat
from dataclasses import dataclass
from pathlib import Path
from typing import Any, cast
from ._fs_safety import atomic_replace_bytes_at, open_bound_directory, require_bound_directory
from .errors import DocForgeError
EXTRACTION_CACHE_SCHEMA_VERSION = 1
MAX_EXTRACTION_CACHE_BYTES = 64_000_000
MAX_EXTRACTION_CACHE_SOURCES = 10_000
@dataclass(frozen=True)
class CachedSource:
"""One adapter-owned cached source contribution."""
source_id: str
source_path: str
fingerprint: str
extractor_version: str
dependencies: tuple[str, ...]
payload: dict[str, object]
@dataclass(frozen=True)
class ExtractionCache:
"""A complete cache generation bound to one adapter identity."""
project_id: str
adapter_id: str
adapter_version: str
sources: tuple[CachedSource, ...]
def load_extraction_cache(
path: Path,
*,
project_id: str,
adapter_id: str,
adapter_version: str,
max_bytes: int = MAX_EXTRACTION_CACHE_BYTES,
max_sources: int = MAX_EXTRACTION_CACHE_SOURCES,
) -> ExtractionCache | None:
"""Read a cache generation, treating malformed or incompatible data as a miss."""
if max_bytes < 1 or max_sources < 1:
raise ValueError("Extraction cache limits must be positive")
try:
descriptor = os.open(
path,
os.O_RDONLY
| getattr(os, "O_CLOEXEC", 0)
| getattr(os, "O_NOFOLLOW", 0)
| getattr(os, "O_NONBLOCK", 0),
)
except OSError:
return None
try:
with os.fdopen(descriptor, "rb") as handle:
before = os.fstat(handle.fileno())
if not stat.S_ISREG(before.st_mode) or before.st_size > max_bytes:
return None
encoded = handle.read(max_bytes + 1)
after = os.fstat(handle.fileno())
if (
len(encoded) > max_bytes
or before.st_dev != after.st_dev
or before.st_ino != after.st_ino
or before.st_size != after.st_size
or before.st_mtime_ns != after.st_mtime_ns
):
return None
raw = json.loads(encoded.decode("utf-8"))
if not isinstance(raw, dict):
return None
document = cast(dict[str, Any], raw)
if (
document.get("schema_version") != EXTRACTION_CACHE_SCHEMA_VERSION
or document.get("project_id") != project_id
or document.get("adapter_id") != adapter_id
or document.get("adapter_version") != adapter_version
):
return None
raw_sources = document.get("sources")
if not isinstance(raw_sources, list):
return None
source_items = cast(list[object], raw_sources)
if len(source_items) > max_sources:
return None
sources: list[CachedSource] = []
for raw_source in source_items:
if not isinstance(raw_source, dict):
return None
item = cast(dict[str, object], raw_source)
if set(item) != {
"source_id",
"source_path",
"fingerprint",
"extractor_version",
"dependencies",
"payload",
}:
return None
source_id = item["source_id"]
source_path = item["source_path"]
fingerprint = item["fingerprint"]
extractor_version = item["extractor_version"]
dependencies = item["dependencies"]
payload = item["payload"]
if (
not isinstance(source_id, str)
or not isinstance(source_path, str)
or not isinstance(fingerprint, str)
or not isinstance(extractor_version, str)
or not isinstance(dependencies, list)
or not all(isinstance(value, str) for value in cast(list[object], dependencies))
or not isinstance(payload, dict)
):
return None
sources.append(
CachedSource(
source_id=source_id,
source_path=source_path,
fingerprint=fingerprint,
extractor_version=extractor_version,
dependencies=tuple(cast(list[str], dependencies)),
payload=cast(dict[str, object], payload),
)
)
ordered = tuple(sorted(sources, key=lambda item: item.source_id))
if tuple(sources) != ordered or len({item.source_id for item in ordered}) != len(ordered):
return None
return ExtractionCache(project_id, adapter_id, adapter_version, ordered)
except (OSError, UnicodeError, json.JSONDecodeError, KeyError, TypeError, ValueError):
return None
def write_extraction_cache(
path: Path,
cache: ExtractionCache,
*,
max_bytes: int = MAX_EXTRACTION_CACHE_BYTES,
max_sources: int = MAX_EXTRACTION_CACHE_SOURCES,
) -> None:
"""Atomically publish one validated extraction-cache generation."""
if max_bytes < 1 or max_sources < 1:
raise ValueError("Extraction cache limits must be positive")
if len(cache.sources) > max_sources:
raise DocForgeError(
"cache_limit",
"Incremental extraction cache exceeds its source limit",
maximum=max_sources,
actual=len(cache.sources),
)
path.parent.mkdir(parents=True, exist_ok=True)
document = {
"schema_version": EXTRACTION_CACHE_SCHEMA_VERSION,
"project_id": cache.project_id,
"adapter_id": cache.adapter_id,
"adapter_version": cache.adapter_version,
"sources": [
{
"source_id": source.source_id,
"source_path": source.source_path,
"fingerprint": source.fingerprint,
"extractor_version": source.extractor_version,
"dependencies": list(source.dependencies),
"payload": source.payload,
}
for source in cache.sources
],
}
encoded = json.dumps(document, sort_keys=True, separators=(",", ":")).encode("utf-8") + b"\n"
if len(encoded) > max_bytes:
raise DocForgeError(
"cache_limit",
"Incremental extraction cache exceeds its byte limit",
maximum=max_bytes,
actual=len(encoded),
)
directory_fd = open_bound_directory(path.parent)
try:
atomic_replace_bytes_at(
path.parent,
directory_fd,
path.name,
encoded,
verify=lambda: require_bound_directory(path.parent, directory_fd),
)
except DocForgeError as error:
if error.code == "path_escape":
raise
raise DocForgeError(
"cache_failure", "Could not publish the incremental extraction cache"
) from error
finally:
os.close(directory_fd)
def affected_sources(
*,
current_dependencies: dict[str, tuple[str, ...]],
cached_dependencies: dict[str, tuple[str, ...]],
changed: set[str],
) -> set[str]:
"""Return the reverse dependency closure of changed, added, or deleted sources."""
reverse: dict[str, set[str]] = {}
for source_id, dependencies in (*cached_dependencies.items(), *current_dependencies.items()):
for dependency in dependencies:
reverse.setdefault(dependency, set()).add(source_id)
affected = set(changed)
pending = list(sorted(changed))
while pending:
source_id = pending.pop()
for dependent in sorted(reverse.get(source_id, ())):
if dependent not in affected:
affected.add(dependent)
pending.append(dependent)
return affected

File diff suppressed because it is too large Load diff

View file

@ -0,0 +1,208 @@
"""Pure manual planning over one immutable validated graph generation."""
from __future__ import annotations
import hashlib
from collections import defaultdict
from collections.abc import Sequence
from .errors import DocForgeError
from .models import Edge, ProjectSnapshot, RenderView
from .project import project_root_fingerprint
from .projection_contract import ManualRenderPlanV1, ProjectionPackageV1
def _edge_dict(edge: Edge) -> dict[str, str]:
return {
"source_id": edge.source_id,
"relation": edge.relation,
"target_id": edge.target_id,
}
def _cycles(node_ids: tuple[str, ...], edges: tuple[Edge, ...]) -> list[list[str]]:
"""Return deterministic strongly connected components that represent cycles."""
adjacency: dict[str, list[str]] = {node_id: [] for node_id in node_ids}
reverse_adjacency: dict[str, list[str]] = {node_id: [] for node_id in node_ids}
for edge in edges:
adjacency[edge.source_id].append(edge.target_id)
reverse_adjacency[edge.target_id].append(edge.source_id)
for targets in (*adjacency.values(), *reverse_adjacency.values()):
targets.sort()
visited: set[str] = set()
finished: list[str] = []
for node_id in node_ids:
if node_id in visited:
continue
visited.add(node_id)
traversal: list[tuple[str, int]] = [(node_id, 0)]
while traversal:
current, position = traversal[-1]
targets = adjacency[current]
if position < len(targets):
target = targets[position]
traversal[-1] = (current, position + 1)
if target not in visited:
visited.add(target)
traversal.append((target, 0))
continue
finished.append(current)
traversal.pop()
assigned: set[str] = set()
components: list[list[str]] = []
for node_id in reversed(finished):
if node_id in assigned:
continue
assigned.add(node_id)
component: list[str] = []
component_stack = [node_id]
while component_stack:
current = component_stack.pop()
component.append(current)
for target in reversed(reverse_adjacency[current]):
if target not in assigned:
assigned.add(target)
component_stack.append(target)
component.sort()
if len(component) > 1 or component[0] in adjacency[component[0]]:
components.append(component)
return sorted(components)
def build_manual_render_plan(
snapshot: ProjectSnapshot,
view: RenderView,
*,
changeset_hash: str | None,
) -> ManualRenderPlanV1:
"""Select and describe a complete manual without rendering markup."""
selected = tuple(
node for node in snapshot.nodes if not view.families or node.family in view.families
)
selected_ids = {node.node_id for node in selected}
edges = tuple(
edge
for edge in snapshot.edges
if edge.source_id in selected_ids and edge.target_id in selected_ids
)
outgoing: dict[str, list[Edge]] = defaultdict(list)
incoming: dict[str, list[Edge]] = defaultdict(list)
for edge in edges:
outgoing[edge.source_id].append(edge)
incoming[edge.target_id].append(edge)
for values in (*outgoing.values(), *incoming.values()):
values.sort(key=lambda edge: (edge.source_id, edge.relation, edge.target_id))
pages = [
{
"node_id": node.node_id,
"title": node.title,
"family": node.family,
"authority": node.authority,
"status": node.status,
"tags": list(node.tags),
"summary": node.summary,
"content": node.content,
"content_hash": node.content_hash,
"components": [
"manual.node-metadata@1",
"manual.summary@1",
"manual.commonmark@1",
"manual.relationships@1",
],
"breadcrumbs": [],
"cross_references": [_edge_dict(edge) for edge in outgoing[node.node_id]],
"backlinks": [_edge_dict(edge) for edge in incoming[node.node_id]],
}
for node in selected
]
node_ids = tuple(node.node_id for node in selected)
connected = {endpoint for edge in edges for endpoint in (edge.source_id, edge.target_id)}
return ManualRenderPlanV1.create(
{
"project": {
"project_id": snapshot.descriptor.project_id,
"project_root_fingerprint": project_root_fingerprint(snapshot.descriptor.root),
"adapter": snapshot.descriptor.adapter,
"revision": snapshot.revision,
"source_hash": snapshot.source_hash,
},
"view": {
"view_id": view.view_id,
"title": view.title,
"families": list(view.families),
"renderer": view.renderer,
},
"changeset_hash": changeset_hash,
"pages": pages,
"navigation": [{"node_id": node.node_id, "title": node.title} for node in selected],
"search_documents": [
{
"node_id": node.node_id,
"title": node.title,
"summary": node.summary,
"family": node.family,
"status": node.status,
"tags": list(node.tags),
}
for node in selected
],
"diagnostics": {
"orphans": [node_id for node_id in node_ids if node_id not in connected],
"cycles": _cycles(node_ids, edges),
},
}
)
def build_manual_projection_package(
plan: ManualRenderPlanV1,
template_bytes: bytes,
*,
renderer_id: str,
renderer_version: str,
max_output_bytes: int,
render_identity: str | None = None,
fragment_records: Sequence[dict[str, object]] = (),
) -> ProjectionPackageV1:
"""Bind one plan and inert template asset for a path-free manual renderer."""
try:
template = template_bytes.decode("utf-8")
except UnicodeDecodeError as error:
raise DocForgeError("invalid_template", "Render template is not valid UTF-8") from error
template_asset: dict[str, object] = {
"asset_id": "manual.template",
"media_type": "text/html; charset=utf-8",
"sha256": hashlib.sha256(template_bytes).hexdigest(),
"text": template,
}
if render_identity is not None:
template_asset["render_identity"] = render_identity
assets = [template_asset]
if fragment_records:
assets.append(
{
"asset_id": "manual.fragments",
"media_type": "application/vnd.docforge.projection-fragments.v1+json",
"records": list(fragment_records),
}
)
return ProjectionPackageV1.create(
kind="manual",
plan=plan,
renderer={"renderer_id": renderer_id, "renderer_version": renderer_version},
components=[
{"component_id": "manual.document@1"},
{"component_id": "manual.commonmark@1"},
],
assets=assets,
output_policy={
"artifact_ids": ["manual.html"],
"max_total_bytes": max_output_bytes,
},
)

File diff suppressed because it is too large Load diff

View file

@ -5,7 +5,7 @@ from __future__ import annotations
from collections.abc import Mapping
from dataclasses import asdict, dataclass
from pathlib import Path
from typing import Protocol
from typing import Literal, Protocol, runtime_checkable
@dataclass(frozen=True)
@ -49,6 +49,33 @@ class RenderConfig:
views: tuple[RenderView, ...]
@dataclass(frozen=True)
class GraphRenderView:
view_id: str
renderer: str
output_path: Path
title: str
root_node_id: str | None
query: str | None
initial_mode: Literal["nodes", "flow", "web", "logic"]
depth: int
max_nodes: int
max_edges: int
max_work: int
families: tuple[str, ...]
relations: tuple[str, ...]
authorities: tuple[str, ...]
statuses: tuple[str, ...]
tags: tuple[str, ...]
include_logic: bool
@dataclass(frozen=True)
class GraphRenderConfig:
output_root: Path
views: tuple[GraphRenderView, ...]
@dataclass(frozen=True)
class ContextProfile:
profile_id: str
@ -78,6 +105,7 @@ class ProjectDescriptor:
allowed_relations: tuple[str, ...]
profiles: tuple[ContextProfile, ...]
limits: Limits
graph_render: GraphRenderConfig | None = None
@dataclass(frozen=True)
@ -111,6 +139,51 @@ class Edge:
return asdict(self)
@dataclass(frozen=True)
class LogicNode:
"""One function-scoped control-flow node kept outside the primary graph."""
logic_id: str
kind: str
label: str
source_anchor: str | None
def as_dict(self) -> dict[str, object]:
return asdict(self)
@dataclass(frozen=True)
class LogicEdge:
"""One directed control-flow transition with an explicit branch label."""
source_id: str
relation: str
target_id: str
label: str | None = None
ordinal: int = 0
def as_dict(self) -> dict[str, object]:
return asdict(self)
@dataclass(frozen=True)
class LogicProjection:
"""A lazy control-flow projection owned by one primary graph symbol."""
owner_node_id: str
source_id: str
nodes: tuple[LogicNode, ...]
edges: tuple[LogicEdge, ...]
def as_dict(self) -> dict[str, object]:
return {
"owner_node_id": self.owner_node_id,
"source_id": self.source_id,
"nodes": [node.as_dict() for node in self.nodes],
"edges": [edge.as_dict() for edge in self.edges],
}
@dataclass(frozen=True)
class ProjectSnapshot:
descriptor: ProjectDescriptor
@ -120,6 +193,14 @@ class ProjectSnapshot:
revision: str
@dataclass(frozen=True)
class ProjectState:
"""Cheap canonical identity used to prove a derived snapshot is current."""
source_hash: str
revision: str
class ProjectService(Protocol):
"""Minimum immutable project boundary required by derived read services."""
@ -137,6 +218,41 @@ class ProjectService(Protocol):
) -> None: ...
@runtime_checkable
class BuildReportingProject(ProjectService, Protocol):
"""Optional project boundary exposing extraction metrics for builds."""
def build_report(self) -> dict[str, object]: ...
@runtime_checkable
class IncrementalStateProject(ProjectService, Protocol):
"""Optional project boundary for manifest-only stale-state checks."""
def incremental_state(self) -> ProjectState | None: ...
@runtime_checkable
class GenerationRecordingProject(IncrementalStateProject, Protocol):
"""Optional project boundary that can persist a verified cheap source generation."""
def record_generation(self, snapshot: ProjectSnapshot) -> None: ...
@runtime_checkable
class RuntimeValidatedProject(ProjectService, Protocol):
"""Optional project boundary that proves its loaded implementation is current."""
def validate_runtime(self) -> None: ...
@runtime_checkable
class LogicProject(ProjectService, Protocol):
"""Optional project boundary exposing logic from its most recent validated load."""
def logic_projections(self) -> tuple[LogicProjection, ...]: ...
@dataclass(frozen=True)
class ContextEntry:
node_id: str

450
src/docforge/onboarding.py Normal file
View file

@ -0,0 +1,450 @@
"""Language-neutral project assessment and safe generic DocForge scaffolding."""
from __future__ import annotations
import json
import os
import re
from dataclasses import dataclass
from pathlib import Path
from typing import cast
from .errors import DocForgeError
_EXCLUDED_DIRECTORIES = frozenset(
{
".cache",
".docforge",
".git",
".gradle",
".idea",
".mypy_cache",
".pytest_cache",
".ruff_cache",
".tox",
".venv",
".vscode",
"__pycache__",
"_deps",
"bin",
"build",
"coverage",
"dist",
"external",
"generated",
"node_modules",
"obj",
"out",
"target",
"third_party",
"vendor",
"venv",
}
)
_PROTECTED_PARTS = frozenset({".git", ".ssh", ".gnupg", "secrets", "credentials"})
_PROJECT_ID_PATTERN = re.compile(r"[a-z0-9][a-z0-9._-]{1,127}")
@dataclass(frozen=True)
class LanguageProfile:
language_id: str
title: str
suffixes: tuple[str, ...]
build_markers: tuple[str, ...]
_LANGUAGE_PROFILES = (
LanguageProfile(
"c",
"C",
(".c",),
("CMakeLists.txt", "meson.build", "Makefile", "configure.ac"),
),
LanguageProfile(
"cpp",
"C++",
(".cc", ".cpp", ".cxx", ".hh", ".hpp", ".hxx"),
("CMakeLists.txt", "meson.build", "Makefile", "conanfile.py", "vcpkg.json"),
),
LanguageProfile("csharp", "C#", (".cs",), (".sln", ".csproj", "global.json")),
LanguageProfile("go", "Go", (".go",), ("go.mod", "go.work")),
LanguageProfile(
"java",
"Java",
(".java",),
("build.gradle", "build.gradle.kts", "pom.xml", "settings.gradle"),
),
LanguageProfile(
"javascript",
"JavaScript",
(".cjs", ".js", ".jsx", ".mjs"),
("package.json",),
),
LanguageProfile(
"kotlin",
"Kotlin",
(".kt", ".kts"),
("build.gradle", "build.gradle.kts", "settings.gradle.kts"),
),
LanguageProfile("lua", "Lua", (".lua",), (".luacheckrc",)),
LanguageProfile("php", "PHP", (".php",), ("composer.json",)),
LanguageProfile(
"python",
"Python",
(".py",),
("pyproject.toml", "requirements.txt", "setup.py", "setup.cfg"),
),
LanguageProfile("ruby", "Ruby", (".rb",), ("Gemfile", ".ruby-version")),
LanguageProfile(
"rust",
"Rust",
(".rs",),
("Cargo.toml", "Cargo.lock", "rust-toolchain.toml"),
),
LanguageProfile("scala", "Scala", (".scala",), ("build.sbt",)),
LanguageProfile("swift", "Swift", (".swift",), ("Package.swift",)),
LanguageProfile(
"typescript",
"TypeScript",
(".ts", ".tsx"),
("package.json", "tsconfig.json"),
),
)
_PROFILES_BY_ID = {profile.language_id: profile for profile in _LANGUAGE_PROFILES}
def _relative_project_path(root: Path, raw: str, *, field: str) -> Path:
candidate = Path(raw)
if candidate.is_absolute() or ".." in candidate.parts or not candidate.parts:
raise DocForgeError("path_escape", f"{field} must stay inside the project root", path=raw)
if any(part.lower() in _PROTECTED_PARTS for part in candidate.parts):
raise DocForgeError("secret_path", f"{field} may not reference a protected path", path=raw)
resolved = (root / candidate).resolve(strict=False)
if not resolved.is_relative_to(root):
raise DocForgeError("path_escape", f"{field} resolves outside the project root", path=raw)
return candidate
def _walk_project_files(root: Path) -> tuple[Path, ...]:
files: list[Path] = []
for directory, directory_names, file_names in os.walk(root, followlinks=False):
current = Path(directory)
directory_names[:] = sorted(
name
for name in directory_names
if name not in _EXCLUDED_DIRECTORIES and not (current / name).is_symlink()
)
for name in sorted(file_names):
path = current / name
if not path.is_symlink():
files.append(path.relative_to(root))
return tuple(files)
def _normalize_requested_languages(requested: tuple[str, ...]) -> tuple[str, ...]:
if not requested or requested == ("auto",):
return ()
values = tuple(sorted(set(item.strip().lower() for item in requested if item.strip())))
if "auto" in values:
raise DocForgeError(
"invalid_onboarding",
"language auto cannot be combined with explicit language profiles",
)
unknown = tuple(item for item in values if item not in _PROFILES_BY_ID)
if unknown:
raise DocForgeError(
"unsupported_language_profile",
"One or more language profiles are not recognized",
languages=list(unknown),
supported=sorted(_PROFILES_BY_ID),
)
return values
def _language_inventory(
root: Path, files: tuple[Path, ...], requested: tuple[str, ...]
) -> tuple[dict[str, object], ...]:
explicit = _normalize_requested_languages(requested)
profiles = tuple(_PROFILES_BY_ID[item] for item in explicit) if explicit else _LANGUAGE_PROFILES
names = {path.name for path in files}
inventory: list[dict[str, object]] = []
for profile in profiles:
source_count = sum(path.suffix.lower() in profile.suffixes for path in files)
markers = sorted(marker for marker in profile.build_markers if marker in names)
if source_count or explicit:
inventory.append(
{
"id": profile.language_id,
"title": profile.title,
"source_files": source_count,
"build_evidence": markers,
"frontend_status": "adapter_required",
}
)
return tuple(sorted(inventory, key=lambda item: str(item["id"])))
def _documentation_inventory(files: tuple[Path, ...]) -> tuple[str, ...]:
candidates = {
path.as_posix()
for path in files
if path.suffix.lower() in {".md", ".mdx", ".rst", ".toml"}
and (
path.name.lower().startswith(("readme", "architecture", "design", "manual"))
or any(part.lower() in {"doc", "docs", "manual"} for part in path.parts[:-1])
)
}
return tuple(sorted(candidates))
def _default_project_id(root: Path) -> str:
value = re.sub(r"[^a-z0-9._-]+", "-", root.name.lower()).strip("-._")
if len(value) < 2:
value = f"{value or 'project'}-docs"
return value[:128]
def assess_project(root: Path, *, requested_languages: tuple[str, ...] = ()) -> dict[str, object]:
"""Return a deterministic, read-only onboarding assessment."""
resolved = root.resolve(strict=True)
if not resolved.is_dir():
raise DocForgeError("invalid_project_root", "Project root must be a directory")
files = _walk_project_files(resolved)
languages = _language_inventory(resolved, files, requested_languages)
existing_descriptor = resolved / ".docforge" / "project.toml"
documentation = _documentation_inventory(files)
return {
"status": "ok",
"mode": "assessment",
"project_root": str(resolved),
"project_id_suggestion": _default_project_id(resolved),
"file_count": len(files),
"languages": list(languages),
"documentation_candidates": list(documentation),
"configured": existing_descriptor.is_file(),
"capabilities": {
"manual_scaffold": "available" if not existing_descriptor.exists() else "configured",
"source_graph": ("adapter_required" if languages else "no_supported_source_detected"),
"incremental_compilation": "available_after_adapter",
"mcp": "available_after_configuration",
"viewer": "available_after_index",
},
"next_actions": [
"Review detected languages and documentation authority.",
(
"Run onboard with --scaffold to create a generic manual when the project is "
"unconfigured."
),
"Implement or select one language frontend per source language.",
"Prove full and incremental projection equivalence.",
"Generate and register the fixed project MCP command.",
],
}
def _toml_string(value: str) -> str:
return json.dumps(value, ensure_ascii=False)
def _descriptor(project_id: str, title: str, content_root: Path) -> str:
content = content_root.as_posix()
return f"""schema_version = 1
project_id = {_toml_string(project_id)}
title = {_toml_string(title)}
adapter = "generic"
[sources]
content_roots = [{_toml_string(content)}]
authority_files = []
[derived]
cache_root = ".docforge/cache"
index = ".docforge/cache/index.sqlite3"
[changesets]
root = ".docforge/changesets"
[[changesets.writers]]
id = "project-editor"
families = ["api", "architecture", "operations", "proof", "roadmap", "system"]
operations = ["create", "update", "move", "delete"]
[render]
template_root = ".docforge/templates"
preview_root = ".docforge/previews"
[[render.views]]
id = "manual"
renderer = "generic_html"
template = "manual.html"
output = ".docforge/rendered/manual.html"
title = {_toml_string(f"{title} Manual")}
families = ["api", "architecture", "operations", "proof", "roadmap", "system"]
[graph]
allowed_relations = [
"calls",
"defines",
"depends_on",
"implements",
"inherits_from",
"owns",
"reads",
"relates_to",
"tested_by",
"writes",
]
[limits]
max_source_bytes = 500000
max_nodes = 10000
max_query_chars = 500
max_results = 100
max_traversal_depth = 6
max_context_tokens = 12000
max_changesets = 100
max_changeset_operations = 100
max_changeset_bytes = 1000000
[[profiles]]
id = "development"
families = ["api", "architecture", "operations", "proof", "roadmap", "system"]
statuses = ["active", "current", "verified"]
required_nodes = ["architecture.overview"]
token_budget = 8000
dependency_depth = 3
"""
def _overview(title: str, languages: tuple[dict[str, object], ...]) -> str:
language_titles = [str(item["title"]) for item in languages]
tags = ["architecture", "onboarding", *[str(item["id"]) for item in languages]]
language_text = ", ".join(language_titles) if language_titles else "No source language selected"
return f"""+++
schema_version = 1
id = "architecture.overview"
title = "Project architecture"
family = "architecture"
authority = "authoritative"
status = "current"
tags = {json.dumps(tags)}
summary = "Introduces the project and its documentation authority."
+++
# {title}
DocForge indexes the canonical documentation under this directory. Derived indexes, rendered
pages, previews, and extraction caches may be deleted and rebuilt.
Detected or selected source languages: {language_text}.
Source-code facts require a language frontend that implements DocForge's adapter contract. Until
that frontend passes full and incremental equivalence checks, this manual remains authoritative
and the source graph remains explicitly unavailable.
"""
_TEMPLATE = """<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<meta name="docforge-render" content="{{ docforge_render_identity }}">
<title>{{ docforge_title }}</title>
<style>
:root { color-scheme: dark; font-family: system-ui, sans-serif; }
body { margin: 0; background: #0e1518; color: #dce9e6; }
header, main { width: min(70rem, calc(100% - 3rem)); margin: auto; }
header { padding: 3rem 0 1rem; border-bottom: 1px solid #31504c; }
main { padding: 2rem 0 5rem; line-height: 1.65; }
article { margin: 0 0 2rem; padding: 1.5rem; background: #14211f; border: 1px solid #31504c; }
a { color: #78ddd0; }
code { color: #f1c779; }
</style>
</head>
<body data-project="{{ docforge_project_id }}" data-view="{{ docforge_view_id }}">
<header><h1>{{ docforge_title }}</h1></header>
<main>{{ docforge_content }}</main>
</body>
</html>
"""
def _write_new(path: Path, content: str) -> None:
if path.exists() or path.is_symlink():
raise DocForgeError(
"onboarding_conflict",
"Onboarding will not replace an existing path",
path=str(path),
)
path.parent.mkdir(parents=True, exist_ok=True)
temporary = path.with_name(f".{path.name}.docforge-new")
try:
temporary.write_text(content, encoding="utf-8")
os.replace(temporary, path)
finally:
temporary.unlink(missing_ok=True)
def scaffold_project(
root: Path,
*,
requested_languages: tuple[str, ...] = (),
project_id: str | None = None,
title: str | None = None,
content_root: str = "docs/docforge/content",
) -> dict[str, object]:
"""Create a minimal generic manual without pretending a source adapter exists."""
resolved = root.resolve(strict=True)
assessment = assess_project(resolved, requested_languages=requested_languages)
language_values = cast(list[object], assessment["languages"])
selected_languages = tuple(
cast(dict[str, object], item) for item in language_values if isinstance(item, dict)
)
resolved_id = project_id or str(assessment["project_id_suggestion"])
if _PROJECT_ID_PATTERN.fullmatch(resolved_id) is None:
raise DocForgeError(
"invalid_onboarding", "project_id is not a stable DocForge ID", project_id=resolved_id
)
resolved_title = title.strip() if title and title.strip() else resolved.name
relative_content = _relative_project_path(resolved, content_root, field="onboard.content_root")
targets = (
resolved / ".docforge" / "project.toml",
resolved / ".docforge" / "templates" / "manual.html",
resolved / relative_content / "project-overview.md",
)
conflicts = [str(path) for path in targets if path.exists() or path.is_symlink()]
if conflicts:
raise DocForgeError(
"onboarding_conflict",
"Onboarding will not replace existing project files",
paths=conflicts,
)
created: list[Path] = []
try:
template = targets[1]
_write_new(template, _TEMPLATE)
created.append(template)
overview = targets[2]
_write_new(overview, _overview(resolved_title, selected_languages))
created.append(overview)
descriptor = targets[0]
_write_new(descriptor, _descriptor(resolved_id, resolved_title, relative_content))
created.append(descriptor)
except Exception:
for path in reversed(created):
path.unlink(missing_ok=True)
raise
return {
**assessment,
"mode": "scaffold",
"configured": True,
"project_id": resolved_id,
"title": resolved_title,
"created": [str(path.relative_to(resolved)) for path in created],
"source_graph_status": "adapter_required",
}

164
src/docforge/pagination.py Normal file
View file

@ -0,0 +1,164 @@
"""Deterministic, generation-bound pagination cursors for bounded public results."""
from __future__ import annotations
import base64
import binascii
import hashlib
import hmac
import json
from collections.abc import Mapping
from typing import cast
from .errors import DocForgeError
CURSOR_SCHEMA_VERSION = 1
MAX_CURSOR_CHARS = 8_192
_CURSOR_DOMAIN = b"docforge-page-cursor-v1\0"
_CURSOR_KEYS = frozenset({"schema_version", "kind", "binding", "position", "checksum"})
def canonical_hash(value: object) -> str:
"""Hash one JSON-compatible value using DocForge's deterministic JSON form."""
try:
encoded = _canonical_bytes(value)
except (TypeError, ValueError) as error:
raise DocForgeError(
"invalid_pagination_source",
"Pagination source data is not deterministic JSON",
) from error
return hashlib.sha256(encoded).hexdigest()
def page_limit(limit: int | None, *, default: int, maximum: int) -> int:
"""Validate one additive page size against the project result policy."""
selected = default if limit is None else limit
if type(selected) is not int or selected < 1 or selected > maximum:
raise DocForgeError(
"invalid_limit",
"Page limit is outside the configured result limit",
maximum=maximum,
)
return selected
def encode_cursor(
*,
kind: str,
binding: Mapping[str, object],
position: int,
) -> str:
"""Encode a corruption-detecting cursor bound to an immutable result identity."""
if not kind or type(position) is not int or position < 0:
raise ValueError("Cursor kind and position must be valid")
body: dict[str, object] = {
"schema_version": CURSOR_SCHEMA_VERSION,
"kind": kind,
"binding": dict(binding),
"position": position,
}
checksum = hashlib.sha256(_CURSOR_DOMAIN + _canonical_bytes(body)).hexdigest()
envelope = {**body, "checksum": checksum}
return base64.urlsafe_b64encode(_canonical_bytes(envelope)).decode("ascii").rstrip("=")
def decode_cursor(
cursor: str | None,
*,
kind: str,
binding: Mapping[str, object],
total_count: int,
) -> int:
"""Return a validated position, rejecting corrupt, foreign, or stale cursors."""
if cursor is None:
return 0
if not cursor or len(cursor) > MAX_CURSOR_CHARS or not cursor.isascii():
raise _invalid_cursor()
padding = "=" * (-len(cursor) % 4)
try:
raw = base64.b64decode(
(cursor + padding).encode("ascii"),
altchars=b"-_",
validate=True,
)
parsed: object = json.loads(raw.decode("utf-8"))
except (binascii.Error, UnicodeDecodeError, json.JSONDecodeError):
raise _invalid_cursor() from None
if base64.urlsafe_b64encode(raw).decode("ascii").rstrip("=") != cursor or not isinstance(
parsed, dict
):
raise _invalid_cursor()
payload = cast(dict[str, object], parsed)
checksum = payload.get("checksum")
position = payload.get("position")
stored_binding = payload.get("binding")
if (
frozenset(payload) != _CURSOR_KEYS
or payload.get("schema_version") != CURSOR_SCHEMA_VERSION
or payload.get("kind") != kind
or not isinstance(stored_binding, dict)
or type(position) is not int
or position < 0
or not isinstance(checksum, str)
or len(checksum) != 64
):
raise _invalid_cursor()
body = {key: payload[key] for key in payload if key != "checksum"}
expected = hashlib.sha256(_CURSOR_DOMAIN + _canonical_bytes(body)).hexdigest()
if not hmac.compare_digest(checksum, expected):
raise _invalid_cursor()
if stored_binding != dict(binding):
raise DocForgeError(
"stale_cursor",
"Pagination cursor does not match the current result generation",
)
if position >= total_count:
raise _invalid_cursor()
return position
def page_receipt(
*,
kind: str,
binding: Mapping[str, object],
position: int,
count: int,
limit: int,
total_count: int,
) -> dict[str, object]:
"""Return one bounded page receipt and the next generation-bound cursor."""
next_position = position + count
has_more = next_position < total_count
return {
"schema_version": CURSOR_SCHEMA_VERSION,
"kind": kind,
"returned_count": count,
"limit": limit,
"total_count": total_count,
"has_more": has_more,
"next_cursor": (
encode_cursor(kind=kind, binding=binding, position=next_position) if has_more else None
),
}
def _canonical_bytes(value: object) -> bytes:
return json.dumps(
value,
sort_keys=True,
separators=(",", ":"),
ensure_ascii=True,
allow_nan=False,
).encode("utf-8")
def _invalid_cursor() -> DocForgeError:
return DocForgeError(
"invalid_cursor",
"Pagination cursor is malformed or does not match its operation",
)

166
src/docforge/policy.py Normal file
View file

@ -0,0 +1,166 @@
"""Versioned immutable policy composition for one project-bound server."""
from __future__ import annotations
from dataclasses import dataclass
from typing import Literal
from .errors import DocForgeError
CapabilityMode = Literal["read", "proposal", "application", "operator"]
CAPABILITY_MODES: tuple[CapabilityMode, ...] = (
"read",
"proposal",
"application",
"operator",
)
POLICY_PRECEDENCE = (
"core_safety",
"explicit_binding",
"no_ast_shorthand",
"resource_availability",
)
def capability_mode(value: str | None, *, default: CapabilityMode) -> CapabilityMode:
"""Validate one additive capability-mode selection."""
selected = default if value is None else value
if selected not in CAPABILITY_MODES:
raise DocForgeError(
"invalid_capability_mode",
"Capability mode is unsupported",
capability_mode=selected,
allowed=list(CAPABILITY_MODES),
)
return selected # type: ignore[return-value]
@dataclass(frozen=True)
class EffectivePolicyV1:
"""One fully composed process policy shared by every public projection."""
capability_mode: CapabilityMode
capability_source: Literal["factory_default", "explicit"]
adapter_evolution: Literal["allowed", "preserve"]
ast_analysis: Literal["allowed", "forbidden"]
logic_indexing: Literal["full", "off"]
synchronization: Literal["automatic"]
integrity: Literal["validated"]
manual_render: Literal["auto", "explicit", "disabled"]
graph_render: Literal["disabled"]
live_viewer: Literal["on-demand"]
profiling: Literal["enabled", "disabled"]
blocked_tools: tuple[str, ...]
prohibitions: tuple[str, ...]
@property
def no_ast(self) -> bool:
return self.ast_analysis == "forbidden"
def as_dict(self) -> dict[str, object]:
return {
"schema_version": 1,
"capability_mode": self.capability_mode,
"capability_source": self.capability_source,
"adapter_evolution": self.adapter_evolution,
"ast_analysis": self.ast_analysis,
"logic_indexing": self.logic_indexing,
"synchronization": self.synchronization,
"integrity": self.integrity,
"manual_render": self.manual_render,
"graph_render": self.graph_render,
"live_viewer": self.live_viewer,
"profiling": self.profiling,
"blocked_tools": list(self.blocked_tools),
"prohibitions": list(self.prohibitions),
"precedence": list(POLICY_PRECEDENCE),
}
def adapter_policy(self) -> dict[str, object]:
"""Preserve the exact legacy adapter-policy projection."""
if not self.no_ast:
return {
"mode": "standard",
"ast_analysis": "allowed",
"logic_projection": "allowed",
"incremental_extraction": "allowed",
"adapter_rewrite": "not_requested",
}
return {
"mode": "preserve-no-ast",
"ast_analysis": "forbidden",
"logic_projection": "forbidden",
"incremental_extraction": "allowed",
"adapter_rewrite": "forbidden",
"blocked_tools": ["docforge_get_logic"],
"instruction": (
"Preserve the existing adapter extraction strategy. Do not add Python AST, "
"Tree-sitter, compiler-AST, or function-Logic extraction. Non-AST incremental "
"fingerprinting and caching remain allowed."
),
}
def compose_effective_policy(
*,
selected_mode: CapabilityMode,
capability_source: Literal["factory_default", "explicit"],
no_ast: bool,
diagnostics: bool,
render_configured: bool,
application_enabled: bool,
) -> EffectivePolicyV1:
"""Compose fixed defaults with restrictive compatibility shorthands."""
if selected_mode == "application" and not application_enabled:
raise DocForgeError(
"capability_unavailable",
"Application capability requires a startup-bound canonical applier",
capability_mode=selected_mode,
required="canonical_applier",
)
prohibitions = [
"arbitrary_file_access",
"arbitrary_renderer_execution",
"shell_execution",
"git_mutation",
"deployment",
"publication",
"project_switching",
]
blocked_tools: tuple[str, ...] = ()
if no_ast:
prohibitions.extend(
(
"adapter_ast_upgrade",
"tree_sitter_upgrade",
"compiler_ast_upgrade",
"function_logic_extraction",
)
)
blocked_tools = ("docforge_get_logic",)
manual_render: Literal["auto", "explicit", "disabled"]
if not render_configured:
manual_render = "disabled"
elif application_enabled and selected_mode in {"application", "operator"}:
manual_render = "auto"
else:
manual_render = "explicit"
return EffectivePolicyV1(
capability_mode=selected_mode,
capability_source=capability_source,
adapter_evolution="preserve" if no_ast else "allowed",
ast_analysis="forbidden" if no_ast else "allowed",
logic_indexing="off" if no_ast else "full",
synchronization="automatic",
integrity="validated",
manual_render=manual_render,
graph_render="disabled",
live_viewer="on-demand",
profiling="enabled" if diagnostics else "disabled",
blocked_tools=blocked_tools,
prohibitions=tuple(prohibitions),
)

Some files were not shown because too many files have changed in this diff Show more