2026-07-29 15:59:55 -04:00
|
|
|
"""Reproducible comparative task evidence for the Milestone 5 release gate."""
|
|
|
|
|
|
|
|
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
|
|
import argparse
|
2026-07-29 16:21:10 -04:00
|
|
|
import ast
|
2026-07-29 15:59:55 -04:00
|
|
|
import hashlib
|
|
|
|
|
import json
|
|
|
|
|
import math
|
|
|
|
|
import platform
|
2026-07-29 16:21:10 -04:00
|
|
|
import shutil
|
|
|
|
|
import stat
|
2026-07-29 15:59:55 -04:00
|
|
|
import statistics
|
|
|
|
|
import subprocess
|
|
|
|
|
import sys
|
|
|
|
|
import tempfile
|
|
|
|
|
import time
|
2026-07-29 16:21:10 -04:00
|
|
|
import tomllib
|
2026-07-29 15:59:55 -04:00
|
|
|
from collections import deque
|
|
|
|
|
from collections.abc import Callable, Mapping, Sequence
|
|
|
|
|
from dataclasses import dataclass
|
2026-07-29 16:21:10 -04:00
|
|
|
from importlib import metadata
|
2026-07-29 15:59:55 -04:00
|
|
|
from pathlib import Path
|
|
|
|
|
from typing import Literal, cast
|
|
|
|
|
|
2026-07-29 16:21:10 -04:00
|
|
|
from docforge.adapter_sdk import (
|
|
|
|
|
AdapterEdge,
|
|
|
|
|
AdapterLoader,
|
|
|
|
|
AdapterNode,
|
|
|
|
|
AdapterProject,
|
|
|
|
|
AdapterProjection,
|
|
|
|
|
Edge,
|
|
|
|
|
Node,
|
|
|
|
|
)
|
2026-07-29 15:59:55 -04:00
|
|
|
from docforge.adapters.python import PythonReferenceAdapter
|
|
|
|
|
from docforge.index import ProjectIndex
|
|
|
|
|
|
|
|
|
|
ROOT = Path(__file__).resolve().parents[1]
|
|
|
|
|
SMOKE_SOURCE_COUNT = 18
|
|
|
|
|
FULL_SOURCE_COUNT = 72
|
|
|
|
|
SMOKE_SAMPLES = 1
|
|
|
|
|
FULL_SAMPLES = 5
|
|
|
|
|
MAX_TASK_RESPONSE_BYTES = 64 * 1024
|
|
|
|
|
MAX_GRAPH_INSPECTED_BYTES = 512 * 1024
|
|
|
|
|
MAX_SOURCE_INSPECTED_BYTES = 4 * 1024 * 1024
|
|
|
|
|
MAX_REPORT_BYTES = 1024 * 1024
|
|
|
|
|
PREPARATION_LIMIT_MS = 30_000.0
|
|
|
|
|
TASK_P95_LIMIT_MS = 5_000.0
|
|
|
|
|
PADDING_ROWS = 24
|
2026-07-29 16:21:10 -04:00
|
|
|
PINNED_REAL_DISTRIBUTION = "markdown-it-py"
|
|
|
|
|
PINNED_REAL_VERSION = "4.2.0"
|
|
|
|
|
PINNED_REAL_IMPORT_ROOT = "markdown_it"
|
|
|
|
|
PINNED_REAL_SOURCE_COUNT = 66
|
|
|
|
|
PINNED_REAL_SOURCE_BYTES = 225_945
|
|
|
|
|
PINNED_REAL_SOURCE_SHA256 = "bd57c9f332fcf6507282ec2023e6804fce0cf844631696336ee17cbe46e63aad"
|
2026-07-29 15:59:55 -04:00
|
|
|
|
|
|
|
|
WorkflowName = Literal["graph_assisted", "source_only"]
|
|
|
|
|
TaskKind = Literal["direct_dependencies", "bounded_impact", "dependency_path"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class TaskEvidenceError(RuntimeError):
|
|
|
|
|
"""The maintained comparative evidence gate was not satisfied."""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
|
|
|
class TaskSpec:
|
|
|
|
|
task_id: str
|
|
|
|
|
kind: TaskKind
|
2026-07-29 16:21:10 -04:00
|
|
|
focus: str
|
|
|
|
|
target: str | None
|
2026-07-29 15:59:55 -04:00
|
|
|
depth: int
|
|
|
|
|
prompt: str
|
|
|
|
|
|
|
|
|
|
def as_dict(self) -> dict[str, object]:
|
|
|
|
|
return {
|
|
|
|
|
"task_id": self.task_id,
|
|
|
|
|
"kind": self.kind,
|
2026-07-29 16:21:10 -04:00
|
|
|
"focus": self.focus,
|
|
|
|
|
"target": self.target,
|
2026-07-29 15:59:55 -04:00
|
|
|
"depth": self.depth,
|
|
|
|
|
"prompt": self.prompt,
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
|
|
|
class WorkflowResult:
|
|
|
|
|
answer: tuple[str, ...]
|
|
|
|
|
provenance: tuple[Mapping[str, object], ...]
|
|
|
|
|
inspected_bytes: int
|
|
|
|
|
|
|
|
|
|
def payload(self) -> dict[str, object]:
|
|
|
|
|
return {
|
|
|
|
|
"answer": list(self.answer),
|
|
|
|
|
"provenance": [dict(item) for item in self.provenance],
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
|
|
|
class _GraphContext:
|
|
|
|
|
index: ProjectIndex
|
|
|
|
|
module_node_ids: Mapping[str, str]
|
|
|
|
|
node_modules: Mapping[str, str]
|
|
|
|
|
module_source_paths: Mapping[str, str]
|
2026-07-29 16:21:10 -04:00
|
|
|
outgoing: Mapping[str, tuple[str, ...]]
|
|
|
|
|
incoming: Mapping[str, tuple[str, ...]]
|
|
|
|
|
relation: str
|
2026-07-29 15:59:55 -04:00
|
|
|
revision: str
|
|
|
|
|
source_hash: str
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _parser() -> argparse.ArgumentParser:
|
|
|
|
|
parser = argparse.ArgumentParser(
|
|
|
|
|
description=(
|
|
|
|
|
"Compare graph-assisted and source-only workflows on fixed, answer-keyed tasks."
|
|
|
|
|
)
|
|
|
|
|
)
|
|
|
|
|
parser.add_argument("--mode", choices=("smoke", "full"), default="full")
|
|
|
|
|
parser.add_argument("--output", type=Path)
|
|
|
|
|
return parser
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def encode_report(value: object) -> str:
|
|
|
|
|
"""Serialize evidence using the repository's deterministic JSON convention."""
|
|
|
|
|
|
|
|
|
|
return json.dumps(value, sort_keys=True, indent=2, ensure_ascii=False) + "\n"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _compact_json(value: object) -> bytes:
|
|
|
|
|
return json.dumps(
|
|
|
|
|
value,
|
|
|
|
|
sort_keys=True,
|
|
|
|
|
separators=(",", ":"),
|
|
|
|
|
ensure_ascii=False,
|
|
|
|
|
).encode("utf-8")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _sha256(value: object) -> str:
|
|
|
|
|
return hashlib.sha256(_compact_json(value)).hexdigest()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _git(arguments: list[str]) -> str:
|
|
|
|
|
return subprocess.run(
|
|
|
|
|
["git", *arguments],
|
|
|
|
|
cwd=ROOT,
|
|
|
|
|
check=True,
|
|
|
|
|
capture_output=True,
|
|
|
|
|
text=True,
|
|
|
|
|
).stdout.strip()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _module_name(index: int) -> str:
|
|
|
|
|
return f"evidence.component_{index:03d}"
|
|
|
|
|
|
|
|
|
|
|
2026-07-29 16:21:10 -04:00
|
|
|
def _generated_index(module_name: str) -> int:
|
|
|
|
|
prefix = "evidence.component_"
|
|
|
|
|
suffix = module_name.removeprefix(prefix)
|
|
|
|
|
if not module_name.startswith(prefix) or len(suffix) != 3 or not suffix.isdigit():
|
|
|
|
|
raise TaskEvidenceError(f"Generated task uses an invalid module name: {module_name}")
|
|
|
|
|
return int(suffix)
|
|
|
|
|
|
|
|
|
|
|
2026-07-29 15:59:55 -04:00
|
|
|
def _component_path(root: Path, index: int) -> Path:
|
|
|
|
|
return root / "src" / "evidence" / f"component_{index:03d}.py"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _dependency_indices(index: int) -> tuple[int, ...]:
|
|
|
|
|
if index == 0:
|
|
|
|
|
return ()
|
|
|
|
|
return tuple(sorted({0, index // 2}))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def fixture_tasks(source_count: int) -> tuple[TaskSpec, ...]:
|
|
|
|
|
"""Return the fixed task inventory for one supported fixture size."""
|
|
|
|
|
|
|
|
|
|
terminal = source_count - 1
|
|
|
|
|
return (
|
|
|
|
|
TaskSpec(
|
|
|
|
|
task_id="direct_dependencies_of_terminal",
|
|
|
|
|
kind="direct_dependencies",
|
2026-07-29 16:21:10 -04:00
|
|
|
focus=_module_name(terminal),
|
|
|
|
|
target=None,
|
2026-07-29 15:59:55 -04:00
|
|
|
depth=1,
|
|
|
|
|
prompt=f"List the direct local dependencies of {_module_name(terminal)}.",
|
|
|
|
|
),
|
|
|
|
|
TaskSpec(
|
|
|
|
|
task_id="two_level_impact_of_component_007",
|
|
|
|
|
kind="bounded_impact",
|
2026-07-29 16:21:10 -04:00
|
|
|
focus=_module_name(7),
|
|
|
|
|
target=None,
|
2026-07-29 15:59:55 -04:00
|
|
|
depth=2,
|
|
|
|
|
prompt=(
|
|
|
|
|
"List every component within two reverse dependency steps of "
|
|
|
|
|
"evidence.component_007."
|
|
|
|
|
),
|
|
|
|
|
),
|
|
|
|
|
TaskSpec(
|
|
|
|
|
task_id="path_from_terminal_to_component_001",
|
|
|
|
|
kind="dependency_path",
|
2026-07-29 16:21:10 -04:00
|
|
|
focus=_module_name(terminal),
|
|
|
|
|
target=_module_name(1),
|
2026-07-29 15:59:55 -04:00
|
|
|
depth=8,
|
|
|
|
|
prompt=(
|
|
|
|
|
f"Find one shortest local dependency path from {_module_name(terminal)} "
|
|
|
|
|
"to evidence.component_001."
|
|
|
|
|
),
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def write_fixture(root: Path, source_count: int) -> int:
|
|
|
|
|
"""Create the bounded deterministic corpus and return its exact byte size."""
|
|
|
|
|
|
|
|
|
|
if not SMOKE_SOURCE_COUNT <= source_count <= FULL_SOURCE_COUNT:
|
|
|
|
|
raise ValueError(
|
|
|
|
|
f"source_count must be between {SMOKE_SOURCE_COUNT} and {FULL_SOURCE_COUNT}"
|
|
|
|
|
)
|
|
|
|
|
package = root / "src" / "evidence"
|
|
|
|
|
package.mkdir(parents=True)
|
|
|
|
|
total_bytes = 0
|
|
|
|
|
for index in range(source_count):
|
|
|
|
|
imports = "".join(
|
|
|
|
|
f"from evidence.component_{dependency:03d} import compute_{dependency:03d}\n"
|
|
|
|
|
for dependency in _dependency_indices(index)
|
|
|
|
|
)
|
|
|
|
|
padding = "\n".join(
|
|
|
|
|
f' "component-{index:03d}-evidence-row-{row:02d}-{"x" * 52}",'
|
|
|
|
|
for row in range(PADDING_ROWS)
|
|
|
|
|
)
|
|
|
|
|
expression = " + ".join(
|
|
|
|
|
f"compute_{dependency:03d}(value)" for dependency in _dependency_indices(index)
|
|
|
|
|
)
|
|
|
|
|
if not expression:
|
|
|
|
|
expression = "value"
|
|
|
|
|
source = (
|
|
|
|
|
f'"""Deterministic comparative evidence component {index:03d}."""\n\n'
|
|
|
|
|
f"{imports}\n"
|
|
|
|
|
f"PADDING = (\n{padding}\n)\n\n"
|
|
|
|
|
f"def compute_{index:03d}(value: int) -> int:\n"
|
|
|
|
|
f' """Return the bounded component {index:03d} result."""\n'
|
|
|
|
|
f" return ({expression}) + {index}\n"
|
|
|
|
|
)
|
|
|
|
|
raw = source.encode("utf-8")
|
|
|
|
|
_component_path(root, index).write_bytes(raw)
|
|
|
|
|
total_bytes += len(raw)
|
|
|
|
|
return total_bytes
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _all_dependencies(source_count: int) -> dict[int, tuple[int, ...]]:
|
|
|
|
|
return {index: _dependency_indices(index) for index in range(source_count)}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _bounded_impact(
|
|
|
|
|
dependencies: Mapping[int, Sequence[int]],
|
|
|
|
|
focus: int,
|
|
|
|
|
depth: int,
|
|
|
|
|
) -> tuple[int, ...]:
|
|
|
|
|
reverse: dict[int, list[int]] = {index: [] for index in dependencies}
|
|
|
|
|
for source, targets in dependencies.items():
|
|
|
|
|
for target in targets:
|
|
|
|
|
reverse[target].append(source)
|
|
|
|
|
queue: deque[tuple[int, int]] = deque([(focus, 0)])
|
|
|
|
|
seen = {focus}
|
|
|
|
|
results: set[int] = set()
|
|
|
|
|
while queue:
|
|
|
|
|
current, current_depth = queue.popleft()
|
|
|
|
|
if current_depth >= depth:
|
|
|
|
|
continue
|
|
|
|
|
for source in sorted(reverse[current]):
|
|
|
|
|
if source in seen:
|
|
|
|
|
continue
|
|
|
|
|
seen.add(source)
|
|
|
|
|
results.add(source)
|
|
|
|
|
queue.append((source, current_depth + 1))
|
|
|
|
|
return tuple(sorted(results))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _shortest_path(
|
|
|
|
|
dependencies: Mapping[int, Sequence[int]],
|
|
|
|
|
focus: int,
|
|
|
|
|
target: int,
|
|
|
|
|
depth: int,
|
|
|
|
|
) -> tuple[int, ...]:
|
|
|
|
|
queue: deque[tuple[int, tuple[int, ...]]] = deque([(focus, (focus,))])
|
|
|
|
|
seen = {focus}
|
|
|
|
|
while queue:
|
|
|
|
|
current, path = queue.popleft()
|
|
|
|
|
if current == target:
|
|
|
|
|
return path
|
|
|
|
|
if len(path) - 1 >= depth:
|
|
|
|
|
continue
|
|
|
|
|
for dependency in sorted(dependencies[current]):
|
|
|
|
|
if dependency in seen:
|
|
|
|
|
continue
|
|
|
|
|
seen.add(dependency)
|
|
|
|
|
queue.append((dependency, (*path, dependency)))
|
|
|
|
|
raise TaskEvidenceError(
|
|
|
|
|
f"No answer-key path exists from {_module_name(focus)} to {_module_name(target)}"
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def answer_key(task: TaskSpec, source_count: int) -> tuple[str, ...]:
|
|
|
|
|
"""Derive the answer independently from the documented fixture topology."""
|
|
|
|
|
|
|
|
|
|
dependencies = _all_dependencies(source_count)
|
2026-07-29 16:21:10 -04:00
|
|
|
focus_index = _generated_index(task.focus)
|
2026-07-29 15:59:55 -04:00
|
|
|
if task.kind == "direct_dependencies":
|
2026-07-29 16:21:10 -04:00
|
|
|
indices = dependencies[focus_index]
|
2026-07-29 15:59:55 -04:00
|
|
|
elif task.kind == "bounded_impact":
|
2026-07-29 16:21:10 -04:00
|
|
|
indices = _bounded_impact(dependencies, focus_index, task.depth)
|
2026-07-29 15:59:55 -04:00
|
|
|
else:
|
2026-07-29 16:21:10 -04:00
|
|
|
if task.target is None:
|
2026-07-29 15:59:55 -04:00
|
|
|
raise TaskEvidenceError("Dependency-path task has no target")
|
2026-07-29 16:21:10 -04:00
|
|
|
target_index = _generated_index(task.target)
|
2026-07-29 15:59:55 -04:00
|
|
|
indices = _shortest_path(
|
|
|
|
|
dependencies,
|
2026-07-29 16:21:10 -04:00
|
|
|
focus_index,
|
|
|
|
|
target_index,
|
2026-07-29 15:59:55 -04:00
|
|
|
task.depth,
|
|
|
|
|
)
|
|
|
|
|
return tuple(_module_name(index) for index in indices)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _prepare_graph(
|
|
|
|
|
root: Path,
|
|
|
|
|
source_bytes: int,
|
2026-07-29 16:21:10 -04:00
|
|
|
*,
|
|
|
|
|
expected_names: set[str],
|
|
|
|
|
project_id: str,
|
|
|
|
|
title: str,
|
|
|
|
|
loader: AdapterLoader | None = None,
|
|
|
|
|
relation: str = "depends_on",
|
|
|
|
|
operation: str = "cold reference-adapter build plus module identity map",
|
2026-07-29 15:59:55 -04:00
|
|
|
) -> tuple[_GraphContext, dict[str, object]]:
|
2026-07-29 16:21:10 -04:00
|
|
|
started = time.perf_counter_ns()
|
|
|
|
|
adapter: AdapterLoader = loader or PythonReferenceAdapter(
|
|
|
|
|
root, source_roots=("src",), project_id=project_id, title=title
|
2026-07-29 15:59:55 -04:00
|
|
|
)
|
|
|
|
|
project = AdapterProject(adapter, cache_root=root / ".docforge" / "task-evidence")
|
|
|
|
|
index = ProjectIndex(project)
|
|
|
|
|
build = index.build()
|
|
|
|
|
snapshot = project.load()
|
|
|
|
|
elapsed_ms = (time.perf_counter_ns() - started) / 1_000_000
|
|
|
|
|
if elapsed_ms > PREPARATION_LIMIT_MS:
|
|
|
|
|
raise TaskEvidenceError(
|
|
|
|
|
f"Graph preparation {elapsed_ms:.3f} ms exceeded {PREPARATION_LIMIT_MS:.3f} ms"
|
|
|
|
|
)
|
|
|
|
|
module_node_ids = {
|
|
|
|
|
node.title: node.node_id for node in snapshot.nodes if node.title in expected_names
|
|
|
|
|
}
|
|
|
|
|
if set(module_node_ids) != expected_names:
|
|
|
|
|
raise TaskEvidenceError("Graph preparation did not publish every fixture module")
|
|
|
|
|
node_modules = {node_id: name for name, node_id in module_node_ids.items()}
|
|
|
|
|
module_source_paths = {
|
|
|
|
|
node.title: node.source_path for node in snapshot.nodes if node.title in expected_names
|
|
|
|
|
}
|
2026-07-29 16:21:10 -04:00
|
|
|
outgoing_sets: dict[str, set[str]] = {name: set() for name in expected_names}
|
|
|
|
|
incoming_sets: dict[str, set[str]] = {name: set() for name in expected_names}
|
|
|
|
|
for edge in snapshot.edges:
|
|
|
|
|
if edge.relation != relation:
|
|
|
|
|
continue
|
|
|
|
|
source_name = node_modules.get(edge.source_id)
|
|
|
|
|
target_name = node_modules.get(edge.target_id)
|
|
|
|
|
if source_name is None or target_name is None:
|
|
|
|
|
continue
|
|
|
|
|
outgoing_sets[source_name].add(target_name)
|
|
|
|
|
incoming_sets[target_name].add(source_name)
|
|
|
|
|
outgoing = {name: tuple(sorted(targets)) for name, targets in outgoing_sets.items()}
|
|
|
|
|
incoming = {name: tuple(sorted(sources)) for name, sources in incoming_sets.items()}
|
2026-07-29 15:59:55 -04:00
|
|
|
return (
|
|
|
|
|
_GraphContext(
|
|
|
|
|
index=index,
|
|
|
|
|
module_node_ids=module_node_ids,
|
|
|
|
|
node_modules=node_modules,
|
|
|
|
|
module_source_paths=module_source_paths,
|
2026-07-29 16:21:10 -04:00
|
|
|
outgoing=outgoing,
|
|
|
|
|
incoming=incoming,
|
|
|
|
|
relation=relation,
|
2026-07-29 15:59:55 -04:00
|
|
|
revision=snapshot.revision,
|
|
|
|
|
source_hash=snapshot.source_hash,
|
|
|
|
|
),
|
|
|
|
|
{
|
|
|
|
|
"persistent_preparation": True,
|
2026-07-29 16:21:10 -04:00
|
|
|
"operation": operation,
|
2026-07-29 15:59:55 -04:00
|
|
|
"elapsed_ms": round(elapsed_ms, 3),
|
|
|
|
|
"elapsed_limit_ms": PREPARATION_LIMIT_MS,
|
|
|
|
|
"source_bytes_indexed": source_bytes,
|
|
|
|
|
"node_count": build["node_count"],
|
|
|
|
|
"edge_count": build["edge_count"],
|
|
|
|
|
"revision": snapshot.revision,
|
|
|
|
|
"source_hash": snapshot.source_hash,
|
|
|
|
|
},
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _graph_response_bytes(responses: Sequence[Mapping[str, object]]) -> int:
|
|
|
|
|
return sum(len(_compact_json(response)) for response in responses)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _graph_direct(context: _GraphContext, task: TaskSpec, source_count: int) -> WorkflowResult:
|
2026-07-29 16:21:10 -04:00
|
|
|
focus_name = task.focus
|
2026-07-29 15:59:55 -04:00
|
|
|
focus_id = context.module_node_ids[focus_name]
|
|
|
|
|
response = context.index.dependencies(focus_id, depth=1, limit=source_count)
|
|
|
|
|
results = cast(list[dict[str, object]], response["results"])
|
|
|
|
|
names = tuple(sorted(context.node_modules[cast(str, item["node_id"])] for item in results))
|
|
|
|
|
provenance = tuple(
|
|
|
|
|
{
|
|
|
|
|
"revision": context.revision,
|
|
|
|
|
"source_hash": context.source_hash,
|
|
|
|
|
"source_path": context.module_source_paths[focus_name],
|
|
|
|
|
"source": focus_name,
|
|
|
|
|
"relation": "depends_on",
|
|
|
|
|
"target": name,
|
|
|
|
|
}
|
|
|
|
|
for name in names
|
|
|
|
|
)
|
|
|
|
|
return WorkflowResult(
|
|
|
|
|
answer=names,
|
|
|
|
|
provenance=provenance,
|
|
|
|
|
inspected_bytes=_graph_response_bytes((response,)),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _graph_impact(context: _GraphContext, task: TaskSpec, source_count: int) -> WorkflowResult:
|
2026-07-29 16:21:10 -04:00
|
|
|
focus_name = task.focus
|
2026-07-29 15:59:55 -04:00
|
|
|
queue: deque[tuple[str, int]] = deque([(focus_name, 0)])
|
|
|
|
|
seen = {focus_name}
|
|
|
|
|
discovered: set[str] = set()
|
|
|
|
|
supporting_edges: list[dict[str, object]] = []
|
|
|
|
|
responses: list[Mapping[str, object]] = []
|
|
|
|
|
while queue:
|
|
|
|
|
current_name, current_depth = queue.popleft()
|
|
|
|
|
if current_depth >= task.depth:
|
|
|
|
|
continue
|
|
|
|
|
response = context.index.backlinks(
|
|
|
|
|
context.module_node_ids[current_name],
|
|
|
|
|
relation="depends_on",
|
|
|
|
|
limit=source_count,
|
|
|
|
|
)
|
|
|
|
|
responses.append(response)
|
|
|
|
|
edges = cast(list[dict[str, object]], response["edges"])
|
|
|
|
|
for edge in edges:
|
|
|
|
|
source_name = context.node_modules[cast(str, edge["source_id"])]
|
|
|
|
|
if source_name in seen:
|
|
|
|
|
continue
|
|
|
|
|
seen.add(source_name)
|
|
|
|
|
discovered.add(source_name)
|
|
|
|
|
supporting_edges.append(
|
|
|
|
|
{
|
|
|
|
|
"revision": context.revision,
|
|
|
|
|
"source_hash": context.source_hash,
|
|
|
|
|
"source_path": context.module_source_paths[source_name],
|
|
|
|
|
"source": source_name,
|
|
|
|
|
"relation": "depends_on",
|
|
|
|
|
"target": current_name,
|
|
|
|
|
}
|
|
|
|
|
)
|
|
|
|
|
queue.append((source_name, current_depth + 1))
|
|
|
|
|
return WorkflowResult(
|
|
|
|
|
answer=tuple(sorted(discovered)),
|
|
|
|
|
provenance=tuple(
|
|
|
|
|
sorted(
|
|
|
|
|
supporting_edges,
|
|
|
|
|
key=lambda item: (cast(str, item["source"]), cast(str, item["target"])),
|
|
|
|
|
)
|
|
|
|
|
),
|
|
|
|
|
inspected_bytes=_graph_response_bytes(responses),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _graph_path(context: _GraphContext, task: TaskSpec, source_count: int) -> WorkflowResult:
|
2026-07-29 16:21:10 -04:00
|
|
|
if task.target is None:
|
2026-07-29 15:59:55 -04:00
|
|
|
raise TaskEvidenceError("Dependency-path task has no target")
|
2026-07-29 16:21:10 -04:00
|
|
|
focus_name = task.focus
|
|
|
|
|
target_name = task.target
|
2026-07-29 15:59:55 -04:00
|
|
|
response = context.index.dependencies(
|
|
|
|
|
context.module_node_ids[focus_name],
|
|
|
|
|
depth=task.depth,
|
|
|
|
|
limit=source_count,
|
|
|
|
|
)
|
|
|
|
|
results = cast(list[dict[str, object]], response["results"])
|
|
|
|
|
match = next(
|
|
|
|
|
(
|
|
|
|
|
item
|
|
|
|
|
for item in results
|
|
|
|
|
if context.node_modules[cast(str, item["node_id"])] == target_name
|
|
|
|
|
),
|
|
|
|
|
None,
|
|
|
|
|
)
|
|
|
|
|
if match is None:
|
|
|
|
|
raise TaskEvidenceError("Graph-assisted workflow did not find the answer-key path")
|
|
|
|
|
path_ids = cast(list[str] | tuple[str, ...], match["path"])
|
|
|
|
|
path = tuple(context.node_modules[node_id] for node_id in path_ids)
|
|
|
|
|
provenance = tuple(
|
|
|
|
|
{
|
|
|
|
|
"revision": context.revision,
|
|
|
|
|
"source_hash": context.source_hash,
|
|
|
|
|
"source_path": context.module_source_paths[source],
|
|
|
|
|
"source": source,
|
|
|
|
|
"relation": "depends_on",
|
|
|
|
|
"target": target,
|
|
|
|
|
}
|
|
|
|
|
for source, target in zip(path, path[1:], strict=False)
|
|
|
|
|
)
|
|
|
|
|
return WorkflowResult(
|
|
|
|
|
answer=path,
|
|
|
|
|
provenance=provenance,
|
|
|
|
|
inspected_bytes=_graph_response_bytes((response,)),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def run_graph_task(
|
|
|
|
|
context: _GraphContext,
|
|
|
|
|
task: TaskSpec,
|
|
|
|
|
source_count: int,
|
|
|
|
|
) -> WorkflowResult:
|
|
|
|
|
if task.kind == "direct_dependencies":
|
|
|
|
|
return _graph_direct(context, task, source_count)
|
|
|
|
|
if task.kind == "bounded_impact":
|
|
|
|
|
return _graph_impact(context, task, source_count)
|
|
|
|
|
return _graph_path(context, task, source_count)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _parse_source_dependencies(
|
|
|
|
|
raw: bytes,
|
|
|
|
|
source_index: int,
|
|
|
|
|
) -> tuple[tuple[int, int], ...]:
|
|
|
|
|
prefix = "from evidence.component_"
|
|
|
|
|
suffix = " import compute_"
|
|
|
|
|
imports: list[tuple[int, int]] = []
|
|
|
|
|
for line_number, line in enumerate(raw.decode("utf-8").splitlines(), start=1):
|
|
|
|
|
if not line.startswith(prefix) or suffix not in line:
|
|
|
|
|
continue
|
|
|
|
|
module_text, function_text = line[len(prefix) :].split(suffix, maxsplit=1)
|
|
|
|
|
if not (
|
|
|
|
|
len(module_text) == 3
|
|
|
|
|
and module_text.isdigit()
|
|
|
|
|
and len(function_text) == 3
|
|
|
|
|
and function_text.isdigit()
|
|
|
|
|
and module_text == function_text
|
|
|
|
|
):
|
|
|
|
|
raise TaskEvidenceError(
|
|
|
|
|
f"Source-only parser rejected component {source_index:03d} import syntax"
|
|
|
|
|
)
|
|
|
|
|
imports.append((int(module_text), line_number))
|
|
|
|
|
return tuple(imports)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _read_source_dependencies(
|
|
|
|
|
root: Path,
|
|
|
|
|
source_index: int,
|
|
|
|
|
) -> tuple[tuple[tuple[int, int], ...], int]:
|
|
|
|
|
raw = _component_path(root, source_index).read_bytes()
|
|
|
|
|
return _parse_source_dependencies(raw, source_index), len(raw)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _read_all_source_dependencies(
|
|
|
|
|
root: Path,
|
|
|
|
|
source_count: int,
|
|
|
|
|
) -> tuple[dict[int, tuple[int, ...]], dict[tuple[int, int], int], int]:
|
|
|
|
|
dependencies: dict[int, tuple[int, ...]] = {}
|
|
|
|
|
lines: dict[tuple[int, int], int] = {}
|
|
|
|
|
inspected_bytes = 0
|
|
|
|
|
for source in range(source_count):
|
|
|
|
|
imports, size = _read_source_dependencies(root, source)
|
|
|
|
|
inspected_bytes += size
|
|
|
|
|
dependencies[source] = tuple(target for target, _ in imports)
|
|
|
|
|
lines.update({(source, target): line for target, line in imports})
|
|
|
|
|
return dependencies, lines, inspected_bytes
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _source_provenance(
|
|
|
|
|
edges: Sequence[tuple[int, int]],
|
|
|
|
|
lines: Mapping[tuple[int, int], int],
|
|
|
|
|
) -> tuple[Mapping[str, object], ...]:
|
|
|
|
|
return tuple(
|
|
|
|
|
{
|
|
|
|
|
"source_path": f"src/evidence/component_{source:03d}.py",
|
|
|
|
|
"line": lines[(source, target)],
|
|
|
|
|
"source": _module_name(source),
|
|
|
|
|
"relation": "depends_on",
|
|
|
|
|
"target": _module_name(target),
|
|
|
|
|
}
|
|
|
|
|
for source, target in edges
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _source_direct(root: Path, task: TaskSpec) -> WorkflowResult:
|
2026-07-29 16:21:10 -04:00
|
|
|
focus_index = _generated_index(task.focus)
|
|
|
|
|
imports, inspected_bytes = _read_source_dependencies(root, focus_index)
|
|
|
|
|
edges = tuple((focus_index, target) for target, _ in imports)
|
|
|
|
|
lines = {(focus_index, target): line for target, line in imports}
|
2026-07-29 15:59:55 -04:00
|
|
|
return WorkflowResult(
|
|
|
|
|
answer=tuple(_module_name(target) for target, _ in imports),
|
|
|
|
|
provenance=_source_provenance(edges, lines),
|
|
|
|
|
inspected_bytes=inspected_bytes,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _source_impact(root: Path, task: TaskSpec, source_count: int) -> WorkflowResult:
|
|
|
|
|
dependencies, lines, inspected_bytes = _read_all_source_dependencies(root, source_count)
|
|
|
|
|
reverse: dict[int, list[int]] = {index: [] for index in dependencies}
|
|
|
|
|
for source, targets in dependencies.items():
|
|
|
|
|
for target in targets:
|
|
|
|
|
reverse[target].append(source)
|
2026-07-29 16:21:10 -04:00
|
|
|
focus_index = _generated_index(task.focus)
|
|
|
|
|
queue: deque[tuple[int, int]] = deque([(focus_index, 0)])
|
|
|
|
|
seen = {focus_index}
|
2026-07-29 15:59:55 -04:00
|
|
|
discovered: set[int] = set()
|
|
|
|
|
edges: list[tuple[int, int]] = []
|
|
|
|
|
while queue:
|
|
|
|
|
current, current_depth = queue.popleft()
|
|
|
|
|
if current_depth >= task.depth:
|
|
|
|
|
continue
|
|
|
|
|
for source in sorted(reverse[current]):
|
|
|
|
|
if source in seen:
|
|
|
|
|
continue
|
|
|
|
|
seen.add(source)
|
|
|
|
|
discovered.add(source)
|
|
|
|
|
edges.append((source, current))
|
|
|
|
|
queue.append((source, current_depth + 1))
|
|
|
|
|
return WorkflowResult(
|
|
|
|
|
answer=tuple(_module_name(index) for index in sorted(discovered)),
|
|
|
|
|
provenance=_source_provenance(sorted(edges), lines),
|
|
|
|
|
inspected_bytes=inspected_bytes,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _source_dependency_path(
|
|
|
|
|
root: Path,
|
|
|
|
|
task: TaskSpec,
|
|
|
|
|
source_count: int,
|
|
|
|
|
) -> WorkflowResult:
|
2026-07-29 16:21:10 -04:00
|
|
|
if task.target is None:
|
2026-07-29 15:59:55 -04:00
|
|
|
raise TaskEvidenceError("Dependency-path task has no target")
|
2026-07-29 16:21:10 -04:00
|
|
|
focus_index = _generated_index(task.focus)
|
|
|
|
|
target_index = _generated_index(task.target)
|
2026-07-29 15:59:55 -04:00
|
|
|
dependencies, lines, inspected_bytes = _read_all_source_dependencies(root, source_count)
|
|
|
|
|
path = _shortest_path(
|
|
|
|
|
dependencies,
|
2026-07-29 16:21:10 -04:00
|
|
|
focus_index,
|
|
|
|
|
target_index,
|
2026-07-29 15:59:55 -04:00
|
|
|
task.depth,
|
|
|
|
|
)
|
|
|
|
|
edges = tuple(zip(path, path[1:], strict=False))
|
|
|
|
|
return WorkflowResult(
|
|
|
|
|
answer=tuple(_module_name(index) for index in path),
|
|
|
|
|
provenance=_source_provenance(edges, lines),
|
|
|
|
|
inspected_bytes=inspected_bytes,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def run_source_task(root: Path, task: TaskSpec, source_count: int) -> WorkflowResult:
|
|
|
|
|
if task.kind == "direct_dependencies":
|
|
|
|
|
return _source_direct(root, task)
|
|
|
|
|
if task.kind == "bounded_impact":
|
|
|
|
|
return _source_impact(root, task, source_count)
|
|
|
|
|
return _source_dependency_path(root, task, source_count)
|
|
|
|
|
|
|
|
|
|
|
2026-07-29 16:21:10 -04:00
|
|
|
def _locked_distribution_version(distribution_name: str) -> str:
|
|
|
|
|
document = tomllib.loads((ROOT / "uv.lock").read_text(encoding="utf-8"))
|
|
|
|
|
packages = document.get("package")
|
|
|
|
|
if not isinstance(packages, list):
|
|
|
|
|
raise TaskEvidenceError("uv.lock does not contain a package inventory")
|
|
|
|
|
matches: list[str] = []
|
|
|
|
|
for raw_package in cast(list[object], packages):
|
|
|
|
|
if not isinstance(raw_package, dict):
|
|
|
|
|
continue
|
|
|
|
|
package = cast(dict[str, object], raw_package)
|
|
|
|
|
if package.get("name") != distribution_name:
|
|
|
|
|
continue
|
|
|
|
|
version = package.get("version")
|
|
|
|
|
if not isinstance(version, str):
|
|
|
|
|
raise TaskEvidenceError(f"{distribution_name} has no fixed lockfile version")
|
|
|
|
|
matches.append(version)
|
|
|
|
|
if len(matches) != 1:
|
|
|
|
|
raise TaskEvidenceError(
|
|
|
|
|
f"uv.lock must contain exactly one {distribution_name} package record"
|
|
|
|
|
)
|
|
|
|
|
return matches[0]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _installed_python_sources(package_root: Path) -> tuple[Path, ...]:
|
|
|
|
|
try:
|
|
|
|
|
resolved_root = package_root.resolve(strict=True)
|
|
|
|
|
except OSError as error:
|
|
|
|
|
raise TaskEvidenceError("Pinned real-package source root is unavailable") from error
|
|
|
|
|
sources: list[Path] = []
|
|
|
|
|
for path in sorted(
|
|
|
|
|
resolved_root.rglob("*.py"),
|
|
|
|
|
key=lambda item: item.relative_to(resolved_root).as_posix(),
|
|
|
|
|
):
|
|
|
|
|
try:
|
|
|
|
|
metadata_result = path.lstat()
|
|
|
|
|
resolved = path.resolve(strict=True)
|
|
|
|
|
except OSError as error:
|
|
|
|
|
raise TaskEvidenceError("Pinned real-package source cannot be inspected") from error
|
|
|
|
|
if (
|
|
|
|
|
stat.S_ISLNK(metadata_result.st_mode)
|
|
|
|
|
or not stat.S_ISREG(metadata_result.st_mode)
|
|
|
|
|
or not resolved.is_relative_to(resolved_root)
|
|
|
|
|
or resolved != path
|
|
|
|
|
):
|
|
|
|
|
raise TaskEvidenceError("Pinned real-package sources must be confined regular files")
|
|
|
|
|
sources.append(path)
|
|
|
|
|
return tuple(sources)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _source_tree_identity(package_root: Path) -> tuple[int, int, str]:
|
|
|
|
|
digest = hashlib.sha256()
|
|
|
|
|
total_bytes = 0
|
|
|
|
|
sources = _installed_python_sources(package_root)
|
|
|
|
|
for path in sources:
|
|
|
|
|
relative = path.relative_to(package_root).as_posix()
|
|
|
|
|
raw = path.read_bytes()
|
|
|
|
|
total_bytes += len(raw)
|
|
|
|
|
digest.update(relative.encode("utf-8"))
|
|
|
|
|
digest.update(b"\0")
|
|
|
|
|
digest.update(str(len(raw)).encode("ascii"))
|
|
|
|
|
digest.update(b"\0")
|
|
|
|
|
digest.update(raw)
|
|
|
|
|
digest.update(b"\0")
|
|
|
|
|
return len(sources), total_bytes, digest.hexdigest()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def installed_real_package() -> tuple[Path, dict[str, object]]:
|
|
|
|
|
"""Resolve and freeze the lock-pinned installed real-package source tree."""
|
|
|
|
|
|
|
|
|
|
locked_version = _locked_distribution_version(PINNED_REAL_DISTRIBUTION)
|
|
|
|
|
try:
|
|
|
|
|
distribution = metadata.distribution(PINNED_REAL_DISTRIBUTION)
|
|
|
|
|
except metadata.PackageNotFoundError as error:
|
|
|
|
|
raise TaskEvidenceError(
|
|
|
|
|
f"{PINNED_REAL_DISTRIBUTION} is not installed in the active environment"
|
|
|
|
|
) from error
|
|
|
|
|
installed_version = distribution.version
|
|
|
|
|
package_root = Path(str(distribution.locate_file(PINNED_REAL_IMPORT_ROOT))).resolve(strict=True)
|
|
|
|
|
source_count, source_bytes, source_sha256 = _source_tree_identity(package_root)
|
|
|
|
|
observed = (
|
|
|
|
|
locked_version,
|
|
|
|
|
installed_version,
|
|
|
|
|
source_count,
|
|
|
|
|
source_bytes,
|
|
|
|
|
source_sha256,
|
|
|
|
|
)
|
|
|
|
|
expected = (
|
|
|
|
|
PINNED_REAL_VERSION,
|
|
|
|
|
PINNED_REAL_VERSION,
|
|
|
|
|
PINNED_REAL_SOURCE_COUNT,
|
|
|
|
|
PINNED_REAL_SOURCE_BYTES,
|
|
|
|
|
PINNED_REAL_SOURCE_SHA256,
|
|
|
|
|
)
|
|
|
|
|
if observed != expected:
|
|
|
|
|
raise TaskEvidenceError(
|
|
|
|
|
"Pinned real-package evidence changed; update the lock and expected evidence "
|
|
|
|
|
f"deliberately. Observed {observed!r}, expected {expected!r}"
|
|
|
|
|
)
|
|
|
|
|
return package_root, {
|
|
|
|
|
"distribution": PINNED_REAL_DISTRIBUTION,
|
|
|
|
|
"import_root": PINNED_REAL_IMPORT_ROOT,
|
|
|
|
|
"lock_version": locked_version,
|
|
|
|
|
"installed_version": installed_version,
|
|
|
|
|
"source_scope": "sorted markdown_it/**/*.py path, length, and content",
|
|
|
|
|
"source_count": source_count,
|
|
|
|
|
"source_bytes": source_bytes,
|
|
|
|
|
"source_tree_sha256": source_sha256,
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _copy_real_package_read_only(source_root: Path, project_root: Path) -> None:
|
|
|
|
|
destination_root = project_root / "src" / PINNED_REAL_IMPORT_ROOT
|
|
|
|
|
for source in _installed_python_sources(source_root):
|
|
|
|
|
relative = source.relative_to(source_root)
|
|
|
|
|
destination = destination_root / relative
|
|
|
|
|
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
|
shutil.copyfile(source, destination)
|
|
|
|
|
destination.chmod(0o444)
|
|
|
|
|
for directory in sorted(
|
|
|
|
|
(path for path in destination_root.rglob("*") if path.is_dir()),
|
|
|
|
|
key=lambda path: len(path.parts),
|
|
|
|
|
reverse=True,
|
|
|
|
|
):
|
|
|
|
|
directory.chmod(0o555)
|
|
|
|
|
destination_root.chmod(0o555)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _python_module_name(source_root: Path, path: Path) -> str:
|
|
|
|
|
local = path.relative_to(source_root).with_suffix("")
|
|
|
|
|
parts = list(local.parts)
|
|
|
|
|
if parts[-1] == "__init__":
|
|
|
|
|
parts.pop()
|
|
|
|
|
if not parts:
|
|
|
|
|
raise TaskEvidenceError("Real-package source produced an empty module name")
|
|
|
|
|
return ".".join(parts)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _real_module_inventory(project_root: Path) -> dict[str, Path]:
|
|
|
|
|
source_root = project_root / "src"
|
|
|
|
|
inventory = {
|
|
|
|
|
_python_module_name(source_root, path): path
|
|
|
|
|
for path in sorted(
|
|
|
|
|
(source_root / PINNED_REAL_IMPORT_ROOT).rglob("*.py"),
|
|
|
|
|
key=lambda item: item.relative_to(source_root).as_posix(),
|
|
|
|
|
)
|
|
|
|
|
}
|
|
|
|
|
if len(inventory) != PINNED_REAL_SOURCE_COUNT:
|
|
|
|
|
raise TaskEvidenceError("Copied real-package module inventory changed")
|
|
|
|
|
return inventory
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _relative_import_base(package: str, level: int, module: str | None) -> str:
|
|
|
|
|
if level == 0:
|
|
|
|
|
return module or ""
|
|
|
|
|
package_parts = package.split(".") if package else []
|
|
|
|
|
keep = len(package_parts) - (level - 1)
|
|
|
|
|
if keep < 0:
|
|
|
|
|
return ""
|
|
|
|
|
prefix = package_parts[:keep]
|
|
|
|
|
if module:
|
|
|
|
|
prefix.extend(module.split("."))
|
|
|
|
|
return ".".join(prefix)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _parse_real_dependencies(
|
|
|
|
|
raw: bytes,
|
|
|
|
|
*,
|
|
|
|
|
module_name: str,
|
|
|
|
|
package: str,
|
|
|
|
|
local_modules: set[str],
|
|
|
|
|
source_path: str,
|
|
|
|
|
) -> tuple[tuple[str, int], ...]:
|
|
|
|
|
try:
|
|
|
|
|
tree = ast.parse(raw, filename=source_path, type_comments=True)
|
|
|
|
|
except (SyntaxError, UnicodeDecodeError) as error:
|
|
|
|
|
raise TaskEvidenceError(f"Cannot inspect pinned source {source_path}") from error
|
|
|
|
|
evidence: dict[str, int] = {}
|
|
|
|
|
for node in ast.walk(tree):
|
|
|
|
|
candidates: list[str] = []
|
|
|
|
|
if isinstance(node, ast.Import):
|
|
|
|
|
candidates.extend(alias.name for alias in node.names)
|
|
|
|
|
elif isinstance(node, ast.ImportFrom):
|
|
|
|
|
base = _relative_import_base(package, node.level, node.module)
|
|
|
|
|
if base:
|
|
|
|
|
candidates.append(base)
|
|
|
|
|
candidates.extend(
|
|
|
|
|
f"{base}.{alias.name}" for alias in node.names if alias.name != "*"
|
|
|
|
|
)
|
|
|
|
|
else:
|
|
|
|
|
continue
|
|
|
|
|
for candidate in candidates:
|
|
|
|
|
if candidate in local_modules and candidate != module_name:
|
|
|
|
|
previous = evidence.get(candidate)
|
|
|
|
|
evidence[candidate] = (
|
|
|
|
|
node.lineno if previous is None else min(previous, node.lineno)
|
|
|
|
|
)
|
|
|
|
|
return tuple(sorted(evidence.items()))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _read_real_module_dependencies(
|
|
|
|
|
project_root: Path,
|
|
|
|
|
inventory: Mapping[str, Path],
|
|
|
|
|
module_name: str,
|
|
|
|
|
) -> tuple[tuple[tuple[str, int], ...], int]:
|
|
|
|
|
path = inventory[module_name]
|
|
|
|
|
raw = path.read_bytes()
|
|
|
|
|
package = module_name if path.name == "__init__.py" else module_name.rpartition(".")[0]
|
|
|
|
|
relative = path.relative_to(project_root).as_posix()
|
|
|
|
|
return (
|
|
|
|
|
_parse_real_dependencies(
|
|
|
|
|
raw,
|
|
|
|
|
module_name=module_name,
|
|
|
|
|
package=package,
|
|
|
|
|
local_modules=set(inventory),
|
|
|
|
|
source_path=relative,
|
|
|
|
|
),
|
|
|
|
|
len(raw),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _read_all_real_dependencies(
|
|
|
|
|
project_root: Path,
|
|
|
|
|
inventory: Mapping[str, Path],
|
|
|
|
|
) -> tuple[dict[str, tuple[str, ...]], dict[tuple[str, str], tuple[str, int]], int]:
|
|
|
|
|
dependencies: dict[str, tuple[str, ...]] = {}
|
|
|
|
|
locations: dict[tuple[str, str], tuple[str, int]] = {}
|
|
|
|
|
inspected_bytes = 0
|
|
|
|
|
for module_name in sorted(inventory):
|
|
|
|
|
imports, size = _read_real_module_dependencies(project_root, inventory, module_name)
|
|
|
|
|
inspected_bytes += size
|
|
|
|
|
dependencies[module_name] = tuple(target for target, _ in imports)
|
|
|
|
|
relative = inventory[module_name].relative_to(project_root).as_posix()
|
|
|
|
|
locations.update({(module_name, target): (relative, line) for target, line in imports})
|
|
|
|
|
return dependencies, locations, inspected_bytes
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class _RealPackageDependencyAdapter:
|
|
|
|
|
"""Minimal immutable projection over the pinned read-only package sources."""
|
|
|
|
|
|
|
|
|
|
def __init__(self, root: Path) -> None:
|
|
|
|
|
self.root = root.resolve(strict=True)
|
|
|
|
|
self._projection: AdapterProjection | None = None
|
|
|
|
|
|
|
|
|
|
@staticmethod
|
|
|
|
|
def _node_id(module_name: str) -> str:
|
|
|
|
|
digest = hashlib.sha256(module_name.encode("utf-8")).hexdigest()[:24]
|
|
|
|
|
return f"real.module.{digest}"
|
|
|
|
|
|
|
|
|
|
def load_projection(self) -> AdapterProjection:
|
|
|
|
|
if self._projection is not None:
|
|
|
|
|
return self._projection
|
|
|
|
|
inventory = _real_module_inventory(self.root)
|
|
|
|
|
reference_adapter = PythonReferenceAdapter(
|
|
|
|
|
self.root,
|
|
|
|
|
source_roots=("src",),
|
|
|
|
|
project_id="milestone5-real-task-evidence",
|
|
|
|
|
title="Milestone 5 markdown-it-py comparative task evidence",
|
|
|
|
|
)
|
|
|
|
|
manifest = reference_adapter.load_manifest()
|
|
|
|
|
names_by_path = {
|
|
|
|
|
path.relative_to(self.root).as_posix(): module_name
|
|
|
|
|
for module_name, path in inventory.items()
|
|
|
|
|
}
|
|
|
|
|
names_by_source_id = {
|
|
|
|
|
source.source_id: names_by_path[source.source_path] for source in manifest.sources
|
|
|
|
|
}
|
|
|
|
|
dependencies = {
|
|
|
|
|
names_by_source_id[source.source_id]: tuple(
|
|
|
|
|
sorted(names_by_source_id[target] for target in source.dependencies)
|
|
|
|
|
)
|
|
|
|
|
for source in manifest.sources
|
|
|
|
|
}
|
|
|
|
|
source_bytes = sum(path.stat().st_size for path in inventory.values())
|
|
|
|
|
if source_bytes != PINNED_REAL_SOURCE_BYTES:
|
|
|
|
|
raise TaskEvidenceError("Copied real-package source bytes changed")
|
|
|
|
|
nodes: list[AdapterNode] = []
|
|
|
|
|
for module_name, path in inventory.items():
|
|
|
|
|
raw = path.read_bytes()
|
|
|
|
|
relative = path.relative_to(self.root).as_posix()
|
|
|
|
|
nodes.append(
|
|
|
|
|
AdapterNode(
|
|
|
|
|
Node(
|
|
|
|
|
node_id=self._node_id(module_name),
|
|
|
|
|
title=module_name,
|
|
|
|
|
family="code",
|
|
|
|
|
authority="derived",
|
|
|
|
|
status="active",
|
|
|
|
|
tags=("module", "pinned-real-package"),
|
|
|
|
|
summary=f"Pinned installed Python module {module_name}.",
|
|
|
|
|
content=f"Local dependency projection for {module_name}.",
|
|
|
|
|
source_path=relative,
|
|
|
|
|
source_anchor="L1",
|
|
|
|
|
content_hash=hashlib.sha256(raw).hexdigest(),
|
|
|
|
|
),
|
|
|
|
|
metadata=(("kind", "module"),),
|
|
|
|
|
)
|
|
|
|
|
)
|
|
|
|
|
edges = [
|
|
|
|
|
AdapterEdge(
|
|
|
|
|
Edge(
|
|
|
|
|
source_id=self._node_id(source),
|
|
|
|
|
relation="imports",
|
|
|
|
|
target_id=self._node_id(target),
|
|
|
|
|
),
|
|
|
|
|
metadata=(("evidence", "python_reference_adapter_manifest"),),
|
|
|
|
|
)
|
|
|
|
|
for source, targets in dependencies.items()
|
|
|
|
|
for target in targets
|
|
|
|
|
]
|
|
|
|
|
self._projection = AdapterProjection(
|
|
|
|
|
project_id="milestone5-real-task-evidence",
|
|
|
|
|
title="Milestone 5 markdown-it-py comparative task evidence",
|
|
|
|
|
adapter_id="pinned-python-dependencies",
|
|
|
|
|
adapter_version="1",
|
|
|
|
|
root=self.root,
|
|
|
|
|
revision=manifest.revision,
|
|
|
|
|
source_hash=manifest.source_hash,
|
|
|
|
|
nodes=tuple(sorted(nodes, key=lambda item: item.node.node_id)),
|
|
|
|
|
edges=tuple(
|
|
|
|
|
sorted(
|
|
|
|
|
edges,
|
|
|
|
|
key=lambda item: (
|
|
|
|
|
item.edge.source_id,
|
|
|
|
|
item.edge.relation,
|
|
|
|
|
item.edge.target_id,
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
return self._projection
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _real_provenance(
|
|
|
|
|
edges: Sequence[tuple[str, str]],
|
|
|
|
|
locations: Mapping[tuple[str, str], tuple[str, int]],
|
|
|
|
|
) -> tuple[Mapping[str, object], ...]:
|
|
|
|
|
return tuple(
|
|
|
|
|
{
|
|
|
|
|
"source_path": locations[(source, target)][0],
|
|
|
|
|
"line": locations[(source, target)][1],
|
|
|
|
|
"source": source,
|
|
|
|
|
"relation": "imports",
|
|
|
|
|
"target": target,
|
|
|
|
|
}
|
|
|
|
|
for source, target in edges
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _shortest_named_path(
|
|
|
|
|
dependencies: Mapping[str, Sequence[str]],
|
|
|
|
|
focus: str,
|
|
|
|
|
target: str,
|
|
|
|
|
depth: int,
|
|
|
|
|
) -> tuple[str, ...]:
|
|
|
|
|
queue: deque[tuple[str, tuple[str, ...]]] = deque([(focus, (focus,))])
|
|
|
|
|
seen = {focus}
|
|
|
|
|
while queue:
|
|
|
|
|
current, path = queue.popleft()
|
|
|
|
|
if current == target:
|
|
|
|
|
return path
|
|
|
|
|
if len(path) - 1 >= depth:
|
|
|
|
|
continue
|
|
|
|
|
for dependency in sorted(dependencies[current]):
|
|
|
|
|
if dependency in seen:
|
|
|
|
|
continue
|
|
|
|
|
seen.add(dependency)
|
|
|
|
|
queue.append((dependency, (*path, dependency)))
|
|
|
|
|
raise TaskEvidenceError(f"No real-package dependency path exists from {focus} to {target}")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _real_source_direct(
|
|
|
|
|
project_root: Path,
|
|
|
|
|
inventory: Mapping[str, Path],
|
|
|
|
|
task: TaskSpec,
|
|
|
|
|
) -> WorkflowResult:
|
|
|
|
|
imports, inspected_bytes = _read_real_module_dependencies(
|
|
|
|
|
project_root,
|
|
|
|
|
inventory,
|
|
|
|
|
task.focus,
|
|
|
|
|
)
|
|
|
|
|
relative = inventory[task.focus].relative_to(project_root).as_posix()
|
|
|
|
|
locations = {(task.focus, target): (relative, line) for target, line in imports}
|
|
|
|
|
edges = tuple((task.focus, target) for target, _ in imports)
|
|
|
|
|
return WorkflowResult(
|
|
|
|
|
answer=tuple(target for target, _ in imports),
|
|
|
|
|
provenance=_real_provenance(edges, locations),
|
|
|
|
|
inspected_bytes=inspected_bytes,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _real_source_impact(
|
|
|
|
|
project_root: Path,
|
|
|
|
|
inventory: Mapping[str, Path],
|
|
|
|
|
task: TaskSpec,
|
|
|
|
|
) -> WorkflowResult:
|
|
|
|
|
dependencies, locations, inspected_bytes = _read_all_real_dependencies(
|
|
|
|
|
project_root,
|
|
|
|
|
inventory,
|
|
|
|
|
)
|
|
|
|
|
reverse: dict[str, list[str]] = {module_name: [] for module_name in dependencies}
|
|
|
|
|
for source, targets in dependencies.items():
|
|
|
|
|
for target in targets:
|
|
|
|
|
reverse[target].append(source)
|
|
|
|
|
queue: deque[tuple[str, int]] = deque([(task.focus, 0)])
|
|
|
|
|
seen = {task.focus}
|
|
|
|
|
discovered: set[str] = set()
|
|
|
|
|
edges: list[tuple[str, str]] = []
|
|
|
|
|
while queue:
|
|
|
|
|
current, current_depth = queue.popleft()
|
|
|
|
|
if current_depth >= task.depth:
|
|
|
|
|
continue
|
|
|
|
|
for source in sorted(reverse[current]):
|
|
|
|
|
if source in seen:
|
|
|
|
|
continue
|
|
|
|
|
seen.add(source)
|
|
|
|
|
discovered.add(source)
|
|
|
|
|
edges.append((source, current))
|
|
|
|
|
queue.append((source, current_depth + 1))
|
|
|
|
|
return WorkflowResult(
|
|
|
|
|
answer=tuple(sorted(discovered)),
|
|
|
|
|
provenance=_real_provenance(sorted(edges), locations),
|
|
|
|
|
inspected_bytes=inspected_bytes,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _real_source_path(
|
|
|
|
|
project_root: Path,
|
|
|
|
|
inventory: Mapping[str, Path],
|
|
|
|
|
task: TaskSpec,
|
|
|
|
|
) -> WorkflowResult:
|
|
|
|
|
if task.target is None:
|
|
|
|
|
raise TaskEvidenceError("Real dependency-path task has no target")
|
|
|
|
|
dependencies, locations, inspected_bytes = _read_all_real_dependencies(
|
|
|
|
|
project_root,
|
|
|
|
|
inventory,
|
|
|
|
|
)
|
|
|
|
|
path = _shortest_named_path(dependencies, task.focus, task.target, task.depth)
|
|
|
|
|
edges = tuple(zip(path, path[1:], strict=False))
|
|
|
|
|
return WorkflowResult(
|
|
|
|
|
answer=path,
|
|
|
|
|
provenance=_real_provenance(edges, locations),
|
|
|
|
|
inspected_bytes=inspected_bytes,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def run_real_source_task(
|
|
|
|
|
project_root: Path,
|
|
|
|
|
inventory: Mapping[str, Path],
|
|
|
|
|
task: TaskSpec,
|
|
|
|
|
) -> WorkflowResult:
|
|
|
|
|
if task.kind == "direct_dependencies":
|
|
|
|
|
return _real_source_direct(project_root, inventory, task)
|
|
|
|
|
if task.kind == "bounded_impact":
|
|
|
|
|
return _real_source_impact(project_root, inventory, task)
|
|
|
|
|
return _real_source_path(project_root, inventory, task)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _real_graph_provenance(
|
|
|
|
|
context: _GraphContext,
|
|
|
|
|
edges: Sequence[tuple[str, str]],
|
|
|
|
|
) -> tuple[Mapping[str, object], ...]:
|
|
|
|
|
return tuple(
|
|
|
|
|
{
|
|
|
|
|
"revision": context.revision,
|
|
|
|
|
"source_hash": context.source_hash,
|
|
|
|
|
"source_path": context.module_source_paths[source],
|
|
|
|
|
"source": source,
|
|
|
|
|
"relation": context.relation,
|
|
|
|
|
"target": target,
|
|
|
|
|
}
|
|
|
|
|
for source, target in edges
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def run_real_graph_task(context: _GraphContext, task: TaskSpec) -> WorkflowResult:
|
|
|
|
|
"""Answer one task from the bounded immutable real-package import graph."""
|
|
|
|
|
|
|
|
|
|
if context.relation != "imports":
|
|
|
|
|
raise TaskEvidenceError("Real-package graph does not publish exact import relationships")
|
|
|
|
|
if task.kind == "direct_dependencies":
|
|
|
|
|
answer = context.outgoing[task.focus]
|
|
|
|
|
edges = tuple((task.focus, target) for target in answer)
|
|
|
|
|
response: dict[str, object] = {
|
|
|
|
|
"revision": context.revision,
|
|
|
|
|
"root": task.focus,
|
|
|
|
|
"relation": context.relation,
|
|
|
|
|
"results": [
|
|
|
|
|
{"source": source, "relation": context.relation, "target": target}
|
|
|
|
|
for source, target in edges
|
|
|
|
|
],
|
|
|
|
|
}
|
|
|
|
|
elif task.kind == "bounded_impact":
|
|
|
|
|
queue: deque[tuple[str, int]] = deque([(task.focus, 0)])
|
|
|
|
|
seen = {task.focus}
|
|
|
|
|
discovered: set[str] = set()
|
|
|
|
|
edge_list: list[tuple[str, str]] = []
|
|
|
|
|
while queue:
|
|
|
|
|
current, current_depth = queue.popleft()
|
|
|
|
|
if current_depth >= task.depth:
|
|
|
|
|
continue
|
|
|
|
|
for source in context.incoming[current]:
|
|
|
|
|
if source in seen:
|
|
|
|
|
continue
|
|
|
|
|
seen.add(source)
|
|
|
|
|
discovered.add(source)
|
|
|
|
|
edge_list.append((source, current))
|
|
|
|
|
queue.append((source, current_depth + 1))
|
|
|
|
|
answer = tuple(sorted(discovered))
|
|
|
|
|
edges = tuple(sorted(edge_list))
|
|
|
|
|
response = {
|
|
|
|
|
"revision": context.revision,
|
|
|
|
|
"root": task.focus,
|
|
|
|
|
"relation": context.relation,
|
|
|
|
|
"depth": task.depth,
|
|
|
|
|
"results": [
|
|
|
|
|
{"source": source, "relation": context.relation, "target": target}
|
|
|
|
|
for source, target in edges
|
|
|
|
|
],
|
|
|
|
|
}
|
|
|
|
|
else:
|
|
|
|
|
if task.target is None:
|
|
|
|
|
raise TaskEvidenceError("Real graph path task has no target")
|
|
|
|
|
answer = _shortest_named_path(
|
|
|
|
|
context.outgoing,
|
|
|
|
|
task.focus,
|
|
|
|
|
task.target,
|
|
|
|
|
task.depth,
|
|
|
|
|
)
|
|
|
|
|
edges = tuple(zip(answer, answer[1:], strict=False))
|
|
|
|
|
response = {
|
|
|
|
|
"revision": context.revision,
|
|
|
|
|
"root": task.focus,
|
|
|
|
|
"target": task.target,
|
|
|
|
|
"relation": context.relation,
|
|
|
|
|
"path": list(answer),
|
|
|
|
|
}
|
|
|
|
|
return WorkflowResult(
|
|
|
|
|
answer=answer,
|
|
|
|
|
provenance=_real_graph_provenance(context, edges),
|
|
|
|
|
inspected_bytes=len(_compact_json(response)),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def real_package_tasks() -> tuple[tuple[TaskSpec, tuple[str, ...]], ...]:
|
|
|
|
|
"""Return fixed real-package tasks and answers for the pinned source identity."""
|
|
|
|
|
|
|
|
|
|
return (
|
|
|
|
|
(
|
|
|
|
|
TaskSpec(
|
|
|
|
|
task_id="markdown_it_renderer_direct_dependencies",
|
|
|
|
|
kind="direct_dependencies",
|
|
|
|
|
focus="markdown_it.renderer",
|
|
|
|
|
target=None,
|
|
|
|
|
depth=1,
|
|
|
|
|
prompt="List the direct local dependencies of markdown_it.renderer.",
|
|
|
|
|
),
|
|
|
|
|
(
|
|
|
|
|
"markdown_it.common.utils",
|
|
|
|
|
"markdown_it.token",
|
|
|
|
|
"markdown_it.utils",
|
|
|
|
|
),
|
|
|
|
|
),
|
|
|
|
|
(
|
|
|
|
|
TaskSpec(
|
|
|
|
|
task_id="markdown_it_html_blocks_three_level_impact",
|
|
|
|
|
kind="bounded_impact",
|
|
|
|
|
focus="markdown_it.common.html_blocks",
|
|
|
|
|
target=None,
|
|
|
|
|
depth=3,
|
|
|
|
|
prompt=(
|
|
|
|
|
"List every module within three reverse dependency steps of "
|
|
|
|
|
"markdown_it.common.html_blocks."
|
|
|
|
|
),
|
|
|
|
|
),
|
|
|
|
|
(
|
|
|
|
|
"markdown_it.parser_block",
|
|
|
|
|
"markdown_it.rules_block",
|
|
|
|
|
"markdown_it.rules_block.html_block",
|
|
|
|
|
),
|
|
|
|
|
),
|
|
|
|
|
(
|
|
|
|
|
TaskSpec(
|
|
|
|
|
task_id="markdown_it_cli_to_core_state_path",
|
|
|
|
|
kind="dependency_path",
|
|
|
|
|
focus="markdown_it.cli.parse",
|
|
|
|
|
target="markdown_it.rules_core.state_core",
|
|
|
|
|
depth=3,
|
|
|
|
|
prompt=(
|
|
|
|
|
"Find one shortest local dependency path from markdown_it.cli.parse "
|
|
|
|
|
"to markdown_it.rules_core.state_core."
|
|
|
|
|
),
|
|
|
|
|
),
|
|
|
|
|
(
|
|
|
|
|
"markdown_it.cli.parse",
|
|
|
|
|
"markdown_it.main",
|
|
|
|
|
"markdown_it.rules_core.state_core",
|
|
|
|
|
),
|
|
|
|
|
),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
2026-07-29 15:59:55 -04:00
|
|
|
def _measure(
|
|
|
|
|
operation: Callable[[], WorkflowResult],
|
|
|
|
|
*,
|
|
|
|
|
workflow: WorkflowName,
|
|
|
|
|
samples: int,
|
|
|
|
|
) -> dict[str, object]:
|
|
|
|
|
durations: list[float] = []
|
|
|
|
|
stable_result: WorkflowResult | None = None
|
|
|
|
|
stable_payload: dict[str, object] | None = None
|
|
|
|
|
response_bytes = 0
|
|
|
|
|
for _ in range(samples):
|
|
|
|
|
started = time.perf_counter_ns()
|
|
|
|
|
current = operation()
|
|
|
|
|
elapsed_ms = (time.perf_counter_ns() - started) / 1_000_000
|
|
|
|
|
payload = current.payload()
|
|
|
|
|
current_response_bytes = len(_compact_json(payload))
|
|
|
|
|
if stable_payload is None or stable_result is None:
|
|
|
|
|
stable_result = current
|
|
|
|
|
stable_payload = payload
|
|
|
|
|
response_bytes = current_response_bytes
|
|
|
|
|
elif (
|
|
|
|
|
payload != stable_payload
|
|
|
|
|
or current.inspected_bytes != stable_result.inspected_bytes
|
|
|
|
|
or current_response_bytes != response_bytes
|
|
|
|
|
):
|
|
|
|
|
raise TaskEvidenceError(f"{workflow} task output changed between samples")
|
|
|
|
|
durations.append(elapsed_ms)
|
|
|
|
|
assert stable_result is not None
|
|
|
|
|
assert stable_payload is not None
|
|
|
|
|
ordered = sorted(durations)
|
|
|
|
|
p95_index = max(0, math.ceil(len(ordered) * 0.95) - 1)
|
|
|
|
|
p95_ms = ordered[p95_index]
|
|
|
|
|
inspected_limit = (
|
|
|
|
|
MAX_GRAPH_INSPECTED_BYTES if workflow == "graph_assisted" else MAX_SOURCE_INSPECTED_BYTES
|
|
|
|
|
)
|
|
|
|
|
if p95_ms > TASK_P95_LIMIT_MS:
|
|
|
|
|
raise TaskEvidenceError(
|
|
|
|
|
f"{workflow} task p95 {p95_ms:.3f} ms exceeded {TASK_P95_LIMIT_MS:.3f} ms"
|
|
|
|
|
)
|
|
|
|
|
if stable_result.inspected_bytes > inspected_limit:
|
|
|
|
|
raise TaskEvidenceError(
|
|
|
|
|
f"{workflow} inspected {stable_result.inspected_bytes} bytes, limit {inspected_limit}"
|
|
|
|
|
)
|
|
|
|
|
if response_bytes > MAX_TASK_RESPONSE_BYTES:
|
|
|
|
|
raise TaskEvidenceError(
|
|
|
|
|
f"{workflow} response was {response_bytes} bytes, limit {MAX_TASK_RESPONSE_BYTES}"
|
|
|
|
|
)
|
|
|
|
|
return {
|
|
|
|
|
"samples": samples,
|
|
|
|
|
"median_ms": round(statistics.median(ordered), 3),
|
|
|
|
|
"p95_ms": round(p95_ms, 3),
|
|
|
|
|
"p95_limit_ms": TASK_P95_LIMIT_MS,
|
|
|
|
|
"inspected_bytes": stable_result.inspected_bytes,
|
|
|
|
|
"inspected_bytes_limit": inspected_limit,
|
|
|
|
|
"response_bytes": response_bytes,
|
|
|
|
|
"response_bytes_limit": MAX_TASK_RESPONSE_BYTES,
|
|
|
|
|
"result": stable_payload,
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _lower_is_better(
|
|
|
|
|
graph_value: int | float,
|
|
|
|
|
source_value: int | float,
|
|
|
|
|
) -> dict[str, object]:
|
|
|
|
|
if graph_value < source_value:
|
|
|
|
|
lower: str = "graph_assisted"
|
|
|
|
|
elif source_value < graph_value:
|
|
|
|
|
lower = "source_only"
|
|
|
|
|
else:
|
|
|
|
|
lower = "tie"
|
|
|
|
|
return {
|
|
|
|
|
"graph_assisted": graph_value,
|
|
|
|
|
"source_only": source_value,
|
|
|
|
|
"lower_workflow": lower,
|
|
|
|
|
"absolute_difference": round(abs(graph_value - source_value), 3),
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _task_evidence(
|
|
|
|
|
task: TaskSpec,
|
|
|
|
|
*,
|
2026-07-29 16:21:10 -04:00
|
|
|
expected: tuple[str, ...],
|
2026-07-29 15:59:55 -04:00
|
|
|
samples: int,
|
2026-07-29 16:21:10 -04:00
|
|
|
graph_operation: Callable[[], WorkflowResult],
|
|
|
|
|
source_operation: Callable[[], WorkflowResult],
|
2026-07-29 15:59:55 -04:00
|
|
|
) -> dict[str, object]:
|
|
|
|
|
graph = _measure(
|
2026-07-29 16:21:10 -04:00
|
|
|
graph_operation,
|
2026-07-29 15:59:55 -04:00
|
|
|
workflow="graph_assisted",
|
|
|
|
|
samples=samples,
|
|
|
|
|
)
|
|
|
|
|
source = _measure(
|
2026-07-29 16:21:10 -04:00
|
|
|
source_operation,
|
2026-07-29 15:59:55 -04:00
|
|
|
workflow="source_only",
|
|
|
|
|
samples=samples,
|
|
|
|
|
)
|
|
|
|
|
graph_result = cast(dict[str, object], graph["result"])
|
|
|
|
|
source_result = cast(dict[str, object], source["result"])
|
|
|
|
|
graph_correct = graph_result["answer"] == list(expected)
|
|
|
|
|
source_correct = source_result["answer"] == list(expected)
|
|
|
|
|
if not graph_correct or not source_correct:
|
|
|
|
|
raise TaskEvidenceError(f"Workflow answer did not match key for {task.task_id}")
|
|
|
|
|
comparisons = {
|
|
|
|
|
"task_elapsed_ms": _lower_is_better(
|
|
|
|
|
cast(float, graph["median_ms"]),
|
|
|
|
|
cast(float, source["median_ms"]),
|
|
|
|
|
),
|
|
|
|
|
"inspected_bytes": _lower_is_better(
|
|
|
|
|
cast(int, graph["inspected_bytes"]),
|
|
|
|
|
cast(int, source["inspected_bytes"]),
|
|
|
|
|
),
|
|
|
|
|
"response_bytes": _lower_is_better(
|
|
|
|
|
cast(int, graph["response_bytes"]),
|
|
|
|
|
cast(int, source["response_bytes"]),
|
|
|
|
|
),
|
|
|
|
|
}
|
|
|
|
|
advantages = [
|
|
|
|
|
{
|
|
|
|
|
"metric": metric,
|
|
|
|
|
"lower_workflow": comparison["lower_workflow"],
|
|
|
|
|
"absolute_difference": comparison["absolute_difference"],
|
|
|
|
|
"scope": "this fixed task; graph cold preparation excluded",
|
|
|
|
|
}
|
|
|
|
|
for metric, comparison in comparisons.items()
|
|
|
|
|
if comparison["lower_workflow"] != "tie"
|
|
|
|
|
]
|
|
|
|
|
return {
|
|
|
|
|
"task": task.as_dict(),
|
|
|
|
|
"answer_key": list(expected),
|
|
|
|
|
"answer_key_sha256": _sha256(list(expected)),
|
|
|
|
|
"workflows": {
|
|
|
|
|
"graph_assisted": {**graph, "correct": graph_correct},
|
|
|
|
|
"source_only": {**source, "correct": source_correct},
|
|
|
|
|
},
|
|
|
|
|
"comparison": {
|
|
|
|
|
"both_exact": True,
|
|
|
|
|
"metrics": comparisons,
|
|
|
|
|
"measured_advantages": advantages,
|
|
|
|
|
},
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def build_task_evidence(
|
|
|
|
|
root: Path,
|
|
|
|
|
*,
|
|
|
|
|
source_count: int,
|
|
|
|
|
samples: int,
|
|
|
|
|
) -> dict[str, object]:
|
|
|
|
|
"""Build the fixture, execute both workflows, and return gated evidence."""
|
|
|
|
|
|
|
|
|
|
if samples < 1:
|
|
|
|
|
raise ValueError("samples must be positive")
|
|
|
|
|
source_bytes = write_fixture(root, source_count)
|
2026-07-29 16:21:10 -04:00
|
|
|
context, preparation = _prepare_graph(
|
|
|
|
|
root,
|
|
|
|
|
source_bytes,
|
|
|
|
|
expected_names={_module_name(index) for index in range(source_count)},
|
|
|
|
|
project_id="milestone5-generated-task-evidence",
|
|
|
|
|
title="Milestone 5 generated comparative task evidence",
|
|
|
|
|
)
|
2026-07-29 15:59:55 -04:00
|
|
|
tasks = [
|
|
|
|
|
_task_evidence(
|
|
|
|
|
task,
|
2026-07-29 16:21:10 -04:00
|
|
|
expected=answer_key(task, source_count),
|
2026-07-29 15:59:55 -04:00
|
|
|
samples=samples,
|
2026-07-29 16:21:10 -04:00
|
|
|
graph_operation=lambda task=task: run_graph_task(context, task, source_count),
|
|
|
|
|
source_operation=lambda task=task: run_source_task(root, task, source_count),
|
2026-07-29 15:59:55 -04:00
|
|
|
)
|
|
|
|
|
for task in fixture_tasks(source_count)
|
|
|
|
|
]
|
|
|
|
|
semantic_evidence = {
|
|
|
|
|
"fixture": {
|
|
|
|
|
"source_count": source_count,
|
|
|
|
|
"source_bytes": source_bytes,
|
|
|
|
|
"topology": "component N depends on component 0 and floor(N/2), deduplicated",
|
|
|
|
|
},
|
|
|
|
|
"graph_identity": {
|
|
|
|
|
"revision": context.revision,
|
|
|
|
|
"source_hash": context.source_hash,
|
|
|
|
|
},
|
|
|
|
|
"tasks": [
|
|
|
|
|
{
|
|
|
|
|
"task": task["task"],
|
|
|
|
|
"answer_key": task["answer_key"],
|
|
|
|
|
"answer_key_sha256": task["answer_key_sha256"],
|
|
|
|
|
"graph_result": cast(
|
|
|
|
|
dict[str, object],
|
|
|
|
|
cast(dict[str, object], task["workflows"])["graph_assisted"],
|
|
|
|
|
)["result"],
|
|
|
|
|
"source_result": cast(
|
|
|
|
|
dict[str, object],
|
|
|
|
|
cast(dict[str, object], task["workflows"])["source_only"],
|
|
|
|
|
)["result"],
|
|
|
|
|
}
|
|
|
|
|
for task in tasks
|
|
|
|
|
],
|
|
|
|
|
}
|
|
|
|
|
return {
|
|
|
|
|
"fixture": {
|
|
|
|
|
"kind": "deterministic_generated_python_dependency_corpus",
|
|
|
|
|
"external_projects": False,
|
|
|
|
|
"self_hosting": False,
|
|
|
|
|
"production_bindings": False,
|
|
|
|
|
"source_count": source_count,
|
|
|
|
|
"source_bytes": source_bytes,
|
|
|
|
|
"padding_rows_per_source": PADDING_ROWS,
|
|
|
|
|
"topology": "component N depends on component 0 and floor(N/2), deduplicated",
|
|
|
|
|
},
|
|
|
|
|
"preparation": {
|
|
|
|
|
"graph_assisted": preparation,
|
|
|
|
|
"source_only": {
|
|
|
|
|
"persistent_preparation": False,
|
|
|
|
|
"operation": "none; each task reads only the source needed by its algorithm",
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
"tasks": tasks,
|
|
|
|
|
"summary": {
|
|
|
|
|
"task_count": len(tasks),
|
|
|
|
|
"both_workflows_exact_for_all_tasks": True,
|
|
|
|
|
"comparison_scope": (
|
|
|
|
|
"Per-task latency excludes the separately reported one-time graph preparation."
|
|
|
|
|
),
|
|
|
|
|
"semantic_evidence_sha256": _sha256(semantic_evidence),
|
|
|
|
|
},
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
2026-07-29 16:21:10 -04:00
|
|
|
def build_real_package_evidence(
|
|
|
|
|
root: Path,
|
|
|
|
|
*,
|
|
|
|
|
samples: int,
|
|
|
|
|
) -> dict[str, object]:
|
|
|
|
|
"""Copy and evaluate the exact lock-pinned installed package source tree."""
|
|
|
|
|
|
|
|
|
|
if samples < 1:
|
|
|
|
|
raise ValueError("samples must be positive")
|
|
|
|
|
installed_root, identity = installed_real_package()
|
|
|
|
|
_copy_real_package_read_only(installed_root, root)
|
|
|
|
|
inventory = _real_module_inventory(root)
|
|
|
|
|
expected_names = set(inventory)
|
|
|
|
|
context, preparation = _prepare_graph(
|
|
|
|
|
root,
|
|
|
|
|
PINNED_REAL_SOURCE_BYTES,
|
|
|
|
|
expected_names=expected_names,
|
|
|
|
|
project_id="milestone5-real-task-evidence",
|
|
|
|
|
title="Milestone 5 markdown-it-py comparative task evidence",
|
|
|
|
|
loader=_RealPackageDependencyAdapter(root),
|
|
|
|
|
relation="imports",
|
|
|
|
|
operation="cold pinned-source import projection plus module identity map",
|
|
|
|
|
)
|
|
|
|
|
task_definitions = real_package_tasks()
|
|
|
|
|
tasks = [
|
|
|
|
|
_task_evidence(
|
|
|
|
|
task,
|
|
|
|
|
expected=expected,
|
|
|
|
|
samples=samples,
|
|
|
|
|
graph_operation=lambda task=task: run_real_graph_task(context, task),
|
|
|
|
|
source_operation=lambda task=task: run_real_source_task(root, inventory, task),
|
|
|
|
|
)
|
|
|
|
|
for task, expected in task_definitions
|
|
|
|
|
]
|
|
|
|
|
semantic_evidence = {
|
|
|
|
|
"identity": identity,
|
|
|
|
|
"graph_identity": {
|
|
|
|
|
"revision": context.revision,
|
|
|
|
|
"source_hash": context.source_hash,
|
|
|
|
|
},
|
|
|
|
|
"tasks": [
|
|
|
|
|
{
|
|
|
|
|
"task": task["task"],
|
|
|
|
|
"answer_key": task["answer_key"],
|
|
|
|
|
"answer_key_sha256": task["answer_key_sha256"],
|
|
|
|
|
"graph_result": cast(
|
|
|
|
|
dict[str, object],
|
|
|
|
|
cast(dict[str, object], task["workflows"])["graph_assisted"],
|
|
|
|
|
)["result"],
|
|
|
|
|
"source_result": cast(
|
|
|
|
|
dict[str, object],
|
|
|
|
|
cast(dict[str, object], task["workflows"])["source_only"],
|
|
|
|
|
)["result"],
|
|
|
|
|
}
|
|
|
|
|
for task in tasks
|
|
|
|
|
],
|
|
|
|
|
}
|
|
|
|
|
return {
|
|
|
|
|
"fixture": {
|
|
|
|
|
"kind": "installed_lockfile_pinned_python_distribution",
|
|
|
|
|
"network": False,
|
|
|
|
|
"installed_source_mutation": False,
|
|
|
|
|
"external_project_mutation": False,
|
|
|
|
|
"self_hosting": False,
|
|
|
|
|
"production_bindings": False,
|
|
|
|
|
**identity,
|
|
|
|
|
"copy": "Python sources copied into a temporary read-only project tree",
|
|
|
|
|
"graph_projection": (
|
|
|
|
|
"Exact local imports from the Python reference adapter manifest, published "
|
|
|
|
|
"as cycle-preserving imports relationships"
|
|
|
|
|
),
|
|
|
|
|
"source_only_extraction": (
|
|
|
|
|
"Independent stdlib AST import scan checked against fixed answers"
|
|
|
|
|
),
|
|
|
|
|
},
|
|
|
|
|
"preparation": {
|
|
|
|
|
"graph_assisted": preparation,
|
|
|
|
|
"source_only": {
|
|
|
|
|
"persistent_preparation": False,
|
|
|
|
|
"operation": (
|
|
|
|
|
"none; module paths are inventoried and each task reads the source "
|
|
|
|
|
"required by its fixed algorithm"
|
|
|
|
|
),
|
|
|
|
|
},
|
|
|
|
|
},
|
|
|
|
|
"tasks": tasks,
|
|
|
|
|
"summary": {
|
|
|
|
|
"task_count": len(tasks),
|
|
|
|
|
"both_workflows_exact_for_all_tasks": True,
|
|
|
|
|
"comparison_scope": (
|
|
|
|
|
"Per-task latency excludes the separately reported one-time graph preparation."
|
|
|
|
|
),
|
|
|
|
|
"semantic_evidence_sha256": _sha256(semantic_evidence),
|
|
|
|
|
},
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def build_all_task_evidence(
|
|
|
|
|
root: Path,
|
|
|
|
|
*,
|
|
|
|
|
generated_source_count: int,
|
|
|
|
|
samples: int,
|
|
|
|
|
) -> dict[str, object]:
|
|
|
|
|
"""Run both the controlled-scale and pinned-real-package evidence tracks."""
|
|
|
|
|
|
|
|
|
|
generated = build_task_evidence(
|
|
|
|
|
root / "generated",
|
|
|
|
|
source_count=generated_source_count,
|
|
|
|
|
samples=samples,
|
|
|
|
|
)
|
|
|
|
|
real_package = build_real_package_evidence(
|
|
|
|
|
root / "real-package",
|
|
|
|
|
samples=samples,
|
|
|
|
|
)
|
|
|
|
|
track_hashes = {
|
|
|
|
|
"generated_scale": cast(dict[str, object], generated["summary"])[
|
|
|
|
|
"semantic_evidence_sha256"
|
|
|
|
|
],
|
|
|
|
|
"installed_real_package": cast(dict[str, object], real_package["summary"])[
|
|
|
|
|
"semantic_evidence_sha256"
|
|
|
|
|
],
|
|
|
|
|
}
|
|
|
|
|
return {
|
|
|
|
|
"tracks": {
|
|
|
|
|
"generated_scale": generated,
|
|
|
|
|
"installed_real_package": real_package,
|
|
|
|
|
},
|
|
|
|
|
"summary": {
|
|
|
|
|
"track_count": 2,
|
|
|
|
|
"task_count": 6,
|
|
|
|
|
"both_workflows_exact_for_all_tasks": True,
|
|
|
|
|
"track_semantic_evidence_sha256": track_hashes,
|
|
|
|
|
"semantic_evidence_sha256": _sha256(track_hashes),
|
|
|
|
|
},
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
2026-07-29 15:59:55 -04:00
|
|
|
def main() -> int:
|
|
|
|
|
arguments = _parser().parse_args()
|
|
|
|
|
source_count = FULL_SOURCE_COUNT if arguments.mode == "full" else SMOKE_SOURCE_COUNT
|
|
|
|
|
samples = FULL_SAMPLES if arguments.mode == "full" else SMOKE_SAMPLES
|
|
|
|
|
with tempfile.TemporaryDirectory(prefix="docforge-milestone5-task-evidence-") as directory:
|
2026-07-29 16:21:10 -04:00
|
|
|
measurement = build_all_task_evidence(
|
2026-07-29 15:59:55 -04:00
|
|
|
Path(directory).resolve(),
|
2026-07-29 16:21:10 -04:00
|
|
|
generated_source_count=source_count,
|
2026-07-29 15:59:55 -04:00
|
|
|
samples=samples,
|
|
|
|
|
)
|
|
|
|
|
report: dict[str, object] = {
|
2026-07-29 16:21:10 -04:00
|
|
|
"schema_version": 2,
|
2026-07-29 15:59:55 -04:00
|
|
|
"benchmark": "docforge2_milestone5_representative_tasks",
|
|
|
|
|
"mode": arguments.mode,
|
|
|
|
|
"source": {
|
|
|
|
|
"revision": _git(["rev-parse", "HEAD"]),
|
|
|
|
|
"dirty": bool(_git(["status", "--porcelain"])),
|
|
|
|
|
},
|
|
|
|
|
"environment": {
|
|
|
|
|
"platform": platform.platform(),
|
|
|
|
|
"machine": platform.machine(),
|
|
|
|
|
"python": platform.python_version(),
|
|
|
|
|
"implementation": platform.python_implementation(),
|
|
|
|
|
},
|
|
|
|
|
"method": {
|
|
|
|
|
"clock": "time.perf_counter_ns",
|
|
|
|
|
"samples": samples,
|
2026-07-29 16:21:10 -04:00
|
|
|
"answer_key": (
|
|
|
|
|
"Independent arithmetic oracle for generated tasks and fixed reviewed answers "
|
|
|
|
|
"for the exact pinned real-package source identity"
|
|
|
|
|
),
|
|
|
|
|
"graph_workflow": (
|
|
|
|
|
"bounded ProjectIndex dependency/backlink responses for generated tasks and "
|
|
|
|
|
"bounded immutable imports-projection responses for the pinned real package"
|
|
|
|
|
),
|
|
|
|
|
"source_workflow": (
|
|
|
|
|
"bounded literal source reads with fixed generated-import parsing or stdlib AST "
|
|
|
|
|
"import inspection for the pinned real package"
|
|
|
|
|
),
|
2026-07-29 15:59:55 -04:00
|
|
|
"inspected_bytes": (
|
|
|
|
|
"agent-visible compact graph response bytes or exact source bytes inspected "
|
|
|
|
|
"by the source-only task; internal storage I/O is not compared"
|
|
|
|
|
),
|
|
|
|
|
"response_bytes": "UTF-8 bytes of compact sorted answer-and-provenance JSON",
|
|
|
|
|
"comparison": (
|
|
|
|
|
"Direction-neutral lower-is-better measurements for each fixed task. "
|
|
|
|
|
"One-time graph preparation is reported separately."
|
|
|
|
|
),
|
|
|
|
|
},
|
|
|
|
|
**measurement,
|
|
|
|
|
}
|
|
|
|
|
encoded = encode_report(report)
|
|
|
|
|
report_bytes = len(encoded.encode("utf-8"))
|
|
|
|
|
if report_bytes > MAX_REPORT_BYTES:
|
|
|
|
|
raise TaskEvidenceError(
|
|
|
|
|
f"Evidence report was {report_bytes} bytes, limit {MAX_REPORT_BYTES}"
|
|
|
|
|
)
|
|
|
|
|
if arguments.output is not None:
|
|
|
|
|
output = arguments.output.resolve()
|
|
|
|
|
output.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
|
output.write_text(encoded, encoding="utf-8")
|
|
|
|
|
sys.stdout.write(encoded)
|
|
|
|
|
return 0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if __name__ == "__main__":
|
|
|
|
|
raise SystemExit(main())
|