fix(cli): stamp freshly-extracted semantic docs in the manifest (#1897)

The #933 manifest filter built its extracted-set from node/edge
source_file values, which are root-relative on a fresh extraction, and
compared them against files_by_type entries, which are absolute (from
detect()). The raw string membership test therefore never matched, so
every freshly-extracted semantic doc was dropped from the manifest and
re-queued as changed on the next run; only code files and cache-replayed
docs got stamped. The filter now lives in _stamped_manifest_files(),
which resolves BOTH sides against the scan root before the membership
test — the same Path/is_absolute/resolve normalization the #1890
reconciliation uses in graphify.llm. Genuinely omitted zero-node docs
still have no source_file entry and stay unstamped, preserving the
intentional #933 re-queue behavior.

Regression tests: a CLI extract with a mocked corpus extractor returning
root-relative source_files lands the doc in manifest.json with a
non-empty semantic_hash while a zero-node doc stays out; a unit test
covers relative (fresh) and absolute (cache-hit) source_file shapes.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
safishamsi
2026-07-15 00:41:37 +01:00
co-authored by Claude Opus 4.8
parent 0cacd708e9
commit b1e313cc5f
2 changed files with 135 additions and 11 deletions
+48 -11
View File
@@ -52,6 +52,50 @@ _GEMINI_NUDGE_TEXT = (
def _default_graph_path() -> str:
return str(Path(_GRAPHIFY_OUT) / "graph.json")
def _stamped_manifest_files(
files_by_type: dict[str, list[str]],
sem_result: dict,
root: Path,
) -> dict[str, list[str]]:
"""Manifest-safe files dict: only stamp semantic files that actually
produced output (cache hit or fresh extraction). Files whose chunk failed
have no source_file entry in sem_result — leaving their semantic_hash
empty so detect_incremental re-queues them (#933).
Both sides of the membership test are resolved against the scan ``root``
before comparing (#1897): node/edge ``source_file`` values are
root-relative on a fresh extraction while ``files_by_type`` entries are
absolute (from detect()), so a raw string comparison never matched and
every freshly-extracted semantic doc was dropped from the manifest.
Mirrors the #1890 path normalization in graphify.llm.
"""
root = Path(root)
def _resolve(value: str) -> Path:
p = Path(value)
if not p.is_absolute():
p = root / p
try:
return p.resolve()
except (OSError, RuntimeError):
return p
sem_extracted: set[Path] = set()
for coll in ("nodes", "edges"):
for item in sem_result.get(coll, []):
sf = item.get("source_file", "")
if sf:
sem_extracted.add(_resolve(sf))
sem_types = {"document", "paper", "image"}
return {
ftype: [
f for f in flist
if ftype not in sem_types or _resolve(f) in sem_extracted
]
for ftype, flist in files_by_type.items()
}
class _StageTimer:
"""Print per-stage wall-clock timings to stderr when --timing is set (#1490).
@@ -2438,17 +2482,10 @@ def dispatch_command(cmd: str) -> None:
# that actually produced output (cache hit or fresh extraction). Files
# whose chunk failed have no source_file entry in sem_result — leaving
# their semantic_hash empty so detect_incremental re-queues them (#933).
_sem_extracted: set[str] = {
n.get("source_file", "") for n in sem_result.get("nodes", [])
} | {
e.get("source_file", "") for e in sem_result.get("edges", [])
}
_sem_extracted.discard("")
_sem_types = {"document", "paper", "image"}
_manifest_files = {
ftype: [f for f in flist if ftype not in _sem_types or f in _sem_extracted]
for ftype, flist in files_by_type.items()
}
# Path normalization against the scan root happens inside the helper
# (#1897) so fresh root-relative source_files match detect()'s
# absolute file lists.
_manifest_files = _stamped_manifest_files(files_by_type, sem_result, target)
if no_cluster:
# --no-cluster: dump the raw merged extraction as graph.json.
+87
View File
@@ -135,6 +135,93 @@ def test_extract_succeeds_when_at_least_one_chunk_completes(
} == {str(corpus / "README.md")}
def test_manifest_stamps_freshly_extracted_semantic_docs(monkeypatch, tmp_path):
"""#1897: fresh extraction returns nodes with ROOT-RELATIVE source_file,
while the #933 manifest filter compared them against detect()'s ABSOLUTE
paths so `f in _sem_extracted` was always False and every freshly
extracted doc was dropped from the manifest (only code/zero-node files
survived). Both sides must be resolved against the scan root; a genuinely
omitted doc (zero nodes) must still stay unstamped (#933 is intentional)."""
import json
corpus = _make_corpus(tmp_path) # main.go + README.md
(corpus / "OMITTED.md").write_text("# never extracted\n")
out_dir = tmp_path / "out"
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-test-fake-key")
def _fresh_relative(paths, **kwargs):
on_chunk = kwargs.get("on_chunk_done")
if on_chunk:
on_chunk(0, 1, {"nodes": [], "edges": [], "hyperedges": []})
# Root-relative source_file, exactly what a fresh extraction produces.
# OMITTED.md gets no nodes/edges — the model skipped it.
return {
"nodes": [{"id": "readme", "source_file": "README.md",
"file_type": "document"}],
"edges": [],
"hyperedges": [],
"input_tokens": 10,
"output_tokens": 5,
}
monkeypatch.setattr("graphify.llm.extract_corpus_parallel", _fresh_relative)
monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None)
monkeypatch.setattr(
mainmod.sys, "argv",
["graphify", "extract", str(corpus), "--backend", "claude",
"--no-cluster", "--out", str(out_dir)],
)
try:
mainmod.main()
except SystemExit as exc:
assert exc.code in (None, 0), f"unexpected exit code {exc.code}"
manifest_path = out_dir / "graphify-out" / "manifest.json"
assert manifest_path.exists()
manifest = json.loads(manifest_path.read_text())
assert "README.md" in manifest, (
f"freshly-extracted doc missing from manifest (#1897): {sorted(manifest)}"
)
assert manifest["README.md"].get("semantic_hash"), (
"freshly-extracted doc must carry a non-empty semantic_hash"
)
# Code files are always stamped.
assert manifest.get("main.go", {}).get("semantic_hash")
# The zero-node doc stays unstamped so detect_incremental re-queues it (#933).
assert "OMITTED.md" not in manifest, (
"zero-node doc must not be stamped in the manifest"
)
def test_stamped_manifest_files_normalizes_both_sides(tmp_path):
"""Unit test for the #1897 helper: relative (fresh) and absolute (cache-hit)
source_file values must both match detect()'s absolute file lists; docs with
no output are filtered; code files pass through untouched."""
from graphify.cli import _stamped_manifest_files
fresh_doc = tmp_path / "fresh.md"; fresh_doc.write_text("# fresh")
cached_doc = tmp_path / "cached.md"; cached_doc.write_text("# cached")
omitted_doc = tmp_path / "omitted.md"; omitted_doc.write_text("# omitted")
code = tmp_path / "app.py"; code.write_text("x = 1")
files_by_type = {
"code": [str(code)],
"document": [str(fresh_doc), str(cached_doc), str(omitted_doc)],
}
sem_result = {
# fresh extraction: root-relative source_file
"nodes": [{"id": "n1", "source_file": "fresh.md"}],
# cache replay: absolute source_file (edge-only coverage counts too)
"edges": [{"source": "a", "target": "b", "source_file": str(cached_doc)}],
}
out = _stamped_manifest_files(files_by_type, sem_result, tmp_path)
assert out["code"] == [str(code)]
assert out["document"] == [str(fresh_doc), str(cached_doc)]
def _code_only_corpus(tmp_path):
"""A corpus with only code — no docs/papers/images."""
(tmp_path / "auth.py").write_text(