mirror of
https://github.com/safishamsi/graphify.git
synced 2026-09-22 21:45:58 +00:00
fix(cli): stamp freshly-extracted semantic docs in the manifest (#1897)
The #933 manifest filter built its extracted-set from node/edge source_file values, which are root-relative on a fresh extraction, and compared them against files_by_type entries, which are absolute (from detect()). The raw string membership test therefore never matched, so every freshly-extracted semantic doc was dropped from the manifest and re-queued as changed on the next run; only code files and cache-replayed docs got stamped. The filter now lives in _stamped_manifest_files(), which resolves BOTH sides against the scan root before the membership test — the same Path/is_absolute/resolve normalization the #1890 reconciliation uses in graphify.llm. Genuinely omitted zero-node docs still have no source_file entry and stay unstamped, preserving the intentional #933 re-queue behavior. Regression tests: a CLI extract with a mocked corpus extractor returning root-relative source_files lands the doc in manifest.json with a non-empty semantic_hash while a zero-node doc stays out; a unit test covers relative (fresh) and absolute (cache-hit) source_file shapes. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 4.8
parent
0cacd708e9
commit
b1e313cc5f
+48
-11
@@ -52,6 +52,50 @@ _GEMINI_NUDGE_TEXT = (
|
||||
|
||||
def _default_graph_path() -> str:
|
||||
return str(Path(_GRAPHIFY_OUT) / "graph.json")
|
||||
|
||||
|
||||
def _stamped_manifest_files(
|
||||
files_by_type: dict[str, list[str]],
|
||||
sem_result: dict,
|
||||
root: Path,
|
||||
) -> dict[str, list[str]]:
|
||||
"""Manifest-safe files dict: only stamp semantic files that actually
|
||||
produced output (cache hit or fresh extraction). Files whose chunk failed
|
||||
have no source_file entry in sem_result — leaving their semantic_hash
|
||||
empty so detect_incremental re-queues them (#933).
|
||||
|
||||
Both sides of the membership test are resolved against the scan ``root``
|
||||
before comparing (#1897): node/edge ``source_file`` values are
|
||||
root-relative on a fresh extraction while ``files_by_type`` entries are
|
||||
absolute (from detect()), so a raw string comparison never matched and
|
||||
every freshly-extracted semantic doc was dropped from the manifest.
|
||||
Mirrors the #1890 path normalization in graphify.llm.
|
||||
"""
|
||||
root = Path(root)
|
||||
|
||||
def _resolve(value: str) -> Path:
|
||||
p = Path(value)
|
||||
if not p.is_absolute():
|
||||
p = root / p
|
||||
try:
|
||||
return p.resolve()
|
||||
except (OSError, RuntimeError):
|
||||
return p
|
||||
|
||||
sem_extracted: set[Path] = set()
|
||||
for coll in ("nodes", "edges"):
|
||||
for item in sem_result.get(coll, []):
|
||||
sf = item.get("source_file", "")
|
||||
if sf:
|
||||
sem_extracted.add(_resolve(sf))
|
||||
sem_types = {"document", "paper", "image"}
|
||||
return {
|
||||
ftype: [
|
||||
f for f in flist
|
||||
if ftype not in sem_types or _resolve(f) in sem_extracted
|
||||
]
|
||||
for ftype, flist in files_by_type.items()
|
||||
}
|
||||
class _StageTimer:
|
||||
"""Print per-stage wall-clock timings to stderr when --timing is set (#1490).
|
||||
|
||||
@@ -2438,17 +2482,10 @@ def dispatch_command(cmd: str) -> None:
|
||||
# that actually produced output (cache hit or fresh extraction). Files
|
||||
# whose chunk failed have no source_file entry in sem_result — leaving
|
||||
# their semantic_hash empty so detect_incremental re-queues them (#933).
|
||||
_sem_extracted: set[str] = {
|
||||
n.get("source_file", "") for n in sem_result.get("nodes", [])
|
||||
} | {
|
||||
e.get("source_file", "") for e in sem_result.get("edges", [])
|
||||
}
|
||||
_sem_extracted.discard("")
|
||||
_sem_types = {"document", "paper", "image"}
|
||||
_manifest_files = {
|
||||
ftype: [f for f in flist if ftype not in _sem_types or f in _sem_extracted]
|
||||
for ftype, flist in files_by_type.items()
|
||||
}
|
||||
# Path normalization against the scan root happens inside the helper
|
||||
# (#1897) so fresh root-relative source_files match detect()'s
|
||||
# absolute file lists.
|
||||
_manifest_files = _stamped_manifest_files(files_by_type, sem_result, target)
|
||||
|
||||
if no_cluster:
|
||||
# --no-cluster: dump the raw merged extraction as graph.json.
|
||||
|
||||
@@ -135,6 +135,93 @@ def test_extract_succeeds_when_at_least_one_chunk_completes(
|
||||
} == {str(corpus / "README.md")}
|
||||
|
||||
|
||||
def test_manifest_stamps_freshly_extracted_semantic_docs(monkeypatch, tmp_path):
|
||||
"""#1897: fresh extraction returns nodes with ROOT-RELATIVE source_file,
|
||||
while the #933 manifest filter compared them against detect()'s ABSOLUTE
|
||||
paths — so `f in _sem_extracted` was always False and every freshly
|
||||
extracted doc was dropped from the manifest (only code/zero-node files
|
||||
survived). Both sides must be resolved against the scan root; a genuinely
|
||||
omitted doc (zero nodes) must still stay unstamped (#933 is intentional)."""
|
||||
import json
|
||||
|
||||
corpus = _make_corpus(tmp_path) # main.go + README.md
|
||||
(corpus / "OMITTED.md").write_text("# never extracted\n")
|
||||
out_dir = tmp_path / "out"
|
||||
monkeypatch.setenv("ANTHROPIC_API_KEY", "sk-test-fake-key")
|
||||
|
||||
def _fresh_relative(paths, **kwargs):
|
||||
on_chunk = kwargs.get("on_chunk_done")
|
||||
if on_chunk:
|
||||
on_chunk(0, 1, {"nodes": [], "edges": [], "hyperedges": []})
|
||||
# Root-relative source_file, exactly what a fresh extraction produces.
|
||||
# OMITTED.md gets no nodes/edges — the model skipped it.
|
||||
return {
|
||||
"nodes": [{"id": "readme", "source_file": "README.md",
|
||||
"file_type": "document"}],
|
||||
"edges": [],
|
||||
"hyperedges": [],
|
||||
"input_tokens": 10,
|
||||
"output_tokens": 5,
|
||||
}
|
||||
|
||||
monkeypatch.setattr("graphify.llm.extract_corpus_parallel", _fresh_relative)
|
||||
monkeypatch.setattr(mainmod, "_check_skill_version", lambda _: None)
|
||||
monkeypatch.setattr(
|
||||
mainmod.sys, "argv",
|
||||
["graphify", "extract", str(corpus), "--backend", "claude",
|
||||
"--no-cluster", "--out", str(out_dir)],
|
||||
)
|
||||
|
||||
try:
|
||||
mainmod.main()
|
||||
except SystemExit as exc:
|
||||
assert exc.code in (None, 0), f"unexpected exit code {exc.code}"
|
||||
|
||||
manifest_path = out_dir / "graphify-out" / "manifest.json"
|
||||
assert manifest_path.exists()
|
||||
manifest = json.loads(manifest_path.read_text())
|
||||
|
||||
assert "README.md" in manifest, (
|
||||
f"freshly-extracted doc missing from manifest (#1897): {sorted(manifest)}"
|
||||
)
|
||||
assert manifest["README.md"].get("semantic_hash"), (
|
||||
"freshly-extracted doc must carry a non-empty semantic_hash"
|
||||
)
|
||||
# Code files are always stamped.
|
||||
assert manifest.get("main.go", {}).get("semantic_hash")
|
||||
# The zero-node doc stays unstamped so detect_incremental re-queues it (#933).
|
||||
assert "OMITTED.md" not in manifest, (
|
||||
"zero-node doc must not be stamped in the manifest"
|
||||
)
|
||||
|
||||
|
||||
def test_stamped_manifest_files_normalizes_both_sides(tmp_path):
|
||||
"""Unit test for the #1897 helper: relative (fresh) and absolute (cache-hit)
|
||||
source_file values must both match detect()'s absolute file lists; docs with
|
||||
no output are filtered; code files pass through untouched."""
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
|
||||
fresh_doc = tmp_path / "fresh.md"; fresh_doc.write_text("# fresh")
|
||||
cached_doc = tmp_path / "cached.md"; cached_doc.write_text("# cached")
|
||||
omitted_doc = tmp_path / "omitted.md"; omitted_doc.write_text("# omitted")
|
||||
code = tmp_path / "app.py"; code.write_text("x = 1")
|
||||
|
||||
files_by_type = {
|
||||
"code": [str(code)],
|
||||
"document": [str(fresh_doc), str(cached_doc), str(omitted_doc)],
|
||||
}
|
||||
sem_result = {
|
||||
# fresh extraction: root-relative source_file
|
||||
"nodes": [{"id": "n1", "source_file": "fresh.md"}],
|
||||
# cache replay: absolute source_file (edge-only coverage counts too)
|
||||
"edges": [{"source": "a", "target": "b", "source_file": str(cached_doc)}],
|
||||
}
|
||||
|
||||
out = _stamped_manifest_files(files_by_type, sem_result, tmp_path)
|
||||
assert out["code"] == [str(code)]
|
||||
assert out["document"] == [str(fresh_doc), str(cached_doc)]
|
||||
|
||||
|
||||
def _code_only_corpus(tmp_path):
|
||||
"""A corpus with only code — no docs/papers/images."""
|
||||
(tmp_path / "auth.py").write_text(
|
||||
|
||||
Reference in New Issue
Block a user