mirror of
https://github.com/safishamsi/graphify.git
synced 2026-09-25 15:05:56 +00:00
fix(skillgen): stamp only produced-output files in the update runbook (#2015)
The --update runbook's Step 9 stamped the entire detected corpus into the manifest, so a semantic file (doc/paper/image) whose chunk failed or was omitted was recorded as done and never re-queued on the next update — losing its content permanently. The runbook now builds the manifest the way the library extract path does: cli._stamped_manifest_files stamps only files that actually produced nodes/edges/hyperedges, dispatched-but-empty files have their stale hash cleared, and scan_corpus drops newly-excluded in-root rows. Applied to the Claude, Aider, and Devin skill bodies and the shared update reference; all 134 artifacts regenerated. gen.py gains a sanctioned monolith- diff predicate for the new stamping lines.
This commit is contained in:
@@ -546,15 +546,37 @@ from graphify.detect import save_manifest
|
||||
|
||||
# Save manifest for --update
|
||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
# In --update mode, 'all_files' carries the full corpus; 'files' is the changed
|
||||
# subset. Full-rebuild mode populates only 'files', so the fallback handles that.
|
||||
# root= relativizes the manifest keys to the scan root (same base as the build),
|
||||
# so the on-disk manifest is portable across clones/machines and a later --update
|
||||
# matches cached files instead of missing every one (#1417).
|
||||
save_manifest(detect.get('all_files') or detect['files'], root='INPUT_PATH')
|
||||
#
|
||||
# Only stamp semantic files (docs/papers/images) that ACTUALLY produced output:
|
||||
# a detected file whose chunk failed or was omitted must stay unstamped so the
|
||||
# next --update re-queues it, otherwise it is marked done and its content is lost
|
||||
# forever (#2015). This mirrors the library extract path exactly
|
||||
# (cli._stamped_manifest_files + clear_semantic + scan_corpus); do not stamp the
|
||||
# raw corpus. Code files are always stamped (AST is deterministic); only semantic
|
||||
# types are gated on output.
|
||||
from graphify.cli import _stamped_manifest_files
|
||||
_corpus = detect.get('all_files') or detect['files']
|
||||
_manifest_files = _stamped_manifest_files(_corpus, extract, Path('INPUT_PATH'))
|
||||
# Files dispatched this run (the changed subset) but NOT stamped above still carry
|
||||
# a stale semantic_hash from a prior run; clear it so detect_incremental re-queues
|
||||
# them instead of reading them as unchanged (#1948).
|
||||
_sem_types = ('document', 'paper', 'image')
|
||||
_dispatched = {f for t, fl in detect['files'].items() if t in _sem_types for f in fl}
|
||||
_stamped = {f for fl in _manifest_files.values() for f in fl}
|
||||
_cleared = _dispatched - _stamped
|
||||
# scan_corpus = the RAW full corpus (not the stamp-filtered subset) so in-root
|
||||
# files newly excluded since last run are dropped rather than masquerading as
|
||||
# deletions; untouched files' prior rows are still preserved (#1908).
|
||||
_scan = {f for fl in _corpus.values() for f in fl}
|
||||
save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
|
||||
|
||||
# Update cumulative cost tracker
|
||||
extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
|
||||
input_tok = extract.get('input_tokens', 0)
|
||||
output_tok = extract.get('output_tokens', 0)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user