fix: surface silently-skipped dirs in enumeration + dedup Pascal edges

Two correctness fixes found while analysing the reported 'graphify update
occasionally writes a partial graph.json' bug.

Enumeration (P0): detect()'s os.walk had no onerror handler, so any os.scandir
failure -- a transient PermissionError, or a directory created/deleted mid-walk
by concurrent writes (e.g. benchmarking racing the scan) -- was silently
swallowed and that entire subtree dropped out of the file list with no log, no
error. Downstream that becomes a silently partial graph.json. The walk now
records each skipped directory (surfaced as walk_errors in detect()'s result)
and warns to stderr, while still enumerating the rest of the tree. This stays
visible even when a --force/GRAPHIFY_FORCE rebuild bypasses the shrink guards.
Relatedly, to_json's #479 anti-shrink guard was fail-OPEN: a non-empty but
unreadable existing graph.json (corrupt or mid-write) proceeded with the
overwrite. It now fails SAFE -- refuse and point at force=True -- while an
empty/whitespace existing file (no nodes to lose) still proceeds. The size-cap
check keeps running before any read, so an oversized existing file is not
loaded into memory.

Pascal edges (P1): a class method declared in the interface section and defined
in the implementation section each emitted a "method" edge to the same node id,
and the edge helpers (unlike the node helpers) did not dedup, so ~half of a
Pascal/Delphi graph's method edges were doubled -- inflating degree/centrality
and tripping the #1739 cross-file resolver's single-owner god-node guard. Both
extractors now dedup edges on (source, target, relation).

Adds regression tests for all three behaviours.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
safishamsi
2026-07-09 01:06:02 +01:00
co-authored by Claude Opus 4.8
parent d89efbf9ef
commit d2d1f68ff9
7 changed files with 172 additions and 6 deletions
+22 -1
View File
@@ -1107,9 +1107,29 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace:
seen: set[Path] = set()
all_files: list[Path] = []
# os.walk swallows os.scandir errors by default (no onerror -> the failing
# directory subtree is silently skipped). That turns a transient
# PermissionError, or a directory created/deleted mid-walk (e.g. concurrent
# writes racing the scan), into a partial file list and, downstream, a
# silently partial graph.json. Record and surface every skipped directory
# so an incomplete enumeration is visible rather than silent.
walk_errors: list[str] = []
def _on_walk_error(err: OSError) -> None:
import sys as _sys
target = getattr(err, "filename", None) or "<unknown>"
walk_errors.append(f"{target}: {err}")
print(
f"[graphify] WARNING: could not scan {target} ({err}); "
f"its files are missing from this run's enumeration.",
file=_sys.stderr,
)
for scan_root in scan_paths:
in_memory_tree = memory_dir.exists() and str(scan_root).startswith(str(memory_dir))
for dirpath, dirnames, filenames in os.walk(scan_root, followlinks=follow_symlinks):
for dirpath, dirnames, filenames in os.walk(
scan_root, followlinks=follow_symlinks, onerror=_on_walk_error
):
dp = Path(dirpath)
if follow_symlinks and os.path.islink(dirpath):
real = os.path.realpath(dirpath)
@@ -1245,6 +1265,7 @@ def detect(root: Path, *, follow_symlinks: bool | None = None, google_workspace:
"warning": warning,
"skipped_sensitive": skipped_sensitive,
"unclassified": sorted(unclassified),
"walk_errors": walk_errors,
"graphifyignore_patterns": len(ignore_patterns),
"scan_root": str(root.resolve()),
}
+36 -5
View File
@@ -184,11 +184,44 @@ def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *,
# Safety check: refuse to silently shrink an existing graph (#479)
existing_path = Path(output_path)
if not force and existing_path.exists():
from graphify.security import check_graph_file_size_cap
try:
from graphify.security import check_graph_file_size_cap
check_graph_file_size_cap(existing_path)
existing_data = json.loads(existing_path.read_text(encoding="utf-8"))
existing_n = len(existing_data.get("nodes", []))
except Exception:
# Existing graph.json trips the size cap; reading it to compare would
# be the very DoS the cap guards against. Can't verify — let the new
# graph replace the oversized file.
oversized = True
else:
oversized = False
if not oversized:
try:
raw = existing_path.read_text(encoding="utf-8")
except Exception:
raw = ""
if not raw.strip():
# Empty/whitespace existing file (e.g. a freshly touched path):
# no nodes to lose, so any new graph is a growth — proceed.
existing_n = 0
else:
try:
existing_data = json.loads(raw)
existing_n = len(existing_data.get("nodes", []))
except Exception as exc:
# Non-empty but unparseable existing graph (corrupt or a
# mid-write): we cannot verify the new graph is not a silent
# shrink. Fail SAFE — refuse rather than overwrite. A
# fail-OPEN here (the prior behavior) is the silent data-loss
# path #479 exists to prevent: a transiently unreadable
# graph.json would let a partial rebuild clobber a good one.
import sys as _sys
print(
f"[graphify] WARNING: existing {existing_path} could not be "
f"read to verify the new graph is not smaller ({exc}). "
f"Refusing to overwrite; pass force=True to override.",
file=_sys.stderr,
)
return False
new_n = G.number_of_nodes()
if new_n < existing_n:
import sys as _sys
@@ -203,8 +236,6 @@ def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *,
file=_sys.stderr,
)
return False
except Exception:
pass # unreadable existing file — proceed with write
node_community = _node_community_map(communities)
_labels: dict[int, str] = {int(k): v for k, v in (community_labels or {}).items()}
+18
View File
@@ -243,6 +243,7 @@ def _extract_pascal_regex(path: Path) -> dict:
edges: list[dict] = []
seen_ids: set[str] = set()
seen_call_pairs: set[tuple[str, str]] = set()
seen_edges: set[tuple[str, str, str]] = set()
def _add_node(nid: str, label: str, line: int) -> None:
if nid not in seen_ids:
@@ -256,6 +257,14 @@ def _extract_pascal_regex(path: Path) -> dict:
})
def _add_edge(src: str, tgt: str, relation: str, line: int, context: str | None = None) -> None:
# A class method declared in the interface section and defined in the
# implementation section both emit a `method` edge to the same node, so
# dedup on (src, tgt, relation) to keep the graph from carrying doubled
# method/contains/inherits edges (mirrors _add_node's seen_ids guard).
key = (src, tgt, relation)
if key in seen_edges:
return
seen_edges.add(key)
edge: dict = {
"source": src,
"target": tgt,
@@ -459,6 +468,7 @@ def extract_pascal(path: Path) -> dict:
nodes: list[dict] = []
edges: list[dict] = []
seen_ids: set[str] = set()
seen_edges: set[tuple[str, str, str]] = set()
proc_bodies: list[tuple[str, Any, str, str]] = []
# (proc_nid, body_node, container, name_lower)
@@ -478,6 +488,14 @@ def extract_pascal(path: Path) -> dict:
confidence: str = "EXTRACTED", weight: float = 1.0,
context: str | None = None,
) -> None:
# A class method declared in the interface section and defined in the
# implementation section both emit a `method` edge to the same node, so
# dedup on (src, tgt, relation) to keep the graph from carrying doubled
# method/contains/inherits edges (mirrors add_node's seen_ids guard).
key = (src, tgt, relation)
if key in seen_edges:
return
seen_edges.add(key)
edge: dict[str, Any] = {
"source": src, "target": tgt, "relation": relation,
"confidence": confidence, "source_file": str_path,