From a380b347397e77282df48b4587630d41a05185f8 Mon Sep 17 00:00:00 2001 From: Safi Date: Sat, 6 Jun 2026 09:35:59 +0100 Subject: [PATCH] fix(extract): three graph quality fixes (#1145 #1146 #1147) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit #1147 — builtin annotation nodes inflate god-node rankings: Add _PYTHON_ANNOTATION_NOISE frozenset (str/int/bool/float/bytes/ MagicMock/Mock/AsyncMock/...) and apply it alongside _PYTHON_TYPE_CONTAINERS in the annotation walker so scalar builtins and test mocks are never created as nodes or emitted as edges. Defense-in-depth guard in analyze.god_nodes filters _BUILTIN_NOISE_LABELS so pre-existing graphs are also protected. #1146 — package-form imports create disconnected islands: from pkg import submod is now resolved to a file-level imports_from edge when submod.py or submod/__init__.py exists on disk. Fix lives in _collect_python_symbol_resolution_facts: when target resolves to a __init__.py, each imported name is checked as a potential submodule file and stored in _SymbolResolutionFacts.module_imports. Applied in _apply_symbol_resolution_facts using stem-based canonical IDs. #1145 — AST vs semantic node ID ghost duplicates: build_from_json now runs a two-pass merge after adding all nodes: collect AST nodes (source_location set) then find semantic ghosts (same basename+label, no source_location). Ghosts are removed and their IDs added to norm_to_id so all edges re-point to the AST node. Co-Authored-By: Claude Sonnet 4.6 --- graphify/analyze.py | 12 ++++++++++++ graphify/build.py | 43 +++++++++++++++++++++++++++++++++++++++++++ graphify/extract.py | 45 ++++++++++++++++++++++++++++++++++++++++++--- 3 files changed, 97 insertions(+), 3 deletions(-) diff --git a/graphify/analyze.py b/graphify/analyze.py index f3e08103d..5f28179d5 100644 --- a/graphify/analyze.py +++ b/graphify/analyze.py @@ -5,6 +5,16 @@ import networkx as nx from graphify.build import edge_data +# Builtin/mock names that can appear as annotation-derived nodes in pre-existing +# graphs. Excluded from god-node ranking so they don't displace real abstractions +# even if they weren't filtered at extraction time (#1147). +_BUILTIN_NOISE_LABELS = frozenset({ + "str", "int", "float", "bool", "bytes", "bytearray", "complex", "object", + "True", "False", + "MagicMock", "Mock", "AsyncMock", "NonCallableMock", + "NonCallableMagicMock", "PropertyMock", "patch", "sentinel", +}) + # Language families — extensions sharing a runtime can legitimately call each other _LANG_FAMILY: dict[str, str] = { **{e: "python" for e in (".py", ".pyw")}, @@ -94,6 +104,8 @@ def god_nodes(G: nx.Graph, top_n: int = 10) -> list[dict]: for node_id, deg in sorted_nodes: if _is_file_node(G, node_id) or _is_concept_node(G, node_id) or _is_json_key_node(G, node_id): continue + if G.nodes[node_id].get("label", "") in _BUILTIN_NOISE_LABELS: + continue result.append({ "id": node_id, "label": G.nodes[node_id].get("label", node_id), diff --git a/graphify/build.py b/graphify/build.py index 07fbb0340..1e040420e 100644 --- a/graphify/build.py +++ b/graphify/build.py @@ -156,10 +156,53 @@ def build_from_json(extraction: dict, *, directed: bool = False, root: str | Pat node["source_file"] = _norm_source_file(node["source_file"], _root) G.add_node(node["id"], **{k: v for k, v in node.items() if k != "id"}) node_set = set(G.nodes()) + + # #1145: merge semantic ghost-duplicate nodes into AST nodes. + # When AST and semantic extractors emit different IDs for the same symbol + # (one has source_location=L, the other has source_location=None), find + # pairs that share (source_file basename, label) and collapse the semantic + # copy into the AST copy so edges re-point to a single node. + # Two passes: first collect all AST (located) nodes, then find ghosts. + _loc_nodes: dict[tuple[str, str], str] = {} # (basename, label) -> AST node id + _noloc_nodes: dict[tuple[str, str], str] = {} # (basename, label) -> semantic node id + for nid in node_set: + attrs = G.nodes[nid] + label = str(attrs.get("label", "")).strip() + sf = str(attrs.get("source_file", "")) + basename = Path(sf).name if sf else "" + if not label or not basename: + continue + if attrs.get("source_location"): + _loc_nodes[(basename, label)] = nid + for nid in node_set: + attrs = G.nodes[nid] + label = str(attrs.get("label", "")).strip() + sf = str(attrs.get("source_file", "")) + basename = Path(sf).name if sf else "" + if not label or not basename or attrs.get("source_location"): + continue + key = (basename, label) + if key in _loc_nodes and _loc_nodes[key] != nid: + _noloc_nodes[key] = nid + # For every ghost that has an AST counterpart, record a remap. + _ghost_remap: dict[str, str] = {} # ghost_id -> canonical_id + for key, sem_id in _noloc_nodes.items(): + ast_id = _loc_nodes.get(key) + if ast_id is not None: + _ghost_remap[sem_id] = ast_id + # Remove ghost nodes from the graph; edges will be re-pointed via norm_to_id. + for ghost_id in _ghost_remap: + G.remove_node(ghost_id) + node_set.discard(ghost_id) + # Normalized ID map: lets edges survive when the LLM generates IDs with # slightly different casing or punctuation than the AST extractor. # e.g. "Session_ValidateToken" maps to "session_validatetoken". norm_to_id: dict[str, str] = {_normalize_id(nid): nid for nid in node_set} + # Also map ghost IDs to their canonical AST replacements. + for ghost_id, canonical_id in _ghost_remap.items(): + norm_to_id[_normalize_id(ghost_id)] = canonical_id + norm_to_id[ghost_id] = canonical_id # Iterate edges in a deterministic order. The graph is undirected and stores # direction in _src/_tgt; when two edges collapse onto the same node pair the # last write wins, so an unstable iteration order flips _src/_tgt run-to-run diff --git a/graphify/extract.py b/graphify/extract.py index 6ec23a089..9cdf36fe9 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -473,6 +473,18 @@ _PYTHON_TYPE_CONTAINERS = frozenset({ "None", "Ellipsis", }) +# Scalar builtins and test-mock names that appear as type annotations but carry +# no useful semantic meaning as graph nodes (#1147). Suppressed at the annotation +# walker level so they are never created as nodes or emitted as edges. +_PYTHON_ANNOTATION_NOISE = frozenset({ + # scalar builtins + "str", "int", "float", "bool", "bytes", "bytearray", "complex", "object", + "True", "False", + # unittest.mock + "MagicMock", "Mock", "AsyncMock", "NonCallableMock", + "NonCallableMagicMock", "PropertyMock", "patch", "sentinel", +}) + def _python_collect_type_refs(node, source: bytes, generic: bool, out: list[tuple[str, str]]) -> None: """Walk a Python type annotation; append (name, role) where role is 'type' or 'generic_arg'. @@ -490,19 +502,19 @@ def _python_collect_type_refs(node, source: bytes, generic: bool, out: list[tupl return if t == "identifier": name = _read_text(node, source) - if name and name not in _PYTHON_TYPE_CONTAINERS: + if name and name not in _PYTHON_TYPE_CONTAINERS and name not in _PYTHON_ANNOTATION_NOISE: out.append((name, "generic_arg" if generic else "type")) return if t == "attribute": tail = _read_text(node, source).rsplit(".", 1)[-1] - if tail and tail not in _PYTHON_TYPE_CONTAINERS: + if tail and tail not in _PYTHON_TYPE_CONTAINERS and tail not in _PYTHON_ANNOTATION_NOISE: out.append((tail, "generic_arg" if generic else "type")) return if t == "generic_type": for c in node.children: if c.type == "identifier": container = _read_text(c, source) - if container and container not in _PYTHON_TYPE_CONTAINERS: + if container and container not in _PYTHON_TYPE_CONTAINERS and container not in _PYTHON_ANNOTATION_NOISE: out.append((container, "generic_arg" if generic else "type")) elif c.type == "type_parameter": for sub in c.children: @@ -6731,6 +6743,9 @@ class _SymbolResolutionFacts: exports: list[_SymbolExportFact] = field(default_factory=list) star_exports: list[_StarExportFact] = field(default_factory=list) uses: list[_SymbolUseFact] = field(default_factory=list) + # File-to-file submodule imports from `from pkg import submod` (#1146). + # Each entry is (importing_file, submodule_file, line). + module_imports: list[tuple[Path, Path, int]] = field(default_factory=list) def _apply_symbol_resolution_facts( @@ -6748,6 +6763,7 @@ def _apply_symbol_resolution_facts( or facts.exports or facts.star_exports or facts.uses + or facts.module_imports ): return @@ -6914,6 +6930,17 @@ def _apply_symbol_resolution_facts( import_fact.file_path, ) + # #1146: emit file-to-file imports_from edges for package-form submodule imports. + for from_path, to_path, line in facts.module_imports: + try: + from_rel = from_path.relative_to(root) + to_rel = to_path.relative_to(root) + except ValueError: + continue + source_id = _make_id(_file_stem(from_rel)) + target_id = _make_id(_file_stem(to_rel)) + add_edge(source_id, target_id, "imports_from", "submodule_import", line, from_path) + for use_fact in facts.uses: file_path = use_fact.file_path.resolve() target_id = None @@ -7540,8 +7567,20 @@ def _collect_python_symbol_resolution_facts( target_path = _resolve_python_module_path(module_name, path, root, level) if target_path is None: continue + # #1146: `from pkg import submod` — if the target is a package + # (__init__.py) and an imported name matches a submodule file on + # disk, emit a file-level import edge to that submodule rather + # than only to the package. + pkg_dir = target_path.parent if target_path.name == "__init__.py" else None for imported_name, local_name in _python_imported_names(node, source): line = node.start_point[0] + 1 + if pkg_dir is not None: + sub_py = pkg_dir / f"{imported_name}.py" + sub_pkg = pkg_dir / imported_name / "__init__.py" + submodule = sub_py if sub_py.is_file() else (sub_pkg if sub_pkg.is_file() else None) + if submodule is not None: + facts.module_imports.append((path, submodule, line)) + continue facts.imports.append( _SymbolImportFact(path, local_name, target_path, imported_name, line) )