From 8e6ba9d714816251e7e441b6ceaf4bfaf43e26ad Mon Sep 17 00:00:00 2001 From: Safi Date: Fri, 19 Jun 2026 10:25:09 +0100 Subject: [PATCH] Parse package manifests into canonical package nodes + depends_on edges (#1377) apm.yml was a .yml document handled by the LLM, so the same package got a different file-anchored node id from its own manifest than from each dependent's dependency reference and split into duplicate nodes. New manifest_ingest module parses apm.yml/pyproject.toml/go.mod/pom.xml deterministically into ONE package node per package, keyed by name via ids.make_id, plus depends_on edges; routed to the AST path (CODE) so the LLM never sees them. Package nodes are exempt from the file-stem prefix remap so the canonical id is stable across manifests and dedup collapses references to a single hub node. Co-Authored-By: Claude Opus 4.8 (1M context) --- graphify/detect.py | 7 + graphify/extract.py | 12 ++ graphify/manifest_ingest.py | 247 ++++++++++++++++++++++++++++++++++ tests/test_manifest_ingest.py | 114 ++++++++++++++++ 4 files changed, 380 insertions(+) create mode 100644 graphify/manifest_ingest.py create mode 100644 tests/test_manifest_ingest.py diff --git a/graphify/detect.py b/graphify/detect.py index ca8bb8505..6c8e71d8f 100644 --- a/graphify/detect.py +++ b/graphify/detect.py @@ -383,6 +383,13 @@ def _shebang_file_type(path: Path) -> FileType | None: def classify_file(path: Path) -> FileType | None: + # Package manifests (apm.yml, pyproject.toml, go.mod, pom.xml) are parsed + # deterministically, so route them to the AST path (CODE) rather than the LLM + # document path — otherwise apm.yml (a .yml "document") would be LLM-extracted + # and a package would split into duplicate file-anchored nodes (#1377). + from graphify.manifest_ingest import is_package_manifest_path + if is_package_manifest_path(path): + return FileType.CODE # Compound extensions must be checked before simple suffix lookup if path.name.lower().endswith(".blade.php"): return FileType.CODE diff --git a/graphify/extract.py b/graphify/extract.py index 0e90566c9..c18ab1e26 100644 --- a/graphify/extract.py +++ b/graphify/extract.py @@ -13,6 +13,7 @@ from typing import Any, Callable from .cache import load_cached, save_cached from .ids import make_id from .mcp_ingest import extract_mcp_config, is_mcp_config_path +from .manifest_ingest import extract_package_manifest, is_package_manifest_path _RECURSION_LIMIT = 10_000 @@ -12379,6 +12380,11 @@ def _get_extractor(path: Path) -> Any | None: # (servers, commands, packages, env vars) instead of opaque JSON keys. if is_mcp_config_path(path): return extract_mcp_config + # Package manifests (apm.yml, pyproject.toml, go.mod, pom.xml) → a canonical + # package node + depends_on edges, by filename before generic suffix dispatch + # (#1377). apm.yml would otherwise be a .yml document handled by the LLM. + if is_package_manifest_path(path): + return extract_package_manifest return _DISPATCH.get(path.suffix) @@ -12684,6 +12690,12 @@ def extract( sf = n.get("source_file") if not sf: continue + # Package nodes carry a canonical name-keyed id (pkg_) that must + # stay identical across every manifest that references the package, so + # they are exempt from the file-stem prefix remap (#1377), like the + # type=module anchors (#1327). + if n.get("type") == "package": + continue try: entry = prefix_remap.get(Path(sf).resolve()) except Exception: diff --git a/graphify/manifest_ingest.py b/graphify/manifest_ingest.py new file mode 100644 index 000000000..ae3aa61fc --- /dev/null +++ b/graphify/manifest_ingest.py @@ -0,0 +1,247 @@ +"""Deterministic package-manifest ingestion (#1377). + +Package manifests (``apm.yml``, ``pyproject.toml``, ``go.mod``, ``pom.xml``) +declare a package and its dependencies. Left to the LLM document path, the same +package gets a different file-anchored node id from its own manifest than from +each dependent's dependency reference, so it splits into duplicate nodes. This +module parses manifests deterministically and emits ONE canonical package node +per package -- keyed by NAME via :func:`graphify.ids.make_id` -- plus +``depends_on`` edges, so a package referenced from N manifests collapses to a +single hub node (the dependency stub and the package's own definition node share +the canonical id and merge at build time). + +Mirrors ``mcp_ingest``: recognized by filename, routed to the deterministic AST +path (never the LLM), so a manifest is extracted exactly once. +""" +from __future__ import annotations + +import re +import xml.etree.ElementTree as ET +from pathlib import Path +from typing import Any + +from graphify.ids import make_id + +__all__ = ["is_package_manifest_path", "extract_package_manifest", "PACKAGE_MANIFEST_NAMES"] + +# manifest filename (lowercased) -> ecosystem tag +PACKAGE_MANIFEST_NAMES: dict[str, str] = { + "apm.yml": "apm", + "apm.yaml": "apm", + "pyproject.toml": "python", + "go.mod": "go", + "pom.xml": "maven", +} + +_MAX_MANIFEST_BYTES = 2_000_000 # 2 MB cap — manifests are small; this rejects junk + + +def is_package_manifest_path(path: Path) -> bool: + """True if ``path`` is a recognized package manifest (by filename).""" + return path.name.lower() in PACKAGE_MANIFEST_NAMES + + +def _pkg_id(name: str) -> str: + """Canonical package node id, keyed by package NAME so every reference to the + same package -- its own manifest and any dependent's dependency line -- maps + to one node.""" + return make_id("pkg", name) + + +def extract_package_manifest(path: Path) -> dict[str, Any]: + """Parse a package manifest into a canonical package node + ``depends_on`` edges.""" + try: + if path.stat().st_size > _MAX_MANIFEST_BYTES: + return {"nodes": [], "edges": [], "error": "manifest too large to index"} + text = path.read_text(encoding="utf-8", errors="replace") + except OSError as exc: + return {"nodes": [], "edges": [], "error": f"manifest read error: {exc}"} + + eco = PACKAGE_MANIFEST_NAMES[path.name.lower()] + try: + info = _PARSERS[eco](text) + except Exception as exc: # noqa: BLE001 — a malformed manifest must not abort extraction + return {"nodes": [], "edges": [], "error": f"manifest parse error: {exc}"} + if not info or not info.get("name"): + return {"nodes": [], "edges": []} + + name = info["name"] + str_path = str(path) + pkg_nid = _pkg_id(name) + node: dict[str, Any] = { + "id": pkg_nid, + "label": name, + "file_type": "code", # valid schema type; `type` distinguishes packages + "type": "package", + "ecosystem": eco, + "source_file": str_path, + "source_location": "L1", + } + if info.get("version"): + node["version"] = info["version"] + nodes: list[dict] = [node] + edges: list[dict] = [] + + seen: set[str] = set() + for dep in info.get("deps", []): + if not dep: + continue + dep_nid = _pkg_id(dep) + if dep_nid == pkg_nid or dep_nid in seen: + continue + seen.add(dep_nid) + # The edge targets the dependency's canonical package id. If that package's + # own manifest is in the corpus, the edge resolves to its (single) node; if + # the dependency is external, build_from_json prunes the dangling edge. We + # deliberately do NOT emit a stub node — a stub with an empty source_file + # would risk clobbering the real node's source_file under id-dedup. + edges.append({ + "source": pkg_nid, + "target": dep_nid, + "relation": "depends_on", + "context": "dependency", + "confidence": "EXTRACTED", + "confidence_score": 1.0, + "source_file": str_path, + "source_location": "L1", + "weight": 1.0, + }) + return {"nodes": nodes, "edges": edges} + + +# ── per-ecosystem parsers: text -> {"name", "version"?, "deps": [str]} | None ── + +def _coerce_deps(value: Any) -> list[str]: + """A dependency block may be a list of names or a name->spec map.""" + if isinstance(value, dict): + return [str(k) for k in value] + if isinstance(value, list): + out: list[str] = [] + for item in value: + if isinstance(item, str): + out.append(item) + elif isinstance(item, dict) and item: + out.append(str(next(iter(item)))) + return out + return [] + + +def _parse_apm(text: str) -> dict | None: + try: + import yaml + except ImportError: + return _parse_apm_fallback(text) + data = yaml.safe_load(text) + if not isinstance(data, dict): + return None + return { + "name": data.get("name"), + "version": data.get("version"), + "deps": _coerce_deps(data.get("dependencies")), + } + + +def _parse_apm_fallback(text: str) -> dict | None: + """Minimal line parser for apm.yml when PyYAML is unavailable: a top-level + ``name:`` plus a simple ``dependencies:`` block (list items or a name map).""" + name = None + deps: list[str] = [] + in_deps = False + for line in text.splitlines(): + if not in_deps: + m = re.match(r'^name:\s*["\']?([^"\'\s#]+)', line) + if m: + name = m.group(1) + continue + if re.match(r'^dependencies:\s*$', line): + in_deps = True + continue + if in_deps: + dm = (re.match(r'^\s*-\s*["\']?([^"\'\s#:]+)', line) + or re.match(r'^\s{2,}([A-Za-z0-9._/@-]+)\s*:', line)) + if dm: + deps.append(dm.group(1)) + elif re.match(r'^\S', line): # next top-level key ends the block + in_deps = False + return {"name": name, "version": None, "deps": deps} if name else None + + +def _pep508_name(spec: str) -> str: + """`requests>=2.0` -> `requests`; `pkg[extra]==1; python_version<'3.9'` -> `pkg`.""" + return re.split(r'[\s<>=!~;\[\(]', spec.strip(), maxsplit=1)[0] + + +def _parse_pyproject(text: str) -> dict | None: + try: + import tomllib as _toml + except ImportError: + try: + import tomli as _toml # type: ignore + except ImportError: + return None + data = _toml.loads(text) + proj = data.get("project", {}) if isinstance(data.get("project"), dict) else {} + poetry = (data.get("tool", {}) or {}).get("poetry", {}) if isinstance(data.get("tool"), dict) else {} + name = proj.get("name") or (poetry.get("name") if isinstance(poetry, dict) else None) + if not name: + return None + deps: list[str] = [_pep508_name(s) for s in (proj.get("dependencies") or []) if isinstance(s, str)] + if isinstance(poetry, dict): + for dep in (poetry.get("dependencies") or {}): + if str(dep).lower() != "python": + deps.append(str(dep)) + return {"name": name, "version": proj.get("version") or (poetry.get("version") if isinstance(poetry, dict) else None), "deps": deps} + + +def _parse_gomod(text: str) -> dict | None: + name = None + deps: list[str] = [] + in_block = False + for line in text.splitlines(): + s = line.strip() + if name is None: + m = re.match(r'^module\s+(\S+)', s) + if m: + name = m.group(1) + continue + if re.match(r'^require\s*\(', s): + in_block = True + continue + if in_block: + if s.startswith(')'): + in_block = False + continue + dm = re.match(r'^(\S+)\s+v\S+', s) + if dm: + deps.append(dm.group(1)) + else: + dm = re.match(r'^require\s+(\S+)\s+v\S+', s) + if dm: + deps.append(dm.group(1)) + return {"name": name, "version": None, "deps": deps} if name else None + + +def _parse_pom(text: str) -> dict | None: + # Drop the default namespace so findtext/findall don't need the {uri} prefix. + text = re.sub(r'\sxmlns="[^"]*"', '', text, count=1) + root = ET.fromstring(text) + aid = root.findtext("artifactId") + gid = root.findtext("groupId") + if not aid: + return None + name = f"{gid}:{aid}" if gid else aid + deps: list[str] = [] + for dep in root.findall(".//dependencies/dependency"): + da = dep.findtext("artifactId") + dg = dep.findtext("groupId") + if da: + deps.append(f"{dg}:{da}" if dg else da) + return {"name": name, "version": root.findtext("version"), "deps": deps} + + +_PARSERS = { + "apm": _parse_apm, + "python": _parse_pyproject, + "go": _parse_gomod, + "maven": _parse_pom, +} diff --git a/tests/test_manifest_ingest.py b/tests/test_manifest_ingest.py new file mode 100644 index 000000000..2b97f5fb9 --- /dev/null +++ b/tests/test_manifest_ingest.py @@ -0,0 +1,114 @@ +from __future__ import annotations + +from pathlib import Path + +from graphify.build import build_from_json +from graphify.detect import FileType, classify_file +from graphify.extract import extract +from graphify.manifest_ingest import ( + extract_package_manifest, + is_package_manifest_path, +) + + +def _write(p: Path, text: str) -> Path: + p.parent.mkdir(parents=True, exist_ok=True) + p.write_text(text, encoding="utf-8") + return p + + +# ── routing: manifests are deterministic (CODE), not LLM documents ─────────── + +def test_manifests_classify_as_code_not_document(tmp_path): + for name in ("apm.yml", "pyproject.toml", "go.mod", "pom.xml"): + p = _write(tmp_path / name, "x") + assert is_package_manifest_path(p) + assert classify_file(p) is FileType.CODE, name + # a generic yaml stays a document + assert classify_file(_write(tmp_path / "config.yaml", "a: 1")) is FileType.DOCUMENT + + +# ── per-format parsing ─────────────────────────────────────────────────────── + +def _pkg_nodes(result): + return [n for n in result["nodes"] if n.get("type") == "package"] + + +def test_apm_parses_name_and_deps(tmp_path): + p = _write(tmp_path / "apm.yml", + "name: my-pkg\nversion: 1.2.3\ndependencies:\n - dep-a\n - dep-b\n") + r = extract_package_manifest(p) + pkg = _pkg_nodes(r)[0] + assert pkg["label"] == "my-pkg" and pkg["version"] == "1.2.3" + deps = {e["target"] for e in r["edges"] if e["relation"] == "depends_on"} + assert {"pkg_dep_a", "pkg_dep_b"} <= deps + + +def test_pyproject_parses_pep508_deps(tmp_path): + p = _write(tmp_path / "pyproject.toml", + '[project]\nname = "cool-lib"\nversion = "0.1"\n' + 'dependencies = ["requests>=2.0", "rich[jupyter]==13.0", "tomli; python_version<\'3.11\'"]\n') + r = extract_package_manifest(p) + assert _pkg_nodes(r)[0]["label"] == "cool-lib" + deps = {e["target"] for e in r["edges"]} + assert {"pkg_requests", "pkg_rich", "pkg_tomli"} <= deps # versions/extras/markers stripped + + +def test_gomod_parses_module_and_requires(tmp_path): + p = _write(tmp_path / "go.mod", + "module example.com/me/app\n\ngo 1.22\n\nrequire (\n" + "\tgithub.com/x/y v1.2.3\n\tgithub.com/a/b v0.4.0\n)\n") + r = extract_package_manifest(p) + assert _pkg_nodes(r)[0]["label"] == "example.com/me/app" + deps = {e["target"] for e in r["edges"]} + assert "pkg_github_com_x_y" in deps and "pkg_github_com_a_b" in deps + + +def test_pom_parses_artifact_and_deps(tmp_path): + p = _write(tmp_path / "pom.xml", + '\n' + ' com.acme\n widget\n 2.0\n' + ' \n org.libcore\n' + ' \n\n') + r = extract_package_manifest(p) + assert _pkg_nodes(r)[0]["label"] == "com.acme:widget" + assert any(e["target"] == "pkg_org_lib_core" for e in r["edges"]) + + +# ── #1377: a package referenced by N manifests is ONE node ─────────────────── + +def test_apm_dependency_collapses_to_single_canonical_node(tmp_path): + base = tmp_path / "packages" + _write(base / "core/apm.yml", "name: coding-standards-core\nversion: 1.0.4\n") + _write(base / "csharp/apm.yml", + "name: coding-standards-csharp\ndependencies:\n - coding-standards-core\n") + _write(base / "python/apm.yml", + 'name: coding-standards-python\ndependencies:\n coding-standards-core: ">=1.0"\n') + files = sorted(base.rglob("apm.yml")) + result = extract(files, cache_root=tmp_path) + + core = [n for n in result["nodes"] + if n.get("type") == "package" and n["label"] == "coding-standards-core"] + assert len(core) == 1, "core package must be a single canonical node" + assert core[0]["id"] == "pkg_coding_standards_core" and core[0]["source_file"] + + g = build_from_json(result) + core_ids = [n for n, d in g.nodes(data=True) if d.get("label") == "coding-standards-core"] + dep_edges = [(u, v) for u, v, d in g.edges(data=True) if d.get("relation") == "depends_on"] + assert len(core_ids) == 1 + assert len(dep_edges) == 2 # both dependents point at the one core node + + +def test_external_dependency_edge_pruned_not_orphaned(tmp_path): + # A dep whose manifest isn't in the corpus: the edge dangles and build prunes it. + p = _write(tmp_path / "apm.yml", "name: leaf\ndependencies:\n - some-external-pkg\n") + result = extract([p], cache_root=tmp_path) + g = build_from_json(result) + assert "pkg_some_external_pkg" not in set(g.nodes()) # no fabricated external node + assert [n for n, d in g.nodes(data=True) if d.get("label") == "leaf"] + + +def test_malformed_manifest_does_not_crash(tmp_path): + p = _write(tmp_path / "pom.xml", " empty, no exception + assert r["nodes"] == [] and r["edges"] == []