mirror of
https://github.com/safishamsi/graphify.git
synced 2026-09-25 15:05:56 +00:00
fix(extract): collect files in a single pruned walk instead of one rglob per extension (#1261)
collect_files ran one full recursive rglob pass per supported extension (~85 walks), descending into node_modules/.git/venv and filtering those paths only after enumerating them, and re-evaluated gitignore patterns per file without the shared cache. Replace the loop with a single os.walk that prunes noise dirs (and, when no negation patterns exist, ignored dirs) in place -- the same pattern the follow_symlinks=True branch and detect.py's scan walk already use -- and pass the shared _is_ignored cache. Extension matching switches to p.suffix in _EXTENSIONS, matching the follow_symlinks branch and preserving the .f/.F Fortran case distinction. Synthetic benchmark (200 source files + 5,000-file node_modules): 0.295s -> 0.007s, identical result set. Tests: parity oracle against the old implementation on the fixtures and on a synthetic tree (noise dirs, hidden dirs, gitignore negation), plus a scandir-counting test asserting each directory is read at most once and noise dirs are never entered. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
co-authored by
Cursor
parent
565026d3d1
commit
d26a32a30c
+23
-7
@@ -11704,18 +11704,34 @@ def collect_files(target: Path, *, follow_symlinks: bool = False, root: Path | N
|
||||
from graphify.detect import _is_ignored, _is_noise_dir, _load_graphifyignore
|
||||
ignore_root = root if root is not None else target
|
||||
patterns = _load_graphifyignore(ignore_root)
|
||||
# Shared across all _is_ignored calls in this scan so ancestor-directory
|
||||
# results are memoised instead of re-evaluated per file.
|
||||
ignore_cache: dict[Path, bool] = {}
|
||||
|
||||
def _ignored(p: Path) -> bool:
|
||||
return bool(patterns and _is_ignored(p, ignore_root, patterns))
|
||||
return bool(patterns and _is_ignored(p, ignore_root, patterns, _cache=ignore_cache))
|
||||
|
||||
if not follow_symlinks:
|
||||
# The old rglob filter rejected paths with a noise component anywhere,
|
||||
# including components of target itself — preserve that.
|
||||
if any(_is_noise_dir(part) for part in target.parts):
|
||||
return []
|
||||
# When negation (!) patterns exist, skip directory-level ignore pruning
|
||||
# so negated files inside ignored dirs can still be reached (same
|
||||
# conservatism as detect's scan walk).
|
||||
has_negation = any(pat.startswith("!") for _, pat in patterns)
|
||||
results: list[Path] = []
|
||||
for ext in sorted(_EXTENSIONS):
|
||||
results.extend(
|
||||
p for p in target.rglob(f"*{ext}")
|
||||
if not any(_is_noise_dir(part) for part in p.parts)
|
||||
and not _ignored(p)
|
||||
)
|
||||
for dirpath, dirnames, filenames in os.walk(target):
|
||||
dp = Path(dirpath)
|
||||
dirnames[:] = [
|
||||
d for d in dirnames
|
||||
if not _is_noise_dir(d)
|
||||
and (has_negation or not _ignored(dp / d))
|
||||
]
|
||||
for fname in filenames:
|
||||
p = dp / fname
|
||||
if p.suffix in _EXTENSIONS and not _ignored(p):
|
||||
results.append(p)
|
||||
return sorted(results)
|
||||
# Walk with symlink following + cycle detection
|
||||
results = []
|
||||
|
||||
@@ -1,4 +1,6 @@
|
||||
import json
|
||||
import os
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
from graphify.extract import extract_python, extract, collect_files, _make_id, extract_bash, extract_json, _DISPATCH
|
||||
|
||||
@@ -247,6 +249,89 @@ def test_collect_files_handles_circular_symlinks(tmp_path):
|
||||
assert any(f.name == "mod.py" for f in files)
|
||||
|
||||
|
||||
def _legacy_collect_files(target, *, root=None):
|
||||
"""The pre-#1261 rglob-per-extension implementation, kept as a parity oracle."""
|
||||
from graphify.detect import _is_ignored, _is_noise_dir, _load_graphifyignore
|
||||
extensions = set(_DISPATCH.keys())
|
||||
ignore_root = root if root is not None else target
|
||||
patterns = _load_graphifyignore(ignore_root)
|
||||
results = []
|
||||
for ext in sorted(extensions):
|
||||
results.extend(
|
||||
p for p in target.rglob(f"*{ext}")
|
||||
if not any(_is_noise_dir(part) for part in p.parts)
|
||||
and not (patterns and _is_ignored(p, ignore_root, patterns))
|
||||
)
|
||||
return sorted(results)
|
||||
|
||||
|
||||
def test_collect_files_parity_with_legacy_on_fixtures():
|
||||
assert collect_files(FIXTURES) == _legacy_collect_files(FIXTURES)
|
||||
|
||||
|
||||
def test_collect_files_parity_with_legacy_synthetic(tmp_path):
|
||||
(tmp_path / "src" / "deep").mkdir(parents=True)
|
||||
(tmp_path / "src" / "app.py").write_text("x = 1")
|
||||
(tmp_path / "src" / "deep" / "lib.ts").write_text("export const x = 1")
|
||||
(tmp_path / "src" / "deep" / "notes.txt").write_text("not code")
|
||||
# Fortran case distinction: .f and .F are distinct dispatch entries
|
||||
(tmp_path / "src" / "legacy.f").write_text(" END")
|
||||
(tmp_path / "src" / "modern.F").write_text(" END")
|
||||
# Hidden dirs are traversed (only noise dirs are skipped)
|
||||
(tmp_path / ".github").mkdir()
|
||||
(tmp_path / ".github" / "ci.sh").write_text("echo hi")
|
||||
# Noise dirs must be excluded entirely
|
||||
(tmp_path / "node_modules" / "pkg").mkdir(parents=True)
|
||||
(tmp_path / "node_modules" / "pkg" / "index.js").write_text("x")
|
||||
(tmp_path / "__pycache__").mkdir()
|
||||
(tmp_path / "__pycache__" / "app.py").write_text("x")
|
||||
# Ignore rules incl. a negation, so directory-level pruning must not
|
||||
# swallow re-included files
|
||||
(tmp_path / "gen").mkdir()
|
||||
(tmp_path / "gen" / "skip.py").write_text("x")
|
||||
(tmp_path / "vendored").mkdir()
|
||||
(tmp_path / "vendored" / "drop.py").write_text("x")
|
||||
(tmp_path / "vendored" / "keep.py").write_text("x")
|
||||
(tmp_path / ".gitignore").write_text("gen/\nvendored/*.py\n!vendored/keep.py\n")
|
||||
|
||||
result = collect_files(tmp_path)
|
||||
assert result == _legacy_collect_files(tmp_path)
|
||||
names = {f.name for f in result}
|
||||
assert names == {"app.py", "lib.ts", "legacy.f", "modern.F", "ci.sh", "keep.py"}
|
||||
|
||||
|
||||
def test_collect_files_walks_each_directory_once(tmp_path, monkeypatch):
|
||||
"""collect_files must scan every directory at most once and never descend
|
||||
into noise dirs (#1261). The old implementation ran one rglob pass per
|
||||
supported extension (~85 walks) and filtered node_modules/.git paths only
|
||||
after descending into them.
|
||||
"""
|
||||
(tmp_path / "src").mkdir()
|
||||
(tmp_path / "src" / "a.py").write_text("x = 1")
|
||||
(tmp_path / "node_modules" / "pkg").mkdir(parents=True)
|
||||
(tmp_path / "node_modules" / "pkg" / "index.js").write_text("x")
|
||||
|
||||
scanned: list[str] = []
|
||||
real_scandir = os.scandir
|
||||
|
||||
def counting_scandir(path=".", *args, **kwargs):
|
||||
scanned.append(os.fspath(path))
|
||||
return real_scandir(path, *args, **kwargs)
|
||||
|
||||
monkeypatch.setattr(os, "scandir", counting_scandir)
|
||||
files = collect_files(tmp_path)
|
||||
monkeypatch.undo()
|
||||
|
||||
assert files == [tmp_path / "src" / "a.py"]
|
||||
# The traversal must be visible as plain os.scandir calls (single os.walk)
|
||||
assert any(s.endswith("src") for s in scanned)
|
||||
# Noise dirs are pruned before descending, not filtered afterwards
|
||||
assert not any("node_modules" in s for s in scanned)
|
||||
# No directory is read more than once
|
||||
counts = Counter(scanned)
|
||||
assert max(counts.values()) == 1
|
||||
|
||||
|
||||
def test_no_dangling_edges_on_extract():
|
||||
"""After merging multiple files, no internal edges should be dangling."""
|
||||
files = list(FIXTURES.glob("*.py"))
|
||||
|
||||
Reference in New Issue
Block a user