fix(dedup): don't over-merge distinct same-file labels differing by a content word (#2576)

Fuzzy dedup compared same-file labels with prefix-weighted Jaro-Winkler, so
two distinct entities differing by one content word (asset contribution
flow vs asset consumption flow) cleared the threshold and one was lost. A
one-token difference is now judged on the differing tokens (any distinct
content word blocks; stopword/typo variants still merge), with a
same-length Damerau-Levenshtein typo escape. Genuine typo and
whitespace/case/punct variants still collapse. The #2532 collision path is
untouched.

Adapts PR #2587 (thanks @wilyan09007) with two hardening deltas.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
safishamsi
2026-08-10 18:06:50 +01:00
co-authored by Claude Opus 4.8
parent 10ad921b42
commit 92274745ad
2 changed files with 304 additions and 2 deletions
+86 -2
View File
@@ -12,7 +12,7 @@ from collections import defaultdict
from pathlib import Path
from graphify._minhash import MinHash, MinHashLSH
from rapidfuzz.distance import Jaro, JaroWinkler
from rapidfuzz.distance import DamerauLevenshtein, Jaro, JaroWinkler
# ── helpers ───────────────────────────────────────────────────────────────────
@@ -114,6 +114,81 @@ def _numeric_tokens_differ(a: str, b: str) -> bool:
sorted(t.lstrip("0") or "0" for t in _DIGIT_RUN.findall(b))
# Function words. A restatement of one entity is what inserts or swaps these
# ("export a read-only ..." vs "export the read-only ..."); a content word
# carries the entity's identity and swapping one names something else.
_STOPWORDS = frozenset({
"a", "an", "the", "and", "or", "of", "for", "to", "in", "on", "at", "by",
"with", "from", "as", "is", "are", "be", "this", "that", "its",
})
def _same_word_variant(x: str, y: str) -> bool:
"""True when tokens x and y read as one word misspelt, not two words (#2576).
A same-length pair within one substitution/transposition is a typo
("manager"/"nanager") -- the same rationale _short_label_blocked applies
to whole short labels, and unlike Jaro-Winkler it holds at position 0,
where the prefix bonus gives no help (JW scores "manager"/"nanager" at
84.92, below threshold, yet it is as much a typo as "managr"). Below 6
chars JW cannot separate two words from a typo ("pane"/"plane" scores
94.0), so short length-differing pairs never read as variants. Longer
pairs fall back to Jaro-Winkler on the merge threshold, so
"manager"/"managr" (97.14) still reads as one word. Accepted trade, per
the never-merge-two-distinct-entities bar: "colour"/"color" (5 chars,
lengths differ) now reads as two words and stays unmerged -- a spelling
variant kept separate beats a fabricated merge.
"""
if len(x) == len(y) and DamerauLevenshtein.distance(x, y) <= 1:
return True # same-length 1-sub/transposition = typo, even at position 0
if min(len(x), len(y)) < 6:
return False # short tokens: JW can't separate pane/plane (94.0) from a typo
return JaroWinkler.normalized_similarity(x, y) * 100 >= _MERGE_THRESHOLD
def _content_token_swap(a: str, b: str) -> bool:
"""True when two equal-token-count labels differ in at least one swapped
content word rather than only typos or function words (#2576, adopted
from @wilyan09007's PR #2587 and generalized from exactly-one to any
number of differing positions).
Whole-string scoring cannot separate a legit restatement from a
distinguishing-token swap: both edit one short run in the middle of a long
shared string, so both land in the same Jaro band (#1243). Which token
differs does separate them. Structured prose names sibling sections from a
template ("Asset Contribution Flow" / "Asset Consumption Flow", four
consecutive headings of one operations doc), and those siblings are densest
inside a single file -- exactly where Jaro-Winkler's prefix bonus still
applies, and where the shared affixes it rewards are boilerplate.
Each same-position differing pair is judged on its own: a function word on
either side is what a restatement swaps, a _same_word_variant pair is one
word misspelt, and anything else is a distinct content word naming a
different entity -- one such pair blocks the merge. A restatement differs
only in stopwords/typos at every position; a template sibling differs in
at least one distinct content word ("... Contribution Flow Handler" vs
"... Consumption Flows Handler" blocks on either position). Pairs with
different token counts are left to the prefix-extension guard (#1201) and
whole-label scoring. Known gap, out of scope here: fused camelCase labels
("AssetContributionFlow" vs "AssetConsumptionFlow") normalize to single
tokens whose only differing "position" is the whole label, so this guard
reduces to whole-token _same_word_variant and long fused pairs can still
clear the JW fallback.
"""
tokens_a, tokens_b = a.split(), b.split()
if len(tokens_a) != len(tokens_b):
return False
for x, y in zip(tokens_a, tokens_b):
if x == y:
continue
if x in _STOPWORDS or y in _STOPWORDS:
continue # restatement: a function word swapped in or out
if _same_word_variant(x, y):
continue # one word misspelt/inflected, not a different word
return True
return False
# file_type values whose identity is anchored to their source location, not
# their label text. Like code (#1205), these must not be label-merged across
# files: rationale = module/class docstrings, document = headings/positional
@@ -614,6 +689,12 @@ def deduplicate_entities(
# regardless of score (#1284).
if _numeric_tokens_differ(norm_label, neighbor_norm):
continue
# Template-named siblings differing in a content word are
# distinct too, on either path: same-file pairs keep the prefix
# bonus, and a cross-file pair can still reach threshold on the
# community boost alone (#2576).
if _content_token_swap(norm_label, neighbor_norm):
continue
if _crossfile_fileanchored_blocked(node, neighbor):
continue
@@ -778,9 +859,12 @@ def _llm_tiebreak(
_lo, _hi = sorted((norm_i, norm_j), key=len)
if _hi.startswith(_lo) and _hi != _lo:
continue
# Mirror pass 2: decisively-distinct pairs never reach the LLM (#1284).
# Mirror pass 2: decisively-distinct pairs never reach the LLM
# (#1284, #2576).
if _numeric_tokens_differ(norm_i, norm_j):
continue
if _content_token_swap(norm_i, norm_j):
continue
if _crossfile_fileanchored_blocked(node, neighbor):
continue
c1 = communities.get(node["id"])
+218
View File
@@ -978,3 +978,221 @@ def test_crossfile_concept_merge_is_transitive():
]
result_nodes, _ = deduplicate_entities(nodes, [], communities={})
assert len(result_nodes) == 1
# ── #2576: same-file labels differing by a content-word swap ──────────────────
# Guard adopted from @wilyan09007's PR #2587, hardened: any same-position
# distinct-content-word pair blocks (not just exactly one), and the differing
# tokens are judged by _same_word_variant instead of bare token-level JW.
def test_dedup_does_not_merge_samefile_sibling_pipeline_stages():
"""Sibling sections of one document, named from a template, share both
affixes and differ in a single content word, so Jaro-Winkler's prefix bonus
clears the threshold on the same-file path (92.74 with the bonus, 87.90 on
plain Jaro). The absorbed node's edges are re-pointed at the survivor, so
the graph asserts relationships the document never states (#2576)."""
src = "docs/pipeline.md"
nodes = [
{"id": "pipeline_asset_contribution_flow", "label": "Asset Contribution Flow",
"file_type": "concept", "source_file": src},
{"id": "pipeline_asset_consumption_flow", "label": "Asset Consumption Flow",
"file_type": "concept", "source_file": src},
{"id": "pipeline_producer_personas", "label": "Producer Personas",
"file_type": "concept", "source_file": src},
{"id": "pipeline_asset_review_flow", "label": "Asset Review & Approval Flow",
"file_type": "concept", "source_file": src},
]
edges = [
{"source": "pipeline_producer_personas",
"target": "pipeline_asset_contribution_flow", "relation": "references"},
{"source": "pipeline_asset_contribution_flow",
"target": "pipeline_asset_review_flow", "relation": "references"},
]
result_nodes, result_edges = deduplicate_entities(nodes, edges, communities={})
assert len(result_nodes) == 4, (
"contribution and consumption are distinct pipeline stages"
)
assert {(e["source"], e["target"]) for e in result_edges} == {
("pipeline_producer_personas", "pipeline_asset_contribution_flow"),
("pipeline_asset_contribution_flow", "pipeline_asset_review_flow"),
}, "edges must stay on the nodes the document actually connects"
def test_dedup_still_merges_samefile_stopword_insertion():
"""The pair #1243 was scoped around: swapping a function word is how a
restatement of the same entity differs, so a stopword in either differing
position exempts the pair and the merge still happens (#2576)."""
nodes = [
{"id": "m1", "file_type": "concept", "source_file": "docs/metrics.md",
"label": "Counts-only metrics export, a read-only aggregation service"},
{"id": "m2", "file_type": "concept", "source_file": "docs/metrics.md",
"label": "Counts-only metrics export, the read-only aggregation service"},
]
result_nodes, _ = deduplicate_entities(nodes, [], communities={})
assert len(result_nodes) == 1
def test_dedup_still_merges_samefile_normalization_variants():
"""Punct/case variants norm-equal in one file (pass 1), and a fused
camelCase symbol vs its spaced spelling (token counts differ, so the
#2576 guard defers to whole-label scoring), both still merge."""
nodes = [
{"id": "am1", "label": "Authentication Manager",
"file_type": "concept", "source_file": "docs/auth.md"},
{"id": "am2", "label": "authentication-manager",
"file_type": "concept", "source_file": "docs/auth.md"},
]
result_nodes, _ = deduplicate_entities(nodes, [], communities={})
assert len(result_nodes) == 1
nodes = [
{"id": "g1", "label": "getUserById",
"file_type": "concept", "source_file": "docs/api.md"},
{"id": "g2", "label": "get user by id",
"file_type": "concept", "source_file": "docs/api.md"},
]
result_nodes, _ = deduplicate_entities(nodes, [], communities={})
assert len(result_nodes) == 1
def test_dedup_still_merges_samefile_typo_in_one_token():
"""A typo inside the differing token leaves it the same word: manager and
managr score 97.14 against each other, so the pair is not a content-word
swap and still merges (#2576)."""
nodes = [
{"id": "a1", "label": "Authentication Manager",
"file_type": "concept", "source_file": "docs/auth.md"},
{"id": "a2", "label": "Authentication Managr",
"file_type": "concept", "source_file": "docs/auth.md"},
]
result_nodes, _ = deduplicate_entities(nodes, [], communities={})
assert len(result_nodes) == 1
def test_dedup_still_merges_single_token_transposition_at_length_boundary():
"""A single-token label at the 12-char _short_label_blocked boundary with a
trailing transposition is a typo: same length, Damerau-Levenshtein 1, so
_same_word_variant exempts it and the merge survives (#2576)."""
nodes = [
{"id": "gb1", "label": "GraphBuilder",
"file_type": "concept", "source_file": "docs/build.md"},
{"id": "gb2", "label": "GraphBuildre",
"file_type": "concept", "source_file": "docs/build.md"},
]
result_nodes, _ = deduplicate_entities(nodes, [], communities={})
assert len(result_nodes) == 1
def test_dedup_recovers_samefile_first_letter_typo():
"""DELTA over #2587: token-level JW scores manager/nanager at 84.92 (the
prefix bonus is zero at position 0), so the PR's bare-JW token test would
have blocked a genuine typo. The same-length Damerau-Levenshtein<=1 branch
of _same_word_variant reads it as one word and the merge happens (#2576).
The label carries a third token so the pair clears MinHash/LSH candidacy
(the two-token pair's shingle Jaccard sits at 0.727, on the 0.7 blocking
threshold, and never reaches the comparator either way)."""
nodes = [
{"id": "a1", "label": "Authentication Session Manager",
"file_type": "concept", "source_file": "docs/auth.md"},
{"id": "a2", "label": "Authentication Session Nanager",
"file_type": "concept", "source_file": "docs/auth.md"},
]
result_nodes, _ = deduplicate_entities(nodes, [], communities={})
assert len(result_nodes) == 1
def test_dedup_does_not_merge_samefile_short_content_word_swap():
"""DELTA over #2587: pane/plane scores 94.0 on token-level JW, above the
merge threshold, so the PR's bare-JW token test read two distinct short
words as a typo. Tokens under 6 chars never pass the JW fallback (#2576)."""
nodes = [
{"id": "p1", "label": "User Profile Pane",
"file_type": "concept", "source_file": "docs/ui.md"},
{"id": "p2", "label": "User Profile Plane",
"file_type": "concept", "source_file": "docs/ui.md"},
]
result_nodes, _ = deduplicate_entities(nodes, [], communities={})
assert len(result_nodes) == 2
def test_dedup_does_not_merge_two_position_content_swap():
"""DELTA over #2587: the PR guard required exactly one differing token, so
a sibling pair that also drifts in a second position (Flow vs Flows) slid
back to whole-label JW (95.44) and merged. Any same-position distinct
content-word pair now blocks (#2576)."""
nodes = [
{"id": "h1", "label": "Customer Asset Contribution Flow Handler",
"file_type": "concept", "source_file": "docs/handlers.md"},
{"id": "h2", "label": "Customer Asset Consumption Flows Handler",
"file_type": "concept", "source_file": "docs/handlers.md"},
]
result_nodes, _ = deduplicate_entities(nodes, [], communities={})
assert len(result_nodes) == 2
def test_dedup_does_not_merge_samefile_sibling_packages():
"""#1243's own first example, with the source file its repro used: two
dependencies both come from one package.json, so the cross-file scoring
change never applied to them and they still merged (#2576)."""
nodes = [
{"id": "jest_native", "label": "@testing-library/jest-native",
"file_type": "concept", "source_file": "package.json"},
{"id": "react_native", "label": "@testing-library/react-native",
"file_type": "concept", "source_file": "package.json"},
]
result_nodes, _ = deduplicate_entities(nodes, [], communities={})
assert len(result_nodes) == 2
def test_dedup_does_not_merge_same_community_dated_doc_slugs():
"""#1243's fourth example, with the communities its repro used: the pair is
cross-file and scores 87.68 on plain Jaro, but both nodes sit in one
community and the +5 boost lifts it back to 92.68. Comparing the differing
tokens (fe/mob) blocks it before the boost applies (#2576)."""
nodes = [
{"id": "fe_doc", "label": "2026-05-17-phase4-fe-api-integration.md",
"file_type": "concept", "source_file": "a.md"},
{"id": "mob_doc", "label": "2026-05-17-phase4-mob-api-integration.md",
"file_type": "concept", "source_file": "b.md"},
]
result_nodes, _ = deduplicate_entities(
nodes, [], communities={"fe_doc": 1, "mob_doc": 1}
)
assert len(result_nodes) == 2
def test_content_token_swap_helper():
"""_content_token_swap fires when any same-position pair is a swap of two
distinct content words (#2576)."""
from graphify.dedup import _content_token_swap
assert _content_token_swap("asset contribution flow", "asset consumption flow")
assert _content_token_swap("testing library jest native",
"testing library react native")
# A function word is what a restatement swaps, so either side exempts.
assert not _content_token_swap("export a read only", "export the read only")
# A typo leaves the differing token the same word.
assert not _content_token_swap("graph extractor", "graph extractar")
# DELTA over #2587: two differing content-word positions block too (the PR
# exempted anything but exactly one).
assert _content_token_swap("alpha beta gamma", "alpha delta epsilon")
assert _content_token_swap("customer asset contribution flow handler",
"customer asset consumption flows handler")
# Differing token counts belong to the prefix-extension guard.
assert not _content_token_swap("graph extractor", "graph extractor service")
assert not _content_token_swap("asset flow", "asset flow")
def test_same_word_variant_helper():
"""_same_word_variant separates one-word misspellings from two words
(#2576 DELTA over #2587's bare token-JW test)."""
from graphify.dedup import _same_word_variant
assert _same_word_variant("manager", "managr") # JW 97.14, min len 6
assert _same_word_variant("manager", "nanager") # JW 84.92 but same-len DL 1
assert _same_word_variant("builder", "buildre") # trailing transposition
assert not _same_word_variant("pane", "plane") # JW 94.0 but short
assert not _same_word_variant("flow", "flows") # short inflection: under-merge
assert not _same_word_variant("contribution", "consumption")
assert not _same_word_variant("jest", "react")
# Accepted trade: a length-differing 5-char spelling variant reads as two
# words and stays unmerged, per the never-merge-distinct-entities bar.
assert not _same_word_variant("colour", "color")