fix(serve): scale per-term exact/prefix score tiers by squared term coverage (#1602)

In a multi-term query, a lone generic term that exactly equals a short leaf
label (`home` vs a `home()` action, `list` vs a `list()` function) received
the full exact-tier bonus and outscored nodes matching several of the query's
terms by ~10x, so `_pick_seeds`' gap cutoff discarded the genuinely relevant
candidates and `path`-style consumers of `scored[0]` had no backstop.

`_score_nodes` now scales the per-term exact/prefix tier contributions by the
squared fraction of query terms the node's label matches
(`tiered * (matched / n_terms) ** 2`). Squared because the exact tier is 10x
the prefix tier; label hits only (source-path hits score but don't count as
coverage, so a colliding leaf in the target's own directory can't win its
exact tier back via path fragments); query tokens deduped order-preserving so
a repeated word can't quadratically restore the collision. Single-term and
full-coverage queries are unchanged (coverage == 1), keeping identifier
exact-match dominance. Guarded against zero-term division.

Dropped the unrelated README badge changes from the PR; kept the scoring fix
and its two regression tests.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
fkhawajagh
2026-07-08 23:56:28 +01:00
committed by safishamsi
co-authored by Claude Opus 4.8
parent e2d47535c3
commit 2f9deae8d5
2 changed files with 95 additions and 3 deletions
+33 -3
View File
@@ -285,7 +285,11 @@ def _trigram_candidates(G: nx.Graph, needles: list[str], *, guard_frac: float =
def _score_nodes(G: nx.Graph, terms: list[str]) -> list[tuple[float, str]]:
scored = []
norm_terms = [tok for t in terms for tok in _search_tokens(t)]
# Dedupe tokens, order-preserving (as _pick_seeds already does): a repeated
# query word must not double-count every tier, and with coverage scaling
# below it would also inflate the matched-term ratio (#1602).
norm_terms = list(dict.fromkeys(tok for t in terms for tok in _search_tokens(t)))
n_terms = len(norm_terms)
idf = _compute_idf(G, norm_terms)
# Whole-query string for full-label matching (mirrors _find_node's `term`).
joined = " ".join(norm_terms)
@@ -328,18 +332,38 @@ def _score_nodes(G: nx.Graph, terms: list[str]) -> list[tuple[float, str]]:
or label_tokens.startswith(joined)
):
score += _PREFIX_MATCH_BONUS * 10 * joined_w
# Term coverage (#1602): scale the per-term exact/prefix tiers by the
# squared fraction of query terms the node's LABEL matches, so a lone
# generic word that happens to equal a short label (query term "home"
# vs. a home() leaf) cannot bury nodes that match several of the
# query's terms. Squaring matters because the exact tier is 10x the
# prefix tier: at linear coverage a 1-of-10-terms exact match still
# outscores a 3-of-10 prefix+substring match. Single-term and
# full-coverage queries are unchanged (coverage == 1), so identifier
# lookups keep exact-match dominance. Source-file hits score but do
# not count as coverage: a colliding leaf whose directory shares
# tokens with the query (common near the intended target) must not
# win back its exact tier via path fragments. The substring/source
# bonuses and the full-query tier above stay unscaled.
matched = 0
tiered = 0.0
for t in norm_terms:
w = idf.get(t, 1.0)
# Three-tier precedence: exact > prefix > substring (take the
# strongest tier per term so a single term cannot double-count).
if t == norm_label or t == bare_label:
score += _EXACT_MATCH_BONUS * w
tiered += _EXACT_MATCH_BONUS * w
matched += 1
elif norm_label.startswith(t) or bare_label.startswith(t):
score += _PREFIX_MATCH_BONUS * w
tiered += _PREFIX_MATCH_BONUS * w
matched += 1
elif t in norm_label:
score += _SUBSTRING_MATCH_BONUS * w
matched += 1
if t in source:
score += _SOURCE_MATCH_BONUS * w
if tiered:
score += tiered * (matched / n_terms) ** 2
if score > 0:
scored.append((score, nid))
# Sort by score desc; break ties toward the shorter label so a concise exact
@@ -378,6 +402,12 @@ def _pick_seeds(
collision cannot starve out the others. Ties within a term are broken by
graph degree (structural centrality), so an isolated incidental match
doesn't out-rank a real, well-connected hub for that term.
Coverage scaling in _score_nodes (#1602) now dampens a lone collision's
exact tier on multi-term queries, which brings label-matching relevant
nodes back inside the gap window; this per-term guarantee remains
load-bearing for relevant nodes matched only via substrings, whose flat
scores a dampened collision can still exceed.
"""
if not scored:
return []
+62
View File
@@ -8,6 +8,8 @@ from graphify.serve import (
_communities_from_graph,
_score_nodes,
_compute_idf,
_EXACT_MATCH_BONUS,
_SOURCE_MATCH_BONUS,
_pick_seeds,
_bfs,
_dfs,
@@ -123,6 +125,66 @@ def test_score_nodes_multiword_exact_label_outranks_superset():
assert scored[0][0] > scored[1][0], "exact label must strictly outrank superset/token-bag matches"
def test_score_nodes_coverage_lone_generic_exact_hit_loses_to_multi_term_match():
"""A lone generic-word exact match must not bury a multi-term match.
Reproduces #1602: in a multi-term query, a single generic term that
exactly equals a short leaf label (query term "list" vs a list() function
node) received the full exact-tier bonus and outranked every node matching
several of the query's terms, even when the query contained the target's
literal identifier. The per-term exact/prefix tiers are now scaled by
squared term coverage, so a 1-of-5-terms collision drops below a
multi-term match. The leaves live in the same directory as the target
(the realistic case) to pin that source-path hits do not count as
coverage and hand the collision its exact tier back.
"""
G = nx.Graph()
def _add(nid, label, src):
G.add_node(nid, label=label, norm_label=label.lower(),
source_file=src, community=0)
_add("target", "ClientLive.Index", "lib/clients_live/index.ex")
_add("form", "ClientLive.Form", "lib/clients_live/form.ex")
_add("show", "ClientLive.Show", "lib/clients_live/show.ex")
# Same-named tiny leaf functions: "list" == bare label fires the exact
# tier. Placed in the target's own directory so their source paths also
# substring-match the query term "clients": a path hit must not inflate
# the coverage that multiplies the exact tier.
for i in range(3):
_add(f"leaf{i}", "list()", f"lib/clients_live/helpers{i}.ex")
# Filler making "list" a common (low-IDF) token, as in a real graph where
# list()/get()/new() style names are ubiquitous.
for i in range(24):
_add(f"filler{i}", f"shopping list {i}", f"lib/filler{i}.ex")
# The user pastes the real identifier plus context words; tokenization
# yields 5 terms: clientlive, index, clients, list, columns.
scored = _score_nodes(G, [t.lower() for t in "ClientLive.Index clients list columns".split()])
by_id = {nid: s for s, nid in scored}
assert scored[0][1] == "target"
assert by_id["target"] > by_id["leaf0"], (
"a 1-of-5-terms exact collision must not outrank the node matching 3 of 5 terms"
)
def test_score_nodes_coverage_full_coverage_query_is_unchanged():
"""Coverage scaling must not touch full-coverage queries (coverage == 1).
A single-term identifier lookup keeps the exact tier's full magnitude, so
`query "FooBarService"` behavior is byte-identical to before #1602.
"""
G = _make_graph()
scored = _score_nodes(G, ["extract"])
w = _compute_idf(G, ["extract"])["extract"]
assert scored[0][1] == "n1"
# Full-query exact tier (10x) + per-term exact tier + source hit
# ("extract" in "extract.py"), all undampened.
expected = (_EXACT_MATCH_BONUS * 10 + _EXACT_MATCH_BONUS + _SOURCE_MATCH_BONUS) * w
assert scored[0][0] == pytest.approx(expected)
def test_find_node_ignores_trailing_punctuation():
G = _make_graph()
assert _find_node(G, "extract?") == ["n1"]