v2: hypergraph support - hyperedges in graph.json, shaded regions in HTML, report section

This commit is contained in:
Safi
2026-04-06 16:06:31 +01:00
parent 6fa4c7e662
commit 3a117c93e8
7 changed files with 299 additions and 3 deletions
+3
View File
@@ -25,6 +25,9 @@ def build_from_json(extraction: dict) -> nx.Graph:
attrs["_src"] = src
attrs["_tgt"] = tgt
G.add_edge(src, tgt, **attrs)
hyperedges = extraction.get("hyperedges", [])
if hyperedges:
G.graph["hyperedges"] = hyperedges
return G
+66
View File
@@ -55,6 +55,58 @@ def _html_styles() -> str:
</style>"""
def _hyperedge_script(hyperedges_json: str) -> str:
return f"""<script>
// Render hyperedges as shaded regions
const hyperedges = {hyperedges_json};
function drawHyperedges() {{
const canvas = network.canvas.frame.canvas;
const ctx = canvas.getContext('2d');
hyperedges.forEach(h => {{
const positions = h.nodes
.map(nid => network.getPositions([nid])[nid])
.filter(p => p !== undefined);
if (positions.length < 2) return;
// Draw convex hull as filled polygon
ctx.save();
ctx.globalAlpha = 0.12;
ctx.fillStyle = '#6366f1';
ctx.strokeStyle = '#6366f1';
ctx.lineWidth = 2;
ctx.beginPath();
const scale = network.getScale();
const offset = network.getViewPosition();
const toCanvas = (p) => ({{
x: (p.x - offset.x) * scale + canvas.width / 2,
y: (p.y - offset.y) * scale + canvas.height / 2
}});
const pts = positions.map(toCanvas);
// Expand hull slightly
const cx = pts.reduce((s, p) => s + p.x, 0) / pts.length;
const cy = pts.reduce((s, p) => s + p.y, 0) / pts.length;
const expanded = pts.map(p => ({{
x: cx + (p.x - cx) * 1.15,
y: cy + (p.y - cy) * 1.15
}}));
ctx.moveTo(expanded[0].x, expanded[0].y);
expanded.slice(1).forEach(p => ctx.lineTo(p.x, p.y));
ctx.closePath();
ctx.fill();
ctx.globalAlpha = 0.4;
ctx.stroke();
// Label
ctx.globalAlpha = 0.8;
ctx.fillStyle = '#4f46e5';
ctx.font = 'bold 11px sans-serif';
ctx.textAlign = 'center';
ctx.fillText(h.label, cx, cy - 5);
ctx.restore();
}});
}}
network.on('afterDrawing', drawHyperedges);
</script>"""
def _html_script(nodes_json: str, edges_json: str, legend_json: str) -> str:
return f"""<script>
const RAW_NODES = {nodes_json};
@@ -198,6 +250,17 @@ LEGEND.forEach(c => {{
_CONFIDENCE_SCORE_DEFAULTS = {"EXTRACTED": 1.0, "INFERRED": 0.5, "AMBIGUOUS": 0.2}
def attach_hyperedges(G: nx.Graph, hyperedges: list) -> None:
"""Store hyperedges in the graph's metadata dict."""
existing = G.graph.get("hyperedges", [])
seen_ids = {h["id"] for h in existing}
for h in hyperedges:
if h.get("id") and h["id"] not in seen_ids:
existing.append(h)
seen_ids.add(h["id"])
G.graph["hyperedges"] = existing
def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str) -> None:
node_community = _node_community_map(communities)
data = json_graph.node_link_data(G, edges="links")
@@ -207,6 +270,7 @@ def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str) ->
if "confidence_score" not in link:
conf = link.get("confidence", "EXTRACTED")
link["confidence_score"] = _CONFIDENCE_SCORE_DEFAULTS.get(conf, 1.0)
data["hyperedges"] = getattr(G, "graph", {}).get("hyperedges", [])
with open(output_path, "w") as f:
json.dump(data, f, indent=2)
@@ -302,6 +366,7 @@ def to_html(
nodes_json = json.dumps(vis_nodes)
edges_json = json.dumps(vis_edges)
legend_json = json.dumps(legend_data)
hyperedges_json = json.dumps(getattr(G, "graph", {}).get("hyperedges", []))
title = sanitize_label(str(output_path))
stats = f"{G.number_of_nodes()} nodes &middot; {G.number_of_edges()} edges &middot; {len(communities)} communities"
@@ -331,6 +396,7 @@ def to_html(
<div id="stats">{stats}</div>
</div>
{_html_script(nodes_json, edges_json, legend_json)}
{_hyperedge_script(hyperedges_json)}
</body>
</html>"""
+10
View File
@@ -74,6 +74,16 @@ def generate(
else:
lines.append("- None detected - all connections are within the same source files.")
hyperedges = G.graph.get("hyperedges", [])
if hyperedges:
lines += ["", "## Hyperedges (group relationships)"]
for h in hyperedges:
node_labels = ", ".join(h.get("nodes", []))
conf = h.get("confidence", "INFERRED")
cscore = h.get("confidence_score")
conf_tag = f"{conf} {cscore:.2f}" if cscore is not None else conf
lines.append(f"- **{h.get('label', h.get('id', ''))}** — {node_labels} [{conf_tag}]")
lines += ["", "## Communities"]
from .analyze import _is_file_node as _ifn
for cid, nodes in communities.items():
+7 -1
View File
@@ -213,6 +213,12 @@ Semantic similarity: if two concepts in this chunk solve the same problem or rep
- Two error types that handle the same failure mode differently
Only add these when the similarity is genuinely non-obvious and cross-cutting. Do not add them for trivially similar things.
Hyperedges: if 3 or more nodes clearly participate together in a shared concept, flow, or pattern that is not captured by pairwise edges alone, add a hyperedge to a top-level `hyperedges` array. Examples:
- All classes that implement a common protocol or interface
- All functions in an authentication flow (even if they don't all call each other)
- All concepts from a paper section that form one coherent idea
Use sparingly — only when the group relationship adds information beyond the pairwise edges. Maximum 3 hyperedges per chunk.
If a file has YAML frontmatter (--- ... ---), copy source_url, captured_at, author,
contributor onto every node from that file.
@@ -224,7 +230,7 @@ confidence_score rules:
- AMBIGUOUS edges: score 0.1-0.3
Output exactly this JSON (no other text):
{"nodes":[{"id":"filestem_entityname","label":"Human Readable Name","file_type":"code|document|paper|image","source_file":"relative/path","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"relative/path","source_location":null,"weight":1.0}],"input_tokens":0,"output_tokens":0}
{"nodes":[{"id":"filestem_entityname","label":"Human Readable Name","file_type":"code|document|paper|image","source_file":"relative/path","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"relative/path","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":"relative/path"}],"input_tokens":0,"output_tokens":0}
```
**Step B3 - Collect, cache, and merge**