From a8559a49049c3ddf3fea73b4fda257c500479c71 Mon Sep 17 00:00:00 2001 From: Andy Tsai Date: Wed, 2 Sep 2026 20:01:37 +0800 Subject: [PATCH] feat(query): surface bridge nodes reached from two or more seeds MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A multi-term question ("AuthModule BillingDB", "what links X to Y") seats several seeds, and the node their traversals share is usually the answer. Today it is rendered as just another depth-1 neighbor — and under a tight `--budget` it loses to a same-layer hub with more edges, so the connective node is cut while the incidental hub survives. `path` covers the exact two-endpoint case, but nothing covers a natural-language question that resolves to several seeds. Add `_bridge_nodes`: walk each seed separately inside the node set the traversal already found (both edge directions, same hop rule as the layering in `_subgraph_to_text`), and return the non-seed nodes reached from two or more seeds, ordered by seeds-reached then id. It never adds a node the traversal did not find, so the answer's node set is unchanged. `_query_graph_text` names them in the header (`Bridges: [...]`) and passes them to `_subgraph_to_text`, which renders them directly after the seed block ahead of the hop/degree-ordered rest. No new flag, no LLM, no extra scoring pass. Single-seed questions are byte-identical: no bridges, no header entry, unchanged ordering. Tests: `_bridge_nodes` (two seeds -> connector; single seed -> empty; a seed is never reported as a bridge), renderer places bridges after seeds, end to end header + survival under a budget that cuts the hub, and the single-seed no-op guard. Co-Authored-By: Claude Fable 5.1 --- graphify/serve.py | 75 ++++++++++++++++++++++++++++++++++++++++++--- tests/test_serve.py | 71 ++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 142 insertions(+), 4 deletions(-) diff --git a/graphify/serve.py b/graphify/serve.py index a9ecd3540..f7dd8accf 100644 --- a/graphify/serve.py +++ b/graphify/serve.py @@ -989,11 +989,66 @@ def _dfs(G: nx.Graph, start_nodes: list[str], depth: int) -> tuple[set[str], lis return visited, edges_seen -def _subgraph_to_text(G: nx.Graph, nodes: set[str], edges: list[tuple], token_budget: int = 2000, *, seeds: list[str] | None = None) -> str: +def _bridge_nodes(G: nx.Graph, nodes: set[str], seeds: list[str], depth: int) -> list[str]: + """Non-seed nodes in `nodes` reachable within `depth` hops from two or more + distinct seeds — the nodes that connect the entities a multi-term question + named, which are usually the answer to "what links X to Y". + + Walks each seed separately inside the already-traversed node set (both edge + directions, like the hop layering in `_subgraph_to_text`), so it never adds + a node the traversal did not find. Ordered by seeds reached (desc) then id, + so the output is deterministic. Empty with fewer than two seeds. + """ + seed_set = set(seeds) + if len(seed_set) < 2: + return [] + + def _adj(n): + if G.is_directed(): + yield from G.successors(n) + yield from G.predecessors(n) + else: + yield from G.neighbors(n) + + reached_by: dict[str, int] = {} + for seed in seed_set: + if seed not in nodes: + continue + seen = {seed} + frontier = [seed] + for _ in range(depth): + nxt = [] + for n in frontier: + for nb in _adj(n): + if nb in nodes and nb not in seen: + seen.add(nb) + nxt.append(nb) + frontier = nxt + for n in seen - seed_set: + reached_by[n] = reached_by.get(n, 0) + 1 + return sorted( + (n for n, k in reached_by.items() if k >= 2), + key=lambda n: (-reached_by[n], str(n)), + ) + + +def _subgraph_to_text( + G: nx.Graph, + nodes: set[str], + edges: list[tuple], + token_budget: int = 2000, + *, + seeds: list[str] | None = None, + bridges: list[str] | None = None, +) -> str: """Render subgraph as text, cutting at token_budget (approx 3 chars/token). seeds: exact-match nodes rendered first before the degree-sorted expansion, so the queried symbol always appears at the top of the output. + bridges: nodes reached from two or more seeds (see `_bridge_nodes`); they + render directly after the seed block so the node connecting the question's + entities survives the budget ahead of same-layer hubs. Missing or empty + leaves the ordering unchanged. """ char_budget = token_budget * 3 lines = [] @@ -1025,8 +1080,12 @@ def _adj(n): dist[nb] = hop nxt.append(nb) frontier = nxt - ordered = seed_hits + sorted( - nodes - seed_set, + # Bridges (reached from ≥2 seeds) follow the seed block in their given + # order; everything else keeps the hop/degree order. + bridge_hits = [n for n in (bridges or []) if n in nodes and n not in seed_set] + bridge_set = set(bridge_hits) + ordered = seed_hits + bridge_hits + sorted( + nodes - seed_set - bridge_set, key=lambda n: (dist.get(n, 1 << 30), -G.degree(n), str(n)), ) for nid in ordered: @@ -1231,10 +1290,16 @@ def _query_graph_text( resolved_filters, filter_source = _resolve_context_filters(question, context_filters) traversal_graph = _filter_graph_by_context(G, resolved_filters) nodes, edges = _dfs(traversal_graph, start_nodes, depth) if mode == "dfs" else _bfs(traversal_graph, start_nodes, depth) + # A multi-term question seats several seeds; the nodes their traversals + # share are the connective tissue the question is asking about. Name them + # up front and render them right after the seeds (see _bridge_nodes). + bridges = _bridge_nodes(traversal_graph, nodes, start_nodes, depth) header_parts = [ f"Traversal: {mode.upper()} depth={depth}", f"Start: {[G.nodes[n].get('label', n) for n in start_nodes]}", ] + if bridges: + header_parts.append(f"Bridges: {[G.nodes[n].get('label', n) for n in bridges]}") # Name the graph this answer came from. `graphify-out/` resolves against the # CWD, so running a query from a parent project while thinking about a # vendored subproject silently answers from the wrong corpus — the output is @@ -1253,7 +1318,9 @@ def _query_graph_text( # Pass the seeds so the queried symbol renders first and survives truncation # (#BUG2): a branch merge had silently dropped this argument, leaving the # seed-first ordering as dead code. - return header + _subgraph_to_text(traversal_graph, nodes, edges, token_budget, seeds=start_nodes) + return header + _subgraph_to_text( + traversal_graph, nodes, edges, token_budget, seeds=start_nodes, bridges=bridges + ) def _find_node_tiers( diff --git a/tests/test_serve.py b/tests/test_serve.py index 87e71f821..e3c29967d 100644 --- a/tests/test_serve.py +++ b/tests/test_serve.py @@ -1735,3 +1735,74 @@ def test_resolve_single_node_shared_by_get_node_and_get_neighbors(): nid, err = _resolve_single_node(G, "nonexistent") assert nid is None assert "No node matching" in err + + +# --- bridge nodes: reached from two or more seeds --- + +def _two_seed_graph(): + """`AuthModule` and `BillingDB` (both exact query hits) are joined only + through `SessionStore`. `Logger` is a hub off `AuthModule` with more + neighbors than `SessionStore`; `TokenCache` hangs off `AuthModule` alone.""" + G = nx.Graph() + for nid, label in [("auth", "AuthModule"), ("bill", "BillingDB"), + ("sess", "SessionStore"), ("log", "Logger"), ("tok", "TokenCache")]: + G.add_node(nid, label=label, source_file=f"{nid}.py") + for u, v in [("auth", "sess"), ("sess", "bill"), ("auth", "log"), ("auth", "tok")]: + G.add_edge(u, v, relation="calls", confidence="EXTRACTED") + for i in range(8): + G.add_node(f"leaf{i}", label=f"Leaf{i}", source_file="leaf.py") + G.add_edge("log", f"leaf{i}", relation="calls", confidence="EXTRACTED") + return G + + +def test_bridge_nodes_returns_nodes_reached_from_two_seeds(): + from graphify.serve import _bridge_nodes + G = _two_seed_graph() + nodes = set(G.nodes) + assert _bridge_nodes(G, nodes, ["auth", "bill"], depth=2) == ["sess"] + + +def test_bridge_nodes_empty_with_single_seed(): + from graphify.serve import _bridge_nodes + G = _two_seed_graph() + assert _bridge_nodes(G, set(G.nodes), ["auth"], depth=2) == [] + + +def test_bridge_nodes_never_reports_a_seed_as_a_bridge(): + """Two adjacent seeds reach each other; a seed is a start point, not a + bridge, so neither may appear in the result.""" + from graphify.serve import _bridge_nodes + G = _two_seed_graph() + G.add_edge("auth", "bill", relation="calls", confidence="EXTRACTED") + assert "auth" not in _bridge_nodes(G, set(G.nodes), ["auth", "bill"], depth=2) + assert "bill" not in _bridge_nodes(G, set(G.nodes), ["auth", "bill"], depth=2) + + +def test_subgraph_to_text_renders_bridges_right_after_seeds(): + G = _two_seed_graph() + text = _subgraph_to_text( + G, set(G.nodes), list(G.edges()), token_budget=2000, + seeds=["auth", "bill"], bridges=["sess"], + ) + labels = [l.split(" [", 1)[0] for l in text.splitlines() if l.startswith("NODE ")] + assert labels[:3] == ["NODE AuthModule", "NODE BillingDB", "NODE SessionStore"], labels + + +def test_query_graph_text_reports_bridge_in_header_and_keeps_it_under_budget(): + """End to end: a two-entity question seats both as seeds; the node that + connects them is named in the header and survives a budget that cuts the + hub sitting in the same hop layer.""" + G = _two_seed_graph() + text = _query_graph_text(G, "AuthModule BillingDB", mode="bfs", depth=2, token_budget=70) + header = text.split("\n\n", 1)[0] + assert "Bridges: ['SessionStore']" in header, header + labels = [l.split(" [", 1)[0] for l in text.splitlines() if l.startswith("NODE ")] + # Seed order is _pick_seeds' business; the bridge must follow the seed block. + assert set(labels[:2]) == {"NODE AuthModule", "NODE BillingDB"}, labels + assert labels[2] == "NODE SessionStore", labels + + +def test_query_graph_text_single_seed_has_no_bridge_header(): + G = _two_seed_graph() + text = _query_graph_text(G, "AuthModule", mode="bfs", depth=2, token_budget=2000) + assert "Bridges" not in text