diff --git a/README.md b/README.md index ecab496d05..fd72e793ad 100644 --- a/README.md +++ b/README.md @@ -860,6 +860,12 @@ graphify label ./my-project # (re)name commun graphify label ./my-project --backend=openai --model gpt-4o # force a specific backend and model ``` +`--no-dedup` also skips coalescing distinct non-AST nodes solely because they +share a file and label. The Python equivalents are `build(chunks, dedup=False)`, +`build_merge(chunks, graph_path, dedup=False)`, and +`build_from_json(extraction, dedup=False)`. AST/semantic twins still reconcile to +the canonical AST node, and document-file twin reconciliation remains enabled. + > **Community names:** inside an agent (Claude Code, Gemini CLI) the agent names communities itself. When you run the bare CLI, `cluster-only` auto-names them with the configured backend (built-in or custom OpenAI-compatible provider) — pass `--no-label` to keep `Community N`, or run `graphify label` to (re)generate names on demand. --- diff --git a/graphify/build.py b/graphify/build.py index 871b927987..151e65bb43 100644 --- a/graphify/build.py +++ b/graphify/build.py @@ -874,13 +874,17 @@ def _doc_twin_remap(nodes: list) -> dict[str, str]: return remap -def build_from_json(extraction: dict, *, directed: bool = False, root: str | Path | None = None) -> nx.Graph: +def build_from_json(extraction: dict, *, directed: bool = False, root: str | Path | None = None, + dedup: bool = True) -> nx.Graph: """Build a NetworkX graph from an extraction dict. directed=True produces a DiGraph that preserves edge direction (source→target). directed=False (default) produces an undirected Graph for backward compatibility. root: if given, absolute source_file paths from semantic subagents are made relative to root so all nodes share a consistent path key (#932). + dedup=False preserves distinct non-AST IDs rather than coalescing nodes by + file and label. AST/semantic and document-file twin reconciliation + remain enabled. """ _root = str(Path(root).resolve()) if root else None # NetworkX <= 3.1 serialised edges as "links"; remap to "edges" for compatibility. @@ -1158,6 +1162,8 @@ def build_from_json(extraction: dict, *, directed: bool = False, root: str | Pat if key in _loc_collisions: continue # ambiguous key: no safe canonical winner, leave ghost intact if key in _loc_nodes and _loc_nodes[key] != nid: + if not dedup and G.nodes[_loc_nodes[key]].get("_origin") != "ast": + continue _noloc_nodes[key] = nid elif key not in _loc_nodes: # Spec-conformant method ghost omitting class segment / leading dot @@ -1599,7 +1605,7 @@ def build( protected_ids=protected_ids, ) _dedup_collapsed = _before_dedup - len(combined["nodes"]) - G = build_from_json(combined, directed=directed, root=_root) + G = build_from_json(combined, directed=directed, root=_root, dedup=dedup) # CLI reads this to tell a dedup shrink from a file deletion (#3774). # Popped before to_json so it is not stored in graph.json. if _dedup_collapsed: diff --git a/tests/test_build_located_semantic_identity.py b/tests/test_build_located_semantic_identity.py new file mode 100644 index 0000000000..b63035e44f --- /dev/null +++ b/tests/test_build_located_semantic_identity.py @@ -0,0 +1,88 @@ +import json + +import pytest + +from graphify.build import build, build_from_json, build_merge + + +def _located_nodes(): + return [ + {"id": "pkg_name", "label": "name", "file_type": "code", + "source_file": "pkg/plugin.json", "source_location": "L2"}, + {"id": "pkg_author_name", "label": "name", "file_type": "code", + "source_file": "pkg/plugin.json", "source_location": "L5"}, + {"id": "doc_s1_notes", "label": "Notas", "file_type": "document", + "source_file": "docs/plan.md", "source_location": "L10"}, + {"id": "doc_s2_notes", "label": "Notas", "file_type": "document", + "source_file": "docs/plan.md", "source_location": "L80"}, + {"id": "other_name", "label": "name", "file_type": "code", + "source_file": "pkg/other.json", "source_location": "L2"}, + ] + + +@pytest.mark.parametrize("reverse", [False, True]) +def test_distinct_same_file_locations_preserve_ids_and_edges(reverse): + nodes = _located_nodes() + if reverse: + nodes.reverse() + graph = build_from_json({ + "nodes": nodes, + "edges": [ + {"source": "pkg_name", "target": "doc_s1_notes", "relation": "references"}, + {"source": "pkg_author_name", "target": "doc_s2_notes", "relation": "references"}, + ], + "hyperedges": [], + }, dedup=False) + assert set(graph.nodes) == {node["id"] for node in nodes} + assert graph.has_edge("pkg_name", "doc_s1_notes") + assert graph.has_edge("pkg_author_name", "doc_s2_notes") + + +def test_empty_no_dedup_merge_keeps_located_graph(tmp_path): + nodes = _located_nodes() + graph_path = tmp_path / "graph.json" + graph_path.write_text(json.dumps({"nodes": nodes, "edges": [], "hyperedges": []}), encoding="utf-8") + graph = build_merge([{"nodes": [], "edges": [], "hyperedges": []}], graph_path, dedup=False) + assert set(graph.nodes) == {node["id"] for node in nodes} + + +def test_disabled_dedup_keeps_distinct_semantic_ids_at_identical_locations(): + nodes = [ + {"id": name, "label": "Notes", "file_type": "document", "_origin": "semantic", + "source_file": "plan.md", "source_location": location} + for name, location in [("a_first", "L2"), ("b_second", "L5"), ("z_first", "L2"), ("z_second", "L5")] + ] + graph = build_from_json({ + "nodes": nodes, + "edges": [{"source": "z_first", "target": "z_second", "relation": "references"}], + "hyperedges": [], + }, dedup=False) + assert set(graph.nodes) == {node["id"] for node in nodes} + assert graph.has_edge("z_first", "z_second") + + +def test_ast_twin_remains_canonical_despite_semantic_location_drift(): + graph = build_from_json({ + "nodes": [ + {"id": "src_render", "label": "render", "file_type": "code", "_origin": "ast", + "source_file": "src/view.py", "source_location": "L2"}, + {"id": "view_render", "label": "render", "file_type": "code", "_origin": "semantic", + "source_file": "src/view.py", "source_location": "L5"}, + ], + "edges": [], "hyperedges": [], + }, dedup=False) + assert set(graph.nodes) == {"src_render"} + + +def test_full_build_honors_disabled_semantic_dedup(): + nodes = _located_nodes() + graph = build([{"nodes": nodes, "edges": [], "hyperedges": []}], dedup=False) + assert set(graph.nodes) == {node["id"] for node in nodes} + + +def test_fresh_merge_honors_disabled_semantic_dedup(tmp_path): + graph_path = tmp_path / "graph.json" + graph_path.write_text(json.dumps({"nodes": [], "edges": [], "hyperedges": []}), encoding="utf-8") + nodes = _located_nodes() + graph = build_merge([{"nodes": nodes, "edges": [], "hyperedges": []}], graph_path, dedup=False) + assert set(graph.nodes) == {node["id"] for node in nodes}