Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -860,6 +860,12 @@ graphify label ./my-project # (re)name commun
graphify label ./my-project --backend=openai --model gpt-4o # force a specific backend and model
```

`--no-dedup` also skips coalescing distinct non-AST nodes solely because they
share a file and label. The Python equivalents are `build(chunks, dedup=False)`,
`build_merge(chunks, graph_path, dedup=False)`, and
`build_from_json(extraction, dedup=False)`. AST/semantic twins still reconcile to
the canonical AST node, and document-file twin reconciliation remains enabled.

> **Community names:** inside an agent (Claude Code, Gemini CLI) the agent names communities itself. When you run the bare CLI, `cluster-only` auto-names them with the configured backend (built-in or custom OpenAI-compatible provider) — pass `--no-label` to keep `Community N`, or run `graphify label` to (re)generate names on demand.

---
Expand Down
10 changes: 8 additions & 2 deletions graphify/build.py
Original file line number Diff line number Diff line change
Expand Up @@ -874,13 +874,17 @@ def _doc_twin_remap(nodes: list) -> dict[str, str]:
return remap


def build_from_json(extraction: dict, *, directed: bool = False, root: str | Path | None = None) -> nx.Graph:
def build_from_json(extraction: dict, *, directed: bool = False, root: str | Path | None = None,

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

⚠️ Health regression — build_from_json()

fans out to 20 callees (efferent coupling); 231 callers depend on it (afferent coupling).

Grounded coupling-delta finding (deterministic), not an LLM guess.

dedup: bool = True) -> nx.Graph:
"""Build a NetworkX graph from an extraction dict.

directed=True produces a DiGraph that preserves edge direction (source→target).
directed=False (default) produces an undirected Graph for backward compatibility.
root: if given, absolute source_file paths from semantic subagents are made
relative to root so all nodes share a consistent path key (#932).
dedup=False preserves distinct non-AST IDs rather than coalescing nodes by
file and label. AST/semantic and document-file twin reconciliation
remain enabled.
"""
_root = str(Path(root).resolve()) if root else None
# NetworkX <= 3.1 serialised edges as "links"; remap to "edges" for compatibility.
Expand Down Expand Up @@ -1158,6 +1162,8 @@ def build_from_json(extraction: dict, *, directed: bool = False, root: str | Pat
if key in _loc_collisions:
continue # ambiguous key: no safe canonical winner, leave ghost intact
if key in _loc_nodes and _loc_nodes[key] != nid:
if not dedup and G.nodes[_loc_nodes[key]].get("_origin") != "ast":
continue
_noloc_nodes[key] = nid
elif key not in _loc_nodes:
# Spec-conformant method ghost omitting class segment / leading dot
Expand Down Expand Up @@ -1599,7 +1605,7 @@ def build(
protected_ids=protected_ids,
)
_dedup_collapsed = _before_dedup - len(combined["nodes"])
G = build_from_json(combined, directed=directed, root=_root)
G = build_from_json(combined, directed=directed, root=_root, dedup=dedup)
# CLI reads this to tell a dedup shrink from a file deletion (#3774).
# Popped before to_json so it is not stored in graph.json.
if _dedup_collapsed:
Expand Down
88 changes: 88 additions & 0 deletions tests/test_build_located_semantic_identity.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,88 @@
import json

import pytest

from graphify.build import build, build_from_json, build_merge


def _located_nodes():
return [
{"id": "pkg_name", "label": "name", "file_type": "code",
"source_file": "pkg/plugin.json", "source_location": "L2"},
{"id": "pkg_author_name", "label": "name", "file_type": "code",
"source_file": "pkg/plugin.json", "source_location": "L5"},
{"id": "doc_s1_notes", "label": "Notas", "file_type": "document",
"source_file": "docs/plan.md", "source_location": "L10"},
{"id": "doc_s2_notes", "label": "Notas", "file_type": "document",
"source_file": "docs/plan.md", "source_location": "L80"},
{"id": "other_name", "label": "name", "file_type": "code",
"source_file": "pkg/other.json", "source_location": "L2"},
]


@pytest.mark.parametrize("reverse", [False, True])
def test_distinct_same_file_locations_preserve_ids_and_edges(reverse):
nodes = _located_nodes()
if reverse:
nodes.reverse()
graph = build_from_json({
"nodes": nodes,
"edges": [
{"source": "pkg_name", "target": "doc_s1_notes", "relation": "references"},
{"source": "pkg_author_name", "target": "doc_s2_notes", "relation": "references"},
],
"hyperedges": [],
}, dedup=False)
assert set(graph.nodes) == {node["id"] for node in nodes}
assert graph.has_edge("pkg_name", "doc_s1_notes")
assert graph.has_edge("pkg_author_name", "doc_s2_notes")


def test_empty_no_dedup_merge_keeps_located_graph(tmp_path):
nodes = _located_nodes()
graph_path = tmp_path / "graph.json"
graph_path.write_text(json.dumps({"nodes": nodes, "edges": [], "hyperedges": []}), encoding="utf-8")
graph = build_merge([{"nodes": [], "edges": [], "hyperedges": []}], graph_path, dedup=False)
assert set(graph.nodes) == {node["id"] for node in nodes}


def test_disabled_dedup_keeps_distinct_semantic_ids_at_identical_locations():
nodes = [
{"id": name, "label": "Notes", "file_type": "document", "_origin": "semantic",
"source_file": "plan.md", "source_location": location}
for name, location in [("a_first", "L2"), ("b_second", "L5"), ("z_first", "L2"), ("z_second", "L5")]
]
graph = build_from_json({
"nodes": nodes,
"edges": [{"source": "z_first", "target": "z_second", "relation": "references"}],
"hyperedges": [],
}, dedup=False)
assert set(graph.nodes) == {node["id"] for node in nodes}
assert graph.has_edge("z_first", "z_second")


def test_ast_twin_remains_canonical_despite_semantic_location_drift():
graph = build_from_json({
"nodes": [
{"id": "src_render", "label": "render", "file_type": "code", "_origin": "ast",
"source_file": "src/view.py", "source_location": "L2"},
{"id": "view_render", "label": "render", "file_type": "code", "_origin": "semantic",
"source_file": "src/view.py", "source_location": "L5"},
],
"edges": [], "hyperedges": [],
}, dedup=False)
assert set(graph.nodes) == {"src_render"}


def test_full_build_honors_disabled_semantic_dedup():
nodes = _located_nodes()
graph = build([{"nodes": nodes, "edges": [], "hyperedges": []}], dedup=False)
assert set(graph.nodes) == {node["id"] for node in nodes}


def test_fresh_merge_honors_disabled_semantic_dedup(tmp_path):
graph_path = tmp_path / "graph.json"
graph_path.write_text(json.dumps({"nodes": [], "edges": [], "hyperedges": []}), encoding="utf-8")
nodes = _located_nodes()
graph = build_merge([{"nodes": nodes, "edges": [], "hyperedges": []}], graph_path, dedup=False)
assert set(graph.nodes) == {node["id"] for node in nodes}
Loading