Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions graphify/__main__.py
Original file line number Diff line number Diff line change
Expand Up @@ -760,6 +760,9 @@ def _run_cli() -> None:
print(" merge-driver <base> <current> <other> git merge driver: union-merge two graph.json files (set up via hook install)")
print(" merge-graphs <g1> <g2> merge two or more graph.json files into one cross-repo graph")
print(" --out <path> output path (default: graphify-out/merged-graph.json)")
print(" --previous <path> restore the previous merged clustering's community ids")
print(" (by node id) instead of each input's own per-source ids,")
print(" so a later cluster-only run reuses labels correctly")
print(" --branch <branch> checkout a specific branch (default: repo default)")
print(" --out <dir> clone to a custom directory (default: ~/.graphify/repos/<owner>/<repo>)")
print(" add <url> fetch a URL and save it to ./raw, then update the graph")
Expand Down
38 changes: 37 additions & 1 deletion graphify/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -2663,20 +2663,44 @@ def _load_graph(p: str):
args = sys.argv[2:]
graph_paths: list[Path] = []
out_path = Path(_GRAPHIFY_OUT) / "merged-graph.json"
previous_path: "Path | None" = None
i = 0
while i < len(args):
if args[i] == "--out" and i + 1 < len(args):
out_path = Path(args[i + 1])
i += 2
elif args[i] == "--previous" and i + 1 < len(args):
previous_path = Path(args[i + 1])
i += 2
else:
graph_paths.append(Path(args[i]))
i += 1
if len(graph_paths) < 2:
print(
"Usage: graphify merge-graphs <graph1.json> <graph2.json> [...] [--out merged.json]",
"Usage: graphify merge-graphs <graph1.json> <graph2.json> [...] "
"[--out merged.json] [--previous previous-merged-graph.json]",
file=sys.stderr,
)
sys.exit(1)
# Each input numbers its own communities from 0, so the merged output
# carries per-source ids (offset since #3014) that have no relation to
# the previous MERGED clustering .graphify_labels.json was written
# against. cluster-only's label-reuse remap reads this file's
# `community` field as "the previous clustering" and silently mismatches
# almost every community when it is actually per-source ids (#3858).
# --previous restores the real previous merged `community` onto each
# node that survives the merge (by id), so a subsequent cluster-only
# run remaps against the clustering its labels were actually built
# from. Per-source ids stay available in `local_community` regardless.
previous_community: dict[str, int] = {}
if previous_path is not None:
if not previous_path.exists():
print(f"error: --previous file not found: {previous_path}", file=sys.stderr)
sys.exit(1)
_prev_raw = json.loads(previous_path.read_text(encoding="utf-8"))
for _n in _prev_raw.get("nodes", []):
if isinstance(_n, dict) and isinstance(_n.get("community"), int) and _n.get("id"):
previous_community[_n["id"]] = _n["community"]
import networkx as _nx
from networkx.readwrite import json_graph as _jg
from graphify.build import prefix_graph_for_global as _prefix, distinct_repo_tags as _repo_tags
Expand Down Expand Up @@ -2781,6 +2805,18 @@ def _to_simple(g: "_nx.Graph") -> "_nx.Graph":
call_links = _link_calls(merged)
if call_links:
print(f" resolved {call_links} member call(s) across repos")
if previous_path is not None:
_restored = 0
for _nid, _data in merged.nodes(data=True):
if _nid in previous_community:
_data["community"] = previous_community[_nid]
_restored += 1
else:
_data.pop("community", None)
print(
f" restored the previous merged community on {_restored} node(s) "
f"from {previous_path}"
)
# Drop whatever compose left behind (the last input's list, possibly
# with internal duplicates) so attach_hyperedges dedups the full
# collection by id from a clean slate.
Expand Down
61 changes: 61 additions & 0 deletions tests/test_merge_graphs_cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -304,3 +304,64 @@ def test_merge_graphs_community_offset_is_byte_reproducible(tmp_path):
assert _run(["merge-graphs", str(a), str(b), "--out", str(out1)], tmp_path).returncode == 0
assert _run(["merge-graphs", str(a), str(b), "--out", str(out2)], tmp_path).returncode == 0
assert out1.read_bytes() == out2.read_bytes(), "same-order merge is not byte-reproducible"


def test_merge_graphs_previous_restores_the_real_merged_community(tmp_path):
"""#3858: each input numbers its own communities from 0, so a fresh merge's
`community` field is per-source ids with no relation to the previous
MERGED clustering a labels sidecar was built against. --previous restores
the real previous merged community (by node id), not the per-source one,
so a later cluster-only run's label-reuse remap aligns correctly."""
a = tmp_path / "alpha" / "graphify-out" / "graph.json"
b = tmp_path / "beta" / "graphify-out" / "graph.json"
_write_with_communities(a, [("a0", 0), ("a1", 1)])
_write_with_communities(b, [("b0", 0), ("b1", 1)])

# A previous merged output, from before alpha's source gained a node: the
# real merged clustering the labels sidecar was actually written against.
previous = tmp_path / "previous-merged.json"
_write_with_communities(previous, [
("alpha::a0", 7), ("alpha::a1", 7), ("beta::b0", 12), ("beta::b1", 13),
])

out = tmp_path / "merged.json"
r = _run(
["merge-graphs", str(a), str(b), "--out", str(out), "--previous", str(previous)],
tmp_path,
)
assert r.returncode == 0, r.stderr
data = json.loads(out.read_text())
by_id = {n["id"]: n for n in data["nodes"]}
# The real previous merged community wins, not the fresh per-source offset.
assert by_id["alpha::a0"]["community"] == 7
assert by_id["alpha::a1"]["community"] == 7
assert by_id["beta::b0"]["community"] == 12
assert by_id["beta::b1"]["community"] == 13
# The per-source partition stays available for anything that wants it.
assert by_id["beta::b0"]["local_community"] == 0
assert by_id["beta::b1"]["local_community"] == 1


def test_merge_graphs_previous_drops_community_for_a_new_node(tmp_path):
"""A node that did not exist in the previous merge has no real previous
community to restore, and must not keep the fresh per-source offset
either — a stale/invented community id is worse than none (cluster-only
already handles a node with no community)."""
a = tmp_path / "alpha" / "graphify-out" / "graph.json"
b = tmp_path / "beta" / "graphify-out" / "graph.json"
_write_with_communities(a, [("a0", 0), ("a1", 1)])
_write_with_communities(b, [("b0", 0)])

previous = tmp_path / "previous-merged.json"
_write_with_communities(previous, [("alpha::a0", 7), ("alpha::a1", 7)])

out = tmp_path / "merged.json"
r = _run(
["merge-graphs", str(a), str(b), "--out", str(out), "--previous", str(previous)],
tmp_path,
)
assert r.returncode == 0, r.stderr
data = json.loads(out.read_text())
by_id = {n["id"]: n for n in data["nodes"]}
assert by_id["alpha::a0"]["community"] == 7
assert "community" not in by_id["beta::b0"]
Loading