diff --git a/graphify/skill-agents.md b/graphify/skill-agents.md index f09e56ca8e..745d133a53 100644 --- a/graphify/skill-agents.md +++ b/graphify/skill-agents.md @@ -298,12 +298,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill-amp.md b/graphify/skill-amp.md index f09e56ca8e..745d133a53 100644 --- a/graphify/skill-amp.md +++ b/graphify/skill-amp.md @@ -298,12 +298,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill-claw.md b/graphify/skill-claw.md index 612da0090a..6b5c28795e 100644 --- a/graphify/skill-claw.md +++ b/graphify/skill-claw.md @@ -301,12 +301,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill-codex.md b/graphify/skill-codex.md index d826d76e61..4bb279c0ea 100644 --- a/graphify/skill-codex.md +++ b/graphify/skill-codex.md @@ -298,12 +298,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill-copilot.md b/graphify/skill-copilot.md index 612da0090a..6b5c28795e 100644 --- a/graphify/skill-copilot.md +++ b/graphify/skill-copilot.md @@ -301,12 +301,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill-droid.md b/graphify/skill-droid.md index ada369b7dd..98b1c68546 100644 --- a/graphify/skill-droid.md +++ b/graphify/skill-droid.md @@ -298,12 +298,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill-kilo.md b/graphify/skill-kilo.md index 9b043233fa..ec6c178806 100644 --- a/graphify/skill-kilo.md +++ b/graphify/skill-kilo.md @@ -301,12 +301,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill-kiro.md b/graphify/skill-kiro.md index 612da0090a..6b5c28795e 100644 --- a/graphify/skill-kiro.md +++ b/graphify/skill-kiro.md @@ -301,12 +301,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill-opencode.md b/graphify/skill-opencode.md index 5cb74e81ae..d7f3b459eb 100644 --- a/graphify/skill-opencode.md +++ b/graphify/skill-opencode.md @@ -293,12 +293,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill-pi.md b/graphify/skill-pi.md index 612da0090a..6b5c28795e 100644 --- a/graphify/skill-pi.md +++ b/graphify/skill-pi.md @@ -301,12 +301,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill-trae.md b/graphify/skill-trae.md index 2037e539f6..ba4813e845 100644 --- a/graphify/skill-trae.md +++ b/graphify/skill-trae.md @@ -299,12 +299,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill-vscode.md b/graphify/skill-vscode.md index 9002650e4a..423ea55834 100644 --- a/graphify/skill-vscode.md +++ b/graphify/skill-vscode.md @@ -297,12 +297,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill-windows.md b/graphify/skill-windows.md index 1246089d50..626613469c 100644 --- a/graphify/skill-windows.md +++ b/graphify/skill-windows.md @@ -329,12 +329,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal @' import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding="utf-8").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding="utf-8")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skill.md b/graphify/skill.md index 612da0090a..6b5c28795e 100644 --- a/graphify/skill.md +++ b/graphify/skill.md @@ -301,12 +301,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/graphify/skills/agents/references/extraction-spec.md b/graphify/skills/agents/references/extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/graphify/skills/agents/references/extraction-spec.md +++ b/graphify/skills/agents/references/extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/graphify/skills/agents/references/update.md b/graphify/skills/agents/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/agents/references/update.md +++ b/graphify/skills/agents/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/amp/references/extraction-spec.md b/graphify/skills/amp/references/extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/graphify/skills/amp/references/extraction-spec.md +++ b/graphify/skills/amp/references/extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/graphify/skills/amp/references/update.md b/graphify/skills/amp/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/amp/references/update.md +++ b/graphify/skills/amp/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/claude/references/extraction-spec.md b/graphify/skills/claude/references/extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/graphify/skills/claude/references/extraction-spec.md +++ b/graphify/skills/claude/references/extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/graphify/skills/claude/references/update.md b/graphify/skills/claude/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/claude/references/update.md +++ b/graphify/skills/claude/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/claw/references/extraction-spec.md b/graphify/skills/claw/references/extraction-spec.md index 4b278b28d3..d1f284b6a1 100644 --- a/graphify/skills/claw/references/extraction-spec.md +++ b/graphify/skills/claw/references/extraction-spec.md @@ -28,4 +28,7 @@ Output exactly this JSON (no other text): {"nodes":[{"id":"auth_session_validatetoken","label":"Human Readable Name","file_type":"code|document|paper|image|rationale|concept","source_file":"","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to|rationale_for","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":""}],"input_tokens":0,"output_tokens":0} source_file RULE: set source_file to the FILE_LIST path for that file VERBATIM (absolute, no shortening to basename, no re-relativizing, no separator change). Keeps full build and --update on one base so build_merge's replace matches instead of duplicating. +Never create stub nodes for files outside FILE_LIST. Use edge-only references +attributed to the originating FILE_LIST file and the complete known target ID; +a stem or ID prefix is never a valid target. Do not guess or invent a target node. ``` diff --git a/graphify/skills/claw/references/update.md b/graphify/skills/claw/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/claw/references/update.md +++ b/graphify/skills/claw/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/codex/references/extraction-spec.md b/graphify/skills/codex/references/extraction-spec.md index 4b278b28d3..d1f284b6a1 100644 --- a/graphify/skills/codex/references/extraction-spec.md +++ b/graphify/skills/codex/references/extraction-spec.md @@ -28,4 +28,7 @@ Output exactly this JSON (no other text): {"nodes":[{"id":"auth_session_validatetoken","label":"Human Readable Name","file_type":"code|document|paper|image|rationale|concept","source_file":"","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to|rationale_for","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":""}],"input_tokens":0,"output_tokens":0} source_file RULE: set source_file to the FILE_LIST path for that file VERBATIM (absolute, no shortening to basename, no re-relativizing, no separator change). Keeps full build and --update on one base so build_merge's replace matches instead of duplicating. +Never create stub nodes for files outside FILE_LIST. Use edge-only references +attributed to the originating FILE_LIST file and the complete known target ID; +a stem or ID prefix is never a valid target. Do not guess or invent a target node. ``` diff --git a/graphify/skills/codex/references/update.md b/graphify/skills/codex/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/codex/references/update.md +++ b/graphify/skills/codex/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/copilot/references/extraction-spec.md b/graphify/skills/copilot/references/extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/graphify/skills/copilot/references/extraction-spec.md +++ b/graphify/skills/copilot/references/extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/graphify/skills/copilot/references/update.md b/graphify/skills/copilot/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/copilot/references/update.md +++ b/graphify/skills/copilot/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/droid/references/extraction-spec.md b/graphify/skills/droid/references/extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/graphify/skills/droid/references/extraction-spec.md +++ b/graphify/skills/droid/references/extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/graphify/skills/droid/references/update.md b/graphify/skills/droid/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/droid/references/update.md +++ b/graphify/skills/droid/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/kilo/references/extraction-spec.md b/graphify/skills/kilo/references/extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/graphify/skills/kilo/references/extraction-spec.md +++ b/graphify/skills/kilo/references/extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/graphify/skills/kilo/references/update.md b/graphify/skills/kilo/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/kilo/references/update.md +++ b/graphify/skills/kilo/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/kiro/references/extraction-spec.md b/graphify/skills/kiro/references/extraction-spec.md index 4b278b28d3..d1f284b6a1 100644 --- a/graphify/skills/kiro/references/extraction-spec.md +++ b/graphify/skills/kiro/references/extraction-spec.md @@ -28,4 +28,7 @@ Output exactly this JSON (no other text): {"nodes":[{"id":"auth_session_validatetoken","label":"Human Readable Name","file_type":"code|document|paper|image|rationale|concept","source_file":"","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to|rationale_for","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":""}],"input_tokens":0,"output_tokens":0} source_file RULE: set source_file to the FILE_LIST path for that file VERBATIM (absolute, no shortening to basename, no re-relativizing, no separator change). Keeps full build and --update on one base so build_merge's replace matches instead of duplicating. +Never create stub nodes for files outside FILE_LIST. Use edge-only references +attributed to the originating FILE_LIST file and the complete known target ID; +a stem or ID prefix is never a valid target. Do not guess or invent a target node. ``` diff --git a/graphify/skills/kiro/references/update.md b/graphify/skills/kiro/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/kiro/references/update.md +++ b/graphify/skills/kiro/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/opencode/references/extraction-spec.md b/graphify/skills/opencode/references/extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/graphify/skills/opencode/references/extraction-spec.md +++ b/graphify/skills/opencode/references/extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/graphify/skills/opencode/references/update.md b/graphify/skills/opencode/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/opencode/references/update.md +++ b/graphify/skills/opencode/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/pi/references/extraction-spec.md b/graphify/skills/pi/references/extraction-spec.md index 4b278b28d3..d1f284b6a1 100644 --- a/graphify/skills/pi/references/extraction-spec.md +++ b/graphify/skills/pi/references/extraction-spec.md @@ -28,4 +28,7 @@ Output exactly this JSON (no other text): {"nodes":[{"id":"auth_session_validatetoken","label":"Human Readable Name","file_type":"code|document|paper|image|rationale|concept","source_file":"","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to|rationale_for","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":""}],"input_tokens":0,"output_tokens":0} source_file RULE: set source_file to the FILE_LIST path for that file VERBATIM (absolute, no shortening to basename, no re-relativizing, no separator change). Keeps full build and --update on one base so build_merge's replace matches instead of duplicating. +Never create stub nodes for files outside FILE_LIST. Use edge-only references +attributed to the originating FILE_LIST file and the complete known target ID; +a stem or ID prefix is never a valid target. Do not guess or invent a target node. ``` diff --git a/graphify/skills/pi/references/update.md b/graphify/skills/pi/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/pi/references/update.md +++ b/graphify/skills/pi/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/trae/references/extraction-spec.md b/graphify/skills/trae/references/extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/graphify/skills/trae/references/extraction-spec.md +++ b/graphify/skills/trae/references/extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/graphify/skills/trae/references/update.md b/graphify/skills/trae/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/trae/references/update.md +++ b/graphify/skills/trae/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/vscode/references/extraction-spec.md b/graphify/skills/vscode/references/extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/graphify/skills/vscode/references/extraction-spec.md +++ b/graphify/skills/vscode/references/extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/graphify/skills/vscode/references/update.md b/graphify/skills/vscode/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/vscode/references/update.md +++ b/graphify/skills/vscode/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/graphify/skills/windows/references/extraction-spec.md b/graphify/skills/windows/references/extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/graphify/skills/windows/references/extraction-spec.md +++ b/graphify/skills/windows/references/extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/graphify/skills/windows/references/update.md b/graphify/skills/windows/references/update.md index 3632fd4126..2817ad2c1e 100644 --- a/graphify/skills/windows/references/update.md +++ b/graphify/skills/windows/references/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tests/test_skill_semantic_scope.py b/tests/test_skill_semantic_scope.py new file mode 100644 index 0000000000..ccf6ed2380 --- /dev/null +++ b/tests/test_skill_semantic_scope.py @@ -0,0 +1,107 @@ +"""The generated skill must scope fresh chunks before destructive update merges.""" +from __future__ import annotations + +import json +import os +import re +import subprocess +import sys +from pathlib import Path + +import pytest + +from graphify.build import build_from_json +from graphify.export import to_json +from tools.skillgen import gen + +REPO_ROOT = Path(__file__).resolve().parent.parent + + +def _python_blocks(body: str, start: str, end: str) -> list[str]: + section = body.split(start, 1)[1].split(end, 1)[0] + sources = [] + for block in re.findall(r"```bash\n(.*?)```", section, re.S): + match = re.search(r' -c "\n(.*)\n"\s*$', block, re.S) + if match: + sources.append(match.group(1).replace('\\"', '"')) + return sources + + +@pytest.mark.parametrize("source_form", ["relative", "absolute"]) +def test_generated_update_retains_undispatched_rich_nodes(tmp_path, source_form): + """A growing graph must not hide replacement of rich foreign nodes by a stub.""" + artifacts = gen.render_all(gen.load_platforms(), only="claude") + body = next(a.content for a in artifacts if a.path == "graphify/skill.md") + update = next(a.content for a in artifacts + if a.path == "graphify/skills/claude/references/update.md") + b3 = _python_blocks(body, "**Step B3", "#### Part C") + part_c = _python_blocks(body, "#### Part C", "\n### ")[:1] + merge = next(src for src in _python_blocks(update, "Then:", "Then run Steps") + if "G = build_merge(" in src) + + corpus = tmp_path / "corpus" + corpus.mkdir() + for filename in ("fresh.md", "foreign.md", "cached.md"): + (corpus / filename).write_text(f"# {filename}\n", encoding="utf-8") + spec = tmp_path / "spec.md" + spec.write_text("# Spec\n", encoding="utf-8") + output = tmp_path / "graphify-out" + output.mkdir() + + def node(nid, filename): + source = str(corpus / filename) if source_form == "absolute" else filename + return {"id": nid, "label": nid, "file_type": "document", "source_file": source} + + rich = [node(f"foreign_rich_{i}", "foreign.md") for i in range(23)] + cached = node("cached_real", "cached.md") + prior = build_from_json({"nodes": rich + [cached], "edges": []}, root=corpus) + assert to_json(prior, {}, output / "graph.json") + fresh = [node(f"fresh_{i}", "fresh.md") for i in range(35)] + payload = { + "nodes": fresh + [node("foreign_stub", "foreign.md")], + "edges": [ + {"source": "fresh_0", "target": "foreign_rich_0", "relation": "references", + "source_file": fresh[0]["source_file"], "confidence": "EXTRACTED"}, + {"source": "fresh_0", "target": "foreign_stub", "relation": "references", + "source_file": fresh[0]["source_file"], "confidence": "INFERRED"}, + ], + "hyperedges": [{"id": "foreign_h", "nodes": ["fresh_0", "fresh_1"], + "source_file": node("unused", "foreign.md")["source_file"]}], + "input_tokens": 10, "output_tokens": 5, + } + files = {"document": [str(corpus / name) + for name in ("fresh.md", "foreign.md", "cached.md")]} + sidecars = { + ".graphify_chunk_01.json": payload, + ".graphify_cached.json": {"nodes": [cached], "edges": [], "hyperedges": []}, + ".graphify_ast.json": {"nodes": [], "edges": []}, + ".graphify_incremental.json": {"files": files, "deleted_files": [], + "new_files": {"document": [str(corpus / "fresh.md")]}}, + } + for filename, data in sidecars.items(): + (output / filename).write_text(json.dumps(data), encoding="utf-8") + (output / ".graphify_uncached.txt").write_text( + str(corpus / "fresh.md") + "\n", encoding="utf-8") + logs = [] + for source in b3 + part_c + [merge]: + source = source.replace("INPUT_PATH", corpus.as_posix()).replace( + "SPEC_PATH", spec.as_posix()).replace("IS_DIRECTED", "False") + result = subprocess.run( + [sys.executable, "-c", source], cwd=tmp_path, check=True, + capture_output=True, text=True, + env={**os.environ, "PYTHONPATH": str(REPO_ROOT)}, + ) + logs.append(result.stdout) + + merged = json.loads((output / ".graphify_extract.json").read_text(encoding="utf-8")) + assert {n["id"] for n in merged["nodes"]} == { + n["id"] for n in rich + [cached] + fresh + } + assert {(e["source"], e["target"]) for e in merged["edges"]} == { + ("fresh_0", "foreign_rich_0") + } + assert merged["hyperedges"] == [] + fresh_result = json.loads( + (output / ".graphify_semantic_new.json").read_text(encoding="utf-8")) + assert {n["id"] for n in fresh_result["nodes"]} == {n["id"] for n in fresh} + assert "foreign.md" in "".join(logs) and "out-of-scope" in "".join(logs) diff --git a/tools/skillgen/expected/graphify__skill-agents.md b/tools/skillgen/expected/graphify__skill-agents.md index f09e56ca8e..745d133a53 100644 --- a/tools/skillgen/expected/graphify__skill-agents.md +++ b/tools/skillgen/expected/graphify__skill-agents.md @@ -298,12 +298,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill-amp.md b/tools/skillgen/expected/graphify__skill-amp.md index f09e56ca8e..745d133a53 100644 --- a/tools/skillgen/expected/graphify__skill-amp.md +++ b/tools/skillgen/expected/graphify__skill-amp.md @@ -298,12 +298,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill-claw.md b/tools/skillgen/expected/graphify__skill-claw.md index 612da0090a..6b5c28795e 100644 --- a/tools/skillgen/expected/graphify__skill-claw.md +++ b/tools/skillgen/expected/graphify__skill-claw.md @@ -301,12 +301,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill-codex.md b/tools/skillgen/expected/graphify__skill-codex.md index d826d76e61..4bb279c0ea 100644 --- a/tools/skillgen/expected/graphify__skill-codex.md +++ b/tools/skillgen/expected/graphify__skill-codex.md @@ -298,12 +298,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill-copilot.md b/tools/skillgen/expected/graphify__skill-copilot.md index 612da0090a..6b5c28795e 100644 --- a/tools/skillgen/expected/graphify__skill-copilot.md +++ b/tools/skillgen/expected/graphify__skill-copilot.md @@ -301,12 +301,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill-droid.md b/tools/skillgen/expected/graphify__skill-droid.md index ada369b7dd..98b1c68546 100644 --- a/tools/skillgen/expected/graphify__skill-droid.md +++ b/tools/skillgen/expected/graphify__skill-droid.md @@ -298,12 +298,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill-kilo.md b/tools/skillgen/expected/graphify__skill-kilo.md index 9b043233fa..ec6c178806 100644 --- a/tools/skillgen/expected/graphify__skill-kilo.md +++ b/tools/skillgen/expected/graphify__skill-kilo.md @@ -301,12 +301,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill-kiro.md b/tools/skillgen/expected/graphify__skill-kiro.md index 612da0090a..6b5c28795e 100644 --- a/tools/skillgen/expected/graphify__skill-kiro.md +++ b/tools/skillgen/expected/graphify__skill-kiro.md @@ -301,12 +301,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill-opencode.md b/tools/skillgen/expected/graphify__skill-opencode.md index 5cb74e81ae..d7f3b459eb 100644 --- a/tools/skillgen/expected/graphify__skill-opencode.md +++ b/tools/skillgen/expected/graphify__skill-opencode.md @@ -293,12 +293,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill-pi.md b/tools/skillgen/expected/graphify__skill-pi.md index 612da0090a..6b5c28795e 100644 --- a/tools/skillgen/expected/graphify__skill-pi.md +++ b/tools/skillgen/expected/graphify__skill-pi.md @@ -301,12 +301,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill-trae.md b/tools/skillgen/expected/graphify__skill-trae.md index 2037e539f6..ba4813e845 100644 --- a/tools/skillgen/expected/graphify__skill-trae.md +++ b/tools/skillgen/expected/graphify__skill-trae.md @@ -299,12 +299,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill-vscode.md b/tools/skillgen/expected/graphify__skill-vscode.md index 9002650e4a..423ea55834 100644 --- a/tools/skillgen/expected/graphify__skill-vscode.md +++ b/tools/skillgen/expected/graphify__skill-vscode.md @@ -297,12 +297,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill-windows.md b/tools/skillgen/expected/graphify__skill-windows.md index 1246089d50..626613469c 100644 --- a/tools/skillgen/expected/graphify__skill-windows.md +++ b/tools/skillgen/expected/graphify__skill-windows.md @@ -329,12 +329,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal @' import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding="utf-8").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding="utf-8")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skill.md b/tools/skillgen/expected/graphify__skill.md index 612da0090a..6b5c28795e 100644 --- a/tools/skillgen/expected/graphify__skill.md +++ b/tools/skillgen/expected/graphify__skill.md @@ -301,12 +301,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/expected/graphify__skills__agents__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__agents__references__extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/tools/skillgen/expected/graphify__skills__agents__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__agents__references__extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/tools/skillgen/expected/graphify__skills__agents__references__update.md b/tools/skillgen/expected/graphify__skills__agents__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__agents__references__update.md +++ b/tools/skillgen/expected/graphify__skills__agents__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__amp__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__amp__references__extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/tools/skillgen/expected/graphify__skills__amp__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__amp__references__extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/tools/skillgen/expected/graphify__skills__amp__references__update.md b/tools/skillgen/expected/graphify__skills__amp__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__amp__references__update.md +++ b/tools/skillgen/expected/graphify__skills__amp__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__claude__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__claude__references__extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/tools/skillgen/expected/graphify__skills__claude__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__claude__references__extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/tools/skillgen/expected/graphify__skills__claude__references__update.md b/tools/skillgen/expected/graphify__skills__claude__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__claude__references__update.md +++ b/tools/skillgen/expected/graphify__skills__claude__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__claw__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__claw__references__extraction-spec.md index 4b278b28d3..d1f284b6a1 100644 --- a/tools/skillgen/expected/graphify__skills__claw__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__claw__references__extraction-spec.md @@ -28,4 +28,7 @@ Output exactly this JSON (no other text): {"nodes":[{"id":"auth_session_validatetoken","label":"Human Readable Name","file_type":"code|document|paper|image|rationale|concept","source_file":"","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to|rationale_for","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":""}],"input_tokens":0,"output_tokens":0} source_file RULE: set source_file to the FILE_LIST path for that file VERBATIM (absolute, no shortening to basename, no re-relativizing, no separator change). Keeps full build and --update on one base so build_merge's replace matches instead of duplicating. +Never create stub nodes for files outside FILE_LIST. Use edge-only references +attributed to the originating FILE_LIST file and the complete known target ID; +a stem or ID prefix is never a valid target. Do not guess or invent a target node. ``` diff --git a/tools/skillgen/expected/graphify__skills__claw__references__update.md b/tools/skillgen/expected/graphify__skills__claw__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__claw__references__update.md +++ b/tools/skillgen/expected/graphify__skills__claw__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__codex__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__codex__references__extraction-spec.md index 4b278b28d3..d1f284b6a1 100644 --- a/tools/skillgen/expected/graphify__skills__codex__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__codex__references__extraction-spec.md @@ -28,4 +28,7 @@ Output exactly this JSON (no other text): {"nodes":[{"id":"auth_session_validatetoken","label":"Human Readable Name","file_type":"code|document|paper|image|rationale|concept","source_file":"","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to|rationale_for","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":""}],"input_tokens":0,"output_tokens":0} source_file RULE: set source_file to the FILE_LIST path for that file VERBATIM (absolute, no shortening to basename, no re-relativizing, no separator change). Keeps full build and --update on one base so build_merge's replace matches instead of duplicating. +Never create stub nodes for files outside FILE_LIST. Use edge-only references +attributed to the originating FILE_LIST file and the complete known target ID; +a stem or ID prefix is never a valid target. Do not guess or invent a target node. ``` diff --git a/tools/skillgen/expected/graphify__skills__codex__references__update.md b/tools/skillgen/expected/graphify__skills__codex__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__codex__references__update.md +++ b/tools/skillgen/expected/graphify__skills__codex__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__copilot__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__copilot__references__extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/tools/skillgen/expected/graphify__skills__copilot__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__copilot__references__extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/tools/skillgen/expected/graphify__skills__copilot__references__update.md b/tools/skillgen/expected/graphify__skills__copilot__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__copilot__references__update.md +++ b/tools/skillgen/expected/graphify__skills__copilot__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__droid__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__droid__references__extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/tools/skillgen/expected/graphify__skills__droid__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__droid__references__extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/tools/skillgen/expected/graphify__skills__droid__references__update.md b/tools/skillgen/expected/graphify__skills__droid__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__droid__references__update.md +++ b/tools/skillgen/expected/graphify__skills__droid__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__kilo__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__kilo__references__extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/tools/skillgen/expected/graphify__skills__kilo__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__kilo__references__extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/tools/skillgen/expected/graphify__skills__kilo__references__update.md b/tools/skillgen/expected/graphify__skills__kilo__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__kilo__references__update.md +++ b/tools/skillgen/expected/graphify__skills__kilo__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__kiro__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__kiro__references__extraction-spec.md index 4b278b28d3..d1f284b6a1 100644 --- a/tools/skillgen/expected/graphify__skills__kiro__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__kiro__references__extraction-spec.md @@ -28,4 +28,7 @@ Output exactly this JSON (no other text): {"nodes":[{"id":"auth_session_validatetoken","label":"Human Readable Name","file_type":"code|document|paper|image|rationale|concept","source_file":"","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to|rationale_for","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":""}],"input_tokens":0,"output_tokens":0} source_file RULE: set source_file to the FILE_LIST path for that file VERBATIM (absolute, no shortening to basename, no re-relativizing, no separator change). Keeps full build and --update on one base so build_merge's replace matches instead of duplicating. +Never create stub nodes for files outside FILE_LIST. Use edge-only references +attributed to the originating FILE_LIST file and the complete known target ID; +a stem or ID prefix is never a valid target. Do not guess or invent a target node. ``` diff --git a/tools/skillgen/expected/graphify__skills__kiro__references__update.md b/tools/skillgen/expected/graphify__skills__kiro__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__kiro__references__update.md +++ b/tools/skillgen/expected/graphify__skills__kiro__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__opencode__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__opencode__references__extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/tools/skillgen/expected/graphify__skills__opencode__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__opencode__references__extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/tools/skillgen/expected/graphify__skills__opencode__references__update.md b/tools/skillgen/expected/graphify__skills__opencode__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__opencode__references__update.md +++ b/tools/skillgen/expected/graphify__skills__opencode__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__pi__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__pi__references__extraction-spec.md index 4b278b28d3..d1f284b6a1 100644 --- a/tools/skillgen/expected/graphify__skills__pi__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__pi__references__extraction-spec.md @@ -28,4 +28,7 @@ Output exactly this JSON (no other text): {"nodes":[{"id":"auth_session_validatetoken","label":"Human Readable Name","file_type":"code|document|paper|image|rationale|concept","source_file":"","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to|rationale_for","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":""}],"input_tokens":0,"output_tokens":0} source_file RULE: set source_file to the FILE_LIST path for that file VERBATIM (absolute, no shortening to basename, no re-relativizing, no separator change). Keeps full build and --update on one base so build_merge's replace matches instead of duplicating. +Never create stub nodes for files outside FILE_LIST. Use edge-only references +attributed to the originating FILE_LIST file and the complete known target ID; +a stem or ID prefix is never a valid target. Do not guess or invent a target node. ``` diff --git a/tools/skillgen/expected/graphify__skills__pi__references__update.md b/tools/skillgen/expected/graphify__skills__pi__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__pi__references__update.md +++ b/tools/skillgen/expected/graphify__skills__pi__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__trae__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__trae__references__extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/tools/skillgen/expected/graphify__skills__trae__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__trae__references__extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/tools/skillgen/expected/graphify__skills__trae__references__update.md b/tools/skillgen/expected/graphify__skills__trae__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__trae__references__update.md +++ b/tools/skillgen/expected/graphify__skills__trae__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__vscode__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__vscode__references__extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/tools/skillgen/expected/graphify__skills__vscode__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__vscode__references__extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/tools/skillgen/expected/graphify__skills__vscode__references__update.md b/tools/skillgen/expected/graphify__skills__vscode__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__vscode__references__update.md +++ b/tools/skillgen/expected/graphify__skills__vscode__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/expected/graphify__skills__windows__references__extraction-spec.md b/tools/skillgen/expected/graphify__skills__windows__references__extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/tools/skillgen/expected/graphify__skills__windows__references__extraction-spec.md +++ b/tools/skillgen/expected/graphify__skills__windows__references__extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/tools/skillgen/expected/graphify__skills__windows__references__update.md b/tools/skillgen/expected/graphify__skills__windows__references__update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/expected/graphify__skills__windows__references__update.md +++ b/tools/skillgen/expected/graphify__skills__windows__references__update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json diff --git a/tools/skillgen/fragments/core/core.md b/tools/skillgen/fragments/core/core.md index 412535088c..7ef91b1643 100644 --- a/tools/skillgen/fragments/core/core.md +++ b/tools/skillgen/fragments/core/core.md @@ -230,12 +230,19 @@ Merge all chunk files into `.graphify_semantic_new.json`. **After each Agent cal $(cat graphify-out/.graphify_python) -c " import json, glob from pathlib import Path +from graphify.cache import scope_semantic_result chunks = sorted(glob.glob('graphify-out/.graphify_chunk_*.json')) +uncached = [line for line in Path('graphify-out/.graphify_uncached.txt').read_text(encoding=\"utf-8\").splitlines() if line] all_nodes, all_edges, all_hyperedges = [], [], [] total_in, total_out = 0, 0 for c in chunks: d = json.loads(Path(c).read_text(encoding=\"utf-8\")) + # source_file is a destructive replacement key during --update. Scope the + # fresh result BEFORE both cache writes and assembly with cached/AST nodes. + dropped_files, dropped_count = scope_semantic_result(d, root=Path('INPUT_PATH'), allowed_source_files=uncached) + if dropped_count: + print(f'[graphify scope] {c}: dropped {dropped_count} out-of-scope items from {sorted(dropped_files)}') all_nodes += d.get('nodes', []) all_edges += d.get('edges', []) all_hyperedges += d.get('hyperedges', []) diff --git a/tools/skillgen/fragments/references/shared/extraction-spec-compact.md b/tools/skillgen/fragments/references/shared/extraction-spec-compact.md index 4b278b28d3..d1f284b6a1 100644 --- a/tools/skillgen/fragments/references/shared/extraction-spec-compact.md +++ b/tools/skillgen/fragments/references/shared/extraction-spec-compact.md @@ -28,4 +28,7 @@ Output exactly this JSON (no other text): {"nodes":[{"id":"auth_session_validatetoken","label":"Human Readable Name","file_type":"code|document|paper|image|rationale|concept","source_file":"","source_location":null,"source_url":null,"captured_at":null,"author":null,"contributor":null}],"edges":[{"source":"node_id","target":"node_id","relation":"calls|implements|references|cites|conceptually_related_to|shares_data_with|semantically_similar_to|rationale_for","confidence":"EXTRACTED|INFERRED|AMBIGUOUS","confidence_score":1.0,"source_file":"","source_location":null,"weight":1.0}],"hyperedges":[{"id":"snake_case_id","label":"Human Readable Label","nodes":["node_id1","node_id2","node_id3"],"relation":"participate_in|implement|form","confidence":"EXTRACTED|INFERRED","confidence_score":0.75,"source_file":""}],"input_tokens":0,"output_tokens":0} source_file RULE: set source_file to the FILE_LIST path for that file VERBATIM (absolute, no shortening to basename, no re-relativizing, no separator change). Keeps full build and --update on one base so build_merge's replace matches instead of duplicating. +Never create stub nodes for files outside FILE_LIST. Use edge-only references +attributed to the originating FILE_LIST file and the complete known target ID; +a stem or ID prefix is never a valid target. Do not guess or invent a target node. ``` diff --git a/tools/skillgen/fragments/references/shared/extraction-spec.md b/tools/skillgen/fragments/references/shared/extraction-spec.md index 388df7674f..f0f6a341b6 100644 --- a/tools/skillgen/fragments/references/shared/extraction-spec.md +++ b/tools/skillgen/fragments/references/shared/extraction-spec.md @@ -65,6 +65,11 @@ Generate the extraction JSON matching this schema exactly: source_file RULE (every node, edge, and hyperedge): set source_file to the path of the originating file EXACTLY as it appears in FILE_LIST — verbatim and absolute. Do NOT shorten to a basename, do NOT re-relativize, do NOT strip any directory prefix, and do NOT change separators (the engine canonicalizes separators and relativizes against the build root downstream). Copy the FILE_LIST entry character-for-character. This keeps the full build and incremental --update on the same base, so build_merge's replace-on-re-extract matches the existing node instead of accumulating a duplicate. +Never create a stub node for a file outside FILE_LIST. Express a reference to an +external entity as an edge attributed to the originating FILE_LIST file, using +the complete known target ID. A stem or ID prefix alone is never a valid target +ID; do not guess one or invent a node to make a reference resolve. + Then write the JSON to disk using the Write tool at this exact absolute path (no relative paths — Write resolves relative paths against an undefined cwd and the file will be silently lost): CHUNK_PATH ``` diff --git a/tools/skillgen/fragments/references/shared/update.md b/tools/skillgen/fragments/references/shared/update.md index 3632fd4126..2817ad2c1e 100644 --- a/tools/skillgen/fragments/references/shared/update.md +++ b/tools/skillgen/fragments/references/shared/update.md @@ -82,6 +82,12 @@ fi Then: +For semantic changes, Step B3's collection commands apply the existing +`scope_semantic_result` guard to fresh semantic chunks using the files dispatched +from `.graphify_uncached.txt`, before caching or mixing them with cached/AST +results. `source_file` is a destructive replacement key: an undispatched file's +stub must never reach `build_merge` and replace that file's existing contribution. + ```bash $(cat graphify-out/.graphify_python) -c " import json