graphitect 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (336) hide show
  1. graphify/__init__.py +30 -0
  2. graphify/__main__.py +757 -0
  3. graphify/_minhash.py +107 -0
  4. graphify/affected.py +318 -0
  5. graphify/always_on/agents-md.md +12 -0
  6. graphify/always_on/antigravity-rules.md +14 -0
  7. graphify/always_on/claude-md.md +9 -0
  8. graphify/always_on/gemini-md.md +9 -0
  9. graphify/always_on/kiro-steering.md +5 -0
  10. graphify/always_on/vscode-instructions.md +17 -0
  11. graphify/analyze.py +769 -0
  12. graphify/benchmark.py +152 -0
  13. graphify/build.py +2300 -0
  14. graphify/cache.py +1746 -0
  15. graphify/callflow_html.py +2051 -0
  16. graphify/cargo_introspect.py +109 -0
  17. graphify/cli.py +4745 -0
  18. graphify/cluster.py +409 -0
  19. graphify/command-kilo.md +15 -0
  20. graphify/cross_repo_calls.py +216 -0
  21. graphify/cross_repo_types.py +75 -0
  22. graphify/csharp_dispatch.py +154 -0
  23. graphify/dedup.py +1213 -0
  24. graphify/detect.py +2566 -0
  25. graphify/diagnostics.py +406 -0
  26. graphify/export.py +1349 -0
  27. graphify/exporters/__init__.py +1 -0
  28. graphify/exporters/base.py +14 -0
  29. graphify/exporters/graphdb.py +173 -0
  30. graphify/exporters/html.py +637 -0
  31. graphify/extract.py +7856 -0
  32. graphify/extractors/MIGRATION.md +107 -0
  33. graphify/extractors/__init__.py +66 -0
  34. graphify/extractors/apex.py +215 -0
  35. graphify/extractors/base.py +85 -0
  36. graphify/extractors/bash.py +579 -0
  37. graphify/extractors/blade.py +53 -0
  38. graphify/extractors/commonlisp.py +540 -0
  39. graphify/extractors/csharp.py +448 -0
  40. graphify/extractors/dart.py +564 -0
  41. graphify/extractors/dm.py +494 -0
  42. graphify/extractors/elixir.py +241 -0
  43. graphify/extractors/engine.py +6509 -0
  44. graphify/extractors/fortran.py +311 -0
  45. graphify/extractors/go.py +527 -0
  46. graphify/extractors/json_config.py +240 -0
  47. graphify/extractors/julia.py +289 -0
  48. graphify/extractors/markdown.py +408 -0
  49. graphify/extractors/models.py +131 -0
  50. graphify/extractors/objc.py +566 -0
  51. graphify/extractors/ocaml.py +289 -0
  52. graphify/extractors/pascal.py +688 -0
  53. graphify/extractors/pascal_forms.py +196 -0
  54. graphify/extractors/powershell.py +522 -0
  55. graphify/extractors/razor.py +192 -0
  56. graphify/extractors/resolution.py +3584 -0
  57. graphify/extractors/robot.py +296 -0
  58. graphify/extractors/rust.py +470 -0
  59. graphify/extractors/sln.py +92 -0
  60. graphify/extractors/sql.py +720 -0
  61. graphify/extractors/terraform.py +181 -0
  62. graphify/extractors/verilog.py +329 -0
  63. graphify/extractors/zig.py +181 -0
  64. graphify/file_slice.py +246 -0
  65. graphify/global_graph.py +194 -0
  66. graphify/google_workspace.py +237 -0
  67. graphify/hooks.py +933 -0
  68. graphify/ids.py +93 -0
  69. graphify/ingest.py +358 -0
  70. graphify/install.py +2366 -0
  71. graphify/llm.py +3544 -0
  72. graphify/manifest.py +4 -0
  73. graphify/manifest_ingest.py +311 -0
  74. graphify/mcp_ingest.py +386 -0
  75. graphify/multigraph_compat.py +212 -0
  76. graphify/pascal_resolution.py +129 -0
  77. graphify/paths.py +436 -0
  78. graphify/pg_introspect.py +165 -0
  79. graphify/prs.py +770 -0
  80. graphify/querylog.py +80 -0
  81. graphify/reflect.py +882 -0
  82. graphify/report.py +346 -0
  83. graphify/resolver_registry.py +85 -0
  84. graphify/ruby_resolution.py +242 -0
  85. graphify/scip_ingest.py +363 -0
  86. graphify/security.py +460 -0
  87. graphify/semantic_cleanup.py +336 -0
  88. graphify/serve.py +2608 -0
  89. graphify/skill-agents.md +710 -0
  90. graphify/skill-aider.md +1283 -0
  91. graphify/skill-amp.md +710 -0
  92. graphify/skill-claw.md +713 -0
  93. graphify/skill-codex.md +710 -0
  94. graphify/skill-copilot.md +713 -0
  95. graphify/skill-devin.md +1410 -0
  96. graphify/skill-droid.md +710 -0
  97. graphify/skill-kilo.md +722 -0
  98. graphify/skill-kiro.md +713 -0
  99. graphify/skill-opencode.md +705 -0
  100. graphify/skill-pi.md +713 -0
  101. graphify/skill-trae.md +711 -0
  102. graphify/skill-vscode.md +709 -0
  103. graphify/skill-windows.md +755 -0
  104. graphify/skill.md +713 -0
  105. graphify/skills/agents/references/add-watch.md +56 -0
  106. graphify/skills/agents/references/exports.md +87 -0
  107. graphify/skills/agents/references/extraction-spec.md +70 -0
  108. graphify/skills/agents/references/github-and-merge.md +46 -0
  109. graphify/skills/agents/references/hooks.md +33 -0
  110. graphify/skills/agents/references/query.md +311 -0
  111. graphify/skills/agents/references/transcribe.md +52 -0
  112. graphify/skills/agents/references/update.md +210 -0
  113. graphify/skills/amp/references/add-watch.md +56 -0
  114. graphify/skills/amp/references/exports.md +87 -0
  115. graphify/skills/amp/references/extraction-spec.md +70 -0
  116. graphify/skills/amp/references/github-and-merge.md +46 -0
  117. graphify/skills/amp/references/hooks.md +33 -0
  118. graphify/skills/amp/references/query.md +311 -0
  119. graphify/skills/amp/references/transcribe.md +52 -0
  120. graphify/skills/amp/references/update.md +210 -0
  121. graphify/skills/claude/references/add-watch.md +56 -0
  122. graphify/skills/claude/references/exports.md +87 -0
  123. graphify/skills/claude/references/extraction-spec.md +70 -0
  124. graphify/skills/claude/references/github-and-merge.md +46 -0
  125. graphify/skills/claude/references/hooks.md +33 -0
  126. graphify/skills/claude/references/query.md +311 -0
  127. graphify/skills/claude/references/transcribe.md +52 -0
  128. graphify/skills/claude/references/update.md +210 -0
  129. graphify/skills/claw/references/add-watch.md +56 -0
  130. graphify/skills/claw/references/exports.md +87 -0
  131. graphify/skills/claw/references/extraction-spec.md +31 -0
  132. graphify/skills/claw/references/github-and-merge.md +46 -0
  133. graphify/skills/claw/references/hooks.md +33 -0
  134. graphify/skills/claw/references/query.md +311 -0
  135. graphify/skills/claw/references/transcribe.md +52 -0
  136. graphify/skills/claw/references/update.md +210 -0
  137. graphify/skills/codex/references/add-watch.md +56 -0
  138. graphify/skills/codex/references/exports.md +87 -0
  139. graphify/skills/codex/references/extraction-spec.md +31 -0
  140. graphify/skills/codex/references/github-and-merge.md +46 -0
  141. graphify/skills/codex/references/hooks.md +33 -0
  142. graphify/skills/codex/references/query.md +311 -0
  143. graphify/skills/codex/references/transcribe.md +52 -0
  144. graphify/skills/codex/references/update.md +210 -0
  145. graphify/skills/copilot/references/add-watch.md +56 -0
  146. graphify/skills/copilot/references/exports.md +87 -0
  147. graphify/skills/copilot/references/extraction-spec.md +70 -0
  148. graphify/skills/copilot/references/github-and-merge.md +46 -0
  149. graphify/skills/copilot/references/hooks.md +33 -0
  150. graphify/skills/copilot/references/query.md +311 -0
  151. graphify/skills/copilot/references/transcribe.md +52 -0
  152. graphify/skills/copilot/references/update.md +210 -0
  153. graphify/skills/droid/references/add-watch.md +56 -0
  154. graphify/skills/droid/references/exports.md +87 -0
  155. graphify/skills/droid/references/extraction-spec.md +70 -0
  156. graphify/skills/droid/references/github-and-merge.md +46 -0
  157. graphify/skills/droid/references/hooks.md +33 -0
  158. graphify/skills/droid/references/query.md +311 -0
  159. graphify/skills/droid/references/transcribe.md +52 -0
  160. graphify/skills/droid/references/update.md +210 -0
  161. graphify/skills/kilo/references/add-watch.md +56 -0
  162. graphify/skills/kilo/references/exports.md +87 -0
  163. graphify/skills/kilo/references/extraction-spec.md +70 -0
  164. graphify/skills/kilo/references/github-and-merge.md +46 -0
  165. graphify/skills/kilo/references/hooks.md +33 -0
  166. graphify/skills/kilo/references/query.md +311 -0
  167. graphify/skills/kilo/references/transcribe.md +52 -0
  168. graphify/skills/kilo/references/update.md +210 -0
  169. graphify/skills/kiro/references/add-watch.md +56 -0
  170. graphify/skills/kiro/references/exports.md +87 -0
  171. graphify/skills/kiro/references/extraction-spec.md +31 -0
  172. graphify/skills/kiro/references/github-and-merge.md +46 -0
  173. graphify/skills/kiro/references/hooks.md +33 -0
  174. graphify/skills/kiro/references/query.md +311 -0
  175. graphify/skills/kiro/references/transcribe.md +52 -0
  176. graphify/skills/kiro/references/update.md +210 -0
  177. graphify/skills/opencode/references/add-watch.md +56 -0
  178. graphify/skills/opencode/references/exports.md +87 -0
  179. graphify/skills/opencode/references/extraction-spec.md +70 -0
  180. graphify/skills/opencode/references/github-and-merge.md +46 -0
  181. graphify/skills/opencode/references/hooks.md +33 -0
  182. graphify/skills/opencode/references/query.md +311 -0
  183. graphify/skills/opencode/references/transcribe.md +52 -0
  184. graphify/skills/opencode/references/update.md +210 -0
  185. graphify/skills/pi/references/add-watch.md +56 -0
  186. graphify/skills/pi/references/exports.md +87 -0
  187. graphify/skills/pi/references/extraction-spec.md +31 -0
  188. graphify/skills/pi/references/github-and-merge.md +46 -0
  189. graphify/skills/pi/references/hooks.md +33 -0
  190. graphify/skills/pi/references/query.md +311 -0
  191. graphify/skills/pi/references/transcribe.md +52 -0
  192. graphify/skills/pi/references/update.md +210 -0
  193. graphify/skills/trae/references/add-watch.md +56 -0
  194. graphify/skills/trae/references/exports.md +87 -0
  195. graphify/skills/trae/references/extraction-spec.md +70 -0
  196. graphify/skills/trae/references/github-and-merge.md +46 -0
  197. graphify/skills/trae/references/hooks.md +35 -0
  198. graphify/skills/trae/references/query.md +311 -0
  199. graphify/skills/trae/references/transcribe.md +52 -0
  200. graphify/skills/trae/references/update.md +210 -0
  201. graphify/skills/vscode/references/add-watch.md +56 -0
  202. graphify/skills/vscode/references/exports.md +87 -0
  203. graphify/skills/vscode/references/extraction-spec.md +70 -0
  204. graphify/skills/vscode/references/github-and-merge.md +46 -0
  205. graphify/skills/vscode/references/hooks.md +33 -0
  206. graphify/skills/vscode/references/query.md +311 -0
  207. graphify/skills/vscode/references/transcribe.md +52 -0
  208. graphify/skills/vscode/references/update.md +210 -0
  209. graphify/skills/windows/references/add-watch.md +56 -0
  210. graphify/skills/windows/references/exports.md +87 -0
  211. graphify/skills/windows/references/extraction-spec.md +70 -0
  212. graphify/skills/windows/references/github-and-merge.md +46 -0
  213. graphify/skills/windows/references/hooks.md +33 -0
  214. graphify/skills/windows/references/query.md +311 -0
  215. graphify/skills/windows/references/transcribe.md +52 -0
  216. graphify/skills/windows/references/update.md +210 -0
  217. graphify/symbol_resolution.py +556 -0
  218. graphify/transcribe.py +186 -0
  219. graphify/tree_html.py +603 -0
  220. graphify/validate.py +95 -0
  221. graphify/watch.py +2280 -0
  222. graphify/wiki.py +405 -0
  223. graphitect/__init__.py +28 -0
  224. graphitect/__main__.py +4 -0
  225. graphitect/_vendor/__init__.py +2 -0
  226. graphitect/_vendor/archify/LICENSE +22 -0
  227. graphitect/_vendor/archify/SKILL.md +137 -0
  228. graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
  229. graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
  230. graphitect/_vendor/archify/assets/template.html +14935 -0
  231. graphitect/_vendor/archify/bin/archify.mjs +2091 -0
  232. graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
  233. graphitect/_vendor/archify/bin/preview.mjs +653 -0
  234. graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
  235. graphitect/_vendor/archify/brand-marks/README.md +31 -0
  236. graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
  237. graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
  238. graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
  239. graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
  240. graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
  241. graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
  242. graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
  243. graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
  244. graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
  245. graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
  246. graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
  247. graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
  248. graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
  249. graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
  250. graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
  251. graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
  252. graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
  253. graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
  254. graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
  255. graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
  256. graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
  257. graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
  258. graphitect/_vendor/archify/package-lock.json +149 -0
  259. graphitect/_vendor/archify/package.json +39 -0
  260. graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
  261. graphitect/_vendor/archify/references/authoring-contract.md +243 -0
  262. graphitect/_vendor/archify/references/brand-marks.md +65 -0
  263. graphitect/_vendor/archify/references/delivery-contract.md +120 -0
  264. graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
  265. graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
  266. graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
  267. graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
  268. graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
  269. graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
  270. graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
  271. graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
  272. graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
  273. graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
  274. graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
  275. graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
  276. graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
  277. graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
  278. graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
  279. graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
  280. graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
  281. graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
  282. graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
  283. graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
  284. graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
  285. graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
  286. graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
  287. graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
  288. graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
  289. graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
  290. graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
  291. graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
  292. graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
  293. graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
  294. graphitect/_vendor/archify/schemas/README.md +211 -0
  295. graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
  296. graphitect/_vendor/archify/schemas/common.schema.json +115 -0
  297. graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
  298. graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
  299. graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
  300. graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
  301. graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
  302. graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
  303. graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
  304. graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
  305. graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
  306. graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
  307. graphitect/_vendor/archify/skill-release.json +10 -0
  308. graphitect/cli.py +981 -0
  309. graphitect/deliver/__init__.py +5 -0
  310. graphitect/deliver/archify_adapter.py +1877 -0
  311. graphitect/deliver/archify_ir.py +160 -0
  312. graphitect/deliver/archify_repair.py +135 -0
  313. graphitect/deliver/doc_compiler.py +916 -0
  314. graphitect/ground/__init__.py +5 -0
  315. graphitect/ground/describe_source.py +27 -0
  316. graphitect/ground/fullread_source.py +56 -0
  317. graphitect/ground/graphify_source.py +107 -0
  318. graphitect/models.py +118 -0
  319. graphitect/skill/SKILL.md +80 -0
  320. graphitect/skill/agents/openai.yaml +4 -0
  321. graphitect/synthesize/__init__.py +5 -0
  322. graphitect/synthesize/engine.py +281 -0
  323. graphitect/synthesize/llm_backend.py +331 -0
  324. graphitect/synthesize/questions.py +139 -0
  325. graphitect/synthesize/rubric.py +104 -0
  326. graphitect-0.2.0.dist-info/METADATA +284 -0
  327. graphitect-0.2.0.dist-info/RECORD +336 -0
  328. graphitect-0.2.0.dist-info/WHEEL +5 -0
  329. graphitect-0.2.0.dist-info/entry_points.txt +2 -0
  330. graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
  331. graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
  332. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
  333. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
  334. graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
  335. graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
  336. graphitect-0.2.0.dist-info/top_level.txt +2 -0
graphify/build.py ADDED
@@ -0,0 +1,2300 @@
1
+ # assemble node+edge dicts into a NetworkX graph, preserving edge direction
2
+ #
3
+ # Node deduplication — three layers:
4
+ #
5
+ # 1. Within a file (AST): each extractor tracks a `seen_ids` set. A node ID is
6
+ # emitted at most once per file, so duplicate class/function definitions in
7
+ # the same source file are collapsed to the first occurrence.
8
+ #
9
+ # 2. Between files (build): NetworkX G.add_node() is idempotent — calling it
10
+ # twice with the same ID overwrites the attributes with the second call's
11
+ # values. Nodes are added in extraction order (AST first, then semantic),
12
+ # so if the same entity is extracted by both passes the semantic node
13
+ # silently overwrites the AST node. This is intentional: semantic nodes
14
+ # carry richer labels and cross-file context, while AST nodes have precise
15
+ # source_location. If you need to change the priority, reorder extractions
16
+ # passed to build().
17
+ #
18
+ # 3. Semantic merge (skill): before calling build(), the skill merges cached
19
+ # and new semantic results using an explicit `seen` set keyed on node["id"],
20
+ # so duplicates across cache hits and new extractions are resolved there
21
+ # before any graph construction happens.
22
+ #
23
+ from __future__ import annotations
24
+ import json
25
+ import math
26
+ import os
27
+ import re
28
+ import sys
29
+ import unicodedata
30
+ from collections.abc import Iterable
31
+ from pathlib import Path
32
+ import networkx as nx
33
+ from .ids import make_id, normalize_id as _normalize_id
34
+ from .paths import default_graph_json as _default_graph_json
35
+ from .paths import is_absolute_any_platform as _is_abs
36
+ from .validate import validate_extraction
37
+
38
+
39
+ # Deterministic (AST) extractors emit source_location "L<line>"; the semantic
40
+ # extraction spec emits null. Used by _is_ast_tier as a shape fallback for
41
+ # legacy items that predate the _origin marker (#2334).
42
+ _AST_LOC_RE = re.compile(r"^L\d")
43
+
44
+
45
+ def _is_ast_tier(item: dict) -> bool:
46
+ """AST vs semantic tier. _origin wins when present; unstamped legacy items
47
+ (pre-0.9.16) fall back to shape: deterministic extractors emit
48
+ source_location 'L<line>', the semantic spec emits null (#2334)."""
49
+ o = item.get("_origin")
50
+ if o is not None:
51
+ return o == "ast"
52
+ loc = item.get("source_location")
53
+ return isinstance(loc, str) and bool(_AST_LOC_RE.match(loc))
54
+
55
+
56
+ # Relations that say only "these two symbols appear together", with no claim about
57
+ # HOW. An extractor that finds a specific fact for a pair — a call, an import, an
58
+ # inheritance — routinely emits one of these for the same pair as well, so when the
59
+ # simple graph collapses the pair to one edge, the generic one must never be the
60
+ # survivor. Deliberately a small denylist rather than a full precedence order over
61
+ # every relation: ranking `contains` against `calls` would be inventing a
62
+ # cross-axis judgement, whereas "specific beats generic" is the only comparison
63
+ # this collapse actually needs.
64
+ _GENERIC_RELATIONS: frozenset[str] = frozenset({"references", "uses", "mentions"})
65
+
66
+ # Language interop families, keyed by extension, for the cross-language phantom-edge
67
+ # guard in the edge loop below. Families group by REAL interop (JS/TS share a module
68
+ # graph; C/C++/ObjC share a compilation unit via headers; JVM langs share bytecode),
69
+ # so a legitimate TS->JS import or C impl->header call survives, while a Python
70
+ # `import time` binding to a `time.ts` (#1749) or a cross-language INFERRED `calls`
71
+ # edge (#1547/#1556) is dropped. Kept local to build.py (not imported from extract.py,
72
+ # which imports build.py — a cycle) and deliberately mirrors extract._LANG_FAMILY_BY_EXT.
73
+ _EDGE_LANG_FAMILY: dict[str, str] = {
74
+ ".py": "py", ".pyi": "py",
75
+ ".js": "js", ".mjs": "js", ".cjs": "js", ".jsx": "js",
76
+ ".ts": "js", ".tsx": "js", ".mts": "js", ".cts": "js",
77
+ ".go": "go", ".rs": "rs",
78
+ ".java": "jvm", ".kt": "jvm", ".scala": "jvm", ".groovy": "jvm",
79
+ ".c": "c", ".h": "c", ".cc": "c", ".cpp": "c", ".hpp": "c",
80
+ ".cxx": "c", ".hh": "c", ".hxx": "c",
81
+ ".cu": "c", ".cuh": "c", ".metal": "c", ".m": "c", ".mm": "c",
82
+ ".rb": "rb", ".rake": "rb", ".php": "php", ".cs": "cs", ".swift": "swift", ".lua": "lua",
83
+ }
84
+
85
+
86
+ # Synonym mapper for known invalid file_type values that LLM subagents commonly
87
+ # emit. Keeps semantic intent close (markdown→document, tool→code) and falls
88
+ # back to "concept" for any other invalid value (see #840).
89
+ _FILE_TYPE_SYNONYMS = {
90
+ "markdown": "document",
91
+ "text": "document",
92
+ "tool": "code",
93
+ "library": "code",
94
+ "pattern": "concept",
95
+ "principle": "concept",
96
+ "constraint": "concept",
97
+ "tech": "concept",
98
+ "technology": "concept",
99
+ "data-source": "concept",
100
+ "data_source": "concept",
101
+ "gotcha": "concept",
102
+ "framework": "concept",
103
+ }
104
+
105
+
106
+ # Hyperedge member lists are canonically keyed `nodes` (see graphify/llm.py
107
+ # extraction spec), but LLM/subagent drift and externally-supplied graph.json
108
+ # sometimes emit `members` or `node_ids`. _normalize_hyperedge_members folds
109
+ # those aliases into `nodes` at ingest so every downstream consumer reads one
110
+ # canonical key — mirroring the `from`/`to` edge-endpoint tolerance below.
111
+ _HE_MEMBER_ALIASES = ("members", "node_ids")
112
+
113
+
114
+ def _coerce_hyperedge_member_refs(he: dict, members: list) -> list:
115
+ """Coerce a hyperedge member list to hashable scalar ids, deduped in order.
116
+
117
+ LLM/subagent drift sometimes emits a member as an object (``{"id": "a_ts"}``)
118
+ instead of a bare id string. Left uncoerced, the dict member is unhashable,
119
+ so the semantic-rekey pass's ``_rekey.get(n, n)`` raised ``TypeError`` and
120
+ aborted the whole merge — destroying a completed extraction (#2486). Object
121
+ members collapse to their non-empty ``id`` (numeric ids str-coerced via
122
+ ``_coerce_id``, matching #2326); members with no usable id are dropped with
123
+ a stderr WARNING naming the hyperedge, never a crash. Hashable scalar refs
124
+ pass through unchanged. A hyperedge that loses every member this way falls
125
+ to the existing no-valid-members drop-with-warning in ``build_from_json``.
126
+ """
127
+ seen: set = set()
128
+ coerced: list = []
129
+ for ref in members:
130
+ if isinstance(ref, dict):
131
+ inner = _coerce_id(ref.get("id"))
132
+ if inner in (None, "") or not _hashable(inner):
133
+ print(
134
+ f"[graphify] WARNING: hyperedge "
135
+ f"'{he.get('id', '?')}' has a member object with no usable "
136
+ f"'id' ({ref!r}); dropping that member.",
137
+ file=sys.stderr,
138
+ )
139
+ continue
140
+ ref = inner
141
+ elif not _hashable(ref):
142
+ print(
143
+ f"[graphify] WARNING: hyperedge "
144
+ f"'{he.get('id', '?')}' has an unusable member reference "
145
+ f"{ref!r}; dropping that member.",
146
+ file=sys.stderr,
147
+ )
148
+ continue
149
+ if ref in seen:
150
+ continue
151
+ seen.add(ref)
152
+ coerced.append(ref)
153
+ return coerced
154
+
155
+
156
+ def _normalize_hyperedge_members(he: object) -> None:
157
+ """Canonicalize a hyperedge's member list onto the `nodes` key, in place.
158
+
159
+ If `nodes` is already a list it wins (canonical), and only stray alias keys
160
+ are dropped. Otherwise the first alias (`members`, then `node_ids`) that is a
161
+ list is moved to `nodes`, with a single stderr WARNING naming the hyperedge
162
+ id and alias used. Leftover alias keys are always removed so downstream code
163
+ never re-reads them. Whichever branch supplied the list, member VALUES are
164
+ coerced to hashable scalar ids and deduped preserving order (#2486) — see
165
+ ``_coerce_hyperedge_member_refs``.
166
+ """
167
+ if not isinstance(he, dict):
168
+ return
169
+ if isinstance(he.get("nodes"), list):
170
+ he["nodes"] = _coerce_hyperedge_member_refs(he, he["nodes"])
171
+ else:
172
+ for alias in _HE_MEMBER_ALIASES:
173
+ val = he.get(alias)
174
+ if isinstance(val, list):
175
+ he["nodes"] = _coerce_hyperedge_member_refs(he, val)
176
+ print(
177
+ f"[graphify] WARNING: hyperedge "
178
+ f"'{he.get('id', '?')}' uses field '{alias}' instead of "
179
+ f"'nodes'; normalizing.",
180
+ file=sys.stderr,
181
+ )
182
+ break
183
+ # Drop any leftover alias keys regardless of which branch ran above.
184
+ for alias in _HE_MEMBER_ALIASES:
185
+ he.pop(alias, None)
186
+
187
+
188
+ def _fold_node_aliases(node: dict) -> None:
189
+ """Fold legacy node field aliases onto canonical keys, in place (#2194).
190
+
191
+ ``name`` -> ``label`` and ``path`` -> ``source_file``. Uses an empty-check
192
+ (not mere key presence) so a node carrying ``label: ""``/``None`` next to a
193
+ real ``name`` is healed too. When the canonical field already holds a value
194
+ it wins and the alias key is left untouched. Without this fold an alias-only
195
+ node enters the graph with no label/source_file: it fails validation, gets
196
+ ``norm_label == ""`` (invisible to query/explain), and is excluded from every
197
+ label-keyed merge/dedup — a permanent ghost that ``graphify update``
198
+ re-feeds through build_from_json forever.
199
+ """
200
+ if not node.get("label") and isinstance(node.get("name"), str) and node["name"]:
201
+ node["label"] = node.pop("name")
202
+ if not node.get("source_file") and isinstance(node.get("path"), str) and node["path"]:
203
+ node["source_file"] = node.pop("path")
204
+
205
+
206
+ def _fold_edge_aliases(edge: dict) -> None:
207
+ """Fold legacy edge field aliases onto canonical keys, in place (#2194).
208
+
209
+ ``type`` -> ``relation``. A ``confidence_score`` float with no ``confidence``
210
+ enum backfills ``confidence: "INFERRED"`` — never EXTRACTED (alias recovery
211
+ is not provenance) and never a threshold mapping of the float. The
212
+ ``confidence_score`` key itself is NOT popped: it is a legitimate companion
213
+ field that the edge loop sanitizes and to_json round-trips.
214
+
215
+ A NUMERIC ``confidence`` (pre-enum graphs stored the LLM pass's float —
216
+ 1.0/0.95/0.9/0.85 — directly in the field) normalizes to ``INFERRED``:
217
+ numeric confidences only ever came from the LLM semantic pass, and
218
+ LLM-derived edges are INFERRED by definition. The original float moves to
219
+ ``confidence_score`` unless an explicit one is already present (the
220
+ companion field is the authority). Without this fold, every reload of a
221
+ pre-enum graph re-warns once per legacy edge, forever. ``bool`` is
222
+ excluded despite subclassing ``int``: ``True`` is not a score.
223
+ """
224
+ if not edge.get("relation") and isinstance(edge.get("type"), str) and edge["type"]:
225
+ edge["relation"] = edge.pop("type")
226
+ _conf = edge.get("confidence")
227
+ if isinstance(_conf, (int, float)) and not isinstance(_conf, bool):
228
+ if edge.get("confidence_score") is None:
229
+ edge["confidence_score"] = float(_conf)
230
+ edge["confidence"] = "INFERRED"
231
+ if not edge.get("confidence") and edge.get("confidence_score") is not None:
232
+ edge["confidence"] = "INFERRED"
233
+
234
+
235
+ def _coerce_id(value: object) -> object:
236
+ """Return a str for a numeric id, else the value unchanged.
237
+
238
+ ``bool`` is excluded despite subclassing ``int``: an id of ``True`` is not a
239
+ number the model meant to name a node, and ``"True"`` would invent a label.
240
+ """
241
+ if isinstance(value, bool) or not isinstance(value, (int, float)):
242
+ return value
243
+ return str(value)
244
+
245
+
246
+ def _hashable(value: object) -> bool:
247
+ """True when value can be a dict key / set member (same probe as the
248
+ inline ``try: hash(m)`` in build_from_json's hyperedge revalidation)."""
249
+ try:
250
+ hash(value)
251
+ except TypeError:
252
+ return False
253
+ return True
254
+
255
+
256
+ def _coerce_non_string_ids(extraction: dict) -> None:
257
+ """Coerce numeric node ids and edge/hyperedge references to str, in place (#2326).
258
+
259
+ A backend can emit ``{"id": 10}`` where the schema says ``{"id": "10"}``.
260
+ Every id consumer downstream assumes ``str``, so one int id aborted the build
261
+ in three places: ``_pick_winner``'s ``_CHUNK_SUFFIX.search(n["id"])`` raised
262
+ ``TypeError: expected string or bytes-like object``, and ``build_from_json``'s
263
+ ``sorted(node_set)`` raised ``'<' not supported between instances of 'str'
264
+ and 'int'`` — the latter for a lone node with nothing to dedup at all.
265
+ Coercing keeps the node and its edges rather than dropping either, which is
266
+ the same tolerate-and-heal treatment loose backend output already gets at the
267
+ parse chokepoint (#1631) and in the alias folds (#2194).
268
+
269
+ Endpoints and hyperedge members are coerced with the nodes, not after: a
270
+ node-only coercion would renumber ``10`` to ``"10"`` and leave every edge
271
+ pointing at the vanished ``10``, trading a loud crash for a silently
272
+ disconnected graph. The legacy ``from``/``to`` endpoint aliases are included
273
+ because dedup reads them directly (#803).
274
+
275
+ Runs in BOTH ``build`` (before dedup, which keys on id) and
276
+ ``build_from_json`` (the direct entry that reloads a persisted graph), for
277
+ the same two-site reason as the ``_fold_node_aliases`` fold (#2194). It is
278
+ idempotent, so the nested call on the ``build`` path is a no-op.
279
+
280
+ Non-numeric non-str ids (``None``, lists, dicts) are left alone for
281
+ ``validate_extraction`` to report: ``str(None) == "None"`` would fabricate a
282
+ node id that no edge references.
283
+ """
284
+ for node in extraction.get("nodes") or ():
285
+ if isinstance(node, dict) and "id" in node:
286
+ node["id"] = _coerce_id(node["id"])
287
+ for edge in extraction.get("edges") or ():
288
+ if not isinstance(edge, dict):
289
+ continue
290
+ for key in ("source", "target", "from", "to"):
291
+ if key in edge:
292
+ edge[key] = _coerce_id(edge[key])
293
+ for he in extraction.get("hyperedges") or ():
294
+ if not isinstance(he, dict):
295
+ continue
296
+ members = he.get("nodes")
297
+ if isinstance(members, list):
298
+ he["nodes"] = [_coerce_id(ref) for ref in members]
299
+
300
+
301
+ def _norm_source_file(p: str | None, root: str | None = None) -> str | None:
302
+ """Normalize path separators and relativize absolute paths.
303
+
304
+ Converts backslashes to forward slashes (Windows compatibility) and, when
305
+ root is provided, strips the absolute prefix from paths produced by semantic
306
+ subagents so source_file is always repo-relative (fixes #932).
307
+ """
308
+ if not p:
309
+ return p
310
+ p = p.replace("\\", "/")
311
+ if root and _is_abs(p):
312
+ try:
313
+ p = Path(p).relative_to(root).as_posix()
314
+ except ValueError:
315
+ # Lexical relative_to failed. Retry with both sides fully resolved:
316
+ # a symlinked scan root (macOS /var -> /private/var, or a symlinked
317
+ # home/worktree) makes the raw prefixes differ even though they point
318
+ # at the same dir, which otherwise silently defeats prune/replace
319
+ # matching. Only the slow path resolves, so the common lexical match
320
+ # stays filesystem-free.
321
+ try:
322
+ p = Path(p).resolve().relative_to(Path(root).resolve()).as_posix()
323
+ except (ValueError, OSError):
324
+ pass
325
+ return p
326
+
327
+
328
+ def _abs_identity(p: str | None, root: str | None = None) -> str | None:
329
+ """Return a form-insensitive absolute identity for a source_file.
330
+
331
+ prune/replace matching in build_merge otherwise compares raw strings against
332
+ ``_norm_source_file`` output, so a node whose source_file survived in a THIRD
333
+ form — absolute where prune_sources is relative, or vice versa, or a symlinked
334
+ root — slips past every equality check and its nodes/edges are never pruned
335
+ (silent survival of a deleted file's graph, #2012). Anchoring relative paths
336
+ at ``root`` and resolving both sides to a canonical absolute posix path gives
337
+ a fallback that matches regardless of which form each side happens to hold.
338
+ """
339
+ if not p:
340
+ return None
341
+ q = p.replace("\\", "/")
342
+ pp = Path(q)
343
+ if not _is_abs(q) and root:
344
+ pp = Path(root) / q
345
+ try:
346
+ return pp.resolve().as_posix()
347
+ except OSError:
348
+ return pp.as_posix()
349
+
350
+
351
+ def _is_file_node_label(label: "str | None", source_file: "str | None") -> bool:
352
+ """Whether *label* is a file node's label for *source_file* — the bare
353
+ basename, OR a directory-qualified suffix produced by the disambiguation pass
354
+ below (#2032). Used both to recognize file nodes when relabeling and by the
355
+ downstream file-node predicates (analyze/tree/serve)."""
356
+ if not label or not source_file:
357
+ return False
358
+ sf = str(source_file).replace("\\", "/")
359
+ lbl = str(label)
360
+ if lbl == sf.rsplit("/", 1)[-1]:
361
+ return True
362
+ return "/" in lbl and (sf == lbl or sf.endswith("/" + lbl))
363
+
364
+
365
+ def _shortest_unique_suffix(sf: str, all_sfs: "set[str]") -> str:
366
+ """Shortest trailing path suffix (basename + k parent dirs) of *sf* that is
367
+ unique among *all_sfs*. `a/b/index.ts` vs `c/b/index.ts` -> `a/b/index.ts`;
368
+ `x/index.ts` vs `y/index.ts` -> `x/index.ts`. Derived from the path (never the
369
+ current label) so relabeling is idempotent across incremental rebuilds."""
370
+ parts = [p for p in sf.replace("\\", "/").split("/") if p]
371
+ others = [
372
+ [p for p in o.replace("\\", "/").split("/") if p]
373
+ for o in all_sfs if o != sf
374
+ ]
375
+ for k in range(1, len(parts) + 1):
376
+ suffix = parts[-k:]
377
+ if all(o[-k:] != suffix for o in others):
378
+ return "/".join(suffix)
379
+ return "/".join(parts)
380
+
381
+
382
+ def _file_label_reassignments(items: "list[tuple]") -> dict:
383
+ """Given (key, label, source_file) triples, return {key: new_label} for file
384
+ nodes whose basename collides with another's — the shortest unique
385
+ directory-qualified suffix (#2032). Keys of non-colliding/basename-unique file
386
+ nodes are omitted (their label stays bare)."""
387
+ from collections import defaultdict
388
+ groups: dict[str, list[tuple]] = defaultdict(list)
389
+ for key, label, sf in items:
390
+ if sf and label and _is_file_node_label(str(label), str(sf)):
391
+ basename = str(sf).replace("\\", "/").rsplit("/", 1)[-1]
392
+ groups[basename].append((key, str(sf)))
393
+ out: dict = {}
394
+ for members in groups.values():
395
+ distinct = {sf for _, sf in members}
396
+ if len(distinct) < 2:
397
+ continue # no collision — leave the bare basename label
398
+ for key, sf in members:
399
+ out[key] = _shortest_unique_suffix(sf, distinct)
400
+ return out
401
+
402
+
403
+ def _disambiguate_file_node_labels(G: "nx.Graph") -> None:
404
+ """Relabel colliding-basename file nodes on a graph (#2032). Ids/edges are
405
+ never changed — only display labels. Idempotent (labels derive from
406
+ source_file, not the current possibly-qualified label)."""
407
+ items = [(nid, a.get("label"), a.get("source_file")) for nid, a in G.nodes(data=True)]
408
+ for nid, new_label in _file_label_reassignments(items).items():
409
+ G.nodes[nid]["label"] = new_label
410
+
411
+
412
+ def disambiguate_file_labels_in_nodes(nodes: "list") -> None:
413
+ """Relabel colliding-basename file nodes on a raw node-dict list, in place
414
+ (#2032). Used by the extract --no-cluster path, which writes the merged
415
+ extraction directly without going through build_from_json."""
416
+ items = [
417
+ (i, n.get("label"), n.get("source_file"))
418
+ for i, n in enumerate(nodes) if isinstance(n, dict)
419
+ ]
420
+ for i, new_label in _file_label_reassignments(items).items():
421
+ nodes[i]["label"] = new_label
422
+
423
+
424
+ def _infer_merge_root(graph_path: Path) -> str | None:
425
+ """Best-effort scan root for relativizing paths in build_merge when the caller
426
+ passes no ``root`` (#1571).
427
+
428
+ Prefers the committed ``graphify-out/.graphify_root`` marker — the authoritative
429
+ scan root graphify records at build/watch time (#686/#1423) — then falls back to
430
+ the directory that contains the output dir (``graph.json``'s grandparent, i.e.
431
+ ``<root>/graphify-out/graph.json`` -> ``<root>``). The grandparent heuristic is
432
+ applied only when ``graph.json``'s own directory actually looks like a graphify
433
+ out-dir (named like one, or holding the marker/manifest); for arbitrary layouts
434
+ (``<root>/graph.json``, #2446) the grandparent is NOT the scan root, and guessing
435
+ it made absolute prune_sources silently no-op. Returns None if neither resolves,
436
+ in which case normalization is a no-op (prior behavior) and prune matching can
437
+ still recover via :func:`_derive_prune_root`.
438
+ """
439
+ parent = graph_path.parent
440
+ try:
441
+ marker = parent / ".graphify_root"
442
+ if marker.exists():
443
+ recorded = marker.read_text(encoding="utf-8-sig").strip()
444
+ if recorded:
445
+ return str(Path(recorded).resolve())
446
+ except OSError:
447
+ pass
448
+ from .paths import GRAPHIFY_OUT
449
+ try:
450
+ if (
451
+ parent.name == Path(GRAPHIFY_OUT).name
452
+ or (parent / ".graphify_root").exists()
453
+ or (parent / "manifest.json").exists()
454
+ ):
455
+ return str(parent.parent.resolve())
456
+ except Exception:
457
+ pass
458
+ return None
459
+
460
+
461
+ def _build_prune_sets(
462
+ prune_sources: "list[str] | None",
463
+ eff_root: "str | None",
464
+ new_sources: "set[str]",
465
+ ) -> "tuple[dict[str, str], dict[str, str]]":
466
+ """Prune match sets for build_merge / merge_raw_extraction.
467
+
468
+ Returns ``(prune_set, prune_abs)`` mapping every match form of each prune
469
+ entry — the raw string, the :func:`_norm_source_file` relative form, and the
470
+ :func:`_abs_identity` absolute form — back to the ORIGINAL entry, so callers
471
+ can report which entries actually matched something (#2446). Membership
472
+ tests read like the old set-based code (``sf in prune_set``).
473
+
474
+ "Replace" wins over a contradictory "delete" of the same source (#1796), in
475
+ both string and absolute-identity space (#2012): every form belonging to a
476
+ source re-extracted this run is removed.
477
+ """
478
+ prune_set: dict[str, str] = {}
479
+ prune_abs: dict[str, str] = {}
480
+ for p in (prune_sources or []):
481
+ if not p:
482
+ continue
483
+ prune_set.setdefault(p, p)
484
+ norm = _norm_source_file(p, eff_root)
485
+ if norm:
486
+ prune_set.setdefault(norm, p)
487
+ a = _abs_identity(p, eff_root)
488
+ if a:
489
+ prune_abs.setdefault(a, p)
490
+ for s in new_sources:
491
+ prune_set.pop(s, None)
492
+ a = _abs_identity(s, eff_root)
493
+ if a:
494
+ prune_abs.pop(a, None)
495
+ return prune_set, prune_abs
496
+
497
+
498
+ def _derive_prune_root(prune_sources: "list[str]", stored_sfs: "set[str]") -> "str | None":
499
+ """Derive the scan root from absolute prune paths that matched nothing (#2446).
500
+
501
+ When build_merge/merge_raw_extraction receive absolute prune_sources but no
502
+ ``root``, :func:`_infer_merge_root`'s guess can be wrong for non-standard
503
+ layouts (``graph.json`` not under ``<root>/graphify-out/``), so every prune
504
+ entry silently no-ops. Each absolute prune path ``P`` that ends with a
505
+ stored RELATIVE source_file ``S`` implies the candidate root
506
+ ``P[:-len(S)-1]``. Within one graph all relative source_files share a single
507
+ scan root, so the candidate is accepted only when it is unique and
508
+ consistent across every suffix-matched entry; on ambiguity (or no suffix
509
+ match at all) returns None and the caller falls through to the zero-match
510
+ warning.
511
+ """
512
+ rel_sfs = [
513
+ sf.replace("\\", "/")
514
+ for sf in stored_sfs
515
+ if sf and isinstance(sf, str) and not _is_abs(sf.replace("\\", "/"))
516
+ ]
517
+ if not rel_sfs:
518
+ return None
519
+ roots: set[str] = set()
520
+ for p in prune_sources:
521
+ if not p or not isinstance(p, str):
522
+ continue
523
+ q = p.replace("\\", "/")
524
+ if not _is_abs(q):
525
+ continue
526
+ hits = {q[: -len(s) - 1] for s in rel_sfs if q.endswith("/" + s)}
527
+ if len(hits) > 1:
528
+ return None # one entry implies two different roots — ambiguous
529
+ roots |= hits
530
+ if len(roots) == 1:
531
+ return next(iter(roots))
532
+ return None
533
+
534
+
535
+ def edge_data(G: nx.Graph, u: str, v: str) -> dict:
536
+ """Return one edge attribute dict for (u, v), tolerating MultiGraph.
537
+
538
+ For MultiGraph/MultiDiGraph there can be multiple parallel edges;
539
+ this returns the first one (sufficient for callers that only need
540
+ relation/confidence for rendering). Fixes #796.
541
+ """
542
+ raw = G[u][v]
543
+ if isinstance(G, (nx.MultiGraph, nx.MultiDiGraph)):
544
+ return next(iter(raw.values()), {})
545
+ return raw
546
+
547
+
548
+ def edge_datas(G: nx.Graph, u: str, v: str) -> list[dict]:
549
+ """Return every edge attribute dict for (u, v); always a list."""
550
+ raw = G[u][v]
551
+ if isinstance(G, (nx.MultiGraph, nx.MultiDiGraph)):
552
+ return list(raw.values())
553
+ return [raw]
554
+
555
+
556
+ def dedupe_nodes(nodes: list[dict]) -> list[dict]:
557
+ """Collapse nodes sharing an ``id``, last-writer-wins on attributes.
558
+
559
+ Mirrors what ``build_from_json``'s ``G.add_node`` does implicitly (idempotent;
560
+ a later node overwrites an earlier one's attributes). The ``--no-cluster``
561
+ write path dumps the raw node list without building a graph, so same-id nodes
562
+ — e.g. a Swift ``type=module`` anchor emitted once per importing file (#1327)
563
+ — would otherwise appear as duplicates. Insertion order follows each id's
564
+ first appearance; the retained dict is the last one seen.
565
+ """
566
+ by_id: dict = {}
567
+ for n in nodes:
568
+ nid = n.get("id")
569
+ if nid is None:
570
+ continue
571
+ by_id[nid] = n
572
+ return list(by_id.values())
573
+
574
+
575
+ def dedupe_edges(edges: list[dict]) -> list[dict]:
576
+ """Collapse exact parallel edges by ``(source, target, relation)``, keeping the
577
+ first occurrence.
578
+
579
+ The clustered build path runs edges through a NetworkX ``DiGraph``, which
580
+ collapses parallel edges automatically. The ``--no-cluster`` and incremental
581
+ ``update`` write paths bypass NetworkX and concatenate edge lists raw, so
582
+ duplicates accumulate and edge counts become non-deterministic across build
583
+ modes / repeated updates (#1317). Deduping on the connectivity identity is
584
+ zero-signal-loss and restores idempotency. Callers that intentionally keep
585
+ parallel edges (multigraph output) must not use this.
586
+ """
587
+ seen: set[tuple] = set()
588
+ out: list[dict] = []
589
+ for e in edges:
590
+ key = (e.get("source"), e.get("target"), e.get("relation"))
591
+ if key in seen:
592
+ continue
593
+ seen.add(key)
594
+ out.append(e)
595
+ return out
596
+
597
+
598
+ def _old_file_stems(rel: Path) -> list[str]:
599
+ """Pre-migration stem forms a semantic fragment may have used for ``rel``.
600
+
601
+ Ordered longest-first so prefix stripping is greedy and unambiguous:
602
+ - one-parent form: ``parent.stem`` (the old _file_stem rule, #550-era)
603
+ - zero-parent form: ``stem`` (the old llm.py prompt rule, #1509)
604
+ """
605
+ forms: list[str] = []
606
+ parent = rel.parent.name
607
+ if parent and parent not in (".", ""):
608
+ forms.append(make_id(f"{parent}.{rel.stem}"))
609
+ forms.append(make_id(rel.stem))
610
+ # Dedupe while preserving order (top-level files collapse both forms).
611
+ seen: set[str] = set()
612
+ return [f for f in forms if f and not (f in seen or seen.add(f))]
613
+
614
+
615
+ def _semantic_id_remap(nodes: list, root: str | None) -> dict:
616
+ """Re-derive non-AST node ids from ``source_file`` using the canonical
617
+ full-path stem, so a cached/LLM fragment carrying a pre-migration short id
618
+ reconciles with the AST node instead of spawning a ghost (#1504/#1509).
619
+
620
+ Drift-proof by construction: the new id is computed from ``source_file`` in
621
+ code, never trusted from the fragment's own ``id`` string. AST-origin nodes
622
+ are skipped (they are already canonical via the extract() post-pass)."""
623
+ from graphify.extractors.base import _file_stem # local: avoid import cost at module load
624
+
625
+ remap: dict[str, str] = {}
626
+ for node in nodes:
627
+ if not isinstance(node, dict):
628
+ continue
629
+ if _is_ast_tier(node):
630
+ continue
631
+ nid = node.get("id")
632
+ sf = node.get("source_file")
633
+ if not nid or not isinstance(nid, str) or not sf:
634
+ continue
635
+ sf_norm = _norm_source_file(str(sf), root) or str(sf)
636
+ rel = Path(sf_norm)
637
+ if _is_abs(sf_norm):
638
+ # Can't relativize (no/failed root) — leave the id untouched rather
639
+ # than bake an on-disk path into it. Tested for BOTH platforms: a
640
+ # graph built on Linux/CI carries POSIX-absolute source_files that
641
+ # WindowsPath.is_absolute() calls relative, which leaked the whole
642
+ # build directory into node IDs when updated on Windows (#2618).
643
+ continue
644
+ if not rel.name:
645
+ # source_file equals the scan root, so _norm_source_file relativized it
646
+ # to Path('.') — a project-level node with no per-file identity to remap.
647
+ # Leave its id untouched (and avoid _file_stem's empty-name crash, #1618).
648
+ continue
649
+ new_stem = make_id(_file_stem(rel))
650
+ if not new_stem:
651
+ continue
652
+ norm_nid = _normalize_id(nid)
653
+ # Idempotency guard (#1917): an id already carrying its canonical stem is
654
+ # done — do not re-run the legacy branch on it. When the canonical stem
655
+ # contains a shorter legacy stem as a prefix (parent dir name == file
656
+ # stem, e.g. `.claude/CLAUDE.md` -> `claude_claude` over legacy `claude`),
657
+ # an already-migrated id like `claude_claude_x` still matches the legacy
658
+ # `claude_` prefix below and would gain another stem segment on every
659
+ # build, defeating the same_topology/no_change short-circuits. Mirrors the
660
+ # canonical check in graph_has_legacy_ids.
661
+ if norm_nid == new_stem or norm_nid.startswith(new_stem + "_"):
662
+ continue
663
+ new_id: str | None = None
664
+ old_forms = _old_file_stems(rel)
665
+ # #2197: on Windows, detect() can emit an ABSOLUTE source_file, and a
666
+ # semantic fragment's id derived from that absolute path (e.g.
667
+ # d_projects_myrepo_docs_dataflow) matches neither the canonical
668
+ # relative stem nor the legacy short forms above — so while source_file
669
+ # itself is healed by _norm_source_file, the id would ghost against the
670
+ # existing graph's docs_dataflow. When the raw path was absolute and
671
+ # relativized under root, treat the raw-absolute stem as one more
672
+ # old-stem form — the semantic-side twin of extract.py's absolute-form
673
+ # id registration. It is the longest form, so it goes first (greedy
674
+ # prefix stripping, same ordering rule as _old_file_stems).
675
+ sf_raw = str(sf).replace("\\", "/")
676
+ if sf_raw != sf_norm and _is_abs(sf_raw):
677
+ abs_stem = make_id(_file_stem(Path(sf_raw)))
678
+ if abs_stem and abs_stem != new_stem and abs_stem not in old_forms:
679
+ old_forms.insert(0, abs_stem)
680
+ for old_stem in old_forms:
681
+ if old_stem == new_stem:
682
+ continue # already canonical for this form
683
+ if norm_nid == old_stem:
684
+ new_id = new_stem # the file node itself
685
+ break
686
+ prefix = old_stem + "_"
687
+ if norm_nid.startswith(prefix):
688
+ entity = norm_nid[len(prefix):]
689
+ new_id = make_id(new_stem, entity)
690
+ break
691
+ if new_id and new_id != nid:
692
+ remap[nid] = new_id
693
+ return remap
694
+
695
+
696
+ # MCP node kinds whose ID is GLOBAL by design — deliberately shared across every
697
+ # config file that mentions them (`mcp_command_npx`, `mcp_package_...`,
698
+ # `env_var_...`), so it is not derived from ``source_file`` at all (#2408). The
699
+ # file-scoped kinds (`mcp_config_file`, `mcp_server`) ARE stem-derived and stay
700
+ # subject to legacy detection.
701
+ _MCP_GLOBAL_ID_KINDS = frozenset({"mcp_command", "mcp_package", "env_var"})
702
+
703
+
704
+ def _has_global_id(node: dict) -> bool:
705
+ """Whether ``node``'s ID is global by construction rather than file-derived."""
706
+ meta = node.get("metadata")
707
+ if not isinstance(meta, dict):
708
+ return False
709
+ return meta.get("mcp_kind") in _MCP_GLOBAL_ID_KINDS
710
+
711
+
712
+ def graph_has_legacy_ids(nodes: list, root: str | Path | None = None, sample: int = 300) -> bool:
713
+ """Whether a loaded graph still uses pre-#1504 node IDs (parent-dir / filename
714
+ stem) rather than the full repo-relative path. Read-only consumers (query,
715
+ serve) use this to nudge the user to rebuild, since they don't re-extract.
716
+
717
+ Heuristic and cheap: only **file-level** nodes (source_location ``L1``) are
718
+ inspected, because their ID is unambiguously the file stem. Symbol nodes are
719
+ skipped — some extractors scope a symbol by package/directory (Go's
720
+ ``_make_id(pkg_dir, name)`` → ``sub_thing``), which can coincide with an old
721
+ file-stem form and would otherwise false-positive. Nodes whose ID is global by
722
+ construction (see ``_MCP_GLOBAL_ID_KINDS``) are skipped for the same reason.
723
+ Returns True as soon as one file node's ID matches an OLD stem form but not the
724
+ canonical full-path form."""
725
+ from graphify.extractors.base import _file_stem
726
+ _r = str(root) if root is not None else None
727
+ checked = 0
728
+ for node in nodes:
729
+ if not isinstance(node, dict):
730
+ continue
731
+ if str(node.get("source_location") or "") != "L1":
732
+ continue # only file-level nodes carry an unambiguous file-stem ID
733
+ if _has_global_id(node):
734
+ # #2408: MCP ingest stamps every node it emits with line 1 (JSON has no
735
+ # line info), so globally-scoped nodes slip past the L1 proxy for
736
+ # "file-level". For `sub/.mcp.json` the old bare stem is `mcp` while the
737
+ # canonical stem is `sub_mcp`, so a perfectly valid `mcp_command_npx`
738
+ # reads as a legacy `mcp_`-prefixed id and warns on every fresh build.
739
+ # (A root-level `.mcp.json` never tripped it: there `mcp` IS canonical.)
740
+ continue
741
+ nid = node.get("id")
742
+ sf = node.get("source_file")
743
+ if not nid or not isinstance(nid, str) or not sf:
744
+ continue
745
+ sf_norm = _norm_source_file(str(sf), _r) or str(sf)
746
+ rel = Path(sf_norm)
747
+ if _is_abs(sf_norm):
748
+ continue
749
+ if not rel.name:
750
+ continue # source_file == scan root -> Path('.'), no file stem (#1618)
751
+ new_stem = make_id(_file_stem(rel))
752
+ if not new_stem:
753
+ continue
754
+ norm = _normalize_id(nid)
755
+ if norm == new_stem or norm.startswith(new_stem + "_"):
756
+ checked += 1
757
+ else:
758
+ for old in _old_file_stems(rel):
759
+ if old != new_stem and (norm == old or norm.startswith(old + "_")):
760
+ return True
761
+ checked += 1
762
+ if checked >= sample:
763
+ break
764
+ return False
765
+
766
+
767
+ def _doc_twin_remap(nodes: list) -> dict[str, str]:
768
+ """Map a markdown quick-scan's bare doc node ``<slug>`` to the semantic
769
+ ``<slug>_doc`` node for the SAME file (#1799).
770
+
771
+ The markdown quick-scan (``extract_markdown``) mints a file node with the
772
+ bare id ``_make_id(path)`` while the semantic pass mints ``<slug>_doc`` for
773
+ the same document. A ``graphify update`` after a semantic build leaves both,
774
+ splitting the file's edges across two disconnected nodes. Canonicalize to the
775
+ semantic ``_doc`` node (it carries the richer references/hyperedges). Gated to
776
+ ``file_type == "document"`` on BOTH twins with an identical ``source_file``,
777
+ so an unrelated code symbol ``foo`` and ``foo_doc`` never merge.
778
+ """
779
+ by_id: dict[str, dict] = {}
780
+ for n in nodes:
781
+ if isinstance(n, dict) and n.get("id"):
782
+ by_id[str(n["id"])] = n
783
+ remap: dict[str, str] = {}
784
+ for nid, node in by_id.items():
785
+ if not nid.endswith("_doc"):
786
+ continue
787
+ bare = by_id.get(nid[:-4])
788
+ if bare is None:
789
+ continue
790
+ sf = node.get("source_file")
791
+ if not sf or bare.get("source_file") != sf:
792
+ continue
793
+ if node.get("file_type") != "document" or bare.get("file_type") != "document":
794
+ continue
795
+ remap[nid[:-4]] = nid
796
+ return remap
797
+
798
+
799
+ def build_from_json(extraction: dict, *, directed: bool = False, root: str | Path | None = None) -> nx.Graph:
800
+ """Build a NetworkX graph from an extraction dict.
801
+
802
+ directed=True produces a DiGraph that preserves edge direction (source→target).
803
+ directed=False (default) produces an undirected Graph for backward compatibility.
804
+ root: if given, absolute source_file paths from semantic subagents are made
805
+ relative to root so all nodes share a consistent path key (#932).
806
+ """
807
+ _root = str(Path(root).resolve()) if root else None
808
+ # NetworkX <= 3.1 serialised edges as "links"; remap to "edges" for compatibility.
809
+ if "edges" not in extraction and "links" in extraction:
810
+ extraction = dict(extraction, edges=extraction["links"])
811
+
812
+ # Hyperedge persistence is dual-slot (#2485): to_json writes BOTH a
813
+ # top-level `hyperedges` key AND the nested `graph.hyperedges` (node_link
814
+ # graph attrs), but node_link_data-only writers emit just the nested slot.
815
+ # Fold the nested slot onto the top-level key ONCE, so every downstream
816
+ # pass (_coerce_non_string_ids, _normalize_hyperedge_members, the member
817
+ # revalidation before G.graph["hyperedges"] is set) reads one location.
818
+ if "hyperedges" not in extraction and isinstance(
819
+ (extraction.get("graph") or {}).get("hyperedges"), list
820
+ ):
821
+ extraction = dict(extraction, hyperedges=extraction["graph"]["hyperedges"])
822
+
823
+ # Numeric ids from a loose backend become str before anything keys on them
824
+ # (#2326) — after the links remap so aliased edges are covered too.
825
+ _coerce_non_string_ids(extraction)
826
+
827
+ # Canonicalize legacy node/edge schema before validation.
828
+ for node in extraction.get("nodes", []):
829
+ if not isinstance(node, dict):
830
+ continue
831
+ if "source" in node and "source_file" not in node:
832
+ # Count edges that reference this node so the warning is actionable (#479)
833
+ node_id = node.get("id", "?")
834
+ affected_edges = sum(
835
+ 1 for e in extraction.get("edges", [])
836
+ if e.get("source") == node_id or e.get("target") == node_id
837
+ )
838
+ print(
839
+ f"[graphify] WARNING: node '{node_id}' uses field 'source' instead of "
840
+ f"'source_file' — {affected_edges} edge(s) may be misrouted. "
841
+ f"Rename the field to 'source_file' to silence this warning.",
842
+ file=sys.stderr,
843
+ )
844
+ node["source_file"] = node.pop("source")
845
+ # Fold the remaining legacy node aliases (`name`->`label`,
846
+ # `path`->`source_file`, #2194) before validation and before the
847
+ # semantic-rekey / ghost-merge passes below, all of which key on
848
+ # label/source_file and would otherwise skip the node entirely.
849
+ _fold_node_aliases(node)
850
+ # Default missing/None file_type to "concept" so legacy graph.json
851
+ # entries (and stub nodes preserved by `_rebuild_code` from older
852
+ # graphify versions that didn't always populate file_type) don't
853
+ # trigger spurious "invalid file_type 'None'" validator warnings (#660).
854
+ if node.get("file_type") in (None, ""):
855
+ node["file_type"] = "concept"
856
+ ft = node.get("file_type", "")
857
+ if ft and ft not in {"code", "document", "paper", "image", "rationale", "concept"}:
858
+ node["file_type"] = _FILE_TYPE_SYNONYMS.get(ft, "concept")
859
+
860
+ # Canonicalize hyperedge member lists (#1561): producers sometimes key the
861
+ # member list `members`/`node_ids` instead of `nodes`. Fold aliases onto
862
+ # `nodes` here — BEFORE validation and the semantic-rekey loop below — so
863
+ # every downstream consumer (rekey, source_file relativize, to_json) reads
864
+ # one canonical key, the same way edge endpoints alias from/to at build.
865
+ for he in extraction.get("hyperedges", []) or []:
866
+ _normalize_hyperedge_members(he)
867
+
868
+ # Fold legacy edge field aliases (`type`->`relation`,
869
+ # `confidence_score`->`confidence`, #2194) BEFORE validation. The existing
870
+ # from/to endpoint fold lives in the edge loop further down, which runs
871
+ # after validate_extraction — too late for fields the validator requires.
872
+ for edge in extraction.get("edges", []):
873
+ if isinstance(edge, dict):
874
+ _fold_edge_aliases(edge)
875
+
876
+ errors = validate_extraction(extraction)
877
+ # Dangling edges (stdlib/external imports) are expected - only warn about real schema errors.
878
+ real_errors = [e for e in errors if "does not match any node id" not in e]
879
+ if real_errors:
880
+ # Break the warning down by cause (#2194): a mixed batch used to surface
881
+ # only real_errors[0], hiding every other failure mode. Group on the
882
+ # "missing required field 'X'" suffix and report per-cause counts plus
883
+ # one example each, so the operator sees the full shape of the damage.
884
+ by_cause: dict[str, list[str]] = {}
885
+ for err in real_errors:
886
+ m = re.search(r"missing required field '[^']*'", err)
887
+ by_cause.setdefault(m.group(0) if m else "other schema issue", []).append(err)
888
+ breakdown = "; ".join(
889
+ f"{len(errs)}x {cause} (e.g. {errs[0]})" for cause, errs in by_cause.items()
890
+ )
891
+ print(
892
+ f"[graphify] Extraction warning ({len(real_errors)} issues): {breakdown}",
893
+ file=sys.stderr,
894
+ )
895
+ # Deterministic semantic re-key (#1504/#1509): the node-ID stem is now the
896
+ # full repo-relative path (docs/v1/api/README.md -> docs_v1_api_readme), but
897
+ # the semantic cache is UNVERSIONED, so a cached/LLM fragment can still carry
898
+ # an OLD short id whose stem was just the immediate parent dir (api_readme),
899
+ # or a prompt-drifting id with zero parent dirs (readme). Rather than trust
900
+ # LLM prose to emit the right stem, we re-derive every non-AST node's id from
901
+ # its own source_file in code, so a drifted fragment physically reconciles
902
+ # with the AST node instead of spawning a ghost / a re-bill. AST-origin nodes
903
+ # already carry canonical ids (the extract() id-remap post-pass guarantees it)
904
+ # and are left untouched.
905
+ _rekey: dict[str, str] = _semantic_id_remap(extraction.get("nodes", []), _root)
906
+ if _rekey:
907
+ for node in extraction.get("nodes", []):
908
+ if isinstance(node, dict) and node.get("id") in _rekey:
909
+ node["id"] = _rekey[node["id"]]
910
+ for edge in extraction.get("edges", []):
911
+ if not isinstance(edge, dict):
912
+ continue
913
+ if edge.get("source") in _rekey:
914
+ edge["source"] = _rekey[edge["source"]]
915
+ if edge.get("target") in _rekey:
916
+ edge["target"] = _rekey[edge["target"]]
917
+ for he in extraction.get("hyperedges", []) or []:
918
+ if isinstance(he, dict) and isinstance(he.get("nodes"), list):
919
+ # Guard on hashability (#2486): _normalize_hyperedge_members
920
+ # has already coerced members above, but a still-unhashable ref
921
+ # must pass through rather than abort the merge on dict.get.
922
+ he["nodes"] = [
923
+ _rekey.get(n, n) if _hashable(n) else n for n in he["nodes"]
924
+ ]
925
+
926
+ # Merge markdown quick-scan bare doc nodes into their semantic `_doc` twin
927
+ # for the same file, so a document is one node regardless of which pipeline
928
+ # touched it last (#1799).
929
+ _doc_remap = _doc_twin_remap(extraction.get("nodes", []))
930
+ if _doc_remap:
931
+ extraction["nodes"] = [
932
+ n for n in extraction.get("nodes", [])
933
+ if not (isinstance(n, dict) and n.get("id") in _doc_remap)
934
+ ]
935
+ _new_edges = []
936
+ for edge in extraction.get("edges", []):
937
+ if isinstance(edge, dict):
938
+ s0, t0 = edge.get("source"), edge.get("target")
939
+ if s0 in _doc_remap:
940
+ edge["source"] = _doc_remap[s0]
941
+ if t0 in _doc_remap:
942
+ edge["target"] = _doc_remap[t0]
943
+ # Drop only self-loops the remap itself collapsed (a bare->_doc
944
+ # link becoming doc->doc); leave any pre-existing self-loop alone.
945
+ if edge.get("source") == edge.get("target") and (s0 in _doc_remap or t0 in _doc_remap):
946
+ continue
947
+ _new_edges.append(edge)
948
+ extraction["edges"] = _new_edges
949
+ for he in extraction.get("hyperedges", []) or []:
950
+ if isinstance(he, dict) and isinstance(he.get("nodes"), list):
951
+ # Same hashability guard as the _rekey pass above (#2486).
952
+ he["nodes"] = [
953
+ _doc_remap.get(n, n) if _hashable(n) else n for n in he["nodes"]
954
+ ]
955
+
956
+ G: nx.Graph = nx.DiGraph() if directed else nx.Graph()
957
+ for node in extraction.get("nodes", []):
958
+ # Skip dict nodes with a missing or non-hashable id (e.g. a list emitted
959
+ # by a buggy LLM extraction) so NetworkX add_node never raises
960
+ # TypeError: unhashable type. Non-dict nodes are deliberately left to
961
+ # raise as before, so callers that probe build for shape errors (e.g.
962
+ # the multigraph diagnostic) still observe the malformed shape.
963
+ if isinstance(node, dict):
964
+ if "id" not in node:
965
+ continue
966
+ try:
967
+ hash(node["id"])
968
+ except TypeError:
969
+ print(
970
+ f"[graphify] WARNING: skipping node with non-hashable id "
971
+ f"{node['id']!r} (must be a string).",
972
+ file=sys.stderr,
973
+ )
974
+ continue
975
+ if "source_file" in node:
976
+ node["source_file"] = _norm_source_file(node["source_file"], _root)
977
+ # definition_file names a file inside the scanned tree exactly like
978
+ # source_file (the #2990 decl/def merge stamps it from the impl's
979
+ # source_file BEFORE this normalization runs), so it must be made
980
+ # portable the same way - it used to ship absolute, leaking the
981
+ # build host's layout into graph.json and MCP get_node (#3223).
982
+ if "definition_file" in node:
983
+ node["definition_file"] = _norm_source_file(node["definition_file"], _root)
984
+ G.add_node(node["id"], **{k: v for k, v in node.items() if k != "id"})
985
+ node_set = set(G.nodes())
986
+
987
+ # #1145 (extended): merge LLM ghost-duplicate nodes into AST canonical nodes.
988
+ # Original bug: AST uses parent-qualified IDs (mingpt_bpe_get_pairs) while LLM
989
+ # uses bare-stem IDs (bpe_get_pairs) — different IDs, same symbol.
990
+ # Original fix only caught LLM nodes with source_location=None; LLM now
991
+ # populates source_location, so those ghosts survived. Extended fix: use
992
+ # _origin=="ast" as the canonical signal. AST nodes always win; any non-AST
993
+ # node sharing (basename, label) with an AST node is a ghost.
994
+ _loc_nodes: dict[tuple[str, str], str] = {} # (source_file, label) -> canonical node id
995
+ _loc_collisions: set[tuple[str, str]] = set() # keys shared by 2+ AST nodes
996
+ _noloc_nodes: dict[tuple[str, str], str] = {} # (source_file, label) -> ghost node id
997
+ _ast_file_nodes: list[tuple[str, str]] = [] # (node_id, source_file) for AST file-self nodes (#3344)
998
+
999
+ # Pass 1: collect canonical nodes — AST-origin nodes take precedence over LLM nodes.
1000
+ # When 2+ AST nodes share a key (same-named symbols in same-named files across
1001
+ # directories, e.g. render in two index.ts), the key is ambiguous: merging a
1002
+ # ghost would pick an arbitrary winner via set-iteration order (#1257). Track
1003
+ # those keys so Pass 2 skips them — same conservatism as
1004
+ # _rewire_unique_stub_nodes, which only merges when exactly one real def exists.
1005
+ # Iterate in a deterministic (sorted) order, not set-iteration order, so the
1006
+ # canonical winner and the ambiguity decisions below don't flip run-to-run
1007
+ # with CPython's per-process string-hash seed (#1753) — the same reason the
1008
+ # edge-iteration loop further down sorts on purpose.
1009
+ for nid in sorted(node_set):
1010
+ attrs = G.nodes[nid]
1011
+ label = str(attrs.get("label", "")).strip()
1012
+ sf = str(attrs.get("source_file", ""))
1013
+ if not label or not sf:
1014
+ continue
1015
+ # Strict _origin check on purpose — NOT _is_ast_tier (#2334): existing
1016
+ # graph items are backfilled with _origin at load time, so inside a
1017
+ # build the only unstamped items are fresh SEMANTIC chunks (extract()
1018
+ # always stamps AST output). Those may carry drifted 'L<line>'
1019
+ # source_locations (the very ghosts #1145-extended collapses), and the
1020
+ # shape fallback would misread them as AST — turning two same-file LLM
1021
+ # duplicates into a fake AST/AST collision that blocks their merge.
1022
+ is_ast = attrs.get("_origin") == "ast"
1023
+ if attrs.get("source_location") or is_ast:
1024
+ # Key on the FULL normalized source_file, not the bare basename
1025
+ # (#2068): the AST/LLM ghost twins of #1145 always share the same
1026
+ # source_file (different ids, same file), so full-path keying still
1027
+ # collapses them, while unrelated same-basename nodes in DIFFERENT
1028
+ # directories (docs/a/index.md vs docs/b/index.md) now get distinct
1029
+ # keys and are never falsely merged. This subsumes the #1753/#1257
1030
+ # cross-file ambiguity guard, which is why the non-AST branch below
1031
+ # no longer needs it.
1032
+ key = (sf, label)
1033
+ if is_ast:
1034
+ # Two AST nodes on the same key (same file, same label) is an
1035
+ # ambiguous collision.
1036
+ if key in _loc_nodes and G.nodes[_loc_nodes[key]].get("_origin") == "ast":
1037
+ _loc_collisions.add(key)
1038
+ # AST-origin nodes always overwrite a prior non-AST entry.
1039
+ _loc_nodes[key] = nid
1040
+ # #3344: a self-referential semantic pass (e.g. re-extracting a
1041
+ # saved graphify-out/memory/*.md query answer) mints a NEW
1042
+ # non-AST node for every bare file path/basename it mentions in
1043
+ # prose ("App.tsx", "customer-app/index.ts"), stamped with
1044
+ # source_file = the memory doc being read, NOT the file named in
1045
+ # the prose. That wrong source_file means such a ghost can never
1046
+ # hit the (sf, label) key above — same underlying bug as #1145,
1047
+ # just with the (source_file, label) *pair* broken instead of
1048
+ # only the id. Record every AST node whose own label already
1049
+ # names its own file (_is_file_node_label — the same "is this a
1050
+ # file node" predicate the label-disambiguation pass uses) so
1051
+ # Pass 2b below can catch these by label alone.
1052
+ if _is_file_node_label(label, sf):
1053
+ _ast_file_nodes.append((nid, sf))
1054
+ else:
1055
+ # First non-AST node for this (file, label) wins as canonical; a
1056
+ # later same-key node is a genuine same-file duplicate and still
1057
+ # collapses in Pass 2.
1058
+ _loc_nodes.setdefault(key, nid)
1059
+
1060
+ # Pass 2: find ghosts — non-AST nodes that have an AST canonical twin.
1061
+ for nid in sorted(node_set):
1062
+ attrs = G.nodes[nid]
1063
+ if attrs.get("_origin") == "ast":
1064
+ continue # AST nodes are never ghosts (strict check — see Pass 1)
1065
+ label = str(attrs.get("label", "")).strip()
1066
+ sf = str(attrs.get("source_file", ""))
1067
+ if not label or not sf:
1068
+ continue
1069
+ key = (sf, label)
1070
+ if key in _loc_collisions:
1071
+ continue # ambiguous key: no safe canonical winner, leave ghost intact
1072
+ if key in _loc_nodes and _loc_nodes[key] != nid:
1073
+ _noloc_nodes[key] = nid
1074
+ # For every ghost that has an AST counterpart, record a remap.
1075
+ _ghost_remap: dict[str, str] = {} # ghost_id -> canonical_id
1076
+ for key, sem_id in _noloc_nodes.items():
1077
+ ast_id = _loc_nodes.get(key)
1078
+ if ast_id is not None:
1079
+ _ghost_remap[sem_id] = ast_id
1080
+
1081
+ # Pass 2b (#3344): catch ghosts the (source_file, label) key above cannot,
1082
+ # because their source_file is simply wrong — a semantic pass over a
1083
+ # document that only *mentions* a file (a saved graphify-out/memory/*.md
1084
+ # query answer, a README, an ADR) stamps the file's own name as a new
1085
+ # node's label but the DOCUMENT's path as source_file, since it has no way
1086
+ # to know the mentioned file's real path. Resolve these by label alone
1087
+ # against every AST file-self node collected in Pass 1, reusing
1088
+ # _is_file_node_label so "App.tsx" matches source_file ".../App.tsx" and
1089
+ # "customer-app/index.ts" matches ".../apps/customer-app/index.ts" (a
1090
+ # directory-qualified suffix, e.g. the leading "apps/" the prose dropped).
1091
+ # Conservative by construction: a label matching 0 or 2+ AST files is left
1092
+ # alone (0 = no known file, 2+ = genuinely ambiguous — same "no safe
1093
+ # canonical winner" rule Pass 2's _loc_collisions already applies).
1094
+ if _ast_file_nodes:
1095
+ for nid in sorted(node_set):
1096
+ if nid in _ghost_remap:
1097
+ continue # already resolved by the exact (sf, label) key
1098
+ attrs = G.nodes[nid]
1099
+ if attrs.get("_origin") == "ast":
1100
+ continue
1101
+ label = str(attrs.get("label", "")).strip()
1102
+ if not label:
1103
+ continue
1104
+ matches = {
1105
+ ast_id for ast_id, ast_sf in _ast_file_nodes
1106
+ if _is_file_node_label(label, ast_sf)
1107
+ }
1108
+ if len(matches) == 1:
1109
+ _ghost_remap[nid] = next(iter(matches))
1110
+
1111
+ # Remove ghost nodes from the graph; edges will be re-pointed via norm_to_id.
1112
+ for ghost_id in _ghost_remap:
1113
+ G.remove_node(ghost_id)
1114
+ node_set.discard(ghost_id)
1115
+
1116
+ # Normalized ID map: lets edges survive when the LLM generates IDs with
1117
+ # slightly different casing or punctuation than the AST extractor.
1118
+ # e.g. "Session_ValidateToken" maps to "session_validatetoken".
1119
+ norm_to_id: dict[str, str] = {_normalize_id(nid): nid for nid in node_set}
1120
+ # Also map ghost IDs to their canonical AST replacements.
1121
+ for ghost_id, canonical_id in _ghost_remap.items():
1122
+ norm_to_id[_normalize_id(ghost_id)] = canonical_id
1123
+ norm_to_id[ghost_id] = canonical_id
1124
+ # Pre-migration alias index (#1504): register each canonical node's OLD-stem id
1125
+ # forms as aliases so a stale-id edge endpoint coming from an un-re-keyed
1126
+ # fragment (e.g. an incremental update whose fragment references a symbol in a
1127
+ # file that was NOT re-extracted) still resolves to the migrated node instead
1128
+ # of dangling. Only fills gaps — never overrides a real node id.
1129
+ #
1130
+ # The old-stem form drops the extension and (for the file node itself) every
1131
+ # directory but the immediate parent, so it collapses easily: "ping.h" and
1132
+ # "ping.php" in different directories both alias to bare "ping". Collecting
1133
+ # every candidate for an alias BEFORE committing any of them — and only
1134
+ # committing when exactly one candidate claims it — keeps this a precise
1135
+ # re-keying aid instead of a silent cross-file (and cross-language) merge.
1136
+ # Without this, a dangling edge to a bare, deliberately-unscoped fallback id
1137
+ # (e.g. the C/C++ extractor's last-resort target for an #include it couldn't
1138
+ # resolve to a real path) could ride this alias onto whichever unrelated
1139
+ # same-stem file happened to be inserted first into ``node_set`` — a Python
1140
+ # set, so "first" is hash-order, not anything meaningful.
1141
+ #
1142
+ # A file node's OWN id is not always a clean ``new_stem`` prefix: when a
1143
+ # same-directory ``.h``/``.cpp`` pair collides on their shared pre-extension
1144
+ # id, _disambiguate_colliding_node_ids salts both apart into ids like
1145
+ # ``tools_aolserver_utility_h_tools_aolserver_utility`` — which no longer
1146
+ # string-prefixes cleanly for the suffix math below. Detecting "this IS the
1147
+ # file node" by label (every file node's label is its own basename,
1148
+ # regardless of id mangling) instead of by id shape keeps a salted file node
1149
+ # in the alias competition, so a genuine collision (a C header AND an
1150
+ # unrelated same-named PHP script) is still caught as ambiguous instead of
1151
+ # the header silently dropping out of the race and leaving the PHP file as
1152
+ # the lone (wrong) "unambiguous" winner.
1153
+ from graphify.extractors.base import _file_stem as _fs
1154
+ _alias_candidates: dict[str, set[str]] = {}
1155
+ for nid in node_set:
1156
+ attrs = G.nodes[nid]
1157
+ sf = attrs.get("source_file")
1158
+ if not sf:
1159
+ continue
1160
+ rel = Path(str(sf))
1161
+ if _is_abs(str(sf)):
1162
+ continue
1163
+ new_stem = make_id(_fs(rel))
1164
+ if str(attrs.get("label", "")) == rel.name:
1165
+ suffix = "" # this node IS the file, whatever its (possibly salted) id
1166
+ else:
1167
+ suffix = ""
1168
+ if _normalize_id(nid).startswith(new_stem):
1169
+ suffix = _normalize_id(nid)[len(new_stem):] # leading "_entity" or ""
1170
+ for old_stem in _old_file_stems(rel):
1171
+ if old_stem == new_stem:
1172
+ continue
1173
+ alias = old_stem + suffix
1174
+ _alias_candidates.setdefault(_normalize_id(alias), set()).add(nid)
1175
+ _alias_candidates.setdefault(alias, set()).add(nid)
1176
+ for alias_key, candidates in _alias_candidates.items():
1177
+ if len(candidates) == 1:
1178
+ norm_to_id.setdefault(alias_key, next(iter(candidates)))
1179
+ # Iterate edges in a deterministic order. The graph is undirected and stores
1180
+ # direction in _src/_tgt; when two edges collapse onto the same node pair the
1181
+ # last write wins, so an unstable iteration order flips _src/_tgt run-to-run
1182
+ # and makes the serialized graph churn. Sorting fixes the last-write outcome.
1183
+ for edge in sorted(
1184
+ extraction.get("edges", []),
1185
+ key=lambda e: (
1186
+ str(e.get("source", e.get("from", ""))),
1187
+ str(e.get("target", e.get("to", ""))),
1188
+ str(e.get("relation", "")),
1189
+ ),
1190
+ ):
1191
+ if "source" not in edge and "from" in edge:
1192
+ edge["source"] = edge["from"]
1193
+ if "target" not in edge and "to" in edge:
1194
+ edge["target"] = edge["to"]
1195
+ if "source" not in edge or "target" not in edge:
1196
+ continue
1197
+ src, tgt = edge["source"], edge["target"]
1198
+ # Skip edges with non-hashable endpoints (e.g. a list emitted by a buggy
1199
+ # LLM extraction) so the `not in node_set` membership test below never
1200
+ # raises TypeError: unhashable type. The validator already reported these.
1201
+ try:
1202
+ hash(src)
1203
+ hash(tgt)
1204
+ except TypeError:
1205
+ print(
1206
+ f"[graphify] WARNING: skipping edge with non-hashable endpoint "
1207
+ f"(source={src!r}, target={tgt!r}).",
1208
+ file=sys.stderr,
1209
+ )
1210
+ continue
1211
+ # Remap mismatched IDs via normalization before dropping the edge.
1212
+ if src not in node_set:
1213
+ src = norm_to_id.get(_normalize_id(src), src)
1214
+ if tgt not in node_set:
1215
+ tgt = norm_to_id.get(_normalize_id(tgt), tgt)
1216
+ if src not in node_set or tgt not in node_set:
1217
+ continue # skip edges to external/stdlib nodes - expected, not an error
1218
+ # `target_file` is a transient import-disambiguation salt hint (#1814)
1219
+ # with no downstream reader; it holds an absolute path, so it must never
1220
+ # be persisted. Disambiguation already pops it off fresh extractions —
1221
+ # dropping it here as well keeps a pre-fix graph's stale absolute hint
1222
+ # from surviving an incremental build_merge, which re-serializes base
1223
+ # edges through here without re-running disambiguation.
1224
+ # `local_alias` is the same shape of transient hint (#2082): it exists only
1225
+ # for the module arm of _resolve_python_member_calls to match an aliased
1226
+ # import receiver, and extract() already drops it once that pass has run.
1227
+ # Dropping it here too covers a stale pre-fix graph re-serialized through
1228
+ # an incremental build_merge, same rationale as target_file above.
1229
+ # Sanitize numeric edge fields (#1960): an explicit ``"weight": null`` in
1230
+ # the extraction JSON survives ``.get("weight", 1.0)`` (the key is present,
1231
+ # so the default never applies) and reaches Louvain/Leiden as None,
1232
+ # crashing modularity arithmetic with a TypeError (graspologic's Leiden
1233
+ # even panics on NaN). Coerce to float and fall back to the schema default
1234
+ # of 1.0 for anything the clustering backends reject — None, non-numeric
1235
+ # strings, NaN/inf, negatives — while numeric strings coerce cleanly.
1236
+ # Repair (not drop) the key so graph.json round-trips a clean value and a
1237
+ # cluster-only/--update reload never re-ingests the null.
1238
+ attrs = {k: v for k, v in edge.items() if k not in ("source", "target", "target_file", "local_alias")}
1239
+ for _num_key in ("weight", "confidence_score"):
1240
+ if _num_key in attrs:
1241
+ try:
1242
+ _num_val = float(attrs[_num_key])
1243
+ except (TypeError, ValueError):
1244
+ _num_val = 1.0
1245
+ if not math.isfinite(_num_val) or _num_val < 0:
1246
+ _num_val = 1.0
1247
+ attrs[_num_key] = _num_val
1248
+ # Backfill source_file from the endpoint nodes (every node carries one).
1249
+ # Semantic/LLM edges occasionally omit it, which downstream validation
1250
+ # flags and leaves query results with no file reference (#1279).
1251
+ if not attrs.get("source_file"):
1252
+ attrs["source_file"] = (
1253
+ G.nodes[src].get("source_file")
1254
+ or G.nodes[tgt].get("source_file")
1255
+ or ""
1256
+ )
1257
+ if "source_file" in attrs:
1258
+ attrs["source_file"] = _norm_source_file(attrs["source_file"], _root)
1259
+ if attrs.get("definition_file"):
1260
+ # Same portability rule as source_file (#3223); heals a graph
1261
+ # written before the fix on its next rebuild.
1262
+ attrs["definition_file"] = _norm_source_file(attrs["definition_file"], _root)
1263
+ # Drop cross-language phantom edges — the same short names (render, parse,
1264
+ # time, ...) recur across language boundaries, so an unresolved target can
1265
+ # bind to a same-named node in another language. The extraction spec forbids
1266
+ # this for `calls`; it is equally invalid for `imports`/`references` (a
1267
+ # Python `import time` must not bind to a `time.ts`, #1749).
1268
+ _edge_rel = attrs.get("relation")
1269
+ if _edge_rel in ("calls", "imports", "imports_from", "references"):
1270
+ src_ext = Path(G.nodes[src].get("source_file") or "").suffix.lower()
1271
+ tgt_ext = Path(G.nodes[tgt].get("source_file") or "").suffix.lower()
1272
+ src_fam = _EDGE_LANG_FAMILY.get(src_ext)
1273
+ tgt_fam = _EDGE_LANG_FAMILY.get(tgt_ext)
1274
+ if _edge_rel == "calls":
1275
+ # Unchanged #1547/#1556 behavior: only INFERRED calls, and drop as
1276
+ # soon as either family differs (an unknown ext counts as different).
1277
+ if (
1278
+ attrs.get("confidence") == "INFERRED"
1279
+ and src_ext and tgt_ext and src_fam != tgt_fam
1280
+ ):
1281
+ continue
1282
+ else:
1283
+ # imports/references: drop only when BOTH endpoints are known code
1284
+ # languages of different families, so a config->code reference
1285
+ # (unknown ext, e.g. a manifest) is never mistaken for a phantom.
1286
+ if src_fam is not None and tgt_fam is not None and src_fam != tgt_fam:
1287
+ continue
1288
+ # A file-level import or re-export cannot carry useful connectivity when
1289
+ # both endpoints resolve to the same node. This most often happens when
1290
+ # the target is an unresolved bare module name (``builtins``, ``poseidon``)
1291
+ # that the legacy-ID alias index above mistakes for the importing file's
1292
+ # own old stem. It also covers a nested module importing its parent file:
1293
+ # at file-node granularity that relationship necessarily collapses. Keep
1294
+ # other self-edges, notably recursive ``calls``, because those are real
1295
+ # program structure rather than import-resolution artifacts.
1296
+ if src == tgt and _edge_rel in ("imports", "imports_from", "re_exports"):
1297
+ continue
1298
+ # Preserve original edge direction - undirected graphs lose it otherwise,
1299
+ # causing display functions to show edges backwards.
1300
+ attrs["_src"] = src
1301
+ attrs["_tgt"] = tgt
1302
+ # When the graph is undirected and the same node pair appears twice with
1303
+ # the same relation but opposite directions (e.g. a `calls` b and b `calls` a),
1304
+ # nx.Graph collapses them into one edge. The deterministic sort above means
1305
+ # the lexicographically-later direction would systematically overwrite the
1306
+ # earlier one's _src/_tgt, silently flipping the surviving edge's caller
1307
+ # and callee. First-seen direction wins instead — drop the redundant
1308
+ # reverse-direction duplicate so the original direction is preserved (#1061).
1309
+ if not G.is_directed() and G.has_edge(src, tgt):
1310
+ existing = edge_data(G, src, tgt)
1311
+ if existing.get("relation") == attrs.get("relation") and (
1312
+ existing.get("_src") == tgt and existing.get("_tgt") == src
1313
+ ):
1314
+ continue
1315
+ # A pair that already carries a SPECIFIC relation must not be downgraded
1316
+ # to a generic one. Only one edge survives per pair here, and the sort
1317
+ # above orders same-pair edges by relation name, so "last write wins"
1318
+ # resolved the winner alphabetically — which put `references` after
1319
+ # `calls` and `uses` after everything. On graphify's own corpus that
1320
+ # rewrote all 144 pairs where the extraction found both `calls` and
1321
+ # `references` into plain `references`, and callflow's relation filter
1322
+ # does not include `references`, so those call sites left the call graph
1323
+ # entirely. Alphabetical order carries no meaning; keeping the specific
1324
+ # fact does. The reverse (specific arriving after generic) still
1325
+ # overwrites, so the outcome no longer depends on edge order at all.
1326
+ if G.has_edge(src, tgt):
1327
+ existing_rel = edge_data(G, src, tgt).get("relation")
1328
+ if (
1329
+ attrs.get("relation") in _GENERIC_RELATIONS
1330
+ and existing_rel is not None
1331
+ and existing_rel not in _GENERIC_RELATIONS
1332
+ ):
1333
+ continue
1334
+ G.add_edge(src, tgt, **attrs)
1335
+ hyperedges = extraction.get("hyperedges", [])
1336
+ if hyperedges:
1337
+ # Relativize hyperedge source_file the same way nodes and edges are
1338
+ # (above), so to_json — which has no root and writes G.graph["hyperedges"]
1339
+ # verbatim — never leaks an absolute path from a semantic subagent (#1418).
1340
+ kept_hyperedges = []
1341
+ for he in hyperedges:
1342
+ if isinstance(he, dict) and he.get("source_file"):
1343
+ he["source_file"] = _norm_source_file(he["source_file"], _root)
1344
+ # Validate members against the built node set (#1916): a hyperedge
1345
+ # member absent from the graph used to be copied into
1346
+ # G.graph["hyperedges"] verbatim and reach graph.json dangling,
1347
+ # even from a live (non-cache) extraction. Mirror the pairwise-edge
1348
+ # handling above: remap mismatched ids via normalization first,
1349
+ # then drop members that still don't resolve; drop the hyperedge
1350
+ # itself when no valid member remains (single-member hyperedges
1351
+ # are legal in this codebase, e.g. a per-file flow, so we prune
1352
+ # rather than require two survivors).
1353
+ if isinstance(he, dict) and isinstance(he.get("nodes"), list):
1354
+ valid_members = []
1355
+ for m in he["nodes"]:
1356
+ try:
1357
+ hash(m)
1358
+ except TypeError:
1359
+ continue
1360
+ if m not in node_set and isinstance(m, str):
1361
+ m = norm_to_id.get(_normalize_id(m), m)
1362
+ if m in node_set:
1363
+ valid_members.append(m)
1364
+ if not valid_members:
1365
+ print(
1366
+ f"[graphify] WARNING: dropping hyperedge "
1367
+ f"{he.get('id', '?')!r} — none of its members "
1368
+ f"{he.get('nodes')!r} match built nodes.",
1369
+ file=sys.stderr,
1370
+ )
1371
+ continue
1372
+ if valid_members != he["nodes"]:
1373
+ he["nodes"] = valid_members
1374
+ kept_hyperedges.append(he)
1375
+ if kept_hyperedges:
1376
+ G.graph["hyperedges"] = kept_hyperedges
1377
+ else:
1378
+ # Full wipeout (#2485): every incoming hyperedge failed member
1379
+ # revalidation. Store an EXPLICIT empty list — distinct from
1380
+ # "this graph never carried hyperedge metadata" — and say loudly
1381
+ # that the persisted set is about to be emptied, so the per-edge
1382
+ # warnings above can't scroll past unnoticed.
1383
+ G.graph["hyperedges"] = []
1384
+ print(
1385
+ f"[graphify] WARNING: all {len(hyperedges)} hyperedge(s) were "
1386
+ f"dropped by member revalidation; graph.json's hyperedge set "
1387
+ f"will be emptied on the next export.",
1388
+ file=sys.stderr,
1389
+ )
1390
+ # Runs LAST, after the alias-competition above (which relies on file-node
1391
+ # labels still being bare basenames): give colliding-basename file nodes a
1392
+ # directory-qualified display label so lookup/discovery can disambiguate
1393
+ # them (#2032). Labels only — ids and edges are untouched.
1394
+ _disambiguate_file_node_labels(G)
1395
+ return G
1396
+
1397
+
1398
+ def build(
1399
+ extractions: list[dict],
1400
+ *,
1401
+ directed: bool = False,
1402
+ dedup: bool = True,
1403
+ dedup_llm_backend: str | None = None,
1404
+ root: str | Path | None = None,
1405
+ protected_ids: "set[str] | None" = None,
1406
+ ) -> nx.Graph:
1407
+ """Merge multiple extraction results into one graph.
1408
+
1409
+ directed=True produces a DiGraph that preserves edge direction (source→target).
1410
+ directed=False (default) produces an undirected Graph for backward compatibility.
1411
+ dedup=True (default) runs entity deduplication before building the graph.
1412
+ dedup_llm_backend: if set (e.g. "gemini", "claude", or "kimi"), uses LLM to resolve
1413
+ ambiguous pairs in the 75–92 Jaro-Winkler score zone.
1414
+ root: if given, absolute source_file paths are made relative to root (#932).
1415
+ protected_ids: optional set of node IDs to protect from being collapsed with
1416
+ other protected nodes during incremental merge (#3477).
1417
+
1418
+ With dedup disabled, extractions are merged in order and the last node's
1419
+ attributes win (NetworkX add_node overwrites). With dedup enabled, nodes
1420
+ sharing an ID use a deterministic survivor and retain missing attributes
1421
+ from duplicate records of the same source entity. Genuine cross-file ID
1422
+ collisions remain isolated and are reported.
1423
+ """
1424
+ from graphify.dedup import deduplicate_entities
1425
+ combined: dict = {"nodes": [], "edges": [], "hyperedges": [], "input_tokens": 0, "output_tokens": 0}
1426
+ for ext in extractions:
1427
+ combined["nodes"].extend(ext.get("nodes", []))
1428
+ combined["edges"].extend(ext.get("edges", []))
1429
+ combined["hyperedges"].extend(ext.get("hyperedges", []))
1430
+ combined["input_tokens"] += ext.get("input_tokens", 0)
1431
+ combined["output_tokens"] += ext.get("output_tokens", 0)
1432
+ _root = str(Path(root).resolve()) if root else None
1433
+ if dedup and combined["nodes"]:
1434
+ # Numeric ids must be str before dedup, which keys on them and would
1435
+ # raise TypeError in _pick_winner's regex search (#2326). build_from_json
1436
+ # coerces too, but that runs after dedup — too late for this path.
1437
+ _coerce_non_string_ids(combined)
1438
+ # Fold legacy node field aliases before dedup (#2194): dedup runs BEFORE
1439
+ # build_from_json and keys on `label`, so a `name`/`path` alias node
1440
+ # would be invisible to it and only label-dedup one build later, after
1441
+ # build_from_json's own fold has healed the persisted graph.json.
1442
+ for n in combined["nodes"]:
1443
+ if isinstance(n, dict):
1444
+ _fold_node_aliases(n)
1445
+ # Normalize source_file and definition_file to the build root before
1446
+ # deduplication (#3472), so exact-ID collision checks and same-file
1447
+ # attribute merging operate on canonical repo-relative paths rather
1448
+ # than false-flagging absolute paths from semantic subagents as
1449
+ # different files.
1450
+ if "source_file" in n:
1451
+ n["source_file"] = _norm_source_file(n["source_file"], _root)
1452
+ if "definition_file" in n:
1453
+ n["definition_file"] = _norm_source_file(n["definition_file"], _root)
1454
+ combined["nodes"], combined["edges"] = deduplicate_entities(
1455
+ combined["nodes"], combined["edges"], communities={},
1456
+ dedup_llm_backend=dedup_llm_backend, root=_root,
1457
+ # Hyperedge members reference node ids too, so they need the same
1458
+ # survivor rewiring the edges get (#2805).
1459
+ hyperedges=combined.get("hyperedges"),
1460
+ protected_ids=protected_ids,
1461
+ )
1462
+ return build_from_json(combined, directed=directed, root=_root)
1463
+
1464
+
1465
+ def _norm_label(label: str | None) -> str:
1466
+ """Canonical dedup key — Unicode-aware, preserves CJK/word characters."""
1467
+ if not isinstance(label, str):
1468
+ label = "" if label is None else str(label)
1469
+ label = unicodedata.normalize("NFKC", label)
1470
+ return re.sub(r"[\W_ ]+", " ", label.casefold(), flags=re.UNICODE).strip()
1471
+
1472
+
1473
+ def deduplicate_by_label(nodes: list[dict], edges: list[dict]) -> tuple[list[dict], list[dict]]:
1474
+ """Merge nodes that share a normalised label, rewriting edge references.
1475
+
1476
+ Prefers IDs without chunk suffixes (_c\\d+) and shorter IDs when tied.
1477
+ Drops self-loops created by the merge.
1478
+
1479
+ Dormant: this is NOT wired into ``build()`` — the active dedup path is
1480
+ ``deduplicate_entities`` (imported and called in ``build``), which supersedes
1481
+ it. The previous "Called in build() automatically" note was never true. It
1482
+ also merges by label alone with no ``file_type`` guard, so it must not be
1483
+ enabled for code nodes: same-label symbols from different files/packages
1484
+ (e.g. two ``Account`` types) would collapse into one — the cross-file
1485
+ conflation ``deduplicate_entities`` deliberately avoids for code (#1205).
1486
+ """
1487
+ _CHUNK_SUFFIX = re.compile(r"_c\d+$")
1488
+ canonical: dict[str, dict] = {} # norm_label -> surviving node
1489
+ remap: dict[str, str] = {} # old_id -> surviving_id
1490
+
1491
+ for node in nodes:
1492
+ key = _norm_label(node.get("label", node.get("id", "")))
1493
+ if not key:
1494
+ continue
1495
+ existing = canonical.get(key)
1496
+ if existing is None:
1497
+ canonical[key] = node
1498
+ else:
1499
+ has_suffix = bool(_CHUNK_SUFFIX.search(node["id"]))
1500
+ existing_has_suffix = bool(_CHUNK_SUFFIX.search(existing["id"]))
1501
+ if has_suffix and not existing_has_suffix:
1502
+ remap[node["id"]] = existing["id"]
1503
+ elif existing_has_suffix and not has_suffix:
1504
+ remap[existing["id"]] = node["id"]
1505
+ canonical[key] = node
1506
+ elif len(node["id"]) < len(existing["id"]):
1507
+ remap[existing["id"]] = node["id"]
1508
+ canonical[key] = node
1509
+ else:
1510
+ remap[node["id"]] = existing["id"]
1511
+
1512
+ if not remap:
1513
+ return nodes, edges
1514
+
1515
+ print(f"[graphify] Deduplicated {len(remap)} duplicate node(s) by label.", file=sys.stderr)
1516
+ deduped_nodes = list(canonical.values())
1517
+ deduped_edges = []
1518
+ for edge in edges:
1519
+ e = dict(edge)
1520
+ e["source"] = remap.get(e["source"], e["source"])
1521
+ e["target"] = remap.get(e["target"], e["target"])
1522
+ if e["source"] != e["target"]:
1523
+ deduped_edges.append(e)
1524
+ return deduped_nodes, deduped_edges
1525
+
1526
+
1527
+ def _load_existing_graph(graph_path: Path) -> "tuple[list, list, list, bool] | None":
1528
+ """Load (nodes, edges, hyperedges, directed) from an existing graph.json for
1529
+ an incremental merge, accepting both the ``links`` and ``edges`` spellings.
1530
+
1531
+ Reads the JSON directly instead of going through node_link_graph().
1532
+ The latter rebuilds an undirected nx.Graph and then enumerating
1533
+ edges() yields endpoints based on node insertion order, which
1534
+ silently flips directional edges (e.g. `calls`) when the callee
1535
+ was inserted before the caller. The _src/_tgt direction-preserving
1536
+ attrs are popped before saving in export.py, so going through the
1537
+ NetworkX round-trip loses direction permanently (#760).
1538
+
1539
+ Returns None when the file does not exist. Raises RuntimeError when it
1540
+ exists but cannot be parsed — callers must refuse to overwrite rather
1541
+ than silently replace a possibly-recoverable graph.
1542
+ """
1543
+ if not graph_path.exists():
1544
+ return None
1545
+ from graphify.security import check_graph_file_size_cap
1546
+ check_graph_file_size_cap(graph_path)
1547
+ try:
1548
+ data = json.loads(graph_path.read_text(encoding="utf-8"))
1549
+ except (json.JSONDecodeError, OSError) as exc:
1550
+ raise RuntimeError(
1551
+ f"Cannot read {graph_path} for incremental merge: {exc}. "
1552
+ "Delete the file and run a full rebuild."
1553
+ ) from exc
1554
+ links_key = "links" if "links" in data else "edges"
1555
+ nodes = list(data.get("nodes", []))
1556
+ edges = list(data.get(links_key, []))
1557
+ # Backfill tier provenance on legacy items (#2334): _origin is stamped at
1558
+ # extraction time only (extract.py for AST, and the semantic path never
1559
+ # stamps), so pre-0.9.16 graphs and externally-merged fragments carry
1560
+ # unstamped items. Stamp them via the _is_ast_tier shape fallback so the
1561
+ # graph self-heals on the next write and every downstream tier decision
1562
+ # (build_merge replace, watch reconcile) reads an explicit marker.
1563
+ for item in nodes:
1564
+ if isinstance(item, dict):
1565
+ item.setdefault("_origin", "ast" if _is_ast_tier(item) else "semantic")
1566
+ for item in edges:
1567
+ if isinstance(item, dict):
1568
+ item.setdefault("_origin", "ast" if _is_ast_tier(item) else "semantic")
1569
+ return (
1570
+ nodes,
1571
+ edges,
1572
+ list(data.get("hyperedges", [])),
1573
+ bool(data.get("directed", False)),
1574
+ )
1575
+
1576
+
1577
+ def _tier_replacement_sources(
1578
+ chunks: "Iterable[dict]",
1579
+ root: "str | Path | None" = None,
1580
+ ast_sources: "Iterable[str | Path] | None" = None,
1581
+ ) -> tuple[set[str], set[str]]:
1582
+ """Compute (new_ast_sources, new_sem_sources) for tier-scoped replacement.
1583
+
1584
+ #3411: AST replacement ownership is derived from explicit extraction
1585
+ provenance (ast_sources or chunk-level "extracted_sources"), so cross-file
1586
+ stub nodes emitted by extractors (e.g. .sln project stubs or ProjectReference
1587
+ stubs) do not pollute the replacement set and wipe the target project's
1588
+ nodes/edges. Falls back to AST node source_file only when no provenance is
1589
+ provided.
1590
+ """
1591
+ explicit_ast_sources: set[str] = set()
1592
+ if ast_sources is not None:
1593
+ for s in ast_sources:
1594
+ if s:
1595
+ explicit_ast_sources.add(str(s))
1596
+ for ch in chunks:
1597
+ if isinstance(ch, dict):
1598
+ for s in (ch.get("extracted_sources") or []):
1599
+ if s:
1600
+ explicit_ast_sources.add(str(s))
1601
+
1602
+ new_ast_sources: set[str] = set()
1603
+ new_sem_sources: set[str] = set()
1604
+
1605
+ if explicit_ast_sources:
1606
+ for sf in explicit_ast_sources:
1607
+ new_ast_sources.add(sf)
1608
+ norm = _norm_source_file(sf, root)
1609
+ if norm:
1610
+ new_ast_sources.add(norm)
1611
+ else:
1612
+ for ch in chunks:
1613
+ if not isinstance(ch, dict):
1614
+ continue
1615
+ for n in ch.get("nodes", []):
1616
+ if not isinstance(n, dict):
1617
+ continue
1618
+ sf = n.get("source_file")
1619
+ if not sf or not _is_ast_tier(n):
1620
+ continue
1621
+ new_ast_sources.add(sf)
1622
+ norm = _norm_source_file(sf, root)
1623
+ if norm:
1624
+ new_ast_sources.add(norm)
1625
+
1626
+ for ch in chunks:
1627
+ if not isinstance(ch, dict):
1628
+ continue
1629
+ for n in ch.get("nodes", []):
1630
+ if not isinstance(n, dict):
1631
+ continue
1632
+ sf = n.get("source_file")
1633
+ if not sf or _is_ast_tier(n):
1634
+ continue
1635
+ new_sem_sources.add(sf)
1636
+ norm = _norm_source_file(sf, root)
1637
+ if norm:
1638
+ new_sem_sources.add(norm)
1639
+
1640
+ return new_ast_sources, new_sem_sources
1641
+
1642
+
1643
+ def merge_raw_extraction(
1644
+ new: dict,
1645
+ graph_path: str | Path,
1646
+ prune_sources: "list[str] | None" = None,
1647
+ root: "str | Path | None" = None,
1648
+ *,
1649
+ ast_sources: "Iterable[str | Path] | None" = None,
1650
+ ) -> dict:
1651
+ """Merge the existing raw graph.json forward into a fresh raw extraction
1652
+ (the ``extract --no-cluster`` incremental path, #2169).
1653
+
1654
+ Replace/prune semantics mirror :func:`build_merge` exactly, so the raw and
1655
+ clustered incremental paths can't drift:
1656
+
1657
+ - sources re-extracted this run REPLACE their prior contribution PER TIER
1658
+ (#2333/#2336, #3411): existing nodes/edges/hyperedges owned by them are dropped
1659
+ only when the new extraction contains the same tier (AST vs semantic,
1660
+ per :func:`_is_ast_tier`) for that source, matched in both raw and
1661
+ :func:`_norm_source_file` form (#1007). AST replacement ownership is derived
1662
+ from explicit extraction provenance (``ast_sources`` or chunk-level
1663
+ ``extracted_sources``), falling back to AST node source_file only when no
1664
+ provenance is provided (#3411);
1665
+ - ``prune_sources`` (deleted / excluded / graph-stale files) are dropped,
1666
+ with the ``_abs_identity`` third-form fallback (#2012), and "replace" wins
1667
+ over a contradictory "delete" of a re-extracted source (#1796);
1668
+ - everything else — nodes/edges/hyperedges owned by unchanged files — is
1669
+ carried forward unchanged.
1670
+
1671
+ Survivors are PREPENDED to ``new``'s lists (existing-first), so the caller's
1672
+ ``dedupe_nodes`` last-writer-wins keeps fresh attributes for re-extracted
1673
+ nodes while ``dedupe_edges`` first-wins never resurrects a replaced edge
1674
+ (replaced sources' edges were already dropped above). Token counters and
1675
+ every other key of ``new`` are left untouched. Returns ``new``, mutated in
1676
+ place. Raises RuntimeError (via :func:`_load_existing_graph`) when the
1677
+ existing graph is present but unparseable — the caller must refuse to
1678
+ overwrite it. No-op when ``graph_path`` does not exist.
1679
+ """
1680
+ graph_path = Path(graph_path)
1681
+ loaded = _load_existing_graph(graph_path)
1682
+ if loaded is None:
1683
+ return new
1684
+ existing_nodes, existing_edges, existing_hyperedges, _ = loaded
1685
+
1686
+ _eff_root = (
1687
+ str(Path(root).resolve()) if root is not None
1688
+ else _infer_merge_root(graph_path)
1689
+ )
1690
+
1691
+ # Tier-scoped replace, mirroring build_merge (#2333/#2336, COEXIST, #3411): a
1692
+ # source re-extracted this run replaces only the tier(s) actually present
1693
+ # in the new extraction, so an AST-only re-extract keeps the file's
1694
+ # semantic layer and vice versa. AST replacement ownership is derived from
1695
+ # explicit extraction provenance (ast_sources or "extracted_sources"),
1696
+ # falling back to AST node source_files only when no provenance is provided.
1697
+ new_ast_sources, new_sem_sources = _tier_replacement_sources(
1698
+ [new], root=_eff_root, ast_sources=ast_sources
1699
+ )
1700
+ new_sources: set[str] = new_ast_sources | new_sem_sources
1701
+
1702
+ # "Replace" wins over a contradictory "delete" of the same source (#1796),
1703
+ # in both string and absolute-identity space (#2012) — as in build_merge.
1704
+ prune_set, prune_abs = _build_prune_sets(prune_sources, _eff_root, new_sources)
1705
+ _prune_root = _eff_root
1706
+
1707
+ def _prune_hit(sf: "str | None") -> bool:
1708
+ if not sf:
1709
+ return False
1710
+ if sf in prune_set:
1711
+ return True
1712
+ norm = _norm_source_file(sf, _prune_root)
1713
+ if norm and norm in prune_set:
1714
+ return True
1715
+ a = _abs_identity(sf, _prune_root)
1716
+ return bool(a) and a in prune_abs
1717
+
1718
+ # #2446: when NONE of the prune entries match anything stored, the guessed
1719
+ # _eff_root is usually wrong (non-standard layout, no marker) and every
1720
+ # prune silently no-ops. Derive the root by suffix-matching the absolute
1721
+ # prune paths against the stored relative source_files and retry — same
1722
+ # fallback as build_merge.
1723
+ if prune_set or prune_abs:
1724
+ _stored_sfs = {
1725
+ item.get("source_file")
1726
+ for seq in (existing_nodes, existing_edges, existing_hyperedges)
1727
+ for item in seq if isinstance(item, dict)
1728
+ }
1729
+ _stored_sfs.discard(None)
1730
+ if not any(_prune_hit(sf) for sf in _stored_sfs):
1731
+ _derived = _derive_prune_root(prune_sources or [], _stored_sfs)
1732
+ if _derived is not None and _derived != _prune_root:
1733
+ _prune_root = _derived
1734
+ prune_set, prune_abs = _build_prune_sets(
1735
+ prune_sources, _prune_root, new_sources
1736
+ )
1737
+
1738
+ def _dropped(item: dict) -> bool:
1739
+ if not isinstance(item, dict):
1740
+ return True
1741
+ sf = item.get("source_file")
1742
+ # Tier-scoped replace: an item is superseded only when ITS OWN tier
1743
+ # re-extracted its source. Hyperedges are semantic-tier (no _origin,
1744
+ # null source_location), so an AST-only re-extract carries them.
1745
+ # Deletion pruning below stays tier-blind.
1746
+ own = new_ast_sources if _is_ast_tier(item) else new_sem_sources
1747
+ if sf in own or _norm_source_file(sf, _eff_root) in own:
1748
+ return True # re-extracted this run — replaced by the new chunk
1749
+ if not sf:
1750
+ return False # unowned — carry forward
1751
+ return _prune_hit(sf)
1752
+
1753
+ # #3203: Check for unverified semantic shrink on re-extracted sources.
1754
+ unverified_semantic_shrink: dict[str, tuple[int, int]] = {}
1755
+ if new_sem_sources:
1756
+ prior_sem_counts: dict[str, int] = {}
1757
+ for n in existing_nodes:
1758
+ if isinstance(n, dict) and not _is_ast_tier(n):
1759
+ sf = n.get("source_file")
1760
+ if sf:
1761
+ canon = _norm_source_file(sf, _eff_root) or sf
1762
+ prior_sem_counts[canon] = prior_sem_counts.get(canon, 0) + 1
1763
+
1764
+ fresh_sem_counts: dict[str, int] = {}
1765
+ for n in new.get("nodes", []):
1766
+ if isinstance(n, dict) and not _is_ast_tier(n):
1767
+ sf = n.get("source_file")
1768
+ if sf:
1769
+ canon = _norm_source_file(sf, _eff_root) or sf
1770
+ fresh_sem_counts[canon] = fresh_sem_counts.get(canon, 0) + 1
1771
+
1772
+ for canon_sf, fresh_count in fresh_sem_counts.items():
1773
+ prior_count = prior_sem_counts.get(canon_sf, 0)
1774
+ if prior_count > 1 and fresh_count < prior_count:
1775
+ unverified_semantic_shrink[canon_sf] = (prior_count, fresh_count)
1776
+
1777
+ new["nodes"] = [n for n in existing_nodes if not _dropped(n)] + list(new.get("nodes", []))
1778
+ new["edges"] = [e for e in existing_edges if not _dropped(e)] + list(new.get("edges", []))
1779
+ carried_hyper = [he for he in existing_hyperedges if not _dropped(he)]
1780
+ if carried_hyper or new.get("hyperedges"):
1781
+ new["hyperedges"] = carried_hyper + list(new.get("hyperedges", []))
1782
+ if unverified_semantic_shrink:
1783
+ new["_unverified_semantic_shrink"] = unverified_semantic_shrink
1784
+ return new
1785
+
1786
+
1787
+ def build_merge(
1788
+ new_chunks: list[dict],
1789
+ graph_path: str | Path | None = None,
1790
+ prune_sources: list[str] | None = None,
1791
+ *,
1792
+ directed: bool | None = None,
1793
+ dedup: bool = True,
1794
+ dedup_llm_backend: str | None = None,
1795
+ root: str | Path | None = None,
1796
+ ast_sources: "Iterable[str | Path] | None" = None,
1797
+ ) -> nx.Graph:
1798
+ """Load existing graph.json and return it merged with ``new_chunks``.
1799
+
1800
+ Does NOT write to disk — the caller persists the result, e.g. via
1801
+ ``export.to_json(G, communities, graph_path, force=True)`` after
1802
+ clustering. ``graph_path`` is read-only here.
1803
+
1804
+ Re-extracted files REPLACE their prior contribution per tier (#2333/#2336, #3411):
1805
+ a source_file present in new_chunks has its existing nodes/edges dropped
1806
+ for each tier (AST vs semantic, per :func:`_is_ast_tier`) the new chunks
1807
+ actually contain, so a changed file's stale nodes/edges don't accumulate
1808
+ while a one-tier re-extract keeps the other tier's layer intact. AST replacement
1809
+ ownership is derived from explicit extraction provenance (``ast_sources`` or
1810
+ chunk-level ``extracted_sources``), falling back to AST node source_file only
1811
+ when no provenance is provided (#3411). Files absent from new_chunks are
1812
+ preserved unchanged; deleted files are removed via prune_sources (tier-blind).
1813
+ Safe to call repeatedly.
1814
+ root: if given, absolute source_file paths in new_chunks are made relative (#932).
1815
+ directed: if None (default), honor the on-disk graph's own ``directed`` flag
1816
+ when one exists, so an incremental merge can't silently flip a directed
1817
+ graph undirected (#2342). Falls back to False when there is no existing
1818
+ graph to inherit from. An explicit True/False always overrides the on-disk
1819
+ flag.
1820
+ """
1821
+ # Iterated more than once below (source sets, the hyperedge carry, the
1822
+ # build itself), so a one-shot iterator must be materialised first.
1823
+ new_chunks = list(new_chunks)
1824
+ graph_path = Path(graph_path if graph_path is not None else _default_graph_json())
1825
+ _loaded = _load_existing_graph(graph_path)
1826
+ if _loaded is not None:
1827
+ existing_nodes, existing_edges, existing_hyperedges, existing_directed = _loaded
1828
+ had_graph = True
1829
+ else:
1830
+ existing_nodes = []
1831
+ existing_edges = []
1832
+ existing_hyperedges = []
1833
+ existing_directed = False
1834
+ had_graph = False
1835
+ if directed is None:
1836
+ directed = existing_directed if had_graph else False
1837
+
1838
+ # Effective root for relativizing absolute source_file / prune paths back to the
1839
+ # stored relative source_file keys. When the caller passes root we use it;
1840
+ # otherwise fall back to the graph's recorded scan root, so absolute
1841
+ # prune_sources and new-chunk paths still match even when a caller omits root
1842
+ # (#1571 — the skill's --update runbook calls build_merge without root, so
1843
+ # absolute deleted-file paths never matched the relative node keys and their
1844
+ # nodes survived as ghosts).
1845
+ _eff_root = (
1846
+ str(Path(root).resolve()) if root is not None
1847
+ else _infer_merge_root(graph_path)
1848
+ )
1849
+
1850
+ # Re-extracted files REPLACE their prior contribution. Every source_file
1851
+ # present in new_chunks is dropped from the loaded base before merging, so a
1852
+ # CHANGED file's stale nodes/edges don't accumulate across incremental
1853
+ # updates. Without this, build() merges old+new for the same file and only
1854
+ # exact-duplicate edges collapse — edges/nodes that disappeared from the new
1855
+ # version survive forever. Brand-new files aren't in base, so this is a no-op
1856
+ # for them; genuinely deleted files are still handled via prune_sources.
1857
+ # Matched in both raw and _norm_source_file form because new_chunks may carry
1858
+ # absolute win32 paths while the stored graph keeps relative posix (#1007).
1859
+ # Replacement is tier-scoped (#2333/#2336, COEXIST): each file has two
1860
+ # producers — the deterministic AST pass and the semantic/LLM pass — whose
1861
+ # node sets coexist in the graph. A re-extract of one tier must replace
1862
+ # only that tier's prior contribution, never the other's (a semantic-only
1863
+ # chunk used to delete the file's AST headings). Which tier a NEW chunk
1864
+ # item belongs to is read via _is_ast_tier (existing items were stamped by
1865
+ # _load_existing_graph above).
1866
+ #
1867
+ # #3411: AST replacement ownership is derived from explicit extraction
1868
+ # provenance (ast_sources or chunk-level "extracted_sources"), so cross-file
1869
+ # stub nodes emitted by extractors (e.g. .sln project stubs or ProjectReference
1870
+ # stubs) do not pollute the replacement set and wipe the target project's
1871
+ # nodes/edges. Falls back to AST node source_file only when no provenance is
1872
+ # provided.
1873
+ _replace_root = _eff_root
1874
+ new_ast_sources, new_sem_sources = _tier_replacement_sources(
1875
+ new_chunks, root=_replace_root, ast_sources=ast_sources
1876
+ )
1877
+ new_sources: set[str] = new_ast_sources | new_sem_sources
1878
+ # True on-disk baseline for the #479 shrink accounting at the end (#2497):
1879
+ # the rebind below removes the re-extracted sources' old nodes from
1880
+ # existing_nodes, so any later size comparison against the rebound list can
1881
+ # never see the loss it is meant to catch.
1882
+ _disk_nodes = existing_nodes
1883
+ _disk_n = len(existing_nodes)
1884
+
1885
+ # #3203: Check for unverified semantic shrink on re-extracted sources.
1886
+ # An existing source with prior semantic nodes (> 1) that produces strictly
1887
+ # fewer semantic nodes in this extraction is flagged on G.graph so the CLI
1888
+ # can arm the shrink guard and leave the source unstamped in the manifest.
1889
+ unverified_semantic_shrink: dict[str, tuple[int, int]] = {}
1890
+ if had_graph and new_sem_sources:
1891
+ prior_sem_counts: dict[str, int] = {}
1892
+ for n in _disk_nodes:
1893
+ if isinstance(n, dict) and not _is_ast_tier(n):
1894
+ sf = n.get("source_file")
1895
+ if sf:
1896
+ canon = _norm_source_file(sf, _replace_root) or sf
1897
+ prior_sem_counts[canon] = prior_sem_counts.get(canon, 0) + 1
1898
+
1899
+ fresh_sem_counts: dict[str, int] = {}
1900
+ for ch in new_chunks:
1901
+ for n in ch.get("nodes", []):
1902
+ if isinstance(n, dict) and not _is_ast_tier(n):
1903
+ sf = n.get("source_file")
1904
+ if sf:
1905
+ canon = _norm_source_file(sf, _replace_root) or sf
1906
+ fresh_sem_counts[canon] = fresh_sem_counts.get(canon, 0) + 1
1907
+
1908
+ for canon_sf, fresh_count in fresh_sem_counts.items():
1909
+ prior_count = prior_sem_counts.get(canon_sf, 0)
1910
+ if prior_count > 1 and fresh_count < prior_count:
1911
+ unverified_semantic_shrink[canon_sf] = (prior_count, fresh_count)
1912
+
1913
+ if new_sources:
1914
+ def _kept(item: dict) -> bool:
1915
+ sf = item.get("source_file")
1916
+ own = new_ast_sources if _is_ast_tier(item) else new_sem_sources
1917
+ return sf not in own and _norm_source_file(sf, _replace_root) not in own
1918
+ existing_nodes = [n for n in existing_nodes if _kept(n)]
1919
+ existing_edges = [e for e in existing_edges if _kept(e)]
1920
+ replaced_n = _disk_n - len(existing_nodes)
1921
+ if replaced_n:
1922
+ print(
1923
+ f"[graphify] Replaced {replaced_n} node(s) from re-extracted "
1924
+ f"source file(s).",
1925
+ file=sys.stderr,
1926
+ )
1927
+
1928
+ # Prune set for deleted source files — both the raw form (matches nodes that
1929
+ # kept absolute source_file) and the normalised relative form (matches nodes
1930
+ # relativised by _norm_source_file at build time). .resolve() (via _eff_root)
1931
+ # handles symlinked roots and ".." / "./" segments so Path.relative_to()
1932
+ # succeeds even when the scan root is a symlink. (#1007, #1571)
1933
+ #
1934
+ # A file that was just re-extracted (present in new_chunks) is being REPLACED,
1935
+ # never deleted — so never prune it, even if the caller also lists it in
1936
+ # prune_sources. Otherwise its fresh, just-built nodes are silently removed
1937
+ # (data loss): common when an edit keeps a node's label and the caller follows
1938
+ # the old edit-workflow of passing the changed file in prune_sources (#1796).
1939
+ # "replace" wins over a contradictory "delete" of the same source. Applied in
1940
+ # both string and absolute-identity space so the third-form fallback below
1941
+ # can't resurrect the delete for a re-extracted file (#2012).
1942
+ _prune_root = _eff_root
1943
+ prune_set, prune_abs = _build_prune_sets(prune_sources, _prune_root, new_sources)
1944
+ _matched_prune_entries: set[str] = set()
1945
+
1946
+ def _prune_match(sf: "str | None") -> bool:
1947
+ # Match a node/edge/hyperedge source_file against the prune set in a
1948
+ # form-insensitive way: exact string, normalised-relative, then the
1949
+ # absolute-identity fallback for the third-form case (#2012). Records
1950
+ # WHICH prune entry matched, so the prune report can count only the
1951
+ # entries that actually hit something (#2446).
1952
+ if not sf:
1953
+ return False
1954
+ hit = prune_set.get(sf)
1955
+ if hit is None:
1956
+ norm = _norm_source_file(sf, _prune_root)
1957
+ if norm:
1958
+ hit = prune_set.get(norm)
1959
+ if hit is None:
1960
+ a = _abs_identity(sf, _prune_root)
1961
+ if a:
1962
+ hit = prune_abs.get(a)
1963
+ if hit is None:
1964
+ return False
1965
+ _matched_prune_entries.add(hit)
1966
+ return True
1967
+
1968
+ # #2446: when NONE of the prune entries match anything stored, the guessed
1969
+ # _eff_root is usually wrong (non-standard layout, no marker) and every
1970
+ # prune would silently no-op. Derive the root by suffix-matching the
1971
+ # absolute prune paths against the stored relative source_files and redo
1972
+ # the prune sets with it; on ambiguity fall through to the zero-match
1973
+ # warning below. Runs before the hyperedge carry so hyperedge pruning
1974
+ # benefits too.
1975
+ if prune_set or prune_abs:
1976
+ _stored_sfs = {
1977
+ item.get("source_file")
1978
+ for seq in (_disk_nodes, existing_edges, existing_hyperedges)
1979
+ for item in seq if isinstance(item, dict)
1980
+ }
1981
+ _stored_sfs.discard(None)
1982
+ if not any(_prune_match(sf) for sf in _stored_sfs):
1983
+ _derived = _derive_prune_root(prune_sources or [], _stored_sfs)
1984
+ if _derived is not None and _derived != _prune_root:
1985
+ _prune_root = _derived
1986
+ prune_set, prune_abs = _build_prune_sets(
1987
+ prune_sources, _prune_root, new_sources
1988
+ )
1989
+
1990
+ # Carry forward hyperedges from files that were neither re-extracted nor
1991
+ # deleted (#1574). build() only sees the new chunks' hyperedges, so without
1992
+ # this every --update collapses the graph's hyperedge set down to just the
1993
+ # changed files'. Re-extracted files' prior hyperedges are dropped (their new
1994
+ # version is already in the new chunks — replace-per-source, like
1995
+ # nodes/edges); deleted files' are dropped via prune_set; id-dedup so a
1996
+ # carried hyperedge never duplicates one the new chunks re-emitted. Mirrors
1997
+ # watch.py, which already preserves existing hyperedges across a rebuild.
1998
+ #
1999
+ # The carried set rides INTO build() on the base chunk rather than being
2000
+ # attached to G afterwards (#3102): entity dedup rewires every edge endpoint
2001
+ # and every hyperedge member it sees onto the survivor (#2805), but a
2002
+ # hyperedge attached after the fact kept naming the merged-away node — a
2003
+ # dangling member with no backing node in the written graph.
2004
+ carried_hyperedges: list[dict] = []
2005
+ if existing_hyperedges:
2006
+ carried = carried_hyperedges
2007
+ _new_hyperedge_ids = {
2008
+ he.get("id")
2009
+ for chunk in new_chunks
2010
+ for he in (chunk.get("hyperedges") or [])
2011
+ if isinstance(he, dict) and he.get("id")
2012
+ }
2013
+ for he in existing_hyperedges:
2014
+ if not isinstance(he, dict):
2015
+ continue
2016
+ sf = he.get("source_file")
2017
+ norm = _norm_source_file(sf, _eff_root)
2018
+ # Hyperedges are semantic-tier: only a SEMANTIC re-extract of the
2019
+ # source replaces them. An AST-only re-extract cannot regenerate
2020
+ # hyperedges, so dropping them there would be data loss (#2336).
2021
+ if sf in new_sem_sources or norm in new_sem_sources:
2022
+ continue # semantically re-extracted — replaced by the new chunk's version
2023
+ if _prune_match(sf):
2024
+ continue # deleted — pruned
2025
+ if he.get("id") and he.get("id") in _new_hyperedge_ids:
2026
+ continue # the new chunks re-emitted it — theirs wins
2027
+ carried.append(he)
2028
+
2029
+ base = (
2030
+ [{"nodes": existing_nodes, "edges": existing_edges, "hyperedges": carried_hyperedges}]
2031
+ if had_graph else []
2032
+ )
2033
+
2034
+ # Untouched existing nodes must not be collapsed with each other during dedup (#3477).
2035
+ _protected_ids = {
2036
+ n["id"] for n in existing_nodes
2037
+ if isinstance(n, dict) and n.get("id")
2038
+ } if had_graph else None
2039
+
2040
+ all_chunks = base + list(new_chunks)
2041
+ G = build(
2042
+ all_chunks,
2043
+ directed=directed,
2044
+ dedup=dedup,
2045
+ dedup_llm_backend=dedup_llm_backend,
2046
+ root=_eff_root,
2047
+ protected_ids=_protected_ids,
2048
+ )
2049
+
2050
+ # Prune nodes and edges from deleted source files
2051
+ if prune_sources:
2052
+ # Source-less nodes that are ALREADY isolated before this prune. They are
2053
+ # not this prune's doing, so they must survive it — the sweep below is
2054
+ # scoped to the ones it orphans itself.
2055
+ _isolated_before = {
2056
+ n for n, d in G.nodes(data=True)
2057
+ if not d.get("source_file") and G.degree(n) == 0
2058
+ }
2059
+ to_remove = [
2060
+ n for n, d in G.nodes(data=True)
2061
+ if _prune_match(d.get("source_file"))
2062
+ ]
2063
+ G.remove_nodes_from(to_remove)
2064
+ n_nodes = len(to_remove)
2065
+
2066
+ edges_to_remove = [
2067
+ (u, v) for u, v, d in G.edges(data=True)
2068
+ if _prune_match(d.get("source_file"))
2069
+ ]
2070
+ if edges_to_remove:
2071
+ G.remove_edges_from(edges_to_remove)
2072
+
2073
+ # Extractors create a per-file node for each IMPORTED EXTERNAL symbol
2074
+ # (`Path` from pathlib, `Counter` from collections), and those carry no
2075
+ # source_file because they are defined outside the corpus. Every edge
2076
+ # they have points at symbols in the one file they were created for, so
2077
+ # pruning that file leaves them at degree 0 — named after a file the
2078
+ # corpus no longer contains, counted in every total that reads the graph,
2079
+ # exported as a note of their own, and unreachable by any future prune
2080
+ # since there is no source_file to match on. Nothing else can collect
2081
+ # them: deletions go through deleted_files, exclusions through
2082
+ # excluded_files (#1908) and _stale_graph_sources (#1909), and all three
2083
+ # match on source_file. A node with neither a source_file nor an edge
2084
+ # names nothing and connects nothing, so dropping it loses no
2085
+ # information (#2807).
2086
+ #
2087
+ # A single pass suffices: external-import stubs are only ever edge
2088
+ # TARGETS (extractors mint them as the target of an imports_from/
2089
+ # references/inherits edge, never as a source), so removing one can
2090
+ # never drop another to degree 0. A future extractor emitting a
2091
+ # stub->stub edge would require iterating this to a fixpoint.
2092
+ orphaned = [
2093
+ n for n, d in G.nodes(data=True)
2094
+ if not d.get("source_file")
2095
+ and G.degree(n) == 0
2096
+ and n not in _isolated_before
2097
+ ]
2098
+ if orphaned:
2099
+ G.remove_nodes_from(orphaned)
2100
+ n_nodes += len(orphaned)
2101
+
2102
+ # Report only the prune entries that ACTUALLY matched something — not
2103
+ # len(prune_sources), which counted every entry as pruned-from even
2104
+ # when a root mismatch made most of them no-ops (#2446).
2105
+ n_files = len(_matched_prune_entries)
2106
+ if n_nodes:
2107
+ print(
2108
+ f"[graphify] Pruned {n_nodes} node(s) from {n_files} deleted or "
2109
+ f"excluded source file(s).",
2110
+ file=sys.stderr,
2111
+ )
2112
+ if edges_to_remove:
2113
+ print(
2114
+ f"[graphify] Pruned {len(edges_to_remove)} edge(s) from deleted or "
2115
+ f"excluded source file(s).",
2116
+ file=sys.stderr,
2117
+ )
2118
+
2119
+ if not n_nodes and not edges_to_remove:
2120
+ if (prune_set or prune_abs) and not _matched_prune_entries:
2121
+ # Live prune entries that matched NOTHING usually mean the
2122
+ # effective root is wrong (and the derived-root fallback above
2123
+ # found no consistent candidate) — warn instead of claiming the
2124
+ # graph is "already clean" (#2446).
2125
+ _p0 = next((p for p in prune_sources if p), None)
2126
+ sample_prune = _norm_source_file(_p0, _prune_root) or _p0
2127
+ sample_sf = next(
2128
+ iter(sorted(
2129
+ str(d.get("source_file"))
2130
+ for _, d in G.nodes(data=True) if d.get("source_file")
2131
+ )),
2132
+ None,
2133
+ )
2134
+ print(
2135
+ f"[graphify] WARNING: {len(prune_sources)} prune source(s) "
2136
+ f"matched no nodes or edges — nothing was removed. Prune "
2137
+ f"entry {sample_prune!r} does not correspond to any stored "
2138
+ f"source_file (e.g. {sample_sf!r}). If these files should "
2139
+ f"have been pruned, pass root= to build_merge so absolute "
2140
+ f"paths relativize to the graph's source_file keys. (#2446)",
2141
+ file=sys.stderr,
2142
+ )
2143
+ else:
2144
+ print(
2145
+ f"[graphify] {len(prune_sources)} source file(s) deleted or "
2146
+ f"excluded since last run — no matching nodes or edges in "
2147
+ f"graph, already clean.",
2148
+ file=sys.stderr,
2149
+ )
2150
+
2151
+ # Safety check: refuse to SILENTLY drop nodes (#479, reworked in #2497).
2152
+ # The old count comparison ran against the post-replace `existing_nodes`,
2153
+ # which had already lost the re-extracted sources' old nodes — so it could
2154
+ # never fire when it mattered, and was skipped outright whenever
2155
+ # prune_sources was passed. Mirror watch._check_shrink instead: diff the
2156
+ # on-disk baseline by node identity and excuse only losses explained by
2157
+ # this run's own re-extraction (same tier) or an explicit prune. Skipped
2158
+ # under dedup, where fuzzy merging collapses ids legitimately.
2159
+ #
2160
+ # Residual tradeoff (accepted): a partial re-extraction that under-produces
2161
+ # for ITS OWN file (>= 1 node still present in new_chunks) is excused here —
2162
+ # that failure mode is owned by the extraction layer's incomplete-build
2163
+ # guard (#1951), and refusing it here would reintroduce the #1116
2164
+ # false-refuse for legitimate edits that remove symbols from a file.
2165
+ if had_graph and not dedup:
2166
+ def _in_new_graph(n: dict) -> bool:
2167
+ nid = n.get("id")
2168
+ if nid is None or not _hashable(nid):
2169
+ return True # untrackable identity — never count as lost
2170
+ if nid in G:
2171
+ return True
2172
+ # Doc-twin heal (#1799): build_from_json merges a markdown
2173
+ # quick-scan's bare doc node into its semantic `<id>_doc` twin for
2174
+ # the same file — a legitimate collapse, not a loss.
2175
+ return (
2176
+ isinstance(nid, str)
2177
+ and not nid.endswith("_doc")
2178
+ and n.get("file_type") == "document"
2179
+ and f"{nid}_doc" in G
2180
+ and G.nodes[f"{nid}_doc"].get("source_file") == n.get("source_file")
2181
+ )
2182
+
2183
+ lost = [
2184
+ n for n in _disk_nodes
2185
+ if isinstance(n, dict) and not _in_new_graph(n)
2186
+ ]
2187
+
2188
+ def _explained(n: dict) -> bool:
2189
+ sf = n.get("source_file")
2190
+ if not sf:
2191
+ return True
2192
+ own = new_ast_sources if _is_ast_tier(n) else new_sem_sources
2193
+ if sf in own or _norm_source_file(sf, _replace_root) in own:
2194
+ return True # replaced by this run's re-extract (same tier)
2195
+ return _prune_match(sf) # deliberately pruned this run
2196
+
2197
+ unexplained = [n for n in lost if not _explained(n)]
2198
+ if unexplained:
2199
+ raise ValueError(
2200
+ f"graphify: build_merge would drop {len(unexplained)} node(s) "
2201
+ f"from sources that were neither re-extracted nor pruned this "
2202
+ f"run (e.g. {unexplained[0].get('id')!r}); graph would go "
2203
+ f"{_disk_n} → {G.number_of_nodes()} nodes. "
2204
+ f"Pass prune_sources explicitly if you intend to remove them. (#479)"
2205
+ )
2206
+
2207
+ if unverified_semantic_shrink:
2208
+ G.graph["_unverified_semantic_shrink"] = unverified_semantic_shrink
2209
+
2210
+ return G
2211
+
2212
+
2213
+ def prefix_graph_for_global(
2214
+ G: nx.Graph, repo_tag: str, community_offset: int = 0
2215
+ ) -> nx.Graph:
2216
+ """Return a copy of G with all node IDs prefixed with repo_tag::.
2217
+
2218
+ Labels are preserved unchanged (for display). A 'local_id' attribute
2219
+ is added to each node so the original ID can be recovered. Edges and
2220
+ their directional attributes (_src/_tgt) are rewritten to match the new
2221
+ prefixed IDs. The 'repo' attribute is set on every node.
2222
+
2223
+ community_offset shifts each node's integer 'community' id into a shared
2224
+ id space and records the original in 'local_community': every input graph
2225
+ numbers its communities from 0, so ids carried across a merge unchanged
2226
+ collide and the aggregated community view fuses unrelated communities
2227
+ into one meta-node (#3014). 0 (the default) leaves communities untouched.
2228
+ """
2229
+ relabel = {n: f"{repo_tag}::{n}" for n in G.nodes}
2230
+ H = nx.relabel_nodes(G, relabel, copy=True)
2231
+ for node, data in H.nodes(data=True):
2232
+ data["repo"] = repo_tag
2233
+ data.setdefault("local_id", node.split("::", 1)[1])
2234
+ cid = data.get("community")
2235
+ if community_offset and isinstance(cid, int):
2236
+ data["local_community"] = cid
2237
+ data["community"] = cid + community_offset
2238
+ for u, v, data in H.edges(data=True):
2239
+ if "_src" in data and data["_src"] in relabel:
2240
+ data["_src"] = relabel[data["_src"]]
2241
+ if "_tgt" in data and data["_tgt"] in relabel:
2242
+ data["_tgt"] = relabel[data["_tgt"]]
2243
+ # Out-of-band hyperedges must be relabeled with the nodes (#2484, after
2244
+ # @oleksii-tumanov's diagnosis in PR #1691): relabel_nodes copies graph
2245
+ # attrs by reference, so member ids kept their pre-prefix form and dangled
2246
+ # after a cross-repo merge. Rebuild the list (fresh dicts — the input
2247
+ # graph's list is shared with H) with member ids mapped through the same
2248
+ # relabel table, and prefix the hyperedge id itself so same-named
2249
+ # hyperedges from different repos cannot collide when merged.
2250
+ hyperedges = H.graph.get("hyperedges")
2251
+ if isinstance(hyperedges, list):
2252
+ rewritten = []
2253
+ for he in hyperedges:
2254
+ if isinstance(he, dict):
2255
+ he = dict(he)
2256
+ if isinstance(he.get("nodes"), list):
2257
+ he["nodes"] = [
2258
+ relabel.get(m, m) if _hashable(m) else m
2259
+ for m in he["nodes"]
2260
+ ]
2261
+ if he.get("id"):
2262
+ he["id"] = f"{repo_tag}::{he['id']}"
2263
+ rewritten.append(he)
2264
+ H.graph["hyperedges"] = rewritten
2265
+ return H
2266
+
2267
+
2268
+ def distinct_repo_tags(graph_paths: "list[Path]") -> "list[str]":
2269
+ """Return a unique, human-meaningful repo tag per input graph for merge-graphs.
2270
+
2271
+ The naive tag (the ``graphify-out`` parent dir name) is NOT unique across
2272
+ inputs: ``src/graphify-out`` and ``frontend/src/graphify-out`` both yield
2273
+ ``src``. Prefixing both node sets with ``src::`` then makes same-stem nodes
2274
+ (a backend ``src/app.js`` and a frontend ``App.jsx``, both bare ``app``)
2275
+ collide, so ``nx.compose`` silently merges two unrelated entities and invents
2276
+ cross-runtime edges (#1729). Colliding tags are widened with their own parent
2277
+ dir (``frontend_src``), then an index suffix guarantees uniqueness so no two
2278
+ graphs ever share a prefix.
2279
+ """
2280
+ repo_dirs = [p.parent.parent for p in graph_paths] # graphify-out/.. → repo dir
2281
+ tags = [d.name or "repo" for d in repo_dirs]
2282
+ if len(set(tags)) != len(tags):
2283
+ widened: list[str] = []
2284
+ for d in repo_dirs:
2285
+ parent = d.parent.name
2286
+ widened.append(f"{parent}_{d.name}" if parent and d.name else (d.name or "repo"))
2287
+ tags = widened
2288
+ seen: dict[str, int] = {}
2289
+ unique: list[str] = []
2290
+ for t in tags:
2291
+ seen[t] = seen.get(t, 0) + 1
2292
+ unique.append(t if seen[t] == 1 else f"{t}-{seen[t]}")
2293
+ return unique
2294
+
2295
+
2296
+ def prune_repo_from_graph(G: nx.Graph, repo_tag: str) -> int:
2297
+ """Remove all nodes tagged with repo_tag from G in-place. Returns count removed."""
2298
+ to_remove = [n for n, d in G.nodes(data=True) if d.get("repo") == repo_tag]
2299
+ G.remove_nodes_from(to_remove)
2300
+ return len(to_remove)