graphitect 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (336) hide show
  1. graphify/__init__.py +30 -0
  2. graphify/__main__.py +757 -0
  3. graphify/_minhash.py +107 -0
  4. graphify/affected.py +318 -0
  5. graphify/always_on/agents-md.md +12 -0
  6. graphify/always_on/antigravity-rules.md +14 -0
  7. graphify/always_on/claude-md.md +9 -0
  8. graphify/always_on/gemini-md.md +9 -0
  9. graphify/always_on/kiro-steering.md +5 -0
  10. graphify/always_on/vscode-instructions.md +17 -0
  11. graphify/analyze.py +769 -0
  12. graphify/benchmark.py +152 -0
  13. graphify/build.py +2300 -0
  14. graphify/cache.py +1746 -0
  15. graphify/callflow_html.py +2051 -0
  16. graphify/cargo_introspect.py +109 -0
  17. graphify/cli.py +4745 -0
  18. graphify/cluster.py +409 -0
  19. graphify/command-kilo.md +15 -0
  20. graphify/cross_repo_calls.py +216 -0
  21. graphify/cross_repo_types.py +75 -0
  22. graphify/csharp_dispatch.py +154 -0
  23. graphify/dedup.py +1213 -0
  24. graphify/detect.py +2566 -0
  25. graphify/diagnostics.py +406 -0
  26. graphify/export.py +1349 -0
  27. graphify/exporters/__init__.py +1 -0
  28. graphify/exporters/base.py +14 -0
  29. graphify/exporters/graphdb.py +173 -0
  30. graphify/exporters/html.py +637 -0
  31. graphify/extract.py +7856 -0
  32. graphify/extractors/MIGRATION.md +107 -0
  33. graphify/extractors/__init__.py +66 -0
  34. graphify/extractors/apex.py +215 -0
  35. graphify/extractors/base.py +85 -0
  36. graphify/extractors/bash.py +579 -0
  37. graphify/extractors/blade.py +53 -0
  38. graphify/extractors/commonlisp.py +540 -0
  39. graphify/extractors/csharp.py +448 -0
  40. graphify/extractors/dart.py +564 -0
  41. graphify/extractors/dm.py +494 -0
  42. graphify/extractors/elixir.py +241 -0
  43. graphify/extractors/engine.py +6509 -0
  44. graphify/extractors/fortran.py +311 -0
  45. graphify/extractors/go.py +527 -0
  46. graphify/extractors/json_config.py +240 -0
  47. graphify/extractors/julia.py +289 -0
  48. graphify/extractors/markdown.py +408 -0
  49. graphify/extractors/models.py +131 -0
  50. graphify/extractors/objc.py +566 -0
  51. graphify/extractors/ocaml.py +289 -0
  52. graphify/extractors/pascal.py +688 -0
  53. graphify/extractors/pascal_forms.py +196 -0
  54. graphify/extractors/powershell.py +522 -0
  55. graphify/extractors/razor.py +192 -0
  56. graphify/extractors/resolution.py +3584 -0
  57. graphify/extractors/robot.py +296 -0
  58. graphify/extractors/rust.py +470 -0
  59. graphify/extractors/sln.py +92 -0
  60. graphify/extractors/sql.py +720 -0
  61. graphify/extractors/terraform.py +181 -0
  62. graphify/extractors/verilog.py +329 -0
  63. graphify/extractors/zig.py +181 -0
  64. graphify/file_slice.py +246 -0
  65. graphify/global_graph.py +194 -0
  66. graphify/google_workspace.py +237 -0
  67. graphify/hooks.py +933 -0
  68. graphify/ids.py +93 -0
  69. graphify/ingest.py +358 -0
  70. graphify/install.py +2366 -0
  71. graphify/llm.py +3544 -0
  72. graphify/manifest.py +4 -0
  73. graphify/manifest_ingest.py +311 -0
  74. graphify/mcp_ingest.py +386 -0
  75. graphify/multigraph_compat.py +212 -0
  76. graphify/pascal_resolution.py +129 -0
  77. graphify/paths.py +436 -0
  78. graphify/pg_introspect.py +165 -0
  79. graphify/prs.py +770 -0
  80. graphify/querylog.py +80 -0
  81. graphify/reflect.py +882 -0
  82. graphify/report.py +346 -0
  83. graphify/resolver_registry.py +85 -0
  84. graphify/ruby_resolution.py +242 -0
  85. graphify/scip_ingest.py +363 -0
  86. graphify/security.py +460 -0
  87. graphify/semantic_cleanup.py +336 -0
  88. graphify/serve.py +2608 -0
  89. graphify/skill-agents.md +710 -0
  90. graphify/skill-aider.md +1283 -0
  91. graphify/skill-amp.md +710 -0
  92. graphify/skill-claw.md +713 -0
  93. graphify/skill-codex.md +710 -0
  94. graphify/skill-copilot.md +713 -0
  95. graphify/skill-devin.md +1410 -0
  96. graphify/skill-droid.md +710 -0
  97. graphify/skill-kilo.md +722 -0
  98. graphify/skill-kiro.md +713 -0
  99. graphify/skill-opencode.md +705 -0
  100. graphify/skill-pi.md +713 -0
  101. graphify/skill-trae.md +711 -0
  102. graphify/skill-vscode.md +709 -0
  103. graphify/skill-windows.md +755 -0
  104. graphify/skill.md +713 -0
  105. graphify/skills/agents/references/add-watch.md +56 -0
  106. graphify/skills/agents/references/exports.md +87 -0
  107. graphify/skills/agents/references/extraction-spec.md +70 -0
  108. graphify/skills/agents/references/github-and-merge.md +46 -0
  109. graphify/skills/agents/references/hooks.md +33 -0
  110. graphify/skills/agents/references/query.md +311 -0
  111. graphify/skills/agents/references/transcribe.md +52 -0
  112. graphify/skills/agents/references/update.md +210 -0
  113. graphify/skills/amp/references/add-watch.md +56 -0
  114. graphify/skills/amp/references/exports.md +87 -0
  115. graphify/skills/amp/references/extraction-spec.md +70 -0
  116. graphify/skills/amp/references/github-and-merge.md +46 -0
  117. graphify/skills/amp/references/hooks.md +33 -0
  118. graphify/skills/amp/references/query.md +311 -0
  119. graphify/skills/amp/references/transcribe.md +52 -0
  120. graphify/skills/amp/references/update.md +210 -0
  121. graphify/skills/claude/references/add-watch.md +56 -0
  122. graphify/skills/claude/references/exports.md +87 -0
  123. graphify/skills/claude/references/extraction-spec.md +70 -0
  124. graphify/skills/claude/references/github-and-merge.md +46 -0
  125. graphify/skills/claude/references/hooks.md +33 -0
  126. graphify/skills/claude/references/query.md +311 -0
  127. graphify/skills/claude/references/transcribe.md +52 -0
  128. graphify/skills/claude/references/update.md +210 -0
  129. graphify/skills/claw/references/add-watch.md +56 -0
  130. graphify/skills/claw/references/exports.md +87 -0
  131. graphify/skills/claw/references/extraction-spec.md +31 -0
  132. graphify/skills/claw/references/github-and-merge.md +46 -0
  133. graphify/skills/claw/references/hooks.md +33 -0
  134. graphify/skills/claw/references/query.md +311 -0
  135. graphify/skills/claw/references/transcribe.md +52 -0
  136. graphify/skills/claw/references/update.md +210 -0
  137. graphify/skills/codex/references/add-watch.md +56 -0
  138. graphify/skills/codex/references/exports.md +87 -0
  139. graphify/skills/codex/references/extraction-spec.md +31 -0
  140. graphify/skills/codex/references/github-and-merge.md +46 -0
  141. graphify/skills/codex/references/hooks.md +33 -0
  142. graphify/skills/codex/references/query.md +311 -0
  143. graphify/skills/codex/references/transcribe.md +52 -0
  144. graphify/skills/codex/references/update.md +210 -0
  145. graphify/skills/copilot/references/add-watch.md +56 -0
  146. graphify/skills/copilot/references/exports.md +87 -0
  147. graphify/skills/copilot/references/extraction-spec.md +70 -0
  148. graphify/skills/copilot/references/github-and-merge.md +46 -0
  149. graphify/skills/copilot/references/hooks.md +33 -0
  150. graphify/skills/copilot/references/query.md +311 -0
  151. graphify/skills/copilot/references/transcribe.md +52 -0
  152. graphify/skills/copilot/references/update.md +210 -0
  153. graphify/skills/droid/references/add-watch.md +56 -0
  154. graphify/skills/droid/references/exports.md +87 -0
  155. graphify/skills/droid/references/extraction-spec.md +70 -0
  156. graphify/skills/droid/references/github-and-merge.md +46 -0
  157. graphify/skills/droid/references/hooks.md +33 -0
  158. graphify/skills/droid/references/query.md +311 -0
  159. graphify/skills/droid/references/transcribe.md +52 -0
  160. graphify/skills/droid/references/update.md +210 -0
  161. graphify/skills/kilo/references/add-watch.md +56 -0
  162. graphify/skills/kilo/references/exports.md +87 -0
  163. graphify/skills/kilo/references/extraction-spec.md +70 -0
  164. graphify/skills/kilo/references/github-and-merge.md +46 -0
  165. graphify/skills/kilo/references/hooks.md +33 -0
  166. graphify/skills/kilo/references/query.md +311 -0
  167. graphify/skills/kilo/references/transcribe.md +52 -0
  168. graphify/skills/kilo/references/update.md +210 -0
  169. graphify/skills/kiro/references/add-watch.md +56 -0
  170. graphify/skills/kiro/references/exports.md +87 -0
  171. graphify/skills/kiro/references/extraction-spec.md +31 -0
  172. graphify/skills/kiro/references/github-and-merge.md +46 -0
  173. graphify/skills/kiro/references/hooks.md +33 -0
  174. graphify/skills/kiro/references/query.md +311 -0
  175. graphify/skills/kiro/references/transcribe.md +52 -0
  176. graphify/skills/kiro/references/update.md +210 -0
  177. graphify/skills/opencode/references/add-watch.md +56 -0
  178. graphify/skills/opencode/references/exports.md +87 -0
  179. graphify/skills/opencode/references/extraction-spec.md +70 -0
  180. graphify/skills/opencode/references/github-and-merge.md +46 -0
  181. graphify/skills/opencode/references/hooks.md +33 -0
  182. graphify/skills/opencode/references/query.md +311 -0
  183. graphify/skills/opencode/references/transcribe.md +52 -0
  184. graphify/skills/opencode/references/update.md +210 -0
  185. graphify/skills/pi/references/add-watch.md +56 -0
  186. graphify/skills/pi/references/exports.md +87 -0
  187. graphify/skills/pi/references/extraction-spec.md +31 -0
  188. graphify/skills/pi/references/github-and-merge.md +46 -0
  189. graphify/skills/pi/references/hooks.md +33 -0
  190. graphify/skills/pi/references/query.md +311 -0
  191. graphify/skills/pi/references/transcribe.md +52 -0
  192. graphify/skills/pi/references/update.md +210 -0
  193. graphify/skills/trae/references/add-watch.md +56 -0
  194. graphify/skills/trae/references/exports.md +87 -0
  195. graphify/skills/trae/references/extraction-spec.md +70 -0
  196. graphify/skills/trae/references/github-and-merge.md +46 -0
  197. graphify/skills/trae/references/hooks.md +35 -0
  198. graphify/skills/trae/references/query.md +311 -0
  199. graphify/skills/trae/references/transcribe.md +52 -0
  200. graphify/skills/trae/references/update.md +210 -0
  201. graphify/skills/vscode/references/add-watch.md +56 -0
  202. graphify/skills/vscode/references/exports.md +87 -0
  203. graphify/skills/vscode/references/extraction-spec.md +70 -0
  204. graphify/skills/vscode/references/github-and-merge.md +46 -0
  205. graphify/skills/vscode/references/hooks.md +33 -0
  206. graphify/skills/vscode/references/query.md +311 -0
  207. graphify/skills/vscode/references/transcribe.md +52 -0
  208. graphify/skills/vscode/references/update.md +210 -0
  209. graphify/skills/windows/references/add-watch.md +56 -0
  210. graphify/skills/windows/references/exports.md +87 -0
  211. graphify/skills/windows/references/extraction-spec.md +70 -0
  212. graphify/skills/windows/references/github-and-merge.md +46 -0
  213. graphify/skills/windows/references/hooks.md +33 -0
  214. graphify/skills/windows/references/query.md +311 -0
  215. graphify/skills/windows/references/transcribe.md +52 -0
  216. graphify/skills/windows/references/update.md +210 -0
  217. graphify/symbol_resolution.py +556 -0
  218. graphify/transcribe.py +186 -0
  219. graphify/tree_html.py +603 -0
  220. graphify/validate.py +95 -0
  221. graphify/watch.py +2280 -0
  222. graphify/wiki.py +405 -0
  223. graphitect/__init__.py +28 -0
  224. graphitect/__main__.py +4 -0
  225. graphitect/_vendor/__init__.py +2 -0
  226. graphitect/_vendor/archify/LICENSE +22 -0
  227. graphitect/_vendor/archify/SKILL.md +137 -0
  228. graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
  229. graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
  230. graphitect/_vendor/archify/assets/template.html +14935 -0
  231. graphitect/_vendor/archify/bin/archify.mjs +2091 -0
  232. graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
  233. graphitect/_vendor/archify/bin/preview.mjs +653 -0
  234. graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
  235. graphitect/_vendor/archify/brand-marks/README.md +31 -0
  236. graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
  237. graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
  238. graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
  239. graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
  240. graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
  241. graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
  242. graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
  243. graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
  244. graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
  245. graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
  246. graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
  247. graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
  248. graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
  249. graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
  250. graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
  251. graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
  252. graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
  253. graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
  254. graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
  255. graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
  256. graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
  257. graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
  258. graphitect/_vendor/archify/package-lock.json +149 -0
  259. graphitect/_vendor/archify/package.json +39 -0
  260. graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
  261. graphitect/_vendor/archify/references/authoring-contract.md +243 -0
  262. graphitect/_vendor/archify/references/brand-marks.md +65 -0
  263. graphitect/_vendor/archify/references/delivery-contract.md +120 -0
  264. graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
  265. graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
  266. graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
  267. graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
  268. graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
  269. graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
  270. graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
  271. graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
  272. graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
  273. graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
  274. graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
  275. graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
  276. graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
  277. graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
  278. graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
  279. graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
  280. graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
  281. graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
  282. graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
  283. graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
  284. graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
  285. graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
  286. graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
  287. graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
  288. graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
  289. graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
  290. graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
  291. graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
  292. graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
  293. graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
  294. graphitect/_vendor/archify/schemas/README.md +211 -0
  295. graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
  296. graphitect/_vendor/archify/schemas/common.schema.json +115 -0
  297. graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
  298. graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
  299. graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
  300. graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
  301. graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
  302. graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
  303. graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
  304. graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
  305. graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
  306. graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
  307. graphitect/_vendor/archify/skill-release.json +10 -0
  308. graphitect/cli.py +981 -0
  309. graphitect/deliver/__init__.py +5 -0
  310. graphitect/deliver/archify_adapter.py +1877 -0
  311. graphitect/deliver/archify_ir.py +160 -0
  312. graphitect/deliver/archify_repair.py +135 -0
  313. graphitect/deliver/doc_compiler.py +916 -0
  314. graphitect/ground/__init__.py +5 -0
  315. graphitect/ground/describe_source.py +27 -0
  316. graphitect/ground/fullread_source.py +56 -0
  317. graphitect/ground/graphify_source.py +107 -0
  318. graphitect/models.py +118 -0
  319. graphitect/skill/SKILL.md +80 -0
  320. graphitect/skill/agents/openai.yaml +4 -0
  321. graphitect/synthesize/__init__.py +5 -0
  322. graphitect/synthesize/engine.py +281 -0
  323. graphitect/synthesize/llm_backend.py +331 -0
  324. graphitect/synthesize/questions.py +139 -0
  325. graphitect/synthesize/rubric.py +104 -0
  326. graphitect-0.2.0.dist-info/METADATA +284 -0
  327. graphitect-0.2.0.dist-info/RECORD +336 -0
  328. graphitect-0.2.0.dist-info/WHEEL +5 -0
  329. graphitect-0.2.0.dist-info/entry_points.txt +2 -0
  330. graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
  331. graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
  332. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
  333. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
  334. graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
  335. graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
  336. graphitect-0.2.0.dist-info/top_level.txt +2 -0
graphify/analyze.py ADDED
@@ -0,0 +1,769 @@
1
+ """Graph analysis: god nodes (most connected), surprising connections (cross-community), suggested questions."""
2
+ from __future__ import annotations
3
+ from pathlib import Path
4
+ import networkx as nx
5
+
6
+ from graphify.build import edge_data
7
+
8
+ # Builtin/mock names that can appear as annotation-derived nodes in pre-existing
9
+ # graphs. Excluded from god-node ranking so they don't displace real abstractions
10
+ # even if they weren't filtered at extraction time (#1147).
11
+ _BUILTIN_NOISE_LABELS = frozenset({
12
+ "str", "int", "float", "bool", "bytes", "bytearray", "complex", "object",
13
+ "True", "False",
14
+ "MagicMock", "Mock", "AsyncMock", "NonCallableMock",
15
+ "NonCallableMagicMock", "PropertyMock", "patch", "sentinel",
16
+ # Python stdlib types commonly confused for project symbols
17
+ "Path", "Any", "Optional", "List", "Dict", "Set", "Tuple", "Union",
18
+ "Callable", "Type", "ClassVar", "Final", "Literal", "Protocol",
19
+ "Counter", "defaultdict", "OrderedDict", "datetime", "Enum",
20
+ "os", "sys", "re", "json", "io", "abc", "typing",
21
+ # Swift / Foundation / SwiftUI framework symbols and module imports that
22
+ # otherwise dominate god-node rankings on Swift codebases (#2147)
23
+ "Foundation", "SwiftUI", "UIKit", "AppKit", "Combine",
24
+ "String", "Int", "Double", "Float", "Bool", "Data", "URL", "Date", "UUID",
25
+ "Sendable", "Codable", "Decodable", "Encodable", "Equatable", "Hashable",
26
+ "Identifiable", "Comparable", "AnyObject", "Error", "LocalizedError",
27
+ "NSObject", "NSString", "NSError", "NSLock",
28
+ "View", "Color", "Font", "DispatchQueue",
29
+ })
30
+
31
+ # Language families — extensions sharing a runtime can legitimately call each other
32
+ _LANG_FAMILY: dict[str, str] = {
33
+ **{e: "python" for e in (".py", ".pyw")},
34
+ **{e: "js" for e in (".js", ".jsx", ".mjs", ".cjs", ".ejs", ".ts", ".tsx", ".mts", ".cts", ".vue", ".svelte")},
35
+ **{e: "go" for e in (".go",)},
36
+ **{e: "rust" for e in (".rs",)},
37
+ **{e: "jvm" for e in (".java", ".kt", ".kts", ".scala")},
38
+ **{e: "c" for e in (".c", ".h", ".cpp", ".cc", ".cxx", ".hpp")},
39
+ **{e: "ruby" for e in (".rb", ".rake")},
40
+ **{e: "swift" for e in (".swift",)},
41
+ **{e: "dotnet" for e in (".cs",)},
42
+ **{e: "php" for e in (".php",)},
43
+ **{e: "r" for e in (".r",)},
44
+ }
45
+
46
+
47
+ def _cross_language(src_a: str, src_b: str) -> bool:
48
+ """Return True if two source files belong to different language families."""
49
+ ext_a = Path(src_a).suffix.lower()
50
+ ext_b = Path(src_b).suffix.lower()
51
+ fam_a = _LANG_FAMILY.get(ext_a)
52
+ fam_b = _LANG_FAMILY.get(ext_b)
53
+ if fam_a is None or fam_b is None:
54
+ return False
55
+ return fam_a != fam_b
56
+
57
+
58
+ def _node_community_map(communities: dict[int, list[str]]) -> dict[str, int]:
59
+ """Invert communities dict: node_id -> community_id."""
60
+ return {n: cid for cid, nodes in communities.items() for n in nodes}
61
+
62
+
63
+ def _is_file_node(G: nx.Graph, node_id: str) -> bool:
64
+ """
65
+ Return True if this node is a file-level hub node (e.g. 'client', 'models')
66
+ or an AST method stub (e.g. '.auth_flow()', '.__init__()').
67
+
68
+ These are synthetic nodes created by the AST extractor and should be excluded
69
+ from god nodes, surprising connections, and knowledge gap reporting.
70
+ """
71
+ attrs = G.nodes[node_id]
72
+ label = attrs.get("label", "")
73
+ if not label:
74
+ return False
75
+ # File-level hub: label matches the actual source filename — bare basename OR
76
+ # the directory-qualified form the #2032 disambiguation pass may assign.
77
+ source_file = attrs.get("source_file", "")
78
+ if source_file:
79
+ from graphify.build import _is_file_node_label
80
+ if _is_file_node_label(label, source_file):
81
+ return True
82
+ # Method stub: AST extractor labels methods as '.method_name()'
83
+ if label.startswith(".") and label.endswith("()"):
84
+ return True
85
+ # Module-level function stub: labeled 'function_name()' - only has a contains edge
86
+ # These are real functions but structurally isolated by definition; not a gap worth flagging
87
+ if label.endswith("()") and G.degree(node_id) <= 1:
88
+ return True
89
+ return False
90
+
91
+
92
+ _JSON_NOISE_LABELS: frozenset[str] = frozenset({
93
+ "start", "end", "name", "id", "type", "properties",
94
+ "value", "key", "data", "items", "title", "description", "version",
95
+ "dependencies", "devdependencies", "peerdependencies",
96
+ "optionaldependencies", "bundleddependencies", "bundledependencies",
97
+ })
98
+
99
+
100
+ def _is_json_key_node(G: nx.Graph, node_id: str) -> bool:
101
+ attrs = G.nodes[node_id]
102
+ src = (attrs.get("source_file") or "").lower()
103
+ if not src.endswith(".json"):
104
+ return False
105
+ label = (attrs.get("label") or "").strip().lower()
106
+ return label in _JSON_NOISE_LABELS
107
+
108
+
109
+ def god_nodes(G: nx.Graph, top_n: int = 10,
110
+ exclude_hubs_percentile: float | None = None) -> list[dict]:
111
+ """Return the top_n most-connected real entities - the core abstractions.
112
+
113
+ File-level hub nodes are excluded: they accumulate import/contains edges
114
+ mechanically and don't represent meaningful architectural abstractions.
115
+
116
+ ``exclude_hubs_percentile`` (0-100) suppresses nodes whose degree exceeds
117
+ that percentile of the graph's degree distribution, using the same
118
+ threshold computation ``cluster()`` applies (#3205) - so the one setting
119
+ suppresses utility hubs in the ranking AND in community resolution,
120
+ instead of only the latter. ``None`` keeps the historical ranking.
121
+ """
122
+ degree = dict(G.degree())
123
+ hub_threshold: float | None = None
124
+ if exclude_hubs_percentile is not None:
125
+ degrees = sorted(degree.values())
126
+ if degrees:
127
+ idx = max(0, int(len(degrees) * exclude_hubs_percentile / 100) - 1)
128
+ hub_threshold = degrees[idx]
129
+ sorted_nodes = sorted(degree.items(), key=lambda x: x[1], reverse=True)
130
+ result = []
131
+ for node_id, deg in sorted_nodes:
132
+ if hub_threshold is not None and deg > hub_threshold:
133
+ continue
134
+ if _is_file_node(G, node_id) or _is_concept_node(G, node_id) or _is_json_key_node(G, node_id):
135
+ continue
136
+ if G.nodes[node_id].get("label", "") in _BUILTIN_NOISE_LABELS:
137
+ continue
138
+ result.append({
139
+ "id": node_id,
140
+ "label": G.nodes[node_id].get("label", node_id),
141
+ "degree": deg,
142
+ })
143
+ if len(result) >= top_n:
144
+ break
145
+ return result
146
+
147
+
148
+ def surprising_connections(
149
+ G: nx.Graph,
150
+ communities: dict[int, list[str]] | None = None,
151
+ top_n: int = 5,
152
+ ) -> list[dict]:
153
+ """
154
+ Find connections that are genuinely surprising - not obvious from file structure.
155
+
156
+ Strategy:
157
+ - Multi-file corpora: cross-file edges between real entities (not concept nodes).
158
+ Sorted AMBIGUOUS → INFERRED → EXTRACTED.
159
+ - Single-file / single-source corpora: cross-community edges that bridge
160
+ distant parts of the graph (betweenness centrality on edges).
161
+ These reveal non-obvious structural couplings.
162
+
163
+ Concept nodes (empty source_file, or injected semantic annotations) are excluded
164
+ from surprising connections because they are intentional, not discovered.
165
+ """
166
+ # Identify unique source files (ignore empty/null source_file)
167
+ source_files = {
168
+ data.get("source_file", "")
169
+ for _, data in G.nodes(data=True)
170
+ if data.get("source_file", "")
171
+ }
172
+ is_multi_source = len(source_files) > 1
173
+
174
+ if is_multi_source:
175
+ return _cross_file_surprises(G, communities or {}, top_n)
176
+ else:
177
+ return _cross_community_surprises(G, communities or {}, top_n)
178
+
179
+
180
+ def _is_concept_node(G: nx.Graph, node_id: str) -> bool:
181
+ """
182
+ Return True if this node is a manually-injected semantic concept node
183
+ rather than a real entity found in source code.
184
+
185
+ Signals:
186
+ - Empty source_file
187
+ - source_file doesn't look like a real file path (no extension)
188
+ """
189
+ data = G.nodes[node_id]
190
+ source = data.get("source_file", "")
191
+ if not source:
192
+ return True
193
+ # Has no file extension → probably a concept label, not a real file
194
+ if "." not in source.split("/")[-1]:
195
+ return True
196
+ return False
197
+
198
+
199
+ from graphify.detect import CODE_EXTENSIONS, DOC_EXTENSIONS, PAPER_EXTENSIONS, IMAGE_EXTENSIONS
200
+
201
+
202
+ def _file_category(path: str) -> str:
203
+ ext = ("." + path.rsplit(".", 1)[-1].lower()) if "." in path else ""
204
+ if ext in CODE_EXTENSIONS:
205
+ return "code"
206
+ if ext in PAPER_EXTENSIONS:
207
+ return "paper"
208
+ if ext in IMAGE_EXTENSIONS:
209
+ return "image"
210
+ return "doc"
211
+
212
+
213
+ def _top_level_dir(path: str) -> str:
214
+ """Return the first path component - used to detect cross-repo edges."""
215
+ return path.split("/")[0] if "/" in path else path
216
+
217
+
218
+ def _surprise_score(
219
+ G: nx.Graph,
220
+ u: str,
221
+ v: str,
222
+ data: dict,
223
+ node_community: dict[str, int],
224
+ u_source: str,
225
+ v_source: str,
226
+ degrees: dict[str, int] | None = None,
227
+ ) -> tuple[int, list[str]]:
228
+ """Score how surprising a cross-file edge is. Returns (score, reasons)."""
229
+ score = 0
230
+ reasons: list[str] = []
231
+
232
+ # 1. Confidence weight - uncertain connections are more noteworthy
233
+ conf = data.get("confidence", "EXTRACTED")
234
+ relation = data.get("relation", "")
235
+ conf_bonus = {"AMBIGUOUS": 3, "INFERRED": 2, "EXTRACTED": 1}.get(conf, 1)
236
+
237
+ cat_u = _file_category(u_source)
238
+ cat_v = _file_category(v_source)
239
+
240
+ # Suppress all structural bonuses for INFERRED calls/uses that cross language
241
+ # boundaries or connect code to a doc file. Both cases are resolver pollution:
242
+ # label-matching fires across language families in monorepos, and code→doc
243
+ # "calls" edges are extraction artefacts, not real architecture.
244
+ # Excludes `semantically_similar_to` (genuine cross-boundary insight) and all
245
+ # AMBIGUOUS/EXTRACTED edges (not from the resolver path).
246
+ _suppress_structural = (
247
+ conf == "INFERRED"
248
+ and relation in ("calls", "uses")
249
+ and (_cross_language(u_source, v_source) or {cat_u, cat_v} == {"code", "doc"})
250
+ )
251
+ if _suppress_structural:
252
+ conf_bonus = 0
253
+
254
+ score += conf_bonus
255
+ if conf in ("AMBIGUOUS", "INFERRED"):
256
+ reasons.append(f"{conf.lower()} connection - not explicitly stated in source")
257
+
258
+ # 2. Cross file-type bonus - code↔paper or code↔image is non-obvious
259
+ if cat_u != cat_v and not _suppress_structural:
260
+ score += 2
261
+ reasons.append(f"crosses file types ({cat_u} ↔ {cat_v})")
262
+
263
+ # 3. Cross-repo bonus - different top-level directory
264
+ if _top_level_dir(u_source) != _top_level_dir(v_source) and not _suppress_structural:
265
+ score += 2
266
+ reasons.append("connects across different repos/directories")
267
+
268
+ # 4. Cross-community bonus - Leiden says these are structurally distant
269
+ cid_u = node_community.get(u)
270
+ cid_v = node_community.get(v)
271
+ if cid_u is not None and cid_v is not None and cid_u != cid_v and not _suppress_structural:
272
+ score += 1
273
+ reasons.append("bridges separate communities")
274
+
275
+ # 4b. Semantic similarity bonus - non-obvious conceptual links score higher
276
+ if data.get("relation") == "semantically_similar_to":
277
+ score = int(score * 1.5)
278
+ reasons.append("semantically similar concepts with no structural link")
279
+
280
+ # 5. Peripheral→hub: a low-degree node connecting to a high-degree one
281
+ deg_u = degrees[u] if degrees is not None else G.degree(u)
282
+ deg_v = degrees[v] if degrees is not None else G.degree(v)
283
+ if min(deg_u, deg_v) <= 2 and max(deg_u, deg_v) >= 5:
284
+ score += 1
285
+ peripheral = G.nodes[u].get("label", u) if deg_u <= 2 else G.nodes[v].get("label", v)
286
+ hub = G.nodes[v].get("label", v) if deg_u <= 2 else G.nodes[u].get("label", u)
287
+ reasons.append(f"peripheral node `{peripheral}` unexpectedly reaches hub `{hub}`")
288
+
289
+ return score, reasons
290
+
291
+
292
+ def _cross_file_surprises(G: nx.Graph, communities: dict[int, list[str]], top_n: int) -> list[dict]:
293
+ """
294
+ Cross-file edges between real code/doc entities, ranked by a composite
295
+ surprise score rather than confidence alone.
296
+
297
+ Surprise score accounts for:
298
+ - Confidence (AMBIGUOUS > INFERRED > EXTRACTED)
299
+ - Cross file-type (code↔paper is more surprising than code↔code)
300
+ - Cross-repo (different top-level directory)
301
+ - Cross-community (Leiden says structurally distant)
302
+ - Peripheral→hub (low-degree node reaching a god node)
303
+
304
+ Each result includes a 'why' field explaining what makes it non-obvious.
305
+ """
306
+ node_community = _node_community_map(communities)
307
+ degrees = dict(G.degree())
308
+ candidates = []
309
+
310
+ for u, v, data in G.edges(data=True):
311
+ relation = data.get("relation", "")
312
+ if relation in ("imports", "imports_from", "contains", "method"):
313
+ continue
314
+ if _is_concept_node(G, u) or _is_concept_node(G, v):
315
+ continue
316
+ if _is_file_node(G, u) or _is_file_node(G, v):
317
+ continue
318
+
319
+ u_source = G.nodes[u].get("source_file", "")
320
+ v_source = G.nodes[v].get("source_file", "")
321
+
322
+ if not u_source or not v_source or u_source == v_source:
323
+ continue
324
+
325
+ score, reasons = _surprise_score(G, u, v, data, node_community, u_source, v_source, degrees)
326
+ src_id = data.get("_src", u)
327
+ if src_id not in G.nodes:
328
+ src_id = u
329
+ tgt_id = data.get("_tgt", v)
330
+ if tgt_id not in G.nodes:
331
+ tgt_id = v
332
+ candidates.append({
333
+ "_score": score,
334
+ "source": G.nodes[src_id].get("label", src_id),
335
+ "target": G.nodes[tgt_id].get("label", tgt_id),
336
+ "source_files": [
337
+ G.nodes[src_id].get("source_file", ""),
338
+ G.nodes[tgt_id].get("source_file", ""),
339
+ ],
340
+ "confidence": data.get("confidence", "EXTRACTED"),
341
+ "relation": relation,
342
+ "why": "; ".join(reasons) if reasons else "cross-file semantic connection",
343
+ })
344
+
345
+ candidates.sort(key=lambda x: x["_score"], reverse=True)
346
+ for c in candidates:
347
+ c.pop("_score")
348
+
349
+ if candidates:
350
+ return candidates[:top_n]
351
+
352
+ return _cross_community_surprises(G, communities, top_n)
353
+
354
+
355
+ def _cross_community_surprises(
356
+ G: nx.Graph,
357
+ communities: dict[int, list[str]],
358
+ top_n: int,
359
+ ) -> list[dict]:
360
+ """
361
+ For single-source corpora: find edges that bridge different communities.
362
+ These are surprising because Leiden grouped everything else tightly -
363
+ these edges cut across the natural structure.
364
+
365
+ Falls back to high-betweenness edges if no community info is provided.
366
+ """
367
+ if not communities:
368
+ # No community info - use edge betweenness centrality
369
+ if G.number_of_edges() == 0:
370
+ return []
371
+ if G.number_of_nodes() > 5000:
372
+ return []
373
+ betweenness = nx.edge_betweenness_centrality(G)
374
+ top_edges = sorted(betweenness.items(), key=lambda x: x[1], reverse=True)[:top_n]
375
+ result = []
376
+ for (u, v), score in top_edges:
377
+ data = edge_data(G, u, v)
378
+ result.append({
379
+ "source": G.nodes[u].get("label", u),
380
+ "target": G.nodes[v].get("label", v),
381
+ "source_files": [
382
+ G.nodes[u].get("source_file", ""),
383
+ G.nodes[v].get("source_file", ""),
384
+ ],
385
+ "confidence": data.get("confidence", "EXTRACTED"),
386
+ "relation": data.get("relation", ""),
387
+ "note": f"Bridges graph structure (betweenness={score:.3f})",
388
+ })
389
+ return result
390
+
391
+ # Build node → community map
392
+ node_community = _node_community_map(communities)
393
+
394
+ surprises = []
395
+ for u, v, data in G.edges(data=True):
396
+ cid_u = node_community.get(u)
397
+ cid_v = node_community.get(v)
398
+ if cid_u is None or cid_v is None or cid_u == cid_v:
399
+ continue
400
+ # Skip file hub nodes and plain structural edges
401
+ if _is_file_node(G, u) or _is_file_node(G, v):
402
+ continue
403
+ relation = data.get("relation", "")
404
+ if relation in ("imports", "imports_from", "contains", "method"):
405
+ continue
406
+ # This edge crosses community boundaries - interesting
407
+ confidence = data.get("confidence", "EXTRACTED")
408
+ src_id = data.get("_src", u)
409
+ if src_id not in G.nodes:
410
+ src_id = u
411
+ tgt_id = data.get("_tgt", v)
412
+ if tgt_id not in G.nodes:
413
+ tgt_id = v
414
+ surprises.append({
415
+ "source": G.nodes[src_id].get("label", src_id),
416
+ "target": G.nodes[tgt_id].get("label", tgt_id),
417
+ "source_files": [
418
+ G.nodes[src_id].get("source_file", ""),
419
+ G.nodes[tgt_id].get("source_file", ""),
420
+ ],
421
+ "confidence": confidence,
422
+ "relation": relation,
423
+ "note": f"Bridges community {cid_u} → community {cid_v}",
424
+ "_pair": tuple(sorted([cid_u, cid_v])),
425
+ })
426
+
427
+ # Sort: AMBIGUOUS first, then INFERRED, then EXTRACTED
428
+ order = {"AMBIGUOUS": 0, "INFERRED": 1, "EXTRACTED": 2}
429
+ surprises.sort(key=lambda x: order.get(x["confidence"], 3))
430
+
431
+ # Deduplicate by community pair - one representative edge per (A→B) boundary.
432
+ # Without this, a single high-betweenness god node dominates all results.
433
+ seen_pairs: set[tuple] = set()
434
+ deduped = []
435
+ for s in surprises:
436
+ pair = s.pop("_pair")
437
+ if pair not in seen_pairs:
438
+ seen_pairs.add(pair)
439
+ deduped.append(s)
440
+ return deduped[:top_n]
441
+
442
+
443
+ def suggest_questions(
444
+ G: nx.Graph,
445
+ communities: dict[int, list[str]],
446
+ community_labels: dict[int, str],
447
+ top_n: int = 7,
448
+ ) -> list[dict]:
449
+ """
450
+ Generate questions the graph is uniquely positioned to answer.
451
+ Based on: AMBIGUOUS edges, bridge nodes, underexplored god nodes, isolated nodes.
452
+ Each question has a 'type', 'question', and 'why' field.
453
+ """
454
+ if community_labels:
455
+ community_labels = {int(k) if isinstance(k, str) else k: v for k, v in community_labels.items()}
456
+
457
+ questions = []
458
+ node_community = _node_community_map(communities)
459
+
460
+ # 1. AMBIGUOUS edges → unresolved relationship questions
461
+ for u, v, data in G.edges(data=True):
462
+ if data.get("confidence") == "AMBIGUOUS":
463
+ ul = G.nodes[u].get("label", u)
464
+ vl = G.nodes[v].get("label", v)
465
+ relation = data.get("relation", "related to")
466
+ questions.append({
467
+ "type": "ambiguous_edge",
468
+ "question": f"What is the exact relationship between `{ul}` and `{vl}`?",
469
+ "why": f"Edge tagged AMBIGUOUS (relation: {relation}) - confidence is low.",
470
+ })
471
+
472
+ # 2. Bridge nodes (high betweenness) → cross-cutting concern questions
473
+ if G.number_of_edges() > 0:
474
+ k = min(100, G.number_of_nodes()) if G.number_of_nodes() > 1000 else None
475
+ betweenness = nx.betweenness_centrality(G, k=k, seed=42)
476
+ # Top bridge nodes that are NOT file-level hubs
477
+ bridges = sorted(
478
+ [(n, s) for n, s in betweenness.items()
479
+ if not _is_file_node(G, n) and not _is_concept_node(G, n) and s > 0],
480
+ key=lambda x: x[1],
481
+ reverse=True,
482
+ )[:3]
483
+ for node_id, score in bridges:
484
+ label = G.nodes[node_id].get("label", node_id)
485
+ cid = node_community.get(node_id)
486
+ comm_label = community_labels.get(cid, f"Community {cid}") if cid is not None else "unknown"
487
+ neighbors = list(G.neighbors(node_id))
488
+ neighbor_comms = {node_community.get(n) for n in neighbors if node_community.get(n) != cid}
489
+ if neighbor_comms:
490
+ other_labels = [community_labels.get(c, f"Community {c}") for c in neighbor_comms]
491
+ questions.append({
492
+ "type": "bridge_node",
493
+ "question": f"Why does `{label}` connect `{comm_label}` to {', '.join(f'`{l}`' for l in other_labels)}?",
494
+ "why": f"High betweenness centrality ({score:.3f}) - this node is a cross-community bridge.",
495
+ })
496
+
497
+ # 3. God nodes with many INFERRED edges → verification questions
498
+ degree = dict(G.degree())
499
+ top_nodes = sorted(
500
+ [(n, d) for n, d in degree.items() if not _is_file_node(G, n)],
501
+ key=lambda x: x[1],
502
+ reverse=True,
503
+ )[:5]
504
+ for node_id, _ in top_nodes:
505
+ inferred = [
506
+ (u, v, d) for u, v, d in G.edges(node_id, data=True)
507
+ if d.get("confidence") == "INFERRED"
508
+ ]
509
+ if len(inferred) >= 2:
510
+ label = G.nodes[node_id].get("label", node_id)
511
+ # Use _src/_tgt to get the correct direction; fall back to v (the other node)
512
+ others = []
513
+ for u, v, d in inferred[:2]:
514
+ src_id = d.get("_src", u)
515
+ if src_id not in G.nodes:
516
+ src_id = u
517
+ tgt_id = d.get("_tgt", v)
518
+ if tgt_id not in G.nodes:
519
+ tgt_id = v
520
+ other_id = tgt_id if src_id == node_id else src_id
521
+ others.append(G.nodes[other_id].get("label", other_id))
522
+ questions.append({
523
+ "type": "verify_inferred",
524
+ "question": f"Are the {len(inferred)} inferred relationships involving `{label}` (e.g. with `{others[0]}` and `{others[1]}`) actually correct?",
525
+ "why": f"`{label}` has {len(inferred)} INFERRED edges - model-reasoned connections that need verification.",
526
+ })
527
+
528
+ # 4. Isolated or weakly-connected nodes → exploration questions
529
+ isolated = [
530
+ n for n in G.nodes()
531
+ if G.degree(n) <= 1
532
+ and not _is_file_node(G, n)
533
+ and not _is_concept_node(G, n)
534
+ and G.nodes[n].get("file_type") != "rationale"
535
+ ]
536
+ if isolated:
537
+ labels = [G.nodes[n].get("label", n) for n in isolated[:3]]
538
+ questions.append({
539
+ "type": "isolated_nodes",
540
+ "question": f"What connects {', '.join(f'`{l}`' for l in labels)} to the rest of the system?",
541
+ "why": f"{len(isolated)} weakly-connected nodes found - possible documentation gaps or missing edges.",
542
+ })
543
+
544
+ # 5. Low-cohesion communities → structural questions
545
+ from .cluster import cohesion_score
546
+ for cid, nodes in communities.items():
547
+ score = cohesion_score(G, nodes)
548
+ if score < 0.15 and len(nodes) >= 5:
549
+ label = community_labels.get(cid, f"Community {cid}")
550
+ questions.append({
551
+ "type": "low_cohesion",
552
+ "question": f"Should `{label}` be split into smaller, more focused modules?",
553
+ "why": f"Cohesion score {score} - nodes in this community are weakly interconnected.",
554
+ })
555
+
556
+ if not questions:
557
+ return [{
558
+ "type": "no_signal",
559
+ "question": None,
560
+ "why": (
561
+ "Not enough signal to generate questions. "
562
+ "This usually means the corpus has no AMBIGUOUS edges, no bridge nodes, "
563
+ "no INFERRED relationships, and all communities are tightly cohesive. "
564
+ "Add more files or run with --mode deep to extract richer edges."
565
+ ),
566
+ }]
567
+
568
+ return questions[:top_n]
569
+
570
+
571
+ def graph_diff(G_old: nx.Graph, G_new: nx.Graph) -> dict:
572
+ """Compare two graph snapshots and return what changed.
573
+
574
+ Returns:
575
+ {
576
+ "new_nodes": [{"id": ..., "label": ...}],
577
+ "removed_nodes": [{"id": ..., "label": ...}],
578
+ "new_edges": [{"source": ..., "target": ..., "relation": ..., "confidence": ...}],
579
+ "removed_edges": [...],
580
+ "summary": "3 new nodes, 5 new edges, 1 node removed"
581
+ }
582
+ """
583
+ old_nodes = set(G_old.nodes())
584
+ new_nodes = set(G_new.nodes())
585
+
586
+ added_node_ids = new_nodes - old_nodes
587
+ removed_node_ids = old_nodes - new_nodes
588
+
589
+ new_nodes_list = [
590
+ {"id": n, "label": G_new.nodes[n].get("label", n)}
591
+ for n in added_node_ids
592
+ ]
593
+ removed_nodes_list = [
594
+ {"id": n, "label": G_old.nodes[n].get("label", n)}
595
+ for n in removed_node_ids
596
+ ]
597
+
598
+ def edge_key(G: nx.Graph, u: str, v: str, data: dict) -> tuple:
599
+ if G.is_directed():
600
+ return (u, v, data.get("relation", ""))
601
+ return (min(u, v), max(u, v), data.get("relation", ""))
602
+
603
+ old_edge_keys = {
604
+ edge_key(G_old, u, v, d)
605
+ for u, v, d in G_old.edges(data=True)
606
+ }
607
+ new_edge_keys = {
608
+ edge_key(G_new, u, v, d)
609
+ for u, v, d in G_new.edges(data=True)
610
+ }
611
+
612
+ added_edge_keys = new_edge_keys - old_edge_keys
613
+ removed_edge_keys = old_edge_keys - new_edge_keys
614
+
615
+ new_edges_list = []
616
+ for u, v, d in G_new.edges(data=True):
617
+ if edge_key(G_new, u, v, d) in added_edge_keys:
618
+ new_edges_list.append({
619
+ "source": u,
620
+ "target": v,
621
+ "relation": d.get("relation", ""),
622
+ "confidence": d.get("confidence", ""),
623
+ })
624
+
625
+ removed_edges_list = []
626
+ for u, v, d in G_old.edges(data=True):
627
+ if edge_key(G_old, u, v, d) in removed_edge_keys:
628
+ removed_edges_list.append({
629
+ "source": u,
630
+ "target": v,
631
+ "relation": d.get("relation", ""),
632
+ "confidence": d.get("confidence", ""),
633
+ })
634
+
635
+ parts = []
636
+ if new_nodes_list:
637
+ parts.append(f"{len(new_nodes_list)} new node{'s' if len(new_nodes_list) != 1 else ''}")
638
+ if new_edges_list:
639
+ parts.append(f"{len(new_edges_list)} new edge{'s' if len(new_edges_list) != 1 else ''}")
640
+ if removed_nodes_list:
641
+ parts.append(f"{len(removed_nodes_list)} node{'s' if len(removed_nodes_list) != 1 else ''} removed")
642
+ if removed_edges_list:
643
+ parts.append(f"{len(removed_edges_list)} edge{'s' if len(removed_edges_list) != 1 else ''} removed")
644
+ summary = ", ".join(parts) if parts else "no changes"
645
+
646
+ return {
647
+ "new_nodes": new_nodes_list,
648
+ "removed_nodes": removed_nodes_list,
649
+ "new_edges": new_edges_list,
650
+ "removed_edges": removed_edges_list,
651
+ "summary": summary,
652
+ }
653
+
654
+
655
+ def find_import_cycles(
656
+ G: nx.Graph,
657
+ max_cycle_length: int = 5,
658
+ top_n: int = 20,
659
+ ) -> list[dict]:
660
+ """Detect circular import dependencies at the file level.
661
+
662
+ Collapses symbol-level nodes to their parent file (using source_file attr
663
+ or 'contains' edges), builds a directed file-level graph from imports_from
664
+ edges, then finds simple cycles.
665
+
666
+ Args:
667
+ G: The full knowledge graph (may be undirected or directed).
668
+ max_cycle_length: Only report cycles with at most this many files.
669
+ top_n: Maximum number of cycles to return (shortest first).
670
+
671
+ Returns:
672
+ List of cycle records with stable structure:
673
+ {
674
+ "cycle": ["a.ts", "b.ts"],
675
+ "length": 2,
676
+ "why": "circular dependency"
677
+ }
678
+ """
679
+ def _endpoint_source_file(node_id: str) -> str:
680
+ attrs = G.nodes.get(node_id, {})
681
+ src_file = attrs.get("source_file", "")
682
+ return src_file if isinstance(src_file, str) else ""
683
+
684
+ # Step 1: Build a directed file-level graph from import/re-export edges.
685
+ # IMPORTANT: resolve endpoints using source_file only; never infer from label/id.
686
+ file_graph = nx.DiGraph()
687
+
688
+ for u, v, data in G.edges(data=True):
689
+ rel = data.get("relation", "")
690
+ if rel not in ("imports_from", "re_exports"):
691
+ continue
692
+
693
+ # Deferred `import(...)` edges are real dependencies but do not form a
694
+ # hard file-level cycle, so they are excluded from cycle detection (#1241).
695
+ if data.get("deferred"):
696
+ continue
697
+ # Type-only imports/re-exports (`import type` / `export type ... from`)
698
+ # are erased at compile time - a cycle that closes through one cannot
699
+ # exist at runtime (#3123). The edge itself stays in the graph.
700
+ if data.get("type_only"):
701
+ continue
702
+
703
+ src_file_attr = data.get("source_file", "")
704
+ if not isinstance(src_file_attr, str) or not src_file_attr:
705
+ continue
706
+
707
+ u_file = _endpoint_source_file(u)
708
+ v_file = _endpoint_source_file(v)
709
+
710
+ # Works for both DiGraph and Graph inputs:
711
+ # orient edge from edge.source_file endpoint to the opposite endpoint.
712
+ if u_file == src_file_attr:
713
+ tgt_file = v_file
714
+ elif v_file == src_file_attr:
715
+ tgt_file = u_file
716
+ else:
717
+ # Fallback: if source endpoint cannot be matched exactly,
718
+ # still treat edge.source_file as source and pick the opposite endpoint
719
+ # only if one endpoint has a real source_file.
720
+ tgt_file = v_file if v_file and v_file != src_file_attr else u_file
721
+
722
+ if not tgt_file:
723
+ continue
724
+
725
+ file_graph.add_edge(src_file_attr, tgt_file)
726
+
727
+ if not file_graph.edges():
728
+ return []
729
+
730
+ # Step 2: Find simple cycles, bounded by length.
731
+ # Pass length_bound so networkx prunes during enumeration rather than
732
+ # enumerating all elementary cycles and post-filtering — avoids exponential
733
+ # blowup on dense graphs with many long cycles (#1196).
734
+ cycles: list[list[str]] = []
735
+ for cycle in nx.simple_cycles(file_graph, length_bound=max_cycle_length):
736
+ if len(cycle) <= max_cycle_length:
737
+ cycles.append(cycle)
738
+ if len(cycles) >= top_n * 10:
739
+ # Stop early to avoid combinatorial explosion
740
+ break
741
+
742
+ # Step 3: Sort by length (shortest = tightest coupling), then deduplicate.
743
+ cycles.sort(key=len)
744
+
745
+ # Deduplicate rotations: normalize each cycle by starting from the
746
+ # lexicographically smallest element.
747
+ seen: set[tuple[str, ...]] = set()
748
+ unique_cycles: list[list[str]] = []
749
+ for cycle in cycles:
750
+ core = list(cycle)
751
+ if not core:
752
+ continue
753
+ min_idx = core.index(min(core))
754
+ normalized = tuple(core[min_idx:] + core[:min_idx])
755
+ if normalized not in seen:
756
+ seen.add(normalized)
757
+ unique_cycles.append(list(normalized))
758
+ if len(unique_cycles) >= top_n:
759
+ break
760
+
761
+ result: list[dict] = []
762
+ for cycle in unique_cycles:
763
+ result.append({
764
+ "cycle": cycle,
765
+ "length": len(cycle),
766
+ "why": "circular dependency",
767
+ })
768
+
769
+ return result