graphitect 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (336) hide show
  1. graphify/__init__.py +30 -0
  2. graphify/__main__.py +757 -0
  3. graphify/_minhash.py +107 -0
  4. graphify/affected.py +318 -0
  5. graphify/always_on/agents-md.md +12 -0
  6. graphify/always_on/antigravity-rules.md +14 -0
  7. graphify/always_on/claude-md.md +9 -0
  8. graphify/always_on/gemini-md.md +9 -0
  9. graphify/always_on/kiro-steering.md +5 -0
  10. graphify/always_on/vscode-instructions.md +17 -0
  11. graphify/analyze.py +769 -0
  12. graphify/benchmark.py +152 -0
  13. graphify/build.py +2300 -0
  14. graphify/cache.py +1746 -0
  15. graphify/callflow_html.py +2051 -0
  16. graphify/cargo_introspect.py +109 -0
  17. graphify/cli.py +4745 -0
  18. graphify/cluster.py +409 -0
  19. graphify/command-kilo.md +15 -0
  20. graphify/cross_repo_calls.py +216 -0
  21. graphify/cross_repo_types.py +75 -0
  22. graphify/csharp_dispatch.py +154 -0
  23. graphify/dedup.py +1213 -0
  24. graphify/detect.py +2566 -0
  25. graphify/diagnostics.py +406 -0
  26. graphify/export.py +1349 -0
  27. graphify/exporters/__init__.py +1 -0
  28. graphify/exporters/base.py +14 -0
  29. graphify/exporters/graphdb.py +173 -0
  30. graphify/exporters/html.py +637 -0
  31. graphify/extract.py +7856 -0
  32. graphify/extractors/MIGRATION.md +107 -0
  33. graphify/extractors/__init__.py +66 -0
  34. graphify/extractors/apex.py +215 -0
  35. graphify/extractors/base.py +85 -0
  36. graphify/extractors/bash.py +579 -0
  37. graphify/extractors/blade.py +53 -0
  38. graphify/extractors/commonlisp.py +540 -0
  39. graphify/extractors/csharp.py +448 -0
  40. graphify/extractors/dart.py +564 -0
  41. graphify/extractors/dm.py +494 -0
  42. graphify/extractors/elixir.py +241 -0
  43. graphify/extractors/engine.py +6509 -0
  44. graphify/extractors/fortran.py +311 -0
  45. graphify/extractors/go.py +527 -0
  46. graphify/extractors/json_config.py +240 -0
  47. graphify/extractors/julia.py +289 -0
  48. graphify/extractors/markdown.py +408 -0
  49. graphify/extractors/models.py +131 -0
  50. graphify/extractors/objc.py +566 -0
  51. graphify/extractors/ocaml.py +289 -0
  52. graphify/extractors/pascal.py +688 -0
  53. graphify/extractors/pascal_forms.py +196 -0
  54. graphify/extractors/powershell.py +522 -0
  55. graphify/extractors/razor.py +192 -0
  56. graphify/extractors/resolution.py +3584 -0
  57. graphify/extractors/robot.py +296 -0
  58. graphify/extractors/rust.py +470 -0
  59. graphify/extractors/sln.py +92 -0
  60. graphify/extractors/sql.py +720 -0
  61. graphify/extractors/terraform.py +181 -0
  62. graphify/extractors/verilog.py +329 -0
  63. graphify/extractors/zig.py +181 -0
  64. graphify/file_slice.py +246 -0
  65. graphify/global_graph.py +194 -0
  66. graphify/google_workspace.py +237 -0
  67. graphify/hooks.py +933 -0
  68. graphify/ids.py +93 -0
  69. graphify/ingest.py +358 -0
  70. graphify/install.py +2366 -0
  71. graphify/llm.py +3544 -0
  72. graphify/manifest.py +4 -0
  73. graphify/manifest_ingest.py +311 -0
  74. graphify/mcp_ingest.py +386 -0
  75. graphify/multigraph_compat.py +212 -0
  76. graphify/pascal_resolution.py +129 -0
  77. graphify/paths.py +436 -0
  78. graphify/pg_introspect.py +165 -0
  79. graphify/prs.py +770 -0
  80. graphify/querylog.py +80 -0
  81. graphify/reflect.py +882 -0
  82. graphify/report.py +346 -0
  83. graphify/resolver_registry.py +85 -0
  84. graphify/ruby_resolution.py +242 -0
  85. graphify/scip_ingest.py +363 -0
  86. graphify/security.py +460 -0
  87. graphify/semantic_cleanup.py +336 -0
  88. graphify/serve.py +2608 -0
  89. graphify/skill-agents.md +710 -0
  90. graphify/skill-aider.md +1283 -0
  91. graphify/skill-amp.md +710 -0
  92. graphify/skill-claw.md +713 -0
  93. graphify/skill-codex.md +710 -0
  94. graphify/skill-copilot.md +713 -0
  95. graphify/skill-devin.md +1410 -0
  96. graphify/skill-droid.md +710 -0
  97. graphify/skill-kilo.md +722 -0
  98. graphify/skill-kiro.md +713 -0
  99. graphify/skill-opencode.md +705 -0
  100. graphify/skill-pi.md +713 -0
  101. graphify/skill-trae.md +711 -0
  102. graphify/skill-vscode.md +709 -0
  103. graphify/skill-windows.md +755 -0
  104. graphify/skill.md +713 -0
  105. graphify/skills/agents/references/add-watch.md +56 -0
  106. graphify/skills/agents/references/exports.md +87 -0
  107. graphify/skills/agents/references/extraction-spec.md +70 -0
  108. graphify/skills/agents/references/github-and-merge.md +46 -0
  109. graphify/skills/agents/references/hooks.md +33 -0
  110. graphify/skills/agents/references/query.md +311 -0
  111. graphify/skills/agents/references/transcribe.md +52 -0
  112. graphify/skills/agents/references/update.md +210 -0
  113. graphify/skills/amp/references/add-watch.md +56 -0
  114. graphify/skills/amp/references/exports.md +87 -0
  115. graphify/skills/amp/references/extraction-spec.md +70 -0
  116. graphify/skills/amp/references/github-and-merge.md +46 -0
  117. graphify/skills/amp/references/hooks.md +33 -0
  118. graphify/skills/amp/references/query.md +311 -0
  119. graphify/skills/amp/references/transcribe.md +52 -0
  120. graphify/skills/amp/references/update.md +210 -0
  121. graphify/skills/claude/references/add-watch.md +56 -0
  122. graphify/skills/claude/references/exports.md +87 -0
  123. graphify/skills/claude/references/extraction-spec.md +70 -0
  124. graphify/skills/claude/references/github-and-merge.md +46 -0
  125. graphify/skills/claude/references/hooks.md +33 -0
  126. graphify/skills/claude/references/query.md +311 -0
  127. graphify/skills/claude/references/transcribe.md +52 -0
  128. graphify/skills/claude/references/update.md +210 -0
  129. graphify/skills/claw/references/add-watch.md +56 -0
  130. graphify/skills/claw/references/exports.md +87 -0
  131. graphify/skills/claw/references/extraction-spec.md +31 -0
  132. graphify/skills/claw/references/github-and-merge.md +46 -0
  133. graphify/skills/claw/references/hooks.md +33 -0
  134. graphify/skills/claw/references/query.md +311 -0
  135. graphify/skills/claw/references/transcribe.md +52 -0
  136. graphify/skills/claw/references/update.md +210 -0
  137. graphify/skills/codex/references/add-watch.md +56 -0
  138. graphify/skills/codex/references/exports.md +87 -0
  139. graphify/skills/codex/references/extraction-spec.md +31 -0
  140. graphify/skills/codex/references/github-and-merge.md +46 -0
  141. graphify/skills/codex/references/hooks.md +33 -0
  142. graphify/skills/codex/references/query.md +311 -0
  143. graphify/skills/codex/references/transcribe.md +52 -0
  144. graphify/skills/codex/references/update.md +210 -0
  145. graphify/skills/copilot/references/add-watch.md +56 -0
  146. graphify/skills/copilot/references/exports.md +87 -0
  147. graphify/skills/copilot/references/extraction-spec.md +70 -0
  148. graphify/skills/copilot/references/github-and-merge.md +46 -0
  149. graphify/skills/copilot/references/hooks.md +33 -0
  150. graphify/skills/copilot/references/query.md +311 -0
  151. graphify/skills/copilot/references/transcribe.md +52 -0
  152. graphify/skills/copilot/references/update.md +210 -0
  153. graphify/skills/droid/references/add-watch.md +56 -0
  154. graphify/skills/droid/references/exports.md +87 -0
  155. graphify/skills/droid/references/extraction-spec.md +70 -0
  156. graphify/skills/droid/references/github-and-merge.md +46 -0
  157. graphify/skills/droid/references/hooks.md +33 -0
  158. graphify/skills/droid/references/query.md +311 -0
  159. graphify/skills/droid/references/transcribe.md +52 -0
  160. graphify/skills/droid/references/update.md +210 -0
  161. graphify/skills/kilo/references/add-watch.md +56 -0
  162. graphify/skills/kilo/references/exports.md +87 -0
  163. graphify/skills/kilo/references/extraction-spec.md +70 -0
  164. graphify/skills/kilo/references/github-and-merge.md +46 -0
  165. graphify/skills/kilo/references/hooks.md +33 -0
  166. graphify/skills/kilo/references/query.md +311 -0
  167. graphify/skills/kilo/references/transcribe.md +52 -0
  168. graphify/skills/kilo/references/update.md +210 -0
  169. graphify/skills/kiro/references/add-watch.md +56 -0
  170. graphify/skills/kiro/references/exports.md +87 -0
  171. graphify/skills/kiro/references/extraction-spec.md +31 -0
  172. graphify/skills/kiro/references/github-and-merge.md +46 -0
  173. graphify/skills/kiro/references/hooks.md +33 -0
  174. graphify/skills/kiro/references/query.md +311 -0
  175. graphify/skills/kiro/references/transcribe.md +52 -0
  176. graphify/skills/kiro/references/update.md +210 -0
  177. graphify/skills/opencode/references/add-watch.md +56 -0
  178. graphify/skills/opencode/references/exports.md +87 -0
  179. graphify/skills/opencode/references/extraction-spec.md +70 -0
  180. graphify/skills/opencode/references/github-and-merge.md +46 -0
  181. graphify/skills/opencode/references/hooks.md +33 -0
  182. graphify/skills/opencode/references/query.md +311 -0
  183. graphify/skills/opencode/references/transcribe.md +52 -0
  184. graphify/skills/opencode/references/update.md +210 -0
  185. graphify/skills/pi/references/add-watch.md +56 -0
  186. graphify/skills/pi/references/exports.md +87 -0
  187. graphify/skills/pi/references/extraction-spec.md +31 -0
  188. graphify/skills/pi/references/github-and-merge.md +46 -0
  189. graphify/skills/pi/references/hooks.md +33 -0
  190. graphify/skills/pi/references/query.md +311 -0
  191. graphify/skills/pi/references/transcribe.md +52 -0
  192. graphify/skills/pi/references/update.md +210 -0
  193. graphify/skills/trae/references/add-watch.md +56 -0
  194. graphify/skills/trae/references/exports.md +87 -0
  195. graphify/skills/trae/references/extraction-spec.md +70 -0
  196. graphify/skills/trae/references/github-and-merge.md +46 -0
  197. graphify/skills/trae/references/hooks.md +35 -0
  198. graphify/skills/trae/references/query.md +311 -0
  199. graphify/skills/trae/references/transcribe.md +52 -0
  200. graphify/skills/trae/references/update.md +210 -0
  201. graphify/skills/vscode/references/add-watch.md +56 -0
  202. graphify/skills/vscode/references/exports.md +87 -0
  203. graphify/skills/vscode/references/extraction-spec.md +70 -0
  204. graphify/skills/vscode/references/github-and-merge.md +46 -0
  205. graphify/skills/vscode/references/hooks.md +33 -0
  206. graphify/skills/vscode/references/query.md +311 -0
  207. graphify/skills/vscode/references/transcribe.md +52 -0
  208. graphify/skills/vscode/references/update.md +210 -0
  209. graphify/skills/windows/references/add-watch.md +56 -0
  210. graphify/skills/windows/references/exports.md +87 -0
  211. graphify/skills/windows/references/extraction-spec.md +70 -0
  212. graphify/skills/windows/references/github-and-merge.md +46 -0
  213. graphify/skills/windows/references/hooks.md +33 -0
  214. graphify/skills/windows/references/query.md +311 -0
  215. graphify/skills/windows/references/transcribe.md +52 -0
  216. graphify/skills/windows/references/update.md +210 -0
  217. graphify/symbol_resolution.py +556 -0
  218. graphify/transcribe.py +186 -0
  219. graphify/tree_html.py +603 -0
  220. graphify/validate.py +95 -0
  221. graphify/watch.py +2280 -0
  222. graphify/wiki.py +405 -0
  223. graphitect/__init__.py +28 -0
  224. graphitect/__main__.py +4 -0
  225. graphitect/_vendor/__init__.py +2 -0
  226. graphitect/_vendor/archify/LICENSE +22 -0
  227. graphitect/_vendor/archify/SKILL.md +137 -0
  228. graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
  229. graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
  230. graphitect/_vendor/archify/assets/template.html +14935 -0
  231. graphitect/_vendor/archify/bin/archify.mjs +2091 -0
  232. graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
  233. graphitect/_vendor/archify/bin/preview.mjs +653 -0
  234. graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
  235. graphitect/_vendor/archify/brand-marks/README.md +31 -0
  236. graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
  237. graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
  238. graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
  239. graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
  240. graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
  241. graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
  242. graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
  243. graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
  244. graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
  245. graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
  246. graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
  247. graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
  248. graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
  249. graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
  250. graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
  251. graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
  252. graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
  253. graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
  254. graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
  255. graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
  256. graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
  257. graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
  258. graphitect/_vendor/archify/package-lock.json +149 -0
  259. graphitect/_vendor/archify/package.json +39 -0
  260. graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
  261. graphitect/_vendor/archify/references/authoring-contract.md +243 -0
  262. graphitect/_vendor/archify/references/brand-marks.md +65 -0
  263. graphitect/_vendor/archify/references/delivery-contract.md +120 -0
  264. graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
  265. graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
  266. graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
  267. graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
  268. graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
  269. graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
  270. graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
  271. graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
  272. graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
  273. graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
  274. graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
  275. graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
  276. graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
  277. graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
  278. graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
  279. graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
  280. graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
  281. graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
  282. graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
  283. graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
  284. graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
  285. graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
  286. graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
  287. graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
  288. graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
  289. graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
  290. graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
  291. graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
  292. graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
  293. graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
  294. graphitect/_vendor/archify/schemas/README.md +211 -0
  295. graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
  296. graphitect/_vendor/archify/schemas/common.schema.json +115 -0
  297. graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
  298. graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
  299. graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
  300. graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
  301. graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
  302. graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
  303. graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
  304. graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
  305. graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
  306. graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
  307. graphitect/_vendor/archify/skill-release.json +10 -0
  308. graphitect/cli.py +981 -0
  309. graphitect/deliver/__init__.py +5 -0
  310. graphitect/deliver/archify_adapter.py +1877 -0
  311. graphitect/deliver/archify_ir.py +160 -0
  312. graphitect/deliver/archify_repair.py +135 -0
  313. graphitect/deliver/doc_compiler.py +916 -0
  314. graphitect/ground/__init__.py +5 -0
  315. graphitect/ground/describe_source.py +27 -0
  316. graphitect/ground/fullread_source.py +56 -0
  317. graphitect/ground/graphify_source.py +107 -0
  318. graphitect/models.py +118 -0
  319. graphitect/skill/SKILL.md +80 -0
  320. graphitect/skill/agents/openai.yaml +4 -0
  321. graphitect/synthesize/__init__.py +5 -0
  322. graphitect/synthesize/engine.py +281 -0
  323. graphitect/synthesize/llm_backend.py +331 -0
  324. graphitect/synthesize/questions.py +139 -0
  325. graphitect/synthesize/rubric.py +104 -0
  326. graphitect-0.2.0.dist-info/METADATA +284 -0
  327. graphitect-0.2.0.dist-info/RECORD +336 -0
  328. graphitect-0.2.0.dist-info/WHEEL +5 -0
  329. graphitect-0.2.0.dist-info/entry_points.txt +2 -0
  330. graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
  331. graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
  332. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
  333. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
  334. graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
  335. graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
  336. graphitect-0.2.0.dist-info/top_level.txt +2 -0
graphify/cli.py ADDED
@@ -0,0 +1,4745 @@
1
+ """graphify command dispatch — every non-install subcommand.
2
+
3
+ Extracted verbatim from __main__.main(); __main__ now calls dispatch_command(cmd)
4
+ after the install/platform dispatch. Kept out of __main__ to shrink the CLI entry
5
+ module. The path-redirect (`graphify <path>` -> extract) re-enters via a lazy
6
+ import of main to avoid a cli<->__main__ import cycle.
7
+ """
8
+ from __future__ import annotations
9
+ import json
10
+ import os
11
+ import re
12
+ import sys
13
+ import time
14
+ from graphify.paths import GRAPHIFY_OUT as _GRAPHIFY_OUT
15
+ from pathlib import Path, PurePosixPath, PureWindowsPath
16
+
17
+
18
+ _SEARCH_NUDGE = json.dumps({
19
+ "hookSpecificOutput": {
20
+ "hookEventName": "PreToolUse",
21
+ "additionalContext": (
22
+ 'MANDATORY: graphify-out/graph.json exists. You MUST run '
23
+ '`graphify query "<question>"` before grepping raw files. Only grep '
24
+ 'after graphify has oriented you, or to modify/debug specific lines.'
25
+ ),
26
+ }
27
+ }, ensure_ascii=False, separators=(",", ":")) + "\n"
28
+ _READ_NUDGE = json.dumps({
29
+ "hookSpecificOutput": {
30
+ "hookEventName": "PreToolUse",
31
+ "additionalContext": (
32
+ 'MANDATORY: graphify-out/graph.json exists. You MUST run graphify '
33
+ 'before reading source files. Use: `graphify query "<question>"` '
34
+ '(scoped subgraph), `graphify explain "<concept>"`, or '
35
+ '`graphify path "<A>" "<B>"`. Only read raw files after graphify has '
36
+ 'oriented you, or to modify/debug specific lines. This rule applies to '
37
+ 'subagents too — include it in every subagent prompt involving code '
38
+ 'exploration.'
39
+ ),
40
+ }
41
+ }, ensure_ascii=False, separators=(",", ":")) + "\n"
42
+ _READ_NUDGE_STALE = json.dumps({
43
+ "hookSpecificOutput": {
44
+ "hookEventName": "PreToolUse",
45
+ "additionalContext": (
46
+ 'graphify-out/graph.json exists but may be STALE for this file (the file '
47
+ 'changed after the last build). Prefer `graphify query "<question>"` for '
48
+ 'orientation, and run `graphify update` to refresh the graph. Reading the '
49
+ 'file directly is fine.'
50
+ ),
51
+ }
52
+ }, ensure_ascii=False, separators=(",", ":")) + "\n"
53
+ # Strict-mode block (opt-in). Claude Code PreToolUse honors
54
+ # hookSpecificOutput.permissionDecision == "deny" and shows permissionDecisionReason
55
+ # to the model. Fires at most once per session (see _mark_session_denied) so it can
56
+ # never strand an agent: the very next read proceeds with the soft nudge.
57
+ _READ_DENY = json.dumps({
58
+ "hookSpecificOutput": {
59
+ "hookEventName": "PreToolUse",
60
+ "permissionDecision": "deny",
61
+ "permissionDecisionReason": (
62
+ 'graphify strict mode: this project has a fresh knowledge graph that covers '
63
+ 'this file. Run `graphify query "<your question>"` (or `graphify explain` / '
64
+ '`graphify path`) FIRST to orient yourself, then re-issue this Read — it '
65
+ 'will be allowed. This block fires at most once per session; reading raw '
66
+ 'files to modify or debug specific lines is fine after one query. Apply the '
67
+ 'same rule in any subagent prompt that explores code.'
68
+ ),
69
+ }
70
+ }, ensure_ascii=False, separators=(",", ":")) + "\n"
71
+ _HOOK_SOURCE_EXTS = (
72
+ '.py', '.js', '.cjs', '.ts', '.tsx', '.jsx', '.astro', '.vue', '.svelte', '.go',
73
+ '.rs', '.java', '.rb', '.c', '.h', '.cpp', '.hpp', '.cc', '.cs', '.kt',
74
+ '.swift', '.php', '.scala', '.lua', '.sh', '.md', '.rst', '.txt', '.mdx',
75
+ )
76
+ _GEMINI_NUDGE_TEXT = (
77
+ 'graphify: knowledge graph at graphify-out/. For focused questions, run '
78
+ '`graphify query "<question>"` (scoped subgraph, usually much smaller than '
79
+ 'GRAPH_REPORT.md) instead of grepping raw files. Read GRAPH_REPORT.md only '
80
+ 'for broad architecture context.'
81
+ )
82
+
83
+
84
+ def _default_graph_path() -> str:
85
+ return str(Path(_GRAPHIFY_OUT) / "graph.json")
86
+
87
+
88
+ def _stamped_manifest_files(
89
+ files_by_type: dict[str, list[str]],
90
+ sem_result: dict,
91
+ root: Path,
92
+ partial_source_files: "set[str] | None" = None,
93
+ failed_ast_sources: "set[str] | list[str] | None" = None,
94
+ unverified_semantic_sources: "set[str] | list[str] | None" = None,
95
+ ) -> dict[str, list[str]]:
96
+ """Manifest-safe files dict: only stamp semantic files that actually
97
+ produced output (cache hit or fresh extraction). Files whose chunk failed
98
+ have no source_file entry in sem_result — leaving their semantic_hash
99
+ empty so detect_incremental re-queues them (#933).
100
+
101
+ A file in ``partial_source_files`` DID produce output this run, but only a
102
+ truncated fragment of it, so it is excluded from stamping too — otherwise
103
+ detect_incremental would see it "done" and never re-dispatch it, leaving the
104
+ incomplete node set live forever on the warm-incremental path. Same #933
105
+ mechanism: leave it unstamped and it is re-queued next run.
106
+
107
+ ``unverified_semantic_sources`` (#3203): files whose semantic extraction
108
+ under-produced compared to their prior representation (e.g. 3 -> 1 nodes).
109
+ They are excluded from stamping unless --allow-partial is set, so the next
110
+ incremental run retries them.
111
+
112
+ Both sides of the membership test are resolved against the scan ``root``
113
+ before comparing (#1897): node/edge/hyperedge ``source_file`` values are
114
+ root-relative on a fresh extraction while ``files_by_type`` entries are
115
+ absolute (from detect()), so a raw string comparison never matched and
116
+ every freshly-extracted semantic doc was dropped from the manifest.
117
+ Mirrors the #1890 path normalization in graphify.llm.
118
+
119
+ Hyperedges are counted as output (#1920): a chunk whose only result for a
120
+ document is a hyperedge (3+ nodes sharing a concept) is valid output that
121
+ the semantic cache persists per-``source_file`` — omitting it here left the
122
+ doc unstamped, so detect_incremental re-queued it on every run. The stamping
123
+ condition mirrors the cache-write keying (a hyperedge carries its own
124
+ ``source_file``); do not derive it from member nodes.
125
+
126
+ ``failed_ast_sources`` (#2543): code files whose AST extractor errored
127
+ (missing optional extra, etc.) or returned zero nodes. They must not be
128
+ stamped as up-to-date or a later install of the extra will never re-run.
129
+ """
130
+ root = Path(root)
131
+
132
+ def _resolve(value: str) -> Path:
133
+ p = Path(value)
134
+ if not p.is_absolute():
135
+ p = root / p
136
+ try:
137
+ return p.resolve()
138
+ except (OSError, RuntimeError):
139
+ return p
140
+
141
+ sem_extracted: set[Path] = set()
142
+ # #2927: only nodes and hyperedges count as valid semantic output that stamps
143
+ # the manifest. An edge-only result has no entity representation in the graph
144
+ # and must be left unstamped so detect_incremental re-queues it (#933/#1666).
145
+ for coll in ("nodes", "hyperedges"):
146
+ for item in sem_result.get(coll, []):
147
+ sf = item.get("source_file", "")
148
+ if sf:
149
+ sem_extracted.add(_resolve(sf))
150
+ partial_resolved = {_resolve(p) for p in (partial_source_files or set())}
151
+ unverified_resolved = {_resolve(p) for p in (unverified_semantic_sources or set())}
152
+ failed_ast_resolved = {_resolve(p) for p in (failed_ast_sources or [])}
153
+ sem_types = {"document", "paper", "image"}
154
+ return {
155
+ ftype: [
156
+ f for f in flist
157
+ if _resolve(f) not in failed_ast_resolved
158
+ and (
159
+ ftype not in sem_types
160
+ or (
161
+ _resolve(f) in sem_extracted
162
+ and _resolve(f) not in partial_resolved
163
+ and _resolve(f) not in unverified_resolved
164
+ )
165
+ )
166
+ ]
167
+ for ftype, flist in files_by_type.items()
168
+ }
169
+
170
+
171
+ def _handle_unverified_semantic_shrink(
172
+ unverified_shrink,
173
+ *,
174
+ cli_allow_partial: bool,
175
+ files_by_type,
176
+ sem_result,
177
+ target,
178
+ partial_semantic_files,
179
+ failed_ast_sources,
180
+ semantic_files,
181
+ ):
182
+ """Shared handling for the #3203 unverified-semantic-shrink guard on both the
183
+ raw and clustered write paths (they differ only in where the flag is read
184
+ from — ``merged`` vs ``G.graph``). Always prints the actionable notice.
185
+
186
+ Returns None when there is no shrink, else ``(incomplete, manifest_files,
187
+ cleared_semantic)`` — the latter two are None unless the guard armed
188
+ (``not cli_allow_partial``), so the caller mirrors the original inline logic.
189
+ """
190
+ if not unverified_shrink:
191
+ return None
192
+ incomplete = False
193
+ manifest_files = None
194
+ cleared_semantic = None
195
+ if not cli_allow_partial:
196
+ incomplete = True
197
+ unverified_sources = set(unverified_shrink.keys())
198
+ manifest_files = _stamped_manifest_files(
199
+ files_by_type,
200
+ sem_result,
201
+ target,
202
+ partial_source_files=partial_semantic_files,
203
+ failed_ast_sources=failed_ast_sources,
204
+ unverified_semantic_sources=unverified_sources,
205
+ )
206
+ stamped = {f for _flist in manifest_files.values() for f in _flist}
207
+ cleared_semantic = {str(p) for p in semantic_files} - stamped
208
+ details = ", ".join(
209
+ f"'{sf}' ({prior} -> {fresh} nodes)"
210
+ for sf, (prior, fresh) in sorted(unverified_shrink.items())
211
+ )
212
+ print(
213
+ f"[graphify extract] semantic extraction is incomplete: unverified semantic "
214
+ f"shrink detected for {details}. The shrink guard stays armed for this write; "
215
+ "pass --allow-partial to overwrite a larger existing graph anyway.",
216
+ file=sys.stderr,
217
+ )
218
+ return incomplete, manifest_files, cleared_semantic
219
+
220
+
221
+ def _stale_graph_sources(
222
+ graph_path: Path,
223
+ scan_root: Path,
224
+ seen_files: set[str],
225
+ detection: dict | None = None,
226
+ ) -> list[str]:
227
+ """Source files graph.json still references but the current scan no longer
228
+ contains (#1909).
229
+
230
+ Incremental extract's prune set was historically derived from the manifest
231
+ alone (``manifest - corpus``), so a file that became EXCLUDED
232
+ (.graphifyignore/.gitignore/--exclude changed) without being listed in the
233
+ manifest kept its stale nodes in graph.json forever. Derive prune
234
+ candidates from the graph's own node ``source_file``s instead: anything
235
+ the graph references that the post-exclude detect corpus no longer
236
+ contains is stale, whether the file was deleted or newly excluded.
237
+
238
+ Only IN-ROOT paths are candidates: out-of-root/absolute entries
239
+ (--include sources, symlinked external corpora) are never walked by
240
+ detect, so their absence from the corpus is not staleness evidence.
241
+ Relative entries are re-anchored against both the scan root and the
242
+ graph's own output root; only anchors that land inside the scan root
243
+ count. Since #1941 extracts always store source_file relative to the SCAN
244
+ root, so the scan-root anchor is the live one; the out-root anchor stays
245
+ for graphs written by <=0.9.16, which stored them relative to the OUT root
246
+ (e.g. ``../project/x.py``, #555/#1899).
247
+ ``seen_files`` must be the FULL detect output including unclassified
248
+ files, so nodes from walked-but-unsupported sources (e.g. introspected
249
+ Cargo.toml manifests) are not misread as stale.
250
+
251
+ Paths are compared NFC-normalized on both sides: macOS reports NFD
252
+ filenames while graph ``source_file`` entries are typically NFC, and a
253
+ raw-string membership test misread every accented live file as stale
254
+ (#2210; same class as the manifest-layer #2221/#2224).
255
+
256
+ Fail-closed liveness guard (#2210, mirrors watch.py's excluded-vs-deleted
257
+ distinction): a source missing from the scan corpus is only pruned when
258
+ the file is gone from disk, or when its exclusion is PROVABLE from the
259
+ same scan that produced ``seen_files`` — ``detection``'s ``ignored`` /
260
+ ``pruned_noise_dirs`` / ``skipped_sensitive`` output, or detect's
261
+ sensitivity predicate. An alive file that merely failed the membership
262
+ test (path-spelling drift the normalization didn't cover, walk errors,
263
+ …) is KEPT and reported, never mass-evicted.
264
+ """
265
+ from graphify.paths import nfc
266
+ try:
267
+ data = json.loads(graph_path.read_text(encoding="utf-8"))
268
+ except Exception:
269
+ return []
270
+ if not isinstance(data, dict):
271
+ return []
272
+ try:
273
+ root_res = scan_root.resolve()
274
+ except (OSError, RuntimeError):
275
+ root_res = scan_root
276
+ # <out>/graphify-out/graph.json — relative source_files may be anchored here.
277
+ out_base = graph_path.parent.parent
278
+ try:
279
+ out_base = out_base.resolve()
280
+ except (OSError, RuntimeError):
281
+ pass
282
+
283
+ def _within_root(p: Path) -> bool:
284
+ try:
285
+ p.relative_to(root_res)
286
+ return True
287
+ except ValueError:
288
+ pass
289
+ try:
290
+ p.resolve().relative_to(root_res)
291
+ return True
292
+ except (ValueError, OSError, RuntimeError):
293
+ return False
294
+
295
+ seen_nfc = {nfc(s) for s in seen_files}
296
+ seen_basenames = {nfc(os.path.basename(s)) for s in seen_files}
297
+
298
+ def _in_seen(p: Path) -> bool:
299
+ if nfc(str(p)) in seen_nfc:
300
+ return True
301
+ try:
302
+ return nfc(str(p.resolve())) in seen_nfc
303
+ except (OSError, RuntimeError):
304
+ return False
305
+
306
+ # Provable-exclusion evidence from the scan that produced seen_files:
307
+ # individually ignored files are exact entries; ignored/noise-pruned
308
+ # directories are recorded once with a trailing separator and cover
309
+ # their whole subtree. skipped_sensitive entries may carry a
310
+ # " [reason]" suffix.
311
+ excluded_exact: set[str] = set()
312
+ excluded_prefixes: list[str] = []
313
+ if detection:
314
+ for entry in list(detection.get("ignored", [])) + list(
315
+ detection.get("pruned_noise_dirs", [])
316
+ ):
317
+ e = nfc(str(entry))
318
+ if e.endswith(os.sep) or e.endswith("/"):
319
+ excluded_prefixes.append(e)
320
+ else:
321
+ excluded_exact.add(e)
322
+ for entry in detection.get("skipped_sensitive", []):
323
+ excluded_exact.add(nfc(str(entry).split(" [", 1)[0]))
324
+
325
+ def _provably_excluded(c: Path) -> bool:
326
+ spellings = [nfc(str(c))]
327
+ try:
328
+ spellings.append(nfc(str(c.resolve())))
329
+ except (OSError, RuntimeError):
330
+ pass
331
+ for s in spellings:
332
+ if s in excluded_exact:
333
+ return True
334
+ if any(s.startswith(pref) for pref in excluded_prefixes):
335
+ return True
336
+ try:
337
+ from graphify.detect import _is_sensitive as _det_sensitive
338
+ if _det_sensitive(c):
339
+ return True
340
+ except Exception:
341
+ pass
342
+ return False
343
+
344
+ stale: list[str] = []
345
+ kept_alive: list[str] = []
346
+ checked: set[str] = set()
347
+ for n in data.get("nodes", []):
348
+ if not isinstance(n, dict):
349
+ continue
350
+ sf = n.get("source_file")
351
+ if not sf or not isinstance(sf, str) or sf in checked:
352
+ continue
353
+ checked.add(sf)
354
+ if "://" in sf:
355
+ continue # remote/virtual source (e.g. Google Workspace), not a scanned path
356
+ p = Path(sf)
357
+ if p.is_absolute():
358
+ candidates = [p]
359
+ else:
360
+ rel = sf.replace("\\", "/")
361
+ bases = [root_res]
362
+ if out_base != root_res:
363
+ bases.append(out_base)
364
+ candidates = [
365
+ Path(os.path.normpath(str(base / rel))) for base in bases
366
+ ]
367
+ in_root = [c for c in candidates if _within_root(c)]
368
+ if not in_root:
369
+ continue # out-of-root under every anchor: never prune
370
+ if any(_in_seen(c) for c in in_root):
371
+ continue # still part of the scan corpus
372
+ # Fail-closed liveness guard (#2210): absence from the corpus is
373
+ # only deletion evidence when the file is actually gone from disk.
374
+ alive = []
375
+ for c in in_root:
376
+ try:
377
+ if c.exists():
378
+ alive.append(c)
379
+ except OSError:
380
+ pass
381
+ if alive:
382
+ if all(_provably_excluded(c) for c in alive):
383
+ stale.append(sf) # alive but excluded under current rules (#1909)
384
+ else:
385
+ kept_alive.append(sf)
386
+ continue
387
+ # No anchored candidate exists, but a legacy bare-basename spelling
388
+ # can't be anchored reliably — a live corpus file with the same name
389
+ # means deletion is unproven; keep.
390
+ rel_sf = sf.replace("\\", "/")
391
+ if "/" not in rel_sf and nfc(rel_sf) in seen_basenames:
392
+ kept_alive.append(sf)
393
+ continue
394
+ stale.append(sf)
395
+ if kept_alive:
396
+ print(
397
+ f"[graphify] fail-closed: kept node(s) from {len(kept_alive)} "
398
+ "source file(s) that left the scan corpus but still exist on disk "
399
+ "(ignore rules or filters changed?). Run a full re-extraction to "
400
+ "purge them if the exclusion is intentional.",
401
+ file=sys.stderr,
402
+ )
403
+ return stale
404
+
405
+
406
+ def _zero_node_stamped_code_sources(
407
+ graph_path: Path,
408
+ scan_root: Path,
409
+ unchanged_code: list[str],
410
+ ) -> list[str]:
411
+ """Manifest-stamped code files with a registered extractor but ZERO nodes
412
+ in the existing graph.json (#2543 heal).
413
+
414
+ The failed-source unstamping only covers failures that happen AFTER it
415
+ shipped; a manifest poisoned by an earlier run (extraction failed, hashes
416
+ stamped anyway) keeps reporting the file unchanged forever, and the only
417
+ documented recovery was deleting graphify-out/. A stamped file that the
418
+ graph has no nodes for, despite an extractor being wired up for it, is
419
+ exactly that state — re-queue it as changed. Bounded by the same no-wedge
420
+ property: if it fails again this run it is now left unstamped, and if it
421
+ succeeds its nodes enter graph.json so the next scan stops re-queuing it.
422
+
423
+ Membership mirrors the ``source_file`` spellings extracts store (#1897/
424
+ #1941: scan-root-relative, forward slash; absolute for out-of-root) and
425
+ compares NFC-normalized (#2210/#2221).
426
+ """
427
+ if not unchanged_code:
428
+ return []
429
+ from graphify.paths import nfc
430
+ try:
431
+ data = json.loads(graph_path.read_text(encoding="utf-8"))
432
+ except Exception:
433
+ return []
434
+ if not isinstance(data, dict):
435
+ return []
436
+ try:
437
+ root_res = scan_root.resolve()
438
+ except (OSError, RuntimeError):
439
+ root_res = scan_root
440
+ # <out>/graphify-out/graph.json — legacy relative source_files may be
441
+ # anchored here instead of the scan root (<=0.9.16, #555/#1899).
442
+ out_base = graph_path.parent.parent
443
+ try:
444
+ out_base = out_base.resolve()
445
+ except (OSError, RuntimeError):
446
+ pass
447
+
448
+ present: set[str] = set()
449
+ for n in data.get("nodes", []):
450
+ if not isinstance(n, dict):
451
+ continue
452
+ sf = n.get("source_file")
453
+ if not sf or not isinstance(sf, str):
454
+ continue
455
+ present.add(nfc(sf))
456
+ p = Path(sf)
457
+ if p.is_absolute():
458
+ try:
459
+ present.add(nfc(str(p.resolve())))
460
+ except (OSError, RuntimeError):
461
+ pass
462
+ else:
463
+ rel = sf.replace("\\", "/")
464
+ for base in (root_res, out_base):
465
+ present.add(nfc(os.path.normpath(str(base / rel))))
466
+
467
+ from graphify.extract import _get_extractor
468
+ healed: list[str] = []
469
+ for f in unchanged_code:
470
+ p = Path(f)
471
+ if _get_extractor(p) is None:
472
+ continue # no extractor: absence from the graph is expected
473
+ spellings = {nfc(str(p))}
474
+ try:
475
+ spellings.add(nfc(str(p.resolve())))
476
+ except (OSError, RuntimeError):
477
+ pass
478
+ try:
479
+ spellings.add(nfc(p.resolve().relative_to(root_res).as_posix()))
480
+ except (ValueError, OSError, RuntimeError):
481
+ pass
482
+ if spellings & present:
483
+ continue # the graph has this file: stamp is honest
484
+ healed.append(f)
485
+ return healed
486
+
487
+
488
+ def _zero_node_stamped_semantic_sources(
489
+ graph_path: Path,
490
+ scan_root: Path,
491
+ unchanged_semantic: list[str],
492
+ ) -> list[str]:
493
+ """Manifest-stamped semantic files (doc/paper/image) with ZERO nodes
494
+ and ZERO hyperedges in the existing graph.json (#2927 heal).
495
+
496
+ A manifest poisoned before #2927 (edge-only result cached and stamped)
497
+ keeps reporting the file unchanged forever, freezing it out of the graph.
498
+ Re-queue any unchanged semantic file that has neither nodes nor hyperedges
499
+ in graph.json. If it succeeds, its nodes enter graph.json; if it produces
500
+ no nodes or fails, it is now left unstamped, so this cannot wedge.
501
+
502
+ Membership mirrors the ``source_file`` spellings extracts store (#1897/
503
+ #1941: scan-root-relative, forward slash; absolute for out-of-root) and
504
+ compares NFC-normalized (#2210/#2221).
505
+ """
506
+ if not unchanged_semantic:
507
+ return []
508
+ from graphify.paths import nfc
509
+ try:
510
+ data = json.loads(graph_path.read_text(encoding="utf-8"))
511
+ except Exception:
512
+ return []
513
+ if not isinstance(data, dict):
514
+ return []
515
+ try:
516
+ root_res = scan_root.resolve()
517
+ except (OSError, RuntimeError):
518
+ root_res = scan_root
519
+ out_base = graph_path.parent.parent
520
+ try:
521
+ out_base = out_base.resolve()
522
+ except (OSError, RuntimeError):
523
+ pass
524
+
525
+ present: set[str] = set()
526
+ for n in data.get("nodes", []):
527
+ if not isinstance(n, dict):
528
+ continue
529
+ sf = n.get("source_file")
530
+ if not sf or not isinstance(sf, str):
531
+ continue
532
+ present.add(nfc(sf))
533
+ p = Path(sf)
534
+ if p.is_absolute():
535
+ try:
536
+ present.add(nfc(str(p.resolve())))
537
+ except (OSError, RuntimeError):
538
+ pass
539
+ else:
540
+ rel = sf.replace("\\", "/")
541
+ for base in (root_res, out_base):
542
+ present.add(nfc(os.path.normpath(str(base / rel))))
543
+
544
+ hyper_items = list(data.get("hyperedges", []) or [])
545
+ if isinstance((data.get("graph") or {}).get("hyperedges"), list):
546
+ hyper_items.extend(data["graph"]["hyperedges"])
547
+ for h in hyper_items:
548
+ if not isinstance(h, dict):
549
+ continue
550
+ sf = h.get("source_file")
551
+ if not sf or not isinstance(sf, str):
552
+ continue
553
+ present.add(nfc(sf))
554
+ p = Path(sf)
555
+ if p.is_absolute():
556
+ try:
557
+ present.add(nfc(str(p.resolve())))
558
+ except (OSError, RuntimeError):
559
+ pass
560
+ else:
561
+ rel = sf.replace("\\", "/")
562
+ for base in (root_res, out_base):
563
+ present.add(nfc(os.path.normpath(str(base / rel))))
564
+
565
+ healed: list[str] = []
566
+ for f in unchanged_semantic:
567
+ p = Path(f)
568
+ spellings = {nfc(str(p))}
569
+ try:
570
+ spellings.add(nfc(str(p.resolve())))
571
+ except (OSError, RuntimeError):
572
+ pass
573
+ try:
574
+ spellings.add(nfc(p.resolve().relative_to(root_res).as_posix()))
575
+ except (ValueError, OSError, RuntimeError):
576
+ pass
577
+ if spellings & present:
578
+ continue # the graph has nodes or hyperedges for this file: stamp is honest
579
+ healed.append(f)
580
+ return healed
581
+
582
+
583
+ def _prune_graph_json_sources(graph_path: Path, stale_sources: list[str]) -> int:
584
+ """Drop nodes/edges/hyperedges owned by ``stale_sources`` from graph.json
585
+ in place. Returns the number of nodes removed.
586
+
587
+ Used by the ``--no-cluster`` incremental early-exit: that path never runs
588
+ ``build_merge`` (it would raw-dump only the new chunks), so an
589
+ exclusion-only change must prune the existing raw graph directly or the
590
+ newly-excluded file's nodes survive forever (#1909).
591
+ ``stale_sources`` comes from :func:`_stale_graph_sources`, i.e. the
592
+ graph's own ``source_file`` spellings, so exact string matching is enough.
593
+ """
594
+ try:
595
+ data = json.loads(graph_path.read_text(encoding="utf-8"))
596
+ except Exception:
597
+ return 0
598
+ if not isinstance(data, dict):
599
+ return 0
600
+ stale = set(stale_sources)
601
+ links_key = "links" if "links" in data else "edges"
602
+ nodes = [n for n in data.get("nodes", []) if isinstance(n, dict)]
603
+ kept_nodes = [n for n in nodes if n.get("source_file") not in stale]
604
+ removed_ids = {
605
+ n.get("id") for n in nodes if n.get("source_file") in stale
606
+ }
607
+ n_removed = len(nodes) - len(kept_nodes)
608
+ kept_edges = [
609
+ e for e in data.get(links_key, [])
610
+ if isinstance(e, dict)
611
+ and e.get("source_file") not in stale
612
+ and e.get("source") not in removed_ids
613
+ and e.get("target") not in removed_ids
614
+ ]
615
+ kept_hyper = [
616
+ h for h in data.get("hyperedges", [])
617
+ if isinstance(h, dict) and h.get("source_file") not in stale
618
+ ]
619
+ if n_removed == 0 and len(kept_edges) == len(data.get(links_key, [])) and (
620
+ len(kept_hyper) == len(data.get("hyperedges", []))
621
+ ):
622
+ return 0
623
+ data["nodes"] = kept_nodes
624
+ data[links_key] = kept_edges
625
+ if "hyperedges" in data:
626
+ data["hyperedges"] = kept_hyper
627
+ from graphify.export import backup_if_protected as _backup
628
+ _backup(graph_path.parent)
629
+ from graphify.paths import write_json_atomic
630
+ write_json_atomic(graph_path, data, indent=2)
631
+ return n_removed
632
+
633
+
634
+ class _StageTimer:
635
+ """Print per-stage wall-clock timings to stderr when --timing is set (#1490).
636
+
637
+ Monotonic (perf_counter), diagnostic-only: emits ``[graphify timing] <stage>:
638
+ N.Ns`` after each stage and a final total. Off by default, so normal output is
639
+ byte-identical and machine-read stdout is untouched.
640
+ """
641
+
642
+ def __init__(self, enabled: bool) -> None:
643
+ import time as _time
644
+ self._now = _time.perf_counter
645
+ self.enabled = enabled
646
+ self.start = self._now()
647
+ self._last = self.start
648
+
649
+ def mark(self, stage: str) -> None:
650
+ now = self._now()
651
+ if self.enabled:
652
+ print(f"[graphify timing] {stage}: {now - self._last:.1f}s", file=sys.stderr)
653
+ self._last = now
654
+
655
+ def total(self) -> None:
656
+ if self.enabled:
657
+ print(f"[graphify timing] total: {self._now() - self.start:.1f}s", file=sys.stderr)
658
+ def _enforce_graph_size_cap_or_exit(gp: Path) -> None:
659
+ """Reject oversized graph files before parsing (CLI exit-on-fail flavor).
660
+
661
+ Delegates to ``graphify.security.check_graph_file_size_cap`` and turns the
662
+ raised ``ValueError`` into a CLI-style ``error: ...`` message + exit 1.
663
+ Use this from ``__main__.py`` subcommands that already use the ``print +
664
+ sys.exit(1)`` idiom. Library/MCP/loader callers (``serve._load_graph``,
665
+ ``build``, ``benchmark``, ``tree_html``, ``callflow_html``, ``prs``,
666
+ ``global_graph``, ``watch``, ``export``) call the security helper directly
667
+ and let the ``ValueError`` propagate.
668
+ """
669
+ from graphify.security import check_graph_file_size_cap
670
+ try:
671
+ check_graph_file_size_cap(gp)
672
+ except ValueError as exc:
673
+ print(f"error: {exc}", file=sys.stderr)
674
+ sys.exit(1)
675
+ def _hook_strict_enabled(flag: bool) -> bool:
676
+ """Resolve strict mode: GRAPHIFY_HOOK_STRICT env overrides the baked-in flag
677
+ (truthy forces on without a reinstall, falsy is the kill switch); unset defers
678
+ to the flag the installed hook command carried."""
679
+ v = os.environ.get("GRAPHIFY_HOOK_STRICT", "").strip().lower()
680
+ if v in ("1", "true", "yes", "on"):
681
+ return True
682
+ if v in ("0", "false", "no", "off"):
683
+ return False
684
+ return flag
685
+
686
+
687
+ def _touch_query_stamp(graph_path: "Path") -> None:
688
+ """Record that graphify oriented the agent recently, next to the queried graph.
689
+ The strict guard suppresses its block while this stamp is fresh. Fail-silent."""
690
+ try:
691
+ from graphify.paths import write_text_atomic
692
+ stamp = Path(graph_path).parent / "cache" / "last_query_stamp"
693
+ stamp.parent.mkdir(parents=True, exist_ok=True)
694
+ write_text_atomic(stamp, str(time.time()))
695
+ except Exception:
696
+ pass
697
+
698
+
699
+ def _query_stamp_fresh() -> bool:
700
+ """True if a query/explain/path ran within GRAPHIFY_HOOK_STRICT_TTL (default
701
+ 1800s) — recent orientation, so strict mode does not block this read."""
702
+ from graphify.paths import out_path
703
+ try:
704
+ ttl = float(os.environ.get("GRAPHIFY_HOOK_STRICT_TTL", "1800"))
705
+ return (time.time() - out_path("cache", "last_query_stamp").stat().st_mtime) < ttl
706
+ except Exception:
707
+ return False
708
+
709
+
710
+ def _mark_session_denied(session_id: str) -> bool:
711
+ """Atomically claim a one-time strict block for this session. Returns True only
712
+ on the FIRST call for a given session id (O_EXCL create wins once); every later
713
+ call — or any error — returns False, so a session is blocked at most once and an
714
+ agent can never be stranded. Best-effort GC of markers older than 24h."""
715
+ from graphify.paths import out_path
716
+ sid = re.sub(r"[^A-Za-z0-9_-]", "_", str(session_id))[:64]
717
+ if not sid:
718
+ return False
719
+ try:
720
+ d = out_path("cache", "hook_sessions")
721
+ d.mkdir(parents=True, exist_ok=True)
722
+ fd = os.open(str(d / f"{sid}.denied"), os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o644)
723
+ os.close(fd)
724
+ try:
725
+ cutoff = time.time() - 86400
726
+ for entry in os.scandir(d):
727
+ try:
728
+ if entry.stat().st_mtime < cutoff:
729
+ os.unlink(entry.path)
730
+ except OSError:
731
+ pass
732
+ except OSError:
733
+ pass
734
+ return True
735
+ except FileExistsError:
736
+ return False
737
+ except Exception:
738
+ return False
739
+
740
+
741
+ _SEARCH_COMMANDS = frozenset({
742
+ "grep", "egrep", "fgrep", "zgrep", "rg", "ripgrep", "find", "fd", "ack", "ag",
743
+ })
744
+ # Prefix words that wrap another command; the real executable follows them.
745
+ _COMMAND_WRAPPERS = frozenset({
746
+ "sudo", "command", "exec", "nohup", "time", "nice", "ionice", "env",
747
+ "xargs", "timeout", "stdbuf", "doas",
748
+ })
749
+ _HEREDOC_OPEN_RE = re.compile(r"<<-?\s*(['\"]?)(\w+)\1")
750
+
751
+
752
+ def _bash_invokes_search(cmd_str: str) -> bool:
753
+ """Whether a Bash command actually RUNS a search tool (#3121).
754
+
755
+ The old test was a plain substring scan over the whole command string,
756
+ including quoted arguments and heredoc bodies - so `git commit -m "add
757
+ flag support"` fired ("flag " contains "ag "), prose containing "find "
758
+ fired, and a design doc written via heredoc fired if its body mentioned
759
+ grep. Every false positive injects a nudge where graphify has nothing to
760
+ contribute and trains the agent to skim the line.
761
+
762
+ Decide on the command's executed tokens instead: drop heredoc bodies and
763
+ quoted spans, split on shell operators, and match the executable at each
764
+ command position (wrappers like sudo/xargs/env skipped; `git grep` and
765
+ `VAR=x grep ...` still count; a search-tool name inside prose does not).
766
+ """
767
+ text = cmd_str
768
+ # Drop heredoc bodies: from the line after `<<WORD` through the line that
769
+ # is exactly WORD. An unterminated heredoc drops to the end of the string.
770
+ m = _HEREDOC_OPEN_RE.search(text)
771
+ while m:
772
+ nl_idx = text.find("\n", m.end())
773
+ if nl_idx == -1:
774
+ break
775
+ term = re.compile(r"^\s*" + re.escape(m.group(2)) + r"\s*$", re.MULTILINE)
776
+ t = term.search(text, nl_idx + 1)
777
+ if t is None:
778
+ # Unterminated heredoc: everything after the opener line is body.
779
+ text = text[: nl_idx + 1]
780
+ break
781
+ text = text[: nl_idx + 1] + text[t.end():]
782
+ m = _HEREDOC_OPEN_RE.search(text, nl_idx + 1)
783
+ # Drop quoted spans (backslash escapes inside double quotes are irrelevant
784
+ # here - anything quoted is an argument, never the executable).
785
+ text = re.sub(r"'[^']*'", " ", text)
786
+ text = re.sub(r'"[^"]*"', " ", text)
787
+ # Split into command segments at shell operators / substitution boundaries.
788
+ for segment in re.split(r"[|;&\n]|\$\(|`|\(|\)|\{|\}", text):
789
+ tokens = segment.split()
790
+ i = 0
791
+ while i < len(tokens):
792
+ tok = tokens[i]
793
+ if "=" in tok.split("/")[-1] and not tok.startswith(("-", "/")):
794
+ i += 1 # VAR=value prefix
795
+ continue
796
+ name = tok.replace("\\", "/").rsplit("/", 1)[-1].lower()
797
+ name = name[:-4] if name.endswith(".exe") else name
798
+ if name in _COMMAND_WRAPPERS:
799
+ i += 1
800
+ # skip the wrapper's own flags (`xargs -0`, `env -i`)
801
+ while i < len(tokens) and tokens[i].startswith("-"):
802
+ i += 1
803
+ continue
804
+ if name in _SEARCH_COMMANDS:
805
+ return True
806
+ if name == "git" and any(
807
+ t2 == "grep" for t2 in tokens[i + 1:i + 4] if not t2.startswith("-")
808
+ ):
809
+ return True
810
+ break # first real token decides this segment
811
+ return False
812
+
813
+
814
+ def _run_hook_guard(kind: str, strict: bool = False) -> None:
815
+ """Shell-agnostic PreToolUse guard (#522).
816
+
817
+ Reads the tool-call JSON from stdin and, when a fresh in-project knowledge graph
818
+ exists, nudges the agent to use graphify instead of grepping/reading raw files.
819
+ Replaces the old inline bash hooks that failed to parse on Windows.
820
+
821
+ Fails open everywhere: any error, or a non-matching tool call, prints nothing
822
+ and the caller exits 0, so a legitimate tool call is never blocked by a bug.
823
+
824
+ In strict mode (opt-in, Claude Code Read only) the FIRST raw read of indexed,
825
+ in-project, fresh code per session is DENIED with a redirect to `graphify query`
826
+ (permissionDecision), then downgrades to the soft nudge — it fires at most once
827
+ per session and can never strand the agent. Search (Bash) and Glob stay
828
+ nudge-only: a compound shell command has no single parseable target and blocking
829
+ file listing would strand navigation. #1840: reads of out-of-project files are
830
+ ignored, and a graph that is stale for the target file softens to a non-mandatory
831
+ nudge instead of blocking or demanding.
832
+ """
833
+ from graphify.paths import out_path, GRAPHIFY_OUT_NAME
834
+ # Gemini's BeforeTool hook takes no stdin and must ALWAYS return a decision so
835
+ # the tool is never blocked; the graph nudge is appended only when a graph
836
+ # exists. Handled before the stdin read below (which the search/read guards need).
837
+ if kind == "gemini":
838
+ payload = {"decision": "allow"}
839
+ try:
840
+ if out_path("graph.json").is_file():
841
+ payload["additionalContext"] = _GEMINI_NUDGE_TEXT
842
+ except Exception:
843
+ pass
844
+ sys.stdout.write(json.dumps(payload, ensure_ascii=False, separators=(",", ":")))
845
+ return
846
+ try:
847
+ d = json.loads(sys.stdin.buffer.read().decode("utf-8", "replace"))
848
+ except Exception:
849
+ return
850
+ if not isinstance(d, dict):
851
+ return
852
+ t = d.get("tool_input", d)
853
+ if not isinstance(t, dict):
854
+ return
855
+ try:
856
+ if kind == "search":
857
+ cmd_str = str(t.get("command", "") or "")
858
+ # Two input shapes reach this guard (matcher "Bash|Grep", #1986):
859
+ # the Bash tool carries `command`, while Claude Code's dedicated
860
+ # Grep tool carries `pattern` (plus optional path/glob) and no
861
+ # command — a Grep call IS a content search by definition, so it
862
+ # nudges whenever a graph exists. For Bash, decide on the
863
+ # command's EXECUTED tokens (#3121): the old whole-string
864
+ # substring scan fired on quoted prose ('add flag support'
865
+ # contains "ag ") and on heredoc bodies that merely mention
866
+ # grep. Nudge-only, even in strict mode — see the docstring.
867
+ is_grep_tool = not cmd_str and bool(t.get("pattern"))
868
+ is_bash_search = bool(cmd_str) and _bash_invokes_search(cmd_str)
869
+ if (is_grep_tool or is_bash_search) and out_path("graph.json").is_file():
870
+ sys.stdout.write(_SEARCH_NUDGE)
871
+ elif kind == "read":
872
+ vals = [str(t.get("file_path") or ""), str(t.get("pattern") or ""), str(t.get("path") or "")]
873
+ j = " ".join(vals).lower().replace("\\", "/")
874
+ tails = [
875
+ "." + seg.rsplit(".", 1)[-1]
876
+ for v in vals if v
877
+ for seg in [v.lower().replace("\\", "/").rsplit("/", 1)[-1]]
878
+ if "." in seg
879
+ ]
880
+ under_out = "graphify-out/" in j or (GRAPHIFY_OUT_NAME.lower() + "/") in j
881
+ if under_out or not any(tl in _HOOK_SOURCE_EXTS for tl in tails):
882
+ return
883
+ # #1840 (a): skip files outside the graph's project. cwd (or
884
+ # CLAUDE_PROJECT_DIR, which Claude Code sets) is the project root, since
885
+ # the guard only triggers when graph.json exists relative to cwd. A path
886
+ # candidate that resolves outside that root is out-of-project.
887
+ root = Path(os.environ.get("CLAUDE_PROJECT_DIR") or os.getcwd())
888
+ try:
889
+ root = root.resolve()
890
+ except (OSError, RuntimeError):
891
+ pass
892
+ path_vals = [str(t.get("file_path") or ""), str(t.get("path") or "")]
893
+ explicit = [v for v in path_vals if v]
894
+ if explicit:
895
+ in_project = False
896
+ for v in explicit:
897
+ p = Path(v)
898
+ if _is_cwd_relative(v):
899
+ in_project = True # relative -> anchored at cwd == in project
900
+ break
901
+ try:
902
+ p.resolve().relative_to(root)
903
+ in_project = True
904
+ break
905
+ except (ValueError, OSError, RuntimeError):
906
+ continue
907
+ if not in_project:
908
+ return
909
+ # One stat for existence + mtime of the graph.
910
+ try:
911
+ gmtime = os.stat(str(out_path("graph.json"))).st_mtime
912
+ except OSError:
913
+ return
914
+ # #1840 (b): stale-for-target -> soften, never block. The target file
915
+ # changed after the last build, or watch flagged the tree.
916
+ stale = False
917
+ fp = str(t.get("file_path") or "")
918
+ if fp:
919
+ try:
920
+ stale = os.stat(fp).st_mtime > gmtime
921
+ except OSError:
922
+ stale = False
923
+ try:
924
+ if out_path("needs_update").exists():
925
+ stale = True
926
+ except Exception:
927
+ pass
928
+ if stale:
929
+ sys.stdout.write(_READ_NUDGE_STALE)
930
+ return
931
+ # Strict block: Read tool only, first time per session, not recently
932
+ # oriented, and the file is demonstrably indexed.
933
+ tool_name = d.get("tool_name")
934
+ if _hook_strict_enabled(strict) and tool_name in (None, "Read") \
935
+ and not _query_stamp_fresh() \
936
+ and _target_is_indexed(fp, root) \
937
+ and _mark_session_denied(str(d.get("session_id") or "")):
938
+ sys.stdout.write(_READ_DENY)
939
+ return
940
+ sys.stdout.write(_READ_NUDGE)
941
+ except Exception:
942
+ pass
943
+
944
+
945
+ def _is_cwd_relative(value: str) -> bool:
946
+ r"""Whether *value* is anchored at the current working directory.
947
+
948
+ The hook's out-of-project guard needs "is this path resolved against cwd?",
949
+ and ``Path.is_absolute()`` is the wrong question for it on Windows. A
950
+ driveless rooted path like ``/tmp/x.py`` — the form POSIX-shaped hosts, WSL
951
+ and Git Bash send — is NOT absolute there (no drive letter), but it is not
952
+ cwd-relative either: Windows anchors it at the current DRIVE root, so
953
+ ``Path("/somewhere/else/x.py").resolve()`` is ``C:\somewhere\else\x.py``,
954
+ which is outside the project unless the project sits at ``C:\``. Reading it
955
+ as cwd-relative made the guard declare it in-project and emit the read nudge
956
+ (and, in strict mode, the once-per-session deny) for files the graph has
957
+ nothing to say about.
958
+
959
+ ``C:x.py`` is the same trap from the other side: drive-relative, anchored at
960
+ that drive's current directory rather than cwd.
961
+
962
+ So the test is "no root and no drive", not "not absolute". These stay the
963
+ host's own rules — the path is about to be resolved against this filesystem,
964
+ so ``paths.is_absolute_any_platform`` (for stored, portable paths) is
965
+ deliberately not used. On POSIX ``root`` is set exactly when the path is
966
+ absolute and ``drive`` is always empty, so this is unchanged there.
967
+ """
968
+ pure = PureWindowsPath(value) if os.name == "nt" else PurePosixPath(value)
969
+ return not pure.root and not pure.drive
970
+
971
+
972
+ def _target_is_indexed(file_path: str, root: "Path") -> bool:
973
+ """Guard the strict deny: only block a read of a file the graph actually indexes.
974
+ Reads manifest.json (cheap, capped); on any doubt (missing/corrupt/oversized
975
+ manifest, unresolvable path) returns True so the once-per-session deny still
976
+ applies — that block is self-limiting, so erring toward it is safe."""
977
+ from graphify.paths import out_path
978
+ if not file_path:
979
+ return True
980
+ try:
981
+ mp = out_path("manifest.json")
982
+ st = mp.stat()
983
+ if st.st_size > 2_000_000:
984
+ return True
985
+ manifest = json.loads(mp.read_text(encoding="utf-8"))
986
+ if not isinstance(manifest, dict) or not manifest:
987
+ return True
988
+ p = Path(file_path)
989
+ rels = set()
990
+ try:
991
+ rels.add(p.resolve().relative_to(root).as_posix())
992
+ except (ValueError, OSError, RuntimeError):
993
+ pass
994
+ rels.add(p.name)
995
+ keys = {str(k).replace("\\", "/") for k in manifest}
996
+ abskey = str(p).replace("\\", "/")
997
+ return abskey in keys or any(r and (r in keys or any(k.endswith("/" + r) or k == r for k in keys)) for r in rels)
998
+ except Exception:
999
+ return True
1000
+ def _clone_repo(
1001
+ url: str, branch: str | None = None, out_dir: Path | None = None
1002
+ ) -> Path:
1003
+ """Clone a GitHub repo to a local cache dir and return the path.
1004
+
1005
+ Clones into ~/.graphify/repos/<owner>/<repo> by default so repeated
1006
+ runs on the same URL reuse the existing clone (git pull instead of clone).
1007
+ """
1008
+ import subprocess as _sp
1009
+ import re as _re
1010
+
1011
+ # Normalise URL — strip trailing .git if present
1012
+ url = url.rstrip("/")
1013
+ if not url.endswith(".git"):
1014
+ git_url = url + ".git"
1015
+ else:
1016
+ git_url = url
1017
+ url = url[:-4]
1018
+
1019
+ # Extract owner/repo from URL
1020
+ m = _re.search(r"github\.com[:/]([^/]+)/([^/]+?)(?:\.git)?$", url)
1021
+ if not m:
1022
+ print(f"error: not a recognised GitHub URL: {url}", file=sys.stderr)
1023
+ sys.exit(1)
1024
+ owner, repo = m.group(1), m.group(2)
1025
+
1026
+ if out_dir:
1027
+ dest = out_dir
1028
+ else:
1029
+ dest = Path.home() / ".graphify" / "repos" / owner / repo
1030
+
1031
+ if branch and branch.startswith("-"):
1032
+ print(f"error: invalid branch name: {branch!r}", file=sys.stderr)
1033
+ sys.exit(1)
1034
+
1035
+ if dest.exists():
1036
+ print(f"Repo already cloned at {dest} - pulling latest...", flush=True)
1037
+ cmd = ["git", "-C", str(dest), "pull"]
1038
+ if branch:
1039
+ cmd += ["origin", "--", branch]
1040
+ result = _sp.run(cmd, capture_output=True, text=True)
1041
+ if result.returncode != 0:
1042
+ print(f"warning: git pull failed:\n{result.stderr}", file=sys.stderr)
1043
+ else:
1044
+ dest.parent.mkdir(parents=True, exist_ok=True)
1045
+ print(f"Cloning {url} -> {dest} ...", flush=True)
1046
+ cmd = ["git", "clone", "--depth", "1"]
1047
+ if branch:
1048
+ cmd += ["--branch", branch]
1049
+ cmd += ["--", git_url, str(dest)]
1050
+ result = _sp.run(cmd, capture_output=True, text=True)
1051
+ if result.returncode != 0:
1052
+ print(f"error: git clone failed:\n{result.stderr}", file=sys.stderr)
1053
+ sys.exit(1)
1054
+
1055
+ print(f"Ready at: {dest}", flush=True)
1056
+ return dest
1057
+
1058
+
1059
+ def _reenter_main() -> None:
1060
+ from graphify.__main__ import main
1061
+ main()
1062
+
1063
+
1064
+ def dispatch_command(cmd: str) -> None:
1065
+ if cmd == "provider":
1066
+ from graphify.llm import _custom_providers_path, BACKENDS
1067
+ import json as _json
1068
+ subcmd = sys.argv[2] if len(sys.argv) > 2 else ""
1069
+ global_path = _custom_providers_path(global_=True)
1070
+
1071
+ if subcmd == "list":
1072
+ global_path.parent.mkdir(parents=True, exist_ok=True)
1073
+ existing: dict = {}
1074
+ if global_path.is_file():
1075
+ try:
1076
+ existing = _json.loads(global_path.read_text(encoding="utf-8"))
1077
+ except Exception:
1078
+ pass
1079
+ if not existing:
1080
+ print("No custom providers registered.")
1081
+ else:
1082
+ for name in existing:
1083
+ print(f" {name} ({existing[name].get('base_url', '')})")
1084
+
1085
+ elif subcmd == "show":
1086
+ name = sys.argv[3] if len(sys.argv) > 3 else ""
1087
+ if not name:
1088
+ print("Usage: graphify provider show <name>", file=sys.stderr)
1089
+ sys.exit(1)
1090
+ existing = {}
1091
+ if global_path.is_file():
1092
+ try:
1093
+ existing = _json.loads(global_path.read_text(encoding="utf-8"))
1094
+ except Exception:
1095
+ pass
1096
+ if name not in existing:
1097
+ print(f"Provider '{name}' not found.", file=sys.stderr)
1098
+ sys.exit(1)
1099
+ print(_json.dumps({name: existing[name]}, indent=2))
1100
+
1101
+ elif subcmd == "add":
1102
+ args = sys.argv[3:]
1103
+ name = args[0] if args and not args[0].startswith("-") else ""
1104
+ if not name:
1105
+ print("Usage: graphify provider add <name> --base-url URL --default-model MODEL --env-key KEY", file=sys.stderr)
1106
+ sys.exit(1)
1107
+ if name in BACKENDS:
1108
+ print(f"Error: '{name}' is a built-in provider and cannot be overridden.", file=sys.stderr)
1109
+ sys.exit(1)
1110
+ base_url = ""
1111
+ default_model = ""
1112
+ env_key = ""
1113
+ pricing_input = 0.0
1114
+ pricing_output = 0.0
1115
+ i = 1
1116
+ while i < len(args):
1117
+ a = args[i]
1118
+ if a == "--base-url" and i + 1 < len(args):
1119
+ base_url = args[i + 1]; i += 2
1120
+ elif a.startswith("--base-url="):
1121
+ base_url = a.split("=", 1)[1]; i += 1
1122
+ elif a == "--default-model" and i + 1 < len(args):
1123
+ default_model = args[i + 1]; i += 2
1124
+ elif a.startswith("--default-model="):
1125
+ default_model = a.split("=", 1)[1]; i += 1
1126
+ elif a == "--env-key" and i + 1 < len(args):
1127
+ env_key = args[i + 1]; i += 2
1128
+ elif a.startswith("--env-key="):
1129
+ env_key = a.split("=", 1)[1]; i += 1
1130
+ elif a == "--pricing-input" and i + 1 < len(args):
1131
+ pricing_input = float(args[i + 1]); i += 2
1132
+ elif a == "--pricing-output" and i + 1 < len(args):
1133
+ pricing_output = float(args[i + 1]); i += 2
1134
+ else:
1135
+ i += 1
1136
+ if not base_url or not default_model or not env_key:
1137
+ print("Error: --base-url, --default-model, and --env-key are required.", file=sys.stderr)
1138
+ sys.exit(1)
1139
+ from graphify.llm import provider_base_url_ok
1140
+ if not provider_base_url_ok(base_url, name):
1141
+ print(f"Error: refusing to add provider with unsafe base_url {base_url!r}.", file=sys.stderr)
1142
+ sys.exit(1)
1143
+ global_path.parent.mkdir(parents=True, exist_ok=True)
1144
+ existing = {}
1145
+ if global_path.is_file():
1146
+ try:
1147
+ existing = _json.loads(global_path.read_text(encoding="utf-8"))
1148
+ except Exception:
1149
+ pass
1150
+ existing[name] = {
1151
+ "base_url": base_url,
1152
+ "default_model": default_model,
1153
+ "env_key": env_key,
1154
+ "pricing": {"input": pricing_input, "output": pricing_output},
1155
+ "temperature": 0,
1156
+ }
1157
+ global_path.write_text(_json.dumps(existing, indent=2) + "\n", encoding="utf-8")
1158
+ print(f"Provider '{name}' added. Use with: graphify extract . --backend {name}")
1159
+
1160
+ elif subcmd == "remove":
1161
+ name = sys.argv[3] if len(sys.argv) > 3 else ""
1162
+ if not name:
1163
+ print("Usage: graphify provider remove <name>", file=sys.stderr)
1164
+ sys.exit(1)
1165
+ existing = {}
1166
+ if global_path.is_file():
1167
+ try:
1168
+ existing = _json.loads(global_path.read_text(encoding="utf-8"))
1169
+ except Exception:
1170
+ pass
1171
+ if name not in existing:
1172
+ print(f"Provider '{name}' not found.", file=sys.stderr)
1173
+ sys.exit(1)
1174
+ del existing[name]
1175
+ global_path.write_text(_json.dumps(existing, indent=2) + "\n", encoding="utf-8")
1176
+ print(f"Provider '{name}' removed.")
1177
+
1178
+ else:
1179
+ print("Usage: graphify provider [add|list|show|remove]", file=sys.stderr)
1180
+ if subcmd:
1181
+ sys.exit(1)
1182
+ elif cmd == "prs":
1183
+ from graphify.prs import cmd_prs
1184
+ cmd_prs(sys.argv[2:])
1185
+ elif cmd == "hook":
1186
+ from graphify.hooks import (
1187
+ install as hook_install,
1188
+ uninstall as hook_uninstall,
1189
+ status as hook_status,
1190
+ )
1191
+
1192
+ subcmd = sys.argv[2] if len(sys.argv) > 2 else ""
1193
+ if subcmd == "install":
1194
+ print(hook_install(Path(".")))
1195
+ elif subcmd == "uninstall":
1196
+ print(hook_uninstall(Path(".")))
1197
+ elif subcmd == "status":
1198
+ print(hook_status(Path(".")))
1199
+ else:
1200
+ print("Usage: graphify hook [install|uninstall|status]", file=sys.stderr)
1201
+ sys.exit(1)
1202
+ elif cmd == "query":
1203
+ if len(sys.argv) < 3:
1204
+ print("Usage: graphify query \"<question>\" [--dfs] [--context C] [--budget N] [--graph path]", file=sys.stderr)
1205
+ sys.exit(1)
1206
+ from graphify.serve import _query_graph_text
1207
+ from graphify.security import sanitize_label
1208
+ from networkx.readwrite import json_graph
1209
+ from graphify import querylog
1210
+
1211
+ question = sys.argv[2]
1212
+ use_dfs = "--dfs" in sys.argv
1213
+ budget = 2000
1214
+ graph_path = _default_graph_path()
1215
+ context_filters: list[str] = []
1216
+ args = sys.argv[3:]
1217
+ i = 0
1218
+ while i < len(args):
1219
+ if args[i] == "--budget" and i + 1 < len(args):
1220
+ try:
1221
+ budget = int(args[i + 1])
1222
+ except ValueError:
1223
+ print(f"error: --budget must be an integer", file=sys.stderr)
1224
+ sys.exit(1)
1225
+ i += 2
1226
+ elif args[i].startswith("--budget="):
1227
+ try:
1228
+ budget = int(args[i].split("=", 1)[1])
1229
+ except ValueError:
1230
+ print(f"error: --budget must be an integer", file=sys.stderr)
1231
+ sys.exit(1)
1232
+ i += 1
1233
+ elif args[i] == "--context" and i + 1 < len(args):
1234
+ context_filters.append(args[i + 1])
1235
+ i += 2
1236
+ elif args[i].startswith("--context="):
1237
+ context_filters.append(args[i].split("=", 1)[1])
1238
+ i += 1
1239
+ elif args[i] == "--graph" and i + 1 < len(args):
1240
+ graph_path = args[i + 1]
1241
+ i += 2
1242
+ else:
1243
+ i += 1
1244
+ gp = Path(graph_path).resolve()
1245
+ if not gp.exists():
1246
+ print(f"error: graph file not found: {gp}", file=sys.stderr)
1247
+ sys.exit(1)
1248
+ if not gp.suffix == ".json":
1249
+ print(f"error: graph file must be a .json file", file=sys.stderr)
1250
+ sys.exit(1)
1251
+ _enforce_graph_size_cap_or_exit(gp)
1252
+ try:
1253
+ import json as _json
1254
+ import networkx as _nx
1255
+
1256
+ _raw = _json.loads(gp.read_text(encoding="utf-8"))
1257
+ if "links" not in _raw and "edges" in _raw:
1258
+ _raw = dict(_raw, links=_raw["edges"])
1259
+ # `query` deliberately keeps the graph undirected (unlike `path` /
1260
+ # `explain`, which force directed=True): BFS/DFS here must explore
1261
+ # both callers and callees of the seed node to build useful
1262
+ # context, and forcing a DiGraph would make G.neighbors() return
1263
+ # successors only, silently dropping every caller-side result for
1264
+ # a seed with no outgoing edges. Direction is instead preserved
1265
+ # per-edge below (mirrors graphify/build.py's _src/_tgt pattern)
1266
+ # so the *rendering* stays correct without narrowing traversal.
1267
+ # Keep in-file markers when present (#2309): unconditionally
1268
+ # overwriting them with source/target would clobber the true
1269
+ # direction of a link persisted in flipped endpoint order.
1270
+ _raw = dict(
1271
+ _raw,
1272
+ links=[
1273
+ {
1274
+ **link,
1275
+ "_src": link.get("_src", link.get("source")),
1276
+ "_tgt": link.get("_tgt", link.get("target")),
1277
+ }
1278
+ for link in _raw.get("links", [])
1279
+ ],
1280
+ )
1281
+ try:
1282
+ G = json_graph.node_link_graph(_raw, edges="links")
1283
+ except TypeError:
1284
+ G = json_graph.node_link_graph(_raw)
1285
+ try:
1286
+ from graphify.build import graph_has_legacy_ids as _legacy
1287
+ if _legacy(_raw.get("nodes", [])):
1288
+ print(
1289
+ "[graphify] note: this graph uses the pre-#1504 node-ID scheme; "
1290
+ "rebuild with `graphify extract --force` to get path-qualified IDs "
1291
+ "(fixes same-name-file collisions).",
1292
+ file=sys.stderr,
1293
+ )
1294
+ except Exception:
1295
+ pass
1296
+ except Exception as exc:
1297
+ print(f"error: could not load graph: {exc}", file=sys.stderr)
1298
+ sys.exit(1)
1299
+ import time as _time
1300
+ _t0 = _time.perf_counter()
1301
+ _mode = "dfs" if use_dfs else "bfs"
1302
+ _result = _query_graph_text(
1303
+ G,
1304
+ question,
1305
+ mode=_mode,
1306
+ depth=2,
1307
+ token_budget=budget,
1308
+ context_filters=context_filters,
1309
+ graph_path=str(gp),
1310
+ )
1311
+ querylog.log_query(
1312
+ kind="query",
1313
+ question=question,
1314
+ corpus=str(gp),
1315
+ result=_result,
1316
+ mode=_mode,
1317
+ depth=2,
1318
+ token_budget=budget,
1319
+ duration_ms=(_time.perf_counter() - _t0) * 1000,
1320
+ )
1321
+ _touch_query_stamp(gp)
1322
+ print(_result)
1323
+ elif cmd == "affected":
1324
+ if len(sys.argv) < 3:
1325
+ print("Usage: graphify affected \"<node-or-label>\" [--relation R] [--depth N] [--graph path]", file=sys.stderr)
1326
+ sys.exit(1)
1327
+ from graphify.affected import DEFAULT_AFFECTED_RELATIONS, format_affected, load_graph
1328
+ query = sys.argv[2]
1329
+ graph_path = _default_graph_path()
1330
+ depth = 2
1331
+ relations: list[str] = []
1332
+ args = sys.argv[3:]
1333
+ i = 0
1334
+ while i < len(args):
1335
+ if args[i] == "--graph" and i + 1 < len(args):
1336
+ graph_path = args[i + 1]
1337
+ i += 2
1338
+ elif args[i].startswith("--graph="):
1339
+ graph_path = args[i].split("=", 1)[1]
1340
+ i += 1
1341
+ elif args[i] == "--depth" and i + 1 < len(args):
1342
+ try:
1343
+ depth = int(args[i + 1])
1344
+ except ValueError:
1345
+ print("error: --depth must be an integer", file=sys.stderr)
1346
+ sys.exit(1)
1347
+ i += 2
1348
+ elif args[i].startswith("--depth="):
1349
+ try:
1350
+ depth = int(args[i].split("=", 1)[1])
1351
+ except ValueError:
1352
+ print("error: --depth must be an integer", file=sys.stderr)
1353
+ sys.exit(1)
1354
+ i += 1
1355
+ elif args[i] == "--relation" and i + 1 < len(args):
1356
+ relations.append(args[i + 1])
1357
+ i += 2
1358
+ elif args[i].startswith("--relation="):
1359
+ relations.append(args[i].split("=", 1)[1])
1360
+ i += 1
1361
+ else:
1362
+ i += 1
1363
+ gp = Path(graph_path).resolve()
1364
+ if not gp.exists():
1365
+ print(f"error: graph file not found: {gp}", file=sys.stderr)
1366
+ sys.exit(1)
1367
+ if not gp.suffix == ".json":
1368
+ print("error: graph file must be a .json file", file=sys.stderr)
1369
+ sys.exit(1)
1370
+ try:
1371
+ graph = load_graph(gp)
1372
+ except Exception as exc:
1373
+ print(f"error: could not load graph: {exc}", file=sys.stderr)
1374
+ sys.exit(1)
1375
+ # Derive the analysed repo root from the graph's own location so an
1376
+ # absolute-path seed resolves without requiring cwd to be that root
1377
+ # (#2706). The graph is written to <root>/<GRAPHIFY_OUT_NAME>/graph.json,
1378
+ # so the root is the output dir's parent; a graph pointed at directly by
1379
+ # --graph falls back to its own directory.
1380
+ from graphify.paths import GRAPHIFY_OUT_NAME
1381
+ graph_root = gp.parent.parent if gp.parent.name == GRAPHIFY_OUT_NAME else gp.parent
1382
+ print(
1383
+ format_affected(
1384
+ graph,
1385
+ query,
1386
+ relations=relations or DEFAULT_AFFECTED_RELATIONS,
1387
+ depth=depth,
1388
+ root=graph_root,
1389
+ )
1390
+ )
1391
+ elif cmd in ("god-nodes", "god_nodes"):
1392
+ # god_nodes has long been an analyzer (analyze.py), an MCP tool, and a
1393
+ # README-advertised capability, but never a CLI subcommand — `graphify
1394
+ # god_nodes` fell through to "unknown command" (#2004). Wire it as a
1395
+ # read-only graph query, mirroring `affected`.
1396
+ from graphify.affected import load_graph
1397
+ from graphify.analyze import god_nodes as _god_nodes
1398
+ from graphify.security import sanitize_label as _sanitize_label
1399
+ graph_path = _default_graph_path()
1400
+ top_n = 10
1401
+ gn_exclude_hubs: float | None = None
1402
+ as_json = "--json" in sys.argv
1403
+ args = sys.argv[2:]
1404
+ i = 0
1405
+ while i < len(args):
1406
+ if args[i] == "--graph" and i + 1 < len(args):
1407
+ graph_path = args[i + 1]
1408
+ i += 2
1409
+ elif args[i].startswith("--graph="):
1410
+ graph_path = args[i].split("=", 1)[1]
1411
+ i += 1
1412
+ elif args[i] == "--top" and i + 1 < len(args):
1413
+ try:
1414
+ top_n = int(args[i + 1])
1415
+ except ValueError:
1416
+ print("error: --top must be an integer", file=sys.stderr)
1417
+ sys.exit(1)
1418
+ i += 2
1419
+ elif args[i].startswith("--top="):
1420
+ try:
1421
+ top_n = int(args[i].split("=", 1)[1])
1422
+ except ValueError:
1423
+ print("error: --top must be an integer", file=sys.stderr)
1424
+ sys.exit(1)
1425
+ i += 1
1426
+ elif args[i] == "--exclude-hubs" and i + 1 < len(args):
1427
+ try:
1428
+ gn_exclude_hubs = float(args[i + 1])
1429
+ except ValueError:
1430
+ print("error: --exclude-hubs must be a number (percentile 0-100)", file=sys.stderr)
1431
+ sys.exit(1)
1432
+ i += 2
1433
+ elif args[i].startswith("--exclude-hubs="):
1434
+ try:
1435
+ gn_exclude_hubs = float(args[i].split("=", 1)[1])
1436
+ except ValueError:
1437
+ print("error: --exclude-hubs must be a number (percentile 0-100)", file=sys.stderr)
1438
+ sys.exit(1)
1439
+ i += 1
1440
+ else:
1441
+ i += 1
1442
+ gp = Path(graph_path).resolve()
1443
+ if not gp.exists():
1444
+ print(f"error: graph file not found: {gp}", file=sys.stderr)
1445
+ sys.exit(1)
1446
+ if not gp.suffix == ".json":
1447
+ print("error: graph file must be a .json file", file=sys.stderr)
1448
+ sys.exit(1)
1449
+ try:
1450
+ G = load_graph(gp)
1451
+ except Exception as exc:
1452
+ print(f"error: could not load graph: {exc}", file=sys.stderr)
1453
+ sys.exit(1)
1454
+ gods = _god_nodes(G, top_n=top_n, exclude_hubs_percentile=gn_exclude_hubs)
1455
+ if as_json:
1456
+ print(json.dumps(gods, indent=2))
1457
+ else:
1458
+ print("God nodes (most connected):")
1459
+ for rank, n in enumerate(gods, 1):
1460
+ print(f" {rank}. {_sanitize_label(str(n['label']))} - {n['degree']} edges")
1461
+ elif cmd == "save-result":
1462
+ # graphify save-result --question Q --answer A [--type T] [--nodes N1 N2 ...]
1463
+ # [--outcome useful|dead_end|corrected] [--correction TEXT]
1464
+ import argparse as _ap
1465
+
1466
+ p = _ap.ArgumentParser(prog="graphify save-result")
1467
+ p.add_argument("--question", required=True)
1468
+ p.add_argument("--answer", default=None)
1469
+ p.add_argument("--answer-file", dest="answer_file", default=None)
1470
+ p.add_argument("--type", dest="query_type", default="query")
1471
+ p.add_argument("--nodes", nargs="*", default=[])
1472
+ p.add_argument("--outcome", choices=("useful", "dead_end", "corrected"), default=None)
1473
+ p.add_argument("--correction", default=None)
1474
+ p.add_argument("--memory-dir", default=str(Path(_GRAPHIFY_OUT) / "memory"))
1475
+ opts = p.parse_args(sys.argv[2:])
1476
+ if opts.answer_file:
1477
+ opts.answer = Path(opts.answer_file).read_text(encoding="utf-8").strip()
1478
+ elif not opts.answer:
1479
+ p.error("--answer or --answer-file is required")
1480
+ from graphify.ingest import save_query_result as _sqr
1481
+
1482
+ out = _sqr(
1483
+ question=opts.question,
1484
+ answer=opts.answer,
1485
+ memory_dir=Path(opts.memory_dir),
1486
+ query_type=opts.query_type,
1487
+ source_nodes=opts.nodes or None,
1488
+ outcome=opts.outcome,
1489
+ correction=opts.correction,
1490
+ )
1491
+ print(f"Saved to {out}")
1492
+ elif cmd == "reflect":
1493
+ import argparse as _ap
1494
+
1495
+ p = _ap.ArgumentParser(prog="graphify reflect")
1496
+ p.add_argument("--memory-dir", default=str(Path(_GRAPHIFY_OUT) / "memory"))
1497
+ p.add_argument(
1498
+ "--out",
1499
+ default=str(Path(_GRAPHIFY_OUT) / "reflections" / "LESSONS.md"),
1500
+ )
1501
+ p.add_argument("--graph", default=None)
1502
+ p.add_argument("--analysis", default=None)
1503
+ p.add_argument("--labels", default=None)
1504
+ p.add_argument("--half-life-days", type=float, default=30.0,
1505
+ help="signal weight halves every N days (default 30)")
1506
+ p.add_argument("--min-corroboration", type=int, default=2,
1507
+ help="distinct useful results to promote a node to preferred (default 2)")
1508
+ p.add_argument("--if-stale", action="store_true",
1509
+ help="skip when LESSONS.md is already newer than every input "
1510
+ "(e.g. the git hook just refreshed it)")
1511
+ opts = p.parse_args(sys.argv[2:])
1512
+ from graphify.reflect import reflect as _reflect, lessons_fresh as _lessons_fresh
1513
+
1514
+ graph_arg = opts.graph
1515
+ if graph_arg is None:
1516
+ default_graph = Path(_GRAPHIFY_OUT) / "graph.json"
1517
+ if default_graph.exists():
1518
+ graph_arg = str(default_graph)
1519
+
1520
+ _gp = Path(graph_arg) if graph_arg else None
1521
+ _analysis_path = None
1522
+ _labels_path = None
1523
+ if _gp is not None:
1524
+ _analysis_path = Path(opts.analysis) if opts.analysis else (
1525
+ _gp.parent / ".graphify_analysis.json")
1526
+ _labels_path = Path(opts.labels) if opts.labels else (
1527
+ _gp.parent / ".graphify_labels.json")
1528
+
1529
+ if opts.if_stale and _lessons_fresh(
1530
+ Path(opts.out), Path(opts.memory_dir), _gp, _analysis_path, _labels_path
1531
+ ):
1532
+ print(f"Lessons already up to date -> {opts.out} (skipped; omit --if-stale to force)")
1533
+ else:
1534
+ out_path, agg = _reflect(
1535
+ memory_dir=Path(opts.memory_dir),
1536
+ out_path=Path(opts.out),
1537
+ graph_path=_gp,
1538
+ analysis_path=_analysis_path,
1539
+ labels_path=_labels_path,
1540
+ half_life_days=opts.half_life_days,
1541
+ min_corroboration=opts.min_corroboration,
1542
+ )
1543
+ c = agg["counts"]
1544
+ print(
1545
+ f"Reflected {agg['total']} memories "
1546
+ f"({c['useful']} useful, {c['dead_end']} dead ends, "
1547
+ f"{c['corrected']} corrected) -> {out_path}"
1548
+ )
1549
+ elif cmd == "path":
1550
+ if len(sys.argv) < 4:
1551
+ print(
1552
+ 'Usage: graphify path "<source>" "<target>" [--graph path] '
1553
+ "[--directed|--undirected]",
1554
+ file=sys.stderr,
1555
+ )
1556
+ sys.exit(1)
1557
+ from graphify.serve import _pick_scored_endpoint, _score_nodes
1558
+ from networkx.readwrite import json_graph
1559
+ import networkx as _nx
1560
+
1561
+ source_label = sys.argv[2]
1562
+ target_label = sys.argv[3]
1563
+ graph_path = _default_graph_path()
1564
+ args = sys.argv[4:]
1565
+ direction_flag = None
1566
+ for i, a in enumerate(args):
1567
+ if a == "--graph" and i + 1 < len(args):
1568
+ graph_path = args[i + 1]
1569
+ elif a == "--directed":
1570
+ if direction_flag == "undirected":
1571
+ print(
1572
+ "error: --directed and --undirected are mutually exclusive",
1573
+ file=sys.stderr,
1574
+ )
1575
+ sys.exit(1)
1576
+ direction_flag = "directed"
1577
+ elif a == "--undirected":
1578
+ if direction_flag == "directed":
1579
+ print(
1580
+ "error: --directed and --undirected are mutually exclusive",
1581
+ file=sys.stderr,
1582
+ )
1583
+ sys.exit(1)
1584
+ direction_flag = "undirected"
1585
+ # Directed by default (#2487): direction truth exists in every
1586
+ # graph.json (arc order on post-#563 files, _src/_tgt markers on legacy
1587
+ # canonicalized files), so respect it unless the caller opts out.
1588
+ undirected = direction_flag == "undirected"
1589
+ gp = Path(graph_path).resolve()
1590
+ if not gp.exists():
1591
+ print(f"error: graph file not found: {gp}", file=sys.stderr)
1592
+ sys.exit(1)
1593
+ _enforce_graph_size_cap_or_exit(gp)
1594
+ _raw = json.loads(gp.read_text(encoding="utf-8"))
1595
+ if "links" not in _raw and "edges" in _raw:
1596
+ _raw = dict(_raw, links=_raw["edges"])
1597
+ # Force directed so the renderer can recover stored caller→callee
1598
+ # direction, and multigraph so exact-pair parallel links (e.g. a
1599
+ # `references` and a `calls` edge between the same two nodes) survive load
1600
+ # instead of being silently collapsed last-writer-wins — otherwise the
1601
+ # printed relation could be one the traversed pair doesn't actually
1602
+ # carry (#2074). Local to this read; serve's shared graph is untouched.
1603
+ _raw = {**_raw, "directed": True, "multigraph": True}
1604
+ try:
1605
+ G = json_graph.node_link_graph(_raw, edges="links")
1606
+ except TypeError:
1607
+ G = json_graph.node_link_graph(_raw)
1608
+ src_scored = _score_nodes(G, [t.lower() for t in source_label.split()])
1609
+ tgt_scored = _score_nodes(G, [t.lower() for t in target_label.split()])
1610
+ if not src_scored:
1611
+ print(f"No node matching '{source_label}' found.", file=sys.stderr)
1612
+ sys.exit(1)
1613
+ if not tgt_scored:
1614
+ print(f"No node matching '{target_label}' found.", file=sys.stderr)
1615
+ sys.exit(1)
1616
+ src_nid = _pick_scored_endpoint(G, src_scored, source_label)
1617
+ tgt_nid = _pick_scored_endpoint(G, tgt_scored, target_label)
1618
+ # Ambiguity guard: when both queries resolve to the same node, the
1619
+ # shortest path is trivially zero hops, which is almost never what the
1620
+ # caller wanted (see bug #828).
1621
+ if src_nid == tgt_nid:
1622
+ print(
1623
+ f"'{source_label}' and '{target_label}' both resolved to the same "
1624
+ f"node '{src_nid}'. Use a more specific label or the exact node ID.",
1625
+ file=sys.stderr,
1626
+ )
1627
+ sys.exit(1)
1628
+ for _name, _scored, _nid in (
1629
+ ("source", src_scored, src_nid),
1630
+ ("target", tgt_scored, tgt_nid),
1631
+ ):
1632
+ # A close runner-up only made the resolution ambiguous when the raw
1633
+ # score head is what got picked; a full-token override was chosen on
1634
+ # token coverage, not score, so the head's margin is irrelevant.
1635
+ if len(_scored) >= 2 and _nid == _scored[0][1]:
1636
+ _top, _runner = _scored[0][0], _scored[1][0]
1637
+ if _top > 0 and (_top - _runner) / _top < 0.10:
1638
+ print(
1639
+ f"warning: {_name} match was ambiguous "
1640
+ f"(top score {_top:g}, runner-up {_runner:g})",
1641
+ file=sys.stderr,
1642
+ )
1643
+ # Deterministic shortest path (#2074): hash-seeded neighbor views
1644
+ # returned an arbitrary route among equal-length paths that varied per
1645
+ # process. Build a sorted, materialized graph so neighbor order — and
1646
+ # thus the chosen path — is canonical for a given graph.json.
1647
+ try:
1648
+ if undirected:
1649
+ _und = _nx.Graph()
1650
+ _und.add_nodes_from(sorted(G.nodes))
1651
+ _und.add_edges_from(sorted((min(u, v), max(u, v)) for u, v in G.edges()))
1652
+ path_nodes = _nx.shortest_path(_und, src_nid, tgt_nid)
1653
+ else:
1654
+ # Directed by default (#2487). True direction is NOT raw arc
1655
+ # order: legacy canonicalized files persist a flipped arc with
1656
+ # _src/_tgt markers (#2309), so build the digraph from _src/_tgt
1657
+ # (falling back to the loaded arc) rather than to_directed().
1658
+ _dg = _nx.DiGraph()
1659
+ _dg.add_nodes_from(sorted(G.nodes))
1660
+ _dg.add_edges_from(sorted(
1661
+ (d.get("_src", u), d.get("_tgt", v)) for u, v, d in G.edges(data=True)
1662
+ ))
1663
+ path_nodes = _nx.shortest_path(_dg, src_nid, tgt_nid)
1664
+ except (_nx.NetworkXNoPath, _nx.NodeNotFound):
1665
+ if undirected:
1666
+ print(f"No path found between '{source_label}' and '{target_label}'.")
1667
+ else:
1668
+ print(
1669
+ f"No directed path found between '{source_label}' and "
1670
+ f"'{target_label}'. Re-run with --undirected to search "
1671
+ "ignoring edge direction."
1672
+ )
1673
+ sys.exit(0)
1674
+ hops = len(path_nodes) - 1
1675
+ segments = []
1676
+ from graphify.build import edge_datas
1677
+ for i in range(len(path_nodes) - 1):
1678
+ u, v = path_nodes[i], path_nodes[i + 1]
1679
+ # Report the ACTUAL stored relation(s) of the traversed pair and
1680
+ # direction — never a fabricated `calls` (#2074). A pair may carry
1681
+ # several parallel relations; show all, and fall back to an honest
1682
+ # "related" when the stored edge has no relation.
1683
+ # Direction truth lives in the per-link _src/_tgt markers (#2309):
1684
+ # undirected NetworkX storage canonicalizes endpoint order, so the
1685
+ # persisted source/target arc can be flipped relative to the real
1686
+ # caller→callee direction. Recover it from _src when present, else
1687
+ # fall back to the loaded arc tail (markerless canonical files keep
1688
+ # today's behavior).
1689
+ fwd, bwd = [], []
1690
+ for a, b in ((u, v), (v, u)):
1691
+ if G.has_edge(a, b):
1692
+ for d in edge_datas(G, a, b):
1693
+ (fwd if d.get("_src", a) == u else bwd).append(d)
1694
+ datas = fwd or bwd
1695
+ forward = bool(fwd)
1696
+ rels = sorted({d.get("relation") for d in datas if d.get("relation")})
1697
+ rel = "/".join(rels) if rels else "related"
1698
+ confs = sorted({d.get("confidence") for d in datas if d.get("confidence")})
1699
+ conf_str = f" [{'/'.join(confs)}]" if confs else ""
1700
+ if i == 0:
1701
+ segments.append(G.nodes[u].get("label", u))
1702
+ if forward:
1703
+ segments.append(f"--{rel}{conf_str}--> {G.nodes[v].get('label', v)}")
1704
+ else:
1705
+ segments.append(f"<--{rel}{conf_str}-- {G.nodes[v].get('label', v)}")
1706
+ print(f"Shortest path ({hops} hops):\n " + " ".join(segments))
1707
+ from graphify import querylog
1708
+ querylog.log_query(
1709
+ kind="path",
1710
+ question=f"{sys.argv[2]} -> {sys.argv[3]}",
1711
+ corpus=str(gp),
1712
+ nodes_returned=hops,
1713
+ )
1714
+ _touch_query_stamp(gp)
1715
+
1716
+ elif cmd == "explain":
1717
+ if len(sys.argv) < 3:
1718
+ print('Usage: graphify explain "<node>" [--graph path]', file=sys.stderr)
1719
+ sys.exit(1)
1720
+ from graphify.serve import _find_node, find_node_ambiguity
1721
+ from networkx.readwrite import json_graph
1722
+
1723
+ label = sys.argv[2]
1724
+ graph_path = _default_graph_path()
1725
+ args = sys.argv[3:]
1726
+ for i, a in enumerate(args):
1727
+ if a == "--graph" and i + 1 < len(args):
1728
+ graph_path = args[i + 1]
1729
+ gp = Path(graph_path).resolve()
1730
+ if not gp.exists():
1731
+ print(f"error: graph file not found: {gp}", file=sys.stderr)
1732
+ sys.exit(1)
1733
+ _enforce_graph_size_cap_or_exit(gp)
1734
+ _raw = json.loads(gp.read_text(encoding="utf-8"))
1735
+ if "links" not in _raw and "edges" in _raw:
1736
+ _raw = dict(_raw, links=_raw["edges"])
1737
+ # Force directed so the renderer can recover stored caller→callee direction.
1738
+ _raw = {**_raw, "directed": True}
1739
+ try:
1740
+ G = json_graph.node_link_graph(_raw, edges="links")
1741
+ except TypeError:
1742
+ G = json_graph.node_link_graph(_raw)
1743
+ matches = _find_node(G, label)
1744
+ if not matches:
1745
+ print(f"No node matching '{label}' found.")
1746
+ sys.exit(0)
1747
+ rivals = find_node_ambiguity(G, label)
1748
+ if rivals:
1749
+ print(f"Ambiguous: '{label}' matches {len(rivals)} nodes in different files.")
1750
+ for rival in rivals:
1751
+ print(f" {G.nodes[rival].get('source_file') or rival}")
1752
+ print(f" id: {rival}")
1753
+ print(
1754
+ f"Retry with path::symbol using one of the paths above (e.g. "
1755
+ f"<path>::{label}) or the full node id."
1756
+ )
1757
+ sys.exit(1)
1758
+ nid = matches[0]
1759
+ d = G.nodes[nid]
1760
+ print(f"Node: {d.get('label', nid)}")
1761
+ print(f" ID: {nid}")
1762
+ print(
1763
+ f" Source: {d.get('source_file', '')} {d.get('source_location', '')}".rstrip()
1764
+ )
1765
+ print(f" Type: {d.get('file_type', '')}")
1766
+ print(f" Community: {d.get('community_name') or d.get('community', '')}")
1767
+ # Work-memory overlay: a derived experiential hint from `graphify reflect`,
1768
+ # merged in display-only from the .graphify_learning.json sidecar next to
1769
+ # graph.json. No line when the node has no overlay entry.
1770
+ try:
1771
+ from graphify.reflect import load_learning_overlay as _llo
1772
+ from graphify.security import sanitize_label as _sl
1773
+ _overlay = _llo(gp)
1774
+ _entry = _overlay.get(str(nid))
1775
+ if _entry:
1776
+ _status = _sl(str(_entry.get("status", "")))
1777
+ if _status == "contested":
1778
+ _line = (f" Lesson: contested (useful {_entry.get('uses', 0)} / "
1779
+ f"dead-end {_entry.get('neg', 0)})")
1780
+ elif _status == "preferred":
1781
+ _line = (f" Lesson: preferred source (start here) — "
1782
+ f"{_entry.get('uses', 0)} useful, score={_entry.get('score', 0)}")
1783
+ else:
1784
+ _line = (f" Lesson: {_status or 'tentative'} — "
1785
+ f"{_entry.get('uses', 0)} useful, score={_entry.get('score', 0)}")
1786
+ if _entry.get("stale"):
1787
+ _line += " [code changed since — re-verify]"
1788
+ print(_line)
1789
+ except Exception:
1790
+ pass
1791
+ print(f" Degree: {G.degree(nid)}")
1792
+ from graphify.build import edge_data
1793
+ connections: list[tuple[str, str, dict]] = [] # (direction, neighbor_id, edge_data)
1794
+ # Classify by the edge's TRUE direction, not the loaded arc order:
1795
+ # a link persisted in flipped endpoint order carries its truth in the
1796
+ # per-edge _src marker (#2309). Markerless edges fall back to the arc
1797
+ # tail (today's behavior).
1798
+ for nb in G.successors(nid):
1799
+ _ed = edge_data(G, nid, nb)
1800
+ connections.append(
1801
+ ("out" if _ed.get("_src", nid) == nid else "in", nb, _ed)
1802
+ )
1803
+ for nb in G.predecessors(nid):
1804
+ _ed = edge_data(G, nb, nid)
1805
+ connections.append(
1806
+ ("in" if _ed.get("_src", nb) == nb else "out", nb, _ed)
1807
+ )
1808
+ if connections:
1809
+ print(f"\nConnections ({len(connections)}):")
1810
+ connections.sort(key=lambda c: G.degree(c[1]), reverse=True)
1811
+ for direction, nb, edata in connections[:20]:
1812
+ rel = edata.get("relation", "")
1813
+ conf = edata.get("confidence", "")
1814
+ arrow = "-->" if direction == "out" else "<--"
1815
+ # Append the edge's location — the actual call/import/reference
1816
+ # SITE (in the caller's file for an incoming call), not a def
1817
+ # line (#BUG1). Labeled by [rel] so the meaning is unambiguous.
1818
+ loc = edata.get("source_location") or ""
1819
+ sfile = edata.get("source_file") or ""
1820
+ at = f" {sfile}:{loc}" if loc else ""
1821
+ print(f" {arrow} {G.nodes[nb].get('label', nb)} [{rel}] [{conf}]{at}")
1822
+ if len(connections) > 20:
1823
+ remainder = connections[20:]
1824
+ print(f" ... and {len(remainder)} more")
1825
+ # #2009: a bare count silently hides the answer on high-degree
1826
+ # nodes ("who calls this, what's the impact?"). Group the cut
1827
+ # connections by direction + file so their shape is visible
1828
+ # without falling back to a repo-wide grep.
1829
+ by_file: dict[tuple[str, str], int] = {}
1830
+ for direction, _nb, edata in remainder:
1831
+ sfile = edata.get("source_file") or "(unknown file)"
1832
+ key = (direction, sfile)
1833
+ by_file[key] = by_file.get(key, 0) + 1
1834
+ # Count desc, then (direction, file) so equal-count groups have a
1835
+ # byte-stable order (not the degree-derived insertion order).
1836
+ grouped = sorted(by_file.items(), key=lambda kv: (-kv[1], kv[0]))
1837
+ print(" Grouped by file:")
1838
+ for (direction, sfile), count in grouped[:20]:
1839
+ arrow = "-->" if direction == "out" else "<--"
1840
+ noun = "connection" if count == 1 else "connections"
1841
+ print(f" {arrow} {sfile}: {count} {noun}")
1842
+ if len(grouped) > 20:
1843
+ print(f" ... and {len(grouped) - 20} more files")
1844
+ from graphify import querylog
1845
+ querylog.log_query(
1846
+ kind="explain",
1847
+ question=sys.argv[2],
1848
+ corpus=str(gp),
1849
+ nodes_returned=len(connections),
1850
+ )
1851
+ _touch_query_stamp(gp)
1852
+
1853
+ elif cmd == "diagnose":
1854
+ subcmd = sys.argv[2] if len(sys.argv) > 2 else ""
1855
+ if subcmd != "multigraph":
1856
+ print(
1857
+ "Usage: graphify diagnose multigraph "
1858
+ "[--graph path] [--json] [--max-examples N] "
1859
+ "[--directed] [--undirected] [--extract-path path]",
1860
+ file=sys.stderr,
1861
+ )
1862
+ sys.exit(1)
1863
+
1864
+ graph_path = Path(_default_graph_path())
1865
+ max_examples = 5
1866
+ directed: bool | None = None
1867
+ direction_flag: str | None = None
1868
+ json_output = False
1869
+ extract_path: Path | None = None
1870
+
1871
+ i = 3
1872
+ while i < len(sys.argv):
1873
+ arg = sys.argv[i]
1874
+ if arg == "--graph":
1875
+ i += 1
1876
+ if i >= len(sys.argv):
1877
+ print("error: --graph requires a path", file=sys.stderr)
1878
+ sys.exit(1)
1879
+ graph_path = Path(sys.argv[i])
1880
+ elif arg == "--json":
1881
+ json_output = True
1882
+ elif arg == "--max-examples":
1883
+ i += 1
1884
+ if i >= len(sys.argv):
1885
+ print("error: --max-examples requires an integer", file=sys.stderr)
1886
+ sys.exit(1)
1887
+ try:
1888
+ max_examples = int(sys.argv[i])
1889
+ except ValueError:
1890
+ print("error: --max-examples requires an integer", file=sys.stderr)
1891
+ sys.exit(1)
1892
+ if max_examples < 0:
1893
+ print("error: --max-examples must be >= 0", file=sys.stderr)
1894
+ sys.exit(1)
1895
+ elif arg == "--directed":
1896
+ if direction_flag == "undirected":
1897
+ print(
1898
+ "error: --directed and --undirected are mutually exclusive",
1899
+ file=sys.stderr,
1900
+ )
1901
+ sys.exit(1)
1902
+ direction_flag = "directed"
1903
+ directed = True
1904
+ elif arg == "--undirected":
1905
+ if direction_flag == "directed":
1906
+ print(
1907
+ "error: --directed and --undirected are mutually exclusive",
1908
+ file=sys.stderr,
1909
+ )
1910
+ sys.exit(1)
1911
+ direction_flag = "undirected"
1912
+ directed = False
1913
+ elif arg == "--extract-path":
1914
+ i += 1
1915
+ if i >= len(sys.argv):
1916
+ print("error: --extract-path requires a path", file=sys.stderr)
1917
+ sys.exit(1)
1918
+ extract_path = Path(sys.argv[i])
1919
+ else:
1920
+ print(f"error: unknown diagnose option {arg}", file=sys.stderr)
1921
+ sys.exit(1)
1922
+ i += 1
1923
+
1924
+ from graphify.diagnostics import (
1925
+ diagnose_file,
1926
+ format_diagnostic_json,
1927
+ format_diagnostic_report,
1928
+ )
1929
+
1930
+ try:
1931
+ summary = diagnose_file(
1932
+ graph_path,
1933
+ directed=directed,
1934
+ root=Path(".").resolve(),
1935
+ max_examples=max_examples,
1936
+ extract_path=extract_path,
1937
+ )
1938
+ except Exception as exc:
1939
+ print(f"error: {exc}", file=sys.stderr)
1940
+ sys.exit(1)
1941
+
1942
+ if json_output:
1943
+ print(json.dumps(format_diagnostic_json(summary), indent=2))
1944
+ else:
1945
+ print(format_diagnostic_report(summary))
1946
+
1947
+ elif cmd == "add":
1948
+ if len(sys.argv) < 3:
1949
+ print(
1950
+ "Usage: graphify add <url> [--author Name] [--contributor Name] [--dir ./raw]",
1951
+ file=sys.stderr,
1952
+ )
1953
+ sys.exit(1)
1954
+ from graphify.ingest import ingest as _ingest
1955
+
1956
+ url = sys.argv[2]
1957
+ author: str | None = None
1958
+ contributor: str | None = None
1959
+ target_dir = Path("raw")
1960
+ args = sys.argv[3:]
1961
+ i = 0
1962
+ while i < len(args):
1963
+ if args[i] == "--author" and i + 1 < len(args):
1964
+ author = args[i + 1]
1965
+ i += 2
1966
+ elif args[i] == "--contributor" and i + 1 < len(args):
1967
+ contributor = args[i + 1]
1968
+ i += 2
1969
+ elif args[i] == "--dir" and i + 1 < len(args):
1970
+ target_dir = Path(args[i + 1])
1971
+ i += 2
1972
+ else:
1973
+ i += 1
1974
+ try:
1975
+ saved = _ingest(url, target_dir, author=author, contributor=contributor)
1976
+ print(f"Saved to {saved}")
1977
+ print("Run /graphify --update in your AI assistant to update the graph.")
1978
+ except Exception as exc:
1979
+ print(f"error: {exc}", file=sys.stderr)
1980
+ sys.exit(1)
1981
+
1982
+ elif cmd == "watch":
1983
+ watch_path = Path(sys.argv[2]) if len(sys.argv) > 2 else Path(".")
1984
+ if not watch_path.exists():
1985
+ print(f"error: path not found: {watch_path}", file=sys.stderr)
1986
+ sys.exit(1)
1987
+ from graphify.watch import watch as _watch
1988
+
1989
+ try:
1990
+ _watch(watch_path)
1991
+ except ImportError as exc:
1992
+ print(f"error: {exc}", file=sys.stderr)
1993
+ sys.exit(1)
1994
+
1995
+ elif cmd in ("cluster-only", "label"):
1996
+ # `label` is `cluster-only` that always (re)generates community names with
1997
+ # the configured backend, even when a .graphify_labels.json already exists.
1998
+ force_relabel = cmd == "label"
1999
+ # Mirror the tree/export arg-parsing pattern: walk argv so flags and
2000
+ # the optional positional path can appear in any order (#724).
2001
+ no_viz = "--no-viz" in sys.argv
2002
+ no_label = "--no-label" in sys.argv
2003
+ missing_only = "--missing-only" in sys.argv
2004
+ co_timing = "--timing" in sys.argv
2005
+ _backend_arg = next((a for a in sys.argv if a.startswith("--backend=")), None)
2006
+ label_backend = _backend_arg.split("=", 1)[1] if _backend_arg else None
2007
+ _model_arg = next((a for a in sys.argv if a.startswith("--model=")), None)
2008
+ label_model = _model_arg.split("=", 1)[1] if _model_arg else None
2009
+ _min_cs_arg = next((a for a in sys.argv if a.startswith("--min-community-size=")), None)
2010
+ min_community_size = int(_min_cs_arg.split("=")[1]) if _min_cs_arg else 3
2011
+ args = sys.argv[2:]
2012
+ watch_path: Path | None = None
2013
+ graph_override: Path | None = None
2014
+ co_resolution: float = 1.0
2015
+ co_exclude_hubs: float | None = None
2016
+ label_max_concurrency: int = 4
2017
+ label_batch_size: int = 100
2018
+ # #2534: defaults make presence undetectable for these, so track whether
2019
+ # the user actually passed them — the label-reuse branch below warns when
2020
+ # labeling flags are silently ignored. (--backend/--model default to None,
2021
+ # so `is not None` already means "explicitly passed" for those.)
2022
+ label_max_concurrency_explicit = False
2023
+ label_batch_size_explicit = False
2024
+ i_arg = 0
2025
+ while i_arg < len(args):
2026
+ a = args[i_arg]
2027
+ if a == "--graph" and i_arg + 1 < len(args):
2028
+ graph_override = Path(args[i_arg + 1]); i_arg += 2
2029
+ elif a == "--backend" and i_arg + 1 < len(args):
2030
+ label_backend = args[i_arg + 1]; i_arg += 2
2031
+ elif a.startswith("--backend="):
2032
+ label_backend = a.split("=", 1)[1]; i_arg += 1
2033
+ elif a == "--model" and i_arg + 1 < len(args):
2034
+ label_model = args[i_arg + 1]; i_arg += 2
2035
+ elif a.startswith("--model="):
2036
+ label_model = a.split("=", 1)[1]; i_arg += 1
2037
+ elif a == "--resolution" and i_arg + 1 < len(args):
2038
+ co_resolution = float(args[i_arg + 1]); i_arg += 2
2039
+ elif a.startswith("--resolution="):
2040
+ co_resolution = float(a.split("=", 1)[1]); i_arg += 1
2041
+ elif a == "--exclude-hubs" and i_arg + 1 < len(args):
2042
+ co_exclude_hubs = float(args[i_arg + 1]); i_arg += 2
2043
+ elif a.startswith("--exclude-hubs="):
2044
+ co_exclude_hubs = float(a.split("=", 1)[1]); i_arg += 1
2045
+ elif a == "--max-concurrency" and i_arg + 1 < len(args):
2046
+ label_max_concurrency = int(args[i_arg + 1]); label_max_concurrency_explicit = True; i_arg += 2
2047
+ elif a.startswith("--max-concurrency="):
2048
+ label_max_concurrency = int(a.split("=", 1)[1]); label_max_concurrency_explicit = True; i_arg += 1
2049
+ elif a == "--batch-size" and i_arg + 1 < len(args):
2050
+ label_batch_size = int(args[i_arg + 1]); label_batch_size_explicit = True; i_arg += 2
2051
+ elif a.startswith("--batch-size="):
2052
+ label_batch_size = int(a.split("=", 1)[1]); label_batch_size_explicit = True; i_arg += 1
2053
+ elif a in ("--no-viz", "--missing-only") or a.startswith("--min-community-size="):
2054
+ i_arg += 1
2055
+ elif a.startswith("--"):
2056
+ i_arg += 1
2057
+ elif watch_path is None:
2058
+ watch_path = Path(a); i_arg += 1
2059
+ else:
2060
+ i_arg += 1
2061
+ if watch_path is None:
2062
+ watch_path = Path(".")
2063
+ graph_json = graph_override if graph_override is not None else watch_path / _GRAPHIFY_OUT / "graph.json"
2064
+ if not graph_json.exists():
2065
+ print(
2066
+ f"error: no graph found at {graph_json} — run /graphify first",
2067
+ file=sys.stderr,
2068
+ )
2069
+ sys.exit(1)
2070
+ from networkx.readwrite import json_graph as _jg
2071
+ from graphify.build import build_from_json
2072
+ from graphify.cluster import cluster, score_all, remap_communities_to_previous
2073
+ from graphify.analyze import (
2074
+ god_nodes,
2075
+ surprising_connections,
2076
+ suggest_questions,
2077
+ )
2078
+ from graphify.report import generate
2079
+ from graphify.export import to_json, to_html
2080
+
2081
+ stages = _StageTimer(co_timing)
2082
+ print("Loading existing graph...")
2083
+ # Solution 3 (#1019): don't hard-exit on an oversized graph.json here.
2084
+ # Core outputs (graph.json + GRAPH_REPORT.md) still get written. The
2085
+ # visualization policy below uses its own node-count limit.
2086
+ from graphify.security import check_graph_file_size_cap as _check_cap
2087
+ try:
2088
+ _check_cap(graph_json)
2089
+ except ValueError:
2090
+ try:
2091
+ _over_cap_bytes = graph_json.stat().st_size
2092
+ except OSError:
2093
+ _over_cap_bytes = -1
2094
+ print(
2095
+ f"warning: graph.json exceeds cap ({_over_cap_bytes} bytes); "
2096
+ "continuing with best-effort visualization",
2097
+ file=sys.stderr,
2098
+ )
2099
+ _raw = json.loads(graph_json.read_text(encoding="utf-8"))
2100
+ _directed = bool(_raw.get("directed", False))
2101
+ G = build_from_json(_raw, directed=_directed)
2102
+ print(f"Graph: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges")
2103
+ stages.mark("load")
2104
+ print("Re-clustering...")
2105
+ communities = cluster(G, resolution=co_resolution, exclude_hubs_percentile=co_exclude_hubs)
2106
+ # Mirror the watch/update path (#822): map new cids to prior ones by
2107
+ # node-overlap so the existing .graphify_labels.json keeps attaching
2108
+ # to the same conceptual community after re-clustering. Without this,
2109
+ # labels follow raw cid index and become misaligned whenever the
2110
+ # graph has changed between labeling and cluster-only (#1027).
2111
+ previous_node_community = {
2112
+ n["id"]: n["community"]
2113
+ for n in _raw.get("nodes", [])
2114
+ if n.get("community") is not None and n.get("id") is not None
2115
+ }
2116
+ if previous_node_community:
2117
+ communities = remap_communities_to_previous(communities, previous_node_community)
2118
+ stages.mark("cluster")
2119
+ cohesion = score_all(G, communities)
2120
+ gods = god_nodes(G, exclude_hubs_percentile=co_exclude_hubs)
2121
+ surprises = surprising_connections(G, communities)
2122
+ stages.mark("analyze")
2123
+ # Where outputs (GRAPH_REPORT.md, re-clustered graph.json, labels,
2124
+ # analysis, html) land. When `--graph` points at a graph INSIDE a
2125
+ # graphify-out/ dir (another project/tenant's output), write beside it,
2126
+ # not into a stray graphify-out/ in the CWD (#1747). But when `--graph`
2127
+ # points at an arbitrary path — e.g. a `backup/graph.json` archived
2128
+ # before re-clustering (#934) — fall back to the CWD's graphify-out/,
2129
+ # which is the restore-into-place workflow that test pins. The default
2130
+ # (no --graph) case already has graph_json under watch_path/graphify-out.
2131
+ _out_name = Path(_GRAPHIFY_OUT).name
2132
+ if graph_override is not None and graph_json.parent.name == _out_name:
2133
+ out = graph_json.parent
2134
+ else:
2135
+ out = watch_path / _GRAPHIFY_OUT
2136
+ out.mkdir(parents=True, exist_ok=True)
2137
+ labels_path = out / ".graphify_labels.json"
2138
+ existing_labels: dict[int, str] = {}
2139
+ if labels_path.exists():
2140
+ try:
2141
+ existing_labels = {
2142
+ int(k): v
2143
+ for k, v in json.loads(labels_path.read_text(encoding="utf-8")).items()
2144
+ if isinstance(v, str)
2145
+ }
2146
+ except Exception:
2147
+ existing_labels = {}
2148
+ # Accumulate token usage from the labeling LLM calls so cluster-only mode
2149
+ # reports real cost instead of a hardcoded zero (#1694). Stays {0, 0} on
2150
+ # the reuse / no-label paths, which make no LLM calls.
2151
+ label_token_usage = {"input": 0, "output": 0}
2152
+ # #2073: a --no-label run produces only "Community N" placeholders.
2153
+ # Persisting them (plus a matching .sig) made the reuse branch treat them
2154
+ # as fresh forever, permanently blocking real labeling on later runs.
2155
+ placeholder_only = False
2156
+ if labels_path.exists() and not force_relabel:
2157
+ # #2534: this branch never calls the LLM, so labeling flags would be
2158
+ # silently ignored. Reuse is still the correct (exit 0) outcome — but
2159
+ # say so, instead of letting `--backend openai` look like it relabeled.
2160
+ _ignored_label_flags = [flag for flag, given in (
2161
+ ("--backend", label_backend is not None),
2162
+ ("--model", label_model is not None),
2163
+ ("--batch-size", label_batch_size_explicit),
2164
+ ("--max-concurrency", label_max_concurrency_explicit),
2165
+ ) if given]
2166
+ if _ignored_label_flags:
2167
+ print(
2168
+ f"[graphify] warning: {'/'.join(_ignored_label_flags)} ignored: "
2169
+ f"reusing saved labels at {labels_path}. "
2170
+ f"Run `graphify label` to relabel, or delete the labels file.",
2171
+ file=sys.stderr,
2172
+ )
2173
+ # Reuse saved labels, but don't blindly trust them: the graph may have
2174
+ # been re-scoped/re-clustered since labeling, in which case a cid now
2175
+ # covers a DIFFERENT community and its old (LLM) name is wrong (#label-stale).
2176
+ # Validate each community against the membership signature saved beside the
2177
+ # labels; any community that changed (or has no saved label) is renamed by
2178
+ # its current hub — deterministic and correct-by-construction — and the user
2179
+ # is told to `graphify label` for fresh LLM names. Unchanged communities keep
2180
+ # their saved label. When no signature sidecar exists (labels predate this),
2181
+ # fall back to hub-filling only the communities missing a label.
2182
+ from graphify.cluster import community_member_sigs, label_communities_by_hub
2183
+ sig_path = labels_path.parent / (labels_path.name + ".sig")
2184
+ saved_sigs: dict[int, str] = {}
2185
+ if sig_path.exists():
2186
+ try:
2187
+ saved_sigs = {
2188
+ int(k): v for k, v in
2189
+ json.loads(sig_path.read_text(encoding="utf-8")).items()
2190
+ if isinstance(v, str)
2191
+ }
2192
+ except Exception:
2193
+ saved_sigs = {}
2194
+ cur_sigs = community_member_sigs(communities)
2195
+ count_mismatch = len(existing_labels) != len(communities)
2196
+ labels = {}
2197
+ hub_labels: dict[int, str] | None = None
2198
+ changed = 0
2199
+ for cid in communities:
2200
+ # A persisted "Community {cid}" is a placeholder, not an earned
2201
+ # label — treat it as absent so the hub labeler replaces it and an
2202
+ # already-polluted sidecar (e.g. from a prior --no-label run) heals
2203
+ # instead of suppressing real labels forever (#2073).
2204
+ have_label = (
2205
+ cid in existing_labels
2206
+ and existing_labels[cid] != f"Community {cid}"
2207
+ )
2208
+ if saved_sigs:
2209
+ # Precise: the membership signature tells us if this exact
2210
+ # community changed since it was labeled.
2211
+ fresh = have_label and saved_sigs.get(cid) == cur_sigs.get(cid)
2212
+ else:
2213
+ # No signature sidecar (labels predate it). A differing community
2214
+ # COUNT means the labels describe a different clustering, so a cid's
2215
+ # old label can't be trusted; equal count is the best "same" signal.
2216
+ fresh = have_label and not count_mismatch
2217
+ if fresh:
2218
+ labels[cid] = existing_labels[cid]
2219
+ else:
2220
+ if hub_labels is None:
2221
+ hub_labels = label_communities_by_hub(G, communities)
2222
+ labels[cid] = hub_labels[cid]
2223
+ if have_label:
2224
+ changed += 1
2225
+ if changed:
2226
+ print(
2227
+ f"[graphify] community set changed since labeling "
2228
+ f"({len(existing_labels)} saved labels, {len(communities)} communities now; "
2229
+ f"renamed {changed} community(ies) by their hub). "
2230
+ f"Run `graphify label` to refresh names with the LLM.",
2231
+ file=sys.stderr,
2232
+ )
2233
+ elif no_label and not force_relabel:
2234
+ labels = {cid: f"Community {cid}" for cid in communities}
2235
+ placeholder_only = True
2236
+ else:
2237
+ # No labels file yet (or `graphify label` forced a refresh). When run
2238
+ # standalone there is no orchestrating agent to do skill.md Step 5, so
2239
+ # auto-name communities rather than leave "Community N" (#1097).
2240
+ from graphify.cluster import label_communities_by_hub
2241
+ from graphify.llm import generate_community_labels
2242
+ print("Labeling communities...")
2243
+ # Deterministic, LLM-free base labels: name each community after its
2244
+ # highest-degree hub, so the report is readable even with no backend
2245
+ # (previously bare "Community N"). A configured LLM backend overrides these
2246
+ # with richer names below; its no-backend placeholder fallback does NOT.
2247
+ hub_labels = label_communities_by_hub(G, communities)
2248
+ label_communities_input = communities
2249
+ labels = dict(hub_labels)
2250
+ if missing_only:
2251
+ labels = {
2252
+ cid: existing_labels.get(cid, hub_labels[cid])
2253
+ for cid in communities
2254
+ }
2255
+ label_communities_input = {
2256
+ cid: members
2257
+ for cid, members in communities.items()
2258
+ if cid not in existing_labels or existing_labels.get(cid) == f"Community {cid}"
2259
+ }
2260
+ generated_labels, _ = generate_community_labels(
2261
+ G, label_communities_input, backend=label_backend, model=label_model, gods=gods,
2262
+ max_concurrency=label_max_concurrency, batch_size=label_batch_size,
2263
+ usage_out=label_token_usage,
2264
+ )
2265
+ # Only let the LLM OVERRIDE where it produced a real name — its no-backend
2266
+ # fallback returns "Community {cid}" placeholders, which must not clobber
2267
+ # the deterministic hub labels. Also reject a model echoing the prompt
2268
+ # key back ("5" for community 5) — that is an id, not a name (#2534).
2269
+ labels.update({
2270
+ cid: v for cid, v in generated_labels.items()
2271
+ if v and v != f"Community {cid}" and v != str(cid)
2272
+ })
2273
+ stages.mark("label")
2274
+ questions = suggest_questions(G, communities, labels)
2275
+ # cluster-only re-clusters an EXISTING graph: the code content is exactly
2276
+ # what extract saw, so keep the extract-time commit stamp instead of
2277
+ # re-deriving it from the shell's cwd — running from another repo used to
2278
+ # re-stamp graph.json with THAT repo's HEAD (#2534). When the loaded
2279
+ # graph predates the stamp, fall back to the analysed repo's HEAD via
2280
+ # the cwd-aware helper (#2316).
2281
+ _commit = _raw.get("built_at_commit")
2282
+ if not _commit:
2283
+ from graphify.watch import _git_head as _gh
2284
+ _commit = _gh(cwd=watch_path)
2285
+ # Snapshot BEFORE any artifact is replaced: GRAPH_REPORT.md was written
2286
+ # first, so the dated folder held the NEW report, not the previous (#2402).
2287
+ from graphify.export import backup_if_protected as _backup
2288
+ from graphify.exporters.html import _HTML_STALE_MARKER
2289
+ _backup(out)
2290
+ html_stale_marker = out / _HTML_STALE_MARKER
2291
+
2292
+ def _clear_html_stale_marker() -> None:
2293
+ try:
2294
+ html_stale_marker.unlink(missing_ok=True)
2295
+ except OSError as exc:
2296
+ print(
2297
+ "warning: graph.html stale marker could not be cleared; "
2298
+ f"regeneration may be retried: {exc}",
2299
+ file=sys.stderr,
2300
+ )
2301
+
2302
+ stale_marker_preexisted = html_stale_marker.exists()
2303
+ # Mark before graph.json advances. Report/sidecar generation or process
2304
+ # interruption must not leave an older HTML looking current.
2305
+ html_stale_marker.touch()
2306
+ # The #479 guard can refuse this write, so it goes before the sidecars —
2307
+ # a report and labels describing a clustering graph.json does not contain
2308
+ # are worse than no run at all (#2436).
2309
+ if not to_json(G, communities, str(out / "graph.json"),
2310
+ community_labels=labels, built_at_commit=_commit):
2311
+ if not stale_marker_preexisted:
2312
+ _clear_html_stale_marker()
2313
+ print(
2314
+ "graph.json NOT written: refusing to overwrite (see warning above). "
2315
+ "GRAPH_REPORT.md, .graphify_labels.json and .graphify_analysis.json "
2316
+ "left untouched.",
2317
+ file=sys.stderr,
2318
+ )
2319
+ sys.exit(1)
2320
+ tokens = label_token_usage
2321
+ from graphify.report import load_learning_for_report as _llfr
2322
+ report = generate(G, communities, cohesion, labels, gods, surprises,
2323
+ {"warning": "cluster-only mode — file stats not available"},
2324
+ tokens, str(watch_path), suggested_questions=questions,
2325
+ min_community_size=min_community_size, built_at_commit=_commit,
2326
+ learning=_llfr(out / "graph.json"))
2327
+ (out / "GRAPH_REPORT.md").write_text(report, encoding="utf-8")
2328
+ stages.mark("report")
2329
+ analysis = {
2330
+ "communities": {str(k): v for k, v in communities.items()},
2331
+ "cohesion": {str(k): v for k, v in cohesion.items()},
2332
+ "gods": gods,
2333
+ "surprises": surprises,
2334
+ "questions": questions,
2335
+ }
2336
+ (out / ".graphify_analysis.json").write_text(
2337
+ json.dumps(analysis, indent=2, ensure_ascii=False),
2338
+ encoding="utf-8",
2339
+ )
2340
+ # Don't persist placeholder-only labels (or their .sig): leaving the
2341
+ # sidecar absent lets a later run generate real labels instead of reading
2342
+ # back "Community N" as authoritative (#2073).
2343
+ if not placeholder_only:
2344
+ from graphify.paths import write_json_atomic as _wja
2345
+ _wja(labels_path, {str(k): v for k, v in labels.items()}, ensure_ascii=False)
2346
+ # Membership signatures beside the labels so a later cluster-only can
2347
+ # detect which communities changed and avoid reusing a stale label
2348
+ # (see reuse above).
2349
+ from graphify.cluster import community_member_sigs as _cms
2350
+ (labels_path.parent / (labels_path.name + ".sig")).write_text(
2351
+ json.dumps({str(k): v for k, v in _cms(communities).items()}), encoding="utf-8")
2352
+
2353
+ # Mirror watch.py pattern: gate to_html so core outputs (graph.json +
2354
+ # GRAPH_REPORT.md) always land. Honor --no-viz explicitly; otherwise
2355
+ # fall back to ValueError handling so an oversized graph doesn't crash
2356
+ # the CLI mid-write and leave a stale graph.html on disk.
2357
+ html_target = out / "graph.html"
2358
+ if no_viz:
2359
+ if html_target.exists():
2360
+ html_target.unlink()
2361
+ _clear_html_stale_marker()
2362
+ stages.mark("export"); stages.total()
2363
+ print(f"Done - {len(communities)} communities. GRAPH_REPORT.md and graph.json updated (--no-viz; graph.html removed).")
2364
+ else:
2365
+ html_written = False
2366
+ skip_reason: str | None = None
2367
+ try:
2368
+ from graphify.exporters.html import _viz_node_limit
2369
+ viz_limit = _viz_node_limit()
2370
+ if viz_limit <= 0:
2371
+ html_target.unlink(missing_ok=True)
2372
+ _clear_html_stale_marker()
2373
+ skip_reason = "GRAPHIFY_VIZ_NODE_LIMIT=0 disables HTML visualization"
2374
+ else:
2375
+ # Passing the positive visualization limit explicitly selects
2376
+ # the community meta-graph when the full graph is too large.
2377
+ html_written = to_html(
2378
+ G,
2379
+ communities,
2380
+ str(html_target),
2381
+ community_labels=labels or None,
2382
+ node_limit=viz_limit,
2383
+ )
2384
+ if html_written:
2385
+ _clear_html_stale_marker()
2386
+ else:
2387
+ skip_reason = "no useful community aggregation could be generated"
2388
+ if html_target.exists():
2389
+ skip_reason += "; existing graph.html left unchanged"
2390
+ except ValueError as viz_err:
2391
+ skip_reason = str(viz_err)
2392
+ if html_target.exists():
2393
+ skip_reason += "; existing graph.html left unchanged"
2394
+
2395
+ if skip_reason:
2396
+ print(f"Skipped graph.html: {skip_reason}")
2397
+ stages.mark("export"); stages.total()
2398
+ if html_written:
2399
+ print(f"Done - {len(communities)} communities. GRAPH_REPORT.md, graph.json and graph.html updated.")
2400
+ else:
2401
+ print(f"Done - {len(communities)} communities. GRAPH_REPORT.md and graph.json updated.")
2402
+
2403
+ elif cmd == "update":
2404
+ force = os.environ.get("GRAPHIFY_FORCE", "").lower() in ("1", "true", "yes")
2405
+ no_cluster = False
2406
+ args = sys.argv[2:]
2407
+ watch_arg: str | None = None
2408
+ for a in args:
2409
+ if a == "--force":
2410
+ force = True
2411
+ continue
2412
+ if a == "--no-cluster":
2413
+ no_cluster = True
2414
+ continue
2415
+ if a.startswith("-"):
2416
+ print(f"error: unknown update option: {a}", file=sys.stderr)
2417
+ sys.exit(2)
2418
+ if watch_arg is not None:
2419
+ print("error: update accepts at most one path argument", file=sys.stderr)
2420
+ sys.exit(2)
2421
+ watch_arg = a
2422
+
2423
+ if watch_arg is not None:
2424
+ watch_path = Path(watch_arg)
2425
+ else:
2426
+ # Try to recover the scan root saved by the last full build
2427
+ saved = Path(_GRAPHIFY_OUT) / ".graphify_root"
2428
+ if saved.exists():
2429
+ # utf-8-sig: a marker written by Windows PowerShell 5.1 carries a
2430
+ # UTF-8 BOM that plain utf-8 keeps as U+FEFF (not stripped by
2431
+ # .strip()), which would make this recovered path fail exists()
2432
+ # (#3028). Match the other .graphify_root readers.
2433
+ watch_path = Path(saved.read_text(encoding="utf-8-sig").strip())
2434
+ else:
2435
+ watch_path = Path(".")
2436
+ if not watch_path.exists():
2437
+ print(f"error: path not found: {watch_path}", file=sys.stderr)
2438
+ sys.exit(1)
2439
+ from graphify.watch import _rebuild_code
2440
+
2441
+ print(f"Re-extracting code files in {watch_path} (no LLM needed)...")
2442
+ # Interactive CLI: block on the per-repo lock rather than skip, so the
2443
+ # user sees their explicit `graphify update` complete instead of
2444
+ # exiting silently when a hook-driven rebuild happens to be running.
2445
+ ok = _rebuild_code(watch_path, force=force, no_cluster=no_cluster, block_on_lock=True)
2446
+ if ok:
2447
+ print("Code graph updated. For doc/paper/image changes run /graphify --update in your AI assistant.")
2448
+ if not (
2449
+ os.environ.get("GEMINI_API_KEY")
2450
+ or os.environ.get("GOOGLE_API_KEY")
2451
+ or os.environ.get("MOONSHOT_API_KEY")
2452
+ or os.environ.get("DEEPSEEK_API_KEY")
2453
+ or os.environ.get("GRAPHIFY_NO_TIPS")
2454
+ ):
2455
+ print("Tip: set GEMINI_API_KEY or GOOGLE_API_KEY to use Gemini for semantic extraction.")
2456
+ else:
2457
+ print(
2458
+ "Nothing to update or rebuild failed — check output above.",
2459
+ file=sys.stderr,
2460
+ )
2461
+ sys.exit(1)
2462
+
2463
+ elif cmd == "hook-check":
2464
+ # Codex Desktop rejects hookSpecificOutput.additionalContext on PreToolUse.
2465
+ # Keep this as a cross-platform no-op so installed hooks never break Bash
2466
+ # tool calls. Graph guidance reaches the agent via AGENTS.md / skill instead.
2467
+ sys.exit(0)
2468
+ elif cmd == "hook-guard":
2469
+ # Shell-agnostic Claude/Codebuddy PreToolUse guard (#522). Replaces the old
2470
+ # inline-bash hooks that failed on Windows. Prints an additionalContext nudge
2471
+ # toward graphify when a fresh in-project graph exists; always exits 0. In
2472
+ # strict mode (opt-in, `hook-guard read --strict`) it blocks the first raw
2473
+ # read per session via the JSON permissionDecision payload — never via exit
2474
+ # code — and downgrades to the nudge thereafter.
2475
+ _run_hook_guard(
2476
+ sys.argv[2] if len(sys.argv) > 2 else "",
2477
+ strict="--strict" in sys.argv[3:],
2478
+ )
2479
+ sys.exit(0)
2480
+ elif cmd == "check-update":
2481
+ if len(sys.argv) < 3:
2482
+ print("Usage: graphify check-update <path>", file=sys.stderr)
2483
+ sys.exit(1)
2484
+ from graphify.watch import check_update
2485
+
2486
+ check_update(Path(sys.argv[2]).resolve())
2487
+ sys.exit(0)
2488
+ elif cmd == "tree":
2489
+ # Emit a D3 v7 collapsible-tree HTML view of graph.json:
2490
+ # expand-all / collapse-all / reset-view buttons, multi-line
2491
+ # wrapText labels with separately-coloured name + count,
2492
+ # depth-based palette, click-to-toggle subtree, hover inspector
2493
+ # showing top-K outbound edges per symbol.
2494
+ from typing import Optional as _Opt
2495
+ from graphify.tree_html import write_tree_html, DEFAULT_MAX_CHILDREN
2496
+ graph_path = Path(_GRAPHIFY_OUT) / "graph.json"
2497
+ output_path: "_Opt[Path]" = None
2498
+ root: "_Opt[str]" = None
2499
+ max_children = DEFAULT_MAX_CHILDREN
2500
+ top_k_edges = 0
2501
+ project_label: "_Opt[str]" = None
2502
+ args = sys.argv[2:]
2503
+ i_arg = 0
2504
+ while i_arg < len(args):
2505
+ a = args[i_arg]
2506
+ if a == "--graph" and i_arg + 1 < len(args):
2507
+ graph_path = Path(args[i_arg + 1]); i_arg += 2
2508
+ elif a == "--output" and i_arg + 1 < len(args):
2509
+ output_path = Path(args[i_arg + 1]); i_arg += 2
2510
+ elif a == "--root" and i_arg + 1 < len(args):
2511
+ root = args[i_arg + 1]; i_arg += 2
2512
+ elif a == "--max-children" and i_arg + 1 < len(args):
2513
+ max_children = int(args[i_arg + 1]); i_arg += 2
2514
+ elif a == "--top-k-edges" and i_arg + 1 < len(args):
2515
+ top_k_edges = int(args[i_arg + 1]); i_arg += 2
2516
+ elif a == "--label" and i_arg + 1 < len(args):
2517
+ project_label = args[i_arg + 1]; i_arg += 2
2518
+ elif a in ("-h", "--help"):
2519
+ print("Usage: graphify tree [--graph PATH] [--output HTML]")
2520
+ print(" --graph PATH path to graph.json (default graphify-out/graph.json)")
2521
+ print(" --output HTML output path (default graphify-out/GRAPH_TREE.html)")
2522
+ print(" --root PATH filesystem root (default: longest common dir of all source_files)")
2523
+ print(" --max-children N cap visible children per node (default 200)")
2524
+ print(" --top-k-edges N pre-compute top-K outbound edges per symbol (default 12)")
2525
+ print(" --label NAME project label shown in the page header")
2526
+ return
2527
+ else:
2528
+ i_arg += 1
2529
+ if not graph_path.is_file():
2530
+ print(f"error: graph.json not found at {graph_path}", file=sys.stderr)
2531
+ sys.exit(1)
2532
+ _enforce_graph_size_cap_or_exit(graph_path)
2533
+ if output_path is None:
2534
+ output_path = graph_path.parent / "GRAPH_TREE.html"
2535
+ try:
2536
+ out = write_tree_html(
2537
+ graph_path=graph_path, output_path=output_path,
2538
+ root=root, max_children=max_children,
2539
+ top_k_edges=top_k_edges, project_label=project_label,
2540
+ )
2541
+ except ValueError as exc:
2542
+ # e.g. an explicit --root that matches no source_file — the tree
2543
+ # would silently flatten, so fail loudly instead (#2534).
2544
+ print(f"error: {exc}", file=sys.stderr)
2545
+ sys.exit(1)
2546
+ size_kb = out.stat().st_size / 1024
2547
+ print(f"wrote {out} ({size_kb:.1f} KB)")
2548
+ print(f"open with: xdg-open {out} (or file://{out.resolve()})")
2549
+ sys.exit(0)
2550
+
2551
+ elif cmd == "merge-driver":
2552
+ # git merge driver for graph.json — takes (base, current, other) and writes
2553
+ # the union of current+other nodes/edges back to current. Exits 1 on
2554
+ # corrupt input so git surfaces the conflict instead of silently
2555
+ # accepting a poisoned merge (see F-005).
2556
+ # Usage: graphify merge-driver %O %A %B (set in .git/config merge driver)
2557
+ if len(sys.argv) < 5:
2558
+ print("Usage: graphify merge-driver <base> <current> <other>", file=sys.stderr)
2559
+ sys.exit(1)
2560
+ _base_path, _current_path, _other_path = sys.argv[2], sys.argv[3], sys.argv[4]
2561
+ # Hard caps so a malicious or corrupted graph.json cannot exhaust memory
2562
+ # at parse time. 50 MB / 100k nodes are well above any realistic graph
2563
+ # (typical graphs are <5 MB / <50k nodes); anything larger should fail
2564
+ # the merge so a human can investigate.
2565
+ _MERGE_MAX_BYTES = 50 * 1024 * 1024
2566
+ _MERGE_MAX_NODES = 100_000
2567
+ import networkx as _nx
2568
+ from networkx.readwrite import json_graph as _jg
2569
+ def _load_graph(p: str):
2570
+ path_obj = Path(p)
2571
+ try:
2572
+ size = path_obj.stat().st_size
2573
+ except OSError as exc:
2574
+ raise RuntimeError(f"cannot stat {p}: {exc}") from exc
2575
+ if size > _MERGE_MAX_BYTES:
2576
+ raise RuntimeError(
2577
+ f"graph.json {p} is {size} bytes, exceeds {_MERGE_MAX_BYTES}-byte cap"
2578
+ )
2579
+ data = json.loads(path_obj.read_text(encoding="utf-8"))
2580
+ # A committed raw (--no-cluster) graph stores edges under "edges";
2581
+ # parse via the shared links/edges-normalizing loader (#2212).
2582
+ from graphify.paths import load_node_link_graph as _lnlg
2583
+ return _lnlg(data), data
2584
+ try:
2585
+ G_cur, _ = _load_graph(_current_path)
2586
+ G_oth, _ = _load_graph(_other_path)
2587
+ except Exception as exc:
2588
+ print(f"[graphify merge-driver] error loading graphs: {exc}", file=sys.stderr)
2589
+ sys.exit(1) # surface the conflict so git doesn't accept a corrupt merge
2590
+ merged = _nx.compose(G_cur, G_oth)
2591
+ if merged.number_of_nodes() > _MERGE_MAX_NODES:
2592
+ print(
2593
+ f"[graphify merge-driver] merged graph has {merged.number_of_nodes()} nodes, "
2594
+ f"exceeds {_MERGE_MAX_NODES}-node cap; aborting merge.",
2595
+ file=sys.stderr,
2596
+ )
2597
+ sys.exit(1)
2598
+ try:
2599
+ out_data = _jg.node_link_data(merged, edges="links")
2600
+ except TypeError:
2601
+ out_data = _jg.node_link_data(merged)
2602
+ from graphify.paths import write_json_atomic
2603
+ write_json_atomic(_current_path, out_data, indent=2)
2604
+ sys.exit(0)
2605
+
2606
+ elif cmd == "merge-graphs":
2607
+ # graphify merge-graphs graph1.json graph2.json ... --out merged.json
2608
+ args = sys.argv[2:]
2609
+ graph_paths: list[Path] = []
2610
+ out_path = Path(_GRAPHIFY_OUT) / "merged-graph.json"
2611
+ i = 0
2612
+ while i < len(args):
2613
+ if args[i] == "--out" and i + 1 < len(args):
2614
+ out_path = Path(args[i + 1])
2615
+ i += 2
2616
+ else:
2617
+ graph_paths.append(Path(args[i]))
2618
+ i += 1
2619
+ if len(graph_paths) < 2:
2620
+ print(
2621
+ "Usage: graphify merge-graphs <graph1.json> <graph2.json> [...] [--out merged.json]",
2622
+ file=sys.stderr,
2623
+ )
2624
+ sys.exit(1)
2625
+ import networkx as _nx
2626
+ from networkx.readwrite import json_graph as _jg
2627
+ from graphify.build import prefix_graph_for_global as _prefix, distinct_repo_tags as _repo_tags
2628
+ graphs = []
2629
+ for gp in graph_paths:
2630
+ if not gp.exists():
2631
+ print(f"error: not found: {gp}", file=sys.stderr)
2632
+ sys.exit(1)
2633
+ _enforce_graph_size_cap_or_exit(gp)
2634
+ data = json.loads(gp.read_text(encoding="utf-8"))
2635
+ # Normalize edges/links key before loading — graphify writes "links"
2636
+ # via node_link_data but older runs may have used "edges" (#738).
2637
+ if "links" not in data and "edges" in data:
2638
+ data = dict(data, links=data["edges"])
2639
+ # Preserve stored edge direction across undirected node_link_graph (#2261).
2640
+ # Mirrors cli.py's query pattern and export.py's _src/_tgt restoration.
2641
+ # Keep in-file markers when present (#2309): unconditionally
2642
+ # overwriting them with source/target would clobber the true
2643
+ # direction of a link persisted in flipped endpoint order.
2644
+ data = dict(
2645
+ data,
2646
+ links=[
2647
+ {
2648
+ **link,
2649
+ "_src": link.get("_src", link.get("source")),
2650
+ "_tgt": link.get("_tgt", link.get("target")),
2651
+ }
2652
+ for link in data.get("links", [])
2653
+ ],
2654
+ )
2655
+ try:
2656
+ G = _jg.node_link_graph(data, edges="links")
2657
+ except TypeError:
2658
+ G = _jg.node_link_graph(data)
2659
+ # node_link_graph restores only the nested `graph.hyperedges` slot;
2660
+ # a graph.json whose hyperedges live only at the top level (the
2661
+ # other half of to_json's dual-slot shape, #2485) would silently
2662
+ # lose them here. Fall back to the top-level key (#2484).
2663
+ if "hyperedges" not in G.graph and isinstance(data.get("hyperedges"), list):
2664
+ G.graph["hyperedges"] = data["hyperedges"]
2665
+ graphs.append(G)
2666
+ # nx.compose requires all graphs to be the same type. When input graphs
2667
+ # come from different sources (e.g. an AST-only run vs a full LLM run) one
2668
+ # may be a MultiGraph and another a Graph. Normalise everything to Graph
2669
+ # (the graphify default) by converting MultiGraphs with nx.Graph().
2670
+ def _to_simple(g: "_nx.Graph") -> "_nx.Graph":
2671
+ # nx.compose requires every graph to be the same type. Inputs may
2672
+ # disagree on BOTH axes — directed vs undirected, and multi vs simple
2673
+ # — because per-repo graph.json files are written by different extract
2674
+ # paths at different times. Normalise everything to a plain undirected
2675
+ # Graph (the merged cross-repo view is undirected anyway), which covers
2676
+ # DiGraph / MultiGraph / MultiDiGraph. Without this a directed input
2677
+ # crashed compose with "All graphs must be directed or undirected" (#1606).
2678
+ if type(g) is not _nx.Graph:
2679
+ return _nx.Graph(g)
2680
+ return g
2681
+ # Unique repo tag per graph. The bare `graphify-out/..` dir name is not
2682
+ # unique across inputs (src/graphify-out and frontend/src/graphify-out both
2683
+ # → "src"), which collides same-stem node ids and silently merges unrelated
2684
+ # entities (#1729). distinct_repo_tags guarantees a distinct prefix per graph.
2685
+ repo_tags = _repo_tags(graph_paths)
2686
+ naive_tags = [gp.parent.parent.name for gp in graph_paths]
2687
+ if len(set(naive_tags)) != len(naive_tags):
2688
+ print(f" note: repo dir names collide; using distinct tags: {', '.join(repo_tags)}")
2689
+ merged = _nx.Graph()
2690
+ # nx.compose merges graph attrs with dict.update, so each iteration
2691
+ # CLOBBERED the previously accumulated hyperedge list — only the last
2692
+ # input's hyperedges survived (#2484, after @oleksii-tumanov's
2693
+ # diagnosis in PR #1691). Collect every input's prefixed hyperedges
2694
+ # and re-attach the union after composing.
2695
+ collected_hyperedges: list = []
2696
+ # Offset each input's community ids into a shared id space as it is
2697
+ # prefixed: every input numbers its communities from 0, so ids carried
2698
+ # across unchanged collide in the merged graph and the aggregated
2699
+ # community view fuses unrelated communities into one meta-node (#3014).
2700
+ # The first input keeps its original ids (offset 0); local_community on
2701
+ # the prefixed nodes preserves each repo's own partition.
2702
+ community_offset = 0
2703
+ for G, repo_tag in zip(graphs, repo_tags):
2704
+ prefixed = _to_simple(_prefix(G, repo_tag, community_offset=community_offset))
2705
+ hes = prefixed.graph.get("hyperedges")
2706
+ if isinstance(hes, list):
2707
+ collected_hyperedges.extend(h for h in hes if isinstance(h, dict))
2708
+ cids = [
2709
+ d["community"]
2710
+ for _, d in prefixed.nodes(data=True)
2711
+ if isinstance(d.get("community"), int)
2712
+ ]
2713
+ if cids:
2714
+ community_offset = max(community_offset, max(cids) + 1)
2715
+ merged = _nx.compose(merged, prefixed)
2716
+ # A contract type both repos declare arrives as two unconnected nodes,
2717
+ # since every id is repo-prefixed. Link them so a traversal can cross
2718
+ # the repo boundary (#3007).
2719
+ from graphify.cross_repo_types import link_shared_type_declarations as _link_shared
2720
+ shared_links = _link_shared(merged)
2721
+ if shared_links:
2722
+ print(f" linked {shared_links} type declaration(s) shared across repos")
2723
+ # A member call whose receiver type lives in another repo was dropped at
2724
+ # extraction; the caller node carries it and this finishes the edge (#3152).
2725
+ from graphify.cross_repo_calls import link_cross_repo_member_calls as _link_calls
2726
+ call_links = _link_calls(merged)
2727
+ if call_links:
2728
+ print(f" resolved {call_links} member call(s) across repos")
2729
+ # Drop whatever compose left behind (the last input's list, possibly
2730
+ # with internal duplicates) so attach_hyperedges dedups the full
2731
+ # collection by id from a clean slate.
2732
+ merged.graph.pop("hyperedges", None)
2733
+ if collected_hyperedges:
2734
+ from graphify.export import attach_hyperedges as _attach
2735
+ _attach(merged, collected_hyperedges)
2736
+ try:
2737
+ out_data = _jg.node_link_data(merged, edges="links")
2738
+ except TypeError:
2739
+ out_data = _jg.node_link_data(merged)
2740
+ # Restore original edge direction from _src/_tgt markers (same pattern as export.py #563/#2261)
2741
+ for link in out_data.get("links", []):
2742
+ tsrc = link.pop("_src", None)
2743
+ ttgt = link.pop("_tgt", None)
2744
+ if tsrc is not None and ttgt is not None:
2745
+ link["source"] = tsrc
2746
+ link["target"] = ttgt
2747
+ # Persist BOTH hyperedge slots (#2484): node_link_data only nests graph
2748
+ # attrs under `graph`, so without this line the union would survive
2749
+ # solely in the slot historic readers ignored (#2485). Mirror to_json's
2750
+ # dual-slot shape so every writer agrees.
2751
+ out_data["hyperedges"] = merged.graph.get("hyperedges", [])
2752
+ out_path.parent.mkdir(parents=True, exist_ok=True)
2753
+ from graphify.paths import write_json_atomic as _wja
2754
+ _wja(out_path, out_data, indent=2)
2755
+ print(f"Merged {len(graphs)} graphs -> {merged.number_of_nodes()} nodes, {merged.number_of_edges()} edges")
2756
+ print(f"Written to: {out_path}")
2757
+
2758
+ elif cmd == "clone":
2759
+ if len(sys.argv) < 3:
2760
+ print(
2761
+ "Usage: graphify clone <github-url> [--branch <branch>] [--out <dir>]",
2762
+ file=sys.stderr,
2763
+ )
2764
+ sys.exit(1)
2765
+ url = sys.argv[2]
2766
+ branch: str | None = None
2767
+ out_dir: Path | None = None
2768
+ args = sys.argv[3:]
2769
+ i = 0
2770
+ while i < len(args):
2771
+ if args[i] == "--branch" and i + 1 < len(args):
2772
+ branch = args[i + 1]
2773
+ i += 2
2774
+ elif args[i] == "--out" and i + 1 < len(args):
2775
+ out_dir = Path(args[i + 1])
2776
+ i += 2
2777
+ else:
2778
+ i += 1
2779
+ local_path = _clone_repo(url, branch=branch, out_dir=out_dir)
2780
+ print(local_path)
2781
+
2782
+ elif cmd == "export":
2783
+ subcmd = sys.argv[2] if len(sys.argv) > 2 else ""
2784
+ if subcmd not in ("html", "callflow-html", "obsidian", "wiki", "svg", "graphml", "neo4j", "falkordb"):
2785
+ print("Usage: graphify export <format>", file=sys.stderr)
2786
+ print(" html [--graph PATH] [--labels PATH] [--node-limit N] [--no-viz]", file=sys.stderr)
2787
+ print(" callflow-html [GRAPH|DIR] [--graph PATH] [--labels PATH] [--report PATH] [--sections PATH] [--output HTML]", file=sys.stderr)
2788
+ print(" [--lang auto|zh-CN|en] [--max-sections N] [--diagram-scale N]", file=sys.stderr)
2789
+ print(" obsidian [--graph PATH] [--labels PATH] [--dir PATH]", file=sys.stderr)
2790
+ print(" wiki [--graph PATH] [--labels PATH]", file=sys.stderr)
2791
+ print(" svg [--graph PATH] [--labels PATH]", file=sys.stderr)
2792
+ print(" graphml [--graph PATH]", file=sys.stderr)
2793
+ print(" neo4j [--graph PATH] [--push URI] [--user U] [--password P]", file=sys.stderr)
2794
+ print(" (or set NEO4J_PASSWORD instead of --password to keep it off argv)", file=sys.stderr)
2795
+ print(" falkordb [--graph PATH] [--push URI] [--user U] [--password P]", file=sys.stderr)
2796
+ print(" (or set FALKORDB_PASSWORD instead of --password to keep it off argv)", file=sys.stderr)
2797
+ sys.exit(1)
2798
+
2799
+ # Parse shared args
2800
+ args = sys.argv[3:]
2801
+ graph_path = Path(_GRAPHIFY_OUT) / "graph.json"
2802
+ graph_path_explicit = False
2803
+ labels_path = Path(_GRAPHIFY_OUT) / ".graphify_labels.json"
2804
+ labels_path_explicit = False
2805
+ report_path = Path(_GRAPHIFY_OUT) / "GRAPH_REPORT.md"
2806
+ report_path_explicit = False
2807
+ sections_path: Path | None = None
2808
+ callflow_output: Path | None = None
2809
+ callflow_lang = "auto"
2810
+ callflow_max_sections = 15
2811
+ callflow_diagram_scale = 1.0
2812
+ callflow_max_diagram_nodes = 18
2813
+ callflow_max_diagram_edges = 24
2814
+ analysis_path = Path(_GRAPHIFY_OUT) / ".graphify_analysis.json"
2815
+ node_limit = 5000
2816
+ no_viz = False
2817
+ obsidian_dir = Path(_GRAPHIFY_OUT) / "obsidian"
2818
+ # Shared push-connection settings for the graph-database sinks (neo4j,
2819
+ # falkordb), parsed from the generic --push/--user/--password flags below.
2820
+ push_uri: str | None = None
2821
+ push_user = "neo4j" # Neo4j default user; FalkorDB auth is optional and ignores it
2822
+ # F-031: prefer an env var so the password never appears on argv (visible
2823
+ # in `ps` output / shell history). The explicit --password flag still
2824
+ # overrides it. Each sink reads its own var: FALKORDB_PASSWORD for falkordb,
2825
+ # NEO4J_PASSWORD otherwise.
2826
+ push_password: str | None = (
2827
+ os.environ.get("FALKORDB_PASSWORD") if subcmd == "falkordb"
2828
+ else os.environ.get("NEO4J_PASSWORD")
2829
+ ) or None
2830
+ i = 0
2831
+ while i < len(args):
2832
+ a = args[i]
2833
+ if a == "--graph" and i + 1 < len(args):
2834
+ graph_path = Path(args[i + 1])
2835
+ graph_path_explicit = True
2836
+ i += 2
2837
+ elif a == "--labels" and i + 1 < len(args):
2838
+ labels_path = Path(args[i + 1])
2839
+ labels_path_explicit = True
2840
+ i += 2
2841
+ elif a == "--report" and i + 1 < len(args):
2842
+ report_path = Path(args[i + 1])
2843
+ report_path_explicit = True
2844
+ i += 2
2845
+ elif a == "--sections" and i + 1 < len(args):
2846
+ sections_path = Path(args[i + 1]); i += 2
2847
+ elif a == "--output" and i + 1 < len(args):
2848
+ callflow_output = Path(args[i + 1]).expanduser()
2849
+ if not callflow_output.is_absolute():
2850
+ callflow_output = Path.cwd() / callflow_output
2851
+ i += 2
2852
+ elif a == "--lang" and i + 1 < len(args):
2853
+ callflow_lang = args[i + 1]; i += 2
2854
+ elif a == "--max-sections" and i + 1 < len(args):
2855
+ callflow_max_sections = int(args[i + 1]); i += 2
2856
+ elif a == "--diagram-scale" and i + 1 < len(args):
2857
+ callflow_diagram_scale = float(args[i + 1]); i += 2
2858
+ elif a == "--max-diagram-nodes" and i + 1 < len(args):
2859
+ callflow_max_diagram_nodes = int(args[i + 1]); i += 2
2860
+ elif a == "--max-diagram-edges" and i + 1 < len(args):
2861
+ callflow_max_diagram_edges = int(args[i + 1]); i += 2
2862
+ elif a in ("-h", "--help") and subcmd == "callflow-html":
2863
+ print("Usage: graphify export callflow-html [GRAPH|DIR] [--graph PATH] [--labels PATH]")
2864
+ print(" --report PATH path to GRAPH_REPORT.md")
2865
+ print(" --sections PATH JSON section definitions")
2866
+ print(" --output HTML output path (default graphify-out/<project>-callflow.html)")
2867
+ print(" --lang LANG auto, zh-CN, en, etc. (default auto)")
2868
+ print(" --max-sections N maximum auto-derived sections (default 15)")
2869
+ print(" --diagram-scale N Mermaid diagram scale (default 1.0)")
2870
+ print(" --max-diagram-nodes N representative nodes per section (default 18)")
2871
+ print(" --max-diagram-edges N representative edges per section (default 24)")
2872
+ sys.exit(0)
2873
+ elif a == "--node-limit" and i + 1 < len(args):
2874
+ node_limit = int(args[i + 1]); i += 2
2875
+ elif a == "--no-viz":
2876
+ no_viz = True; i += 1
2877
+ elif a == "--dir" and i + 1 < len(args):
2878
+ obsidian_dir = Path(args[i + 1]); i += 2
2879
+ elif a == "--push" and i + 1 < len(args):
2880
+ push_uri = args[i + 1]; i += 2
2881
+ elif a == "--user" and i + 1 < len(args):
2882
+ push_user = args[i + 1]; i += 2
2883
+ elif a == "--password" and i + 1 < len(args):
2884
+ push_password = args[i + 1]; i += 2
2885
+ elif subcmd == "callflow-html" and not a.startswith("-") and not graph_path_explicit:
2886
+ candidate = Path(a)
2887
+ if candidate.name == "graph.json" or candidate.suffix.lower() == ".json":
2888
+ graph_path = candidate
2889
+ elif (candidate / "graph.json").exists():
2890
+ graph_path = candidate / "graph.json"
2891
+ else:
2892
+ graph_path = candidate / _GRAPHIFY_OUT / "graph.json"
2893
+ graph_path_explicit = True
2894
+ i += 1
2895
+ else:
2896
+ i += 1
2897
+
2898
+ graph_path = graph_path.expanduser()
2899
+ if graph_path_explicit:
2900
+ graph_out_dir = graph_path.parent
2901
+ if not labels_path_explicit:
2902
+ labels_path = graph_out_dir / ".graphify_labels.json"
2903
+ if not report_path_explicit:
2904
+ report_path = graph_out_dir / "GRAPH_REPORT.md"
2905
+ labels_path = labels_path.expanduser()
2906
+ report_path = report_path.expanduser()
2907
+
2908
+ if not graph_path.exists():
2909
+ print(f"error: graph not found: {graph_path}. Run /graphify <path> first.", file=sys.stderr)
2910
+ sys.exit(1)
2911
+
2912
+ if subcmd == "callflow-html":
2913
+ from graphify.callflow_html import write_callflow_html as _write_callflow_html
2914
+ out = _write_callflow_html(
2915
+ graph=graph_path,
2916
+ report=report_path,
2917
+ labels=labels_path,
2918
+ sections=sections_path,
2919
+ output=callflow_output,
2920
+ lang=callflow_lang,
2921
+ max_sections=callflow_max_sections,
2922
+ diagram_scale=callflow_diagram_scale,
2923
+ max_diagram_nodes=callflow_max_diagram_nodes,
2924
+ max_diagram_edges=callflow_max_diagram_edges,
2925
+ verbose=True,
2926
+ )
2927
+ print(f"callflow HTML written - open in any browser: {out}")
2928
+ sys.exit(0)
2929
+
2930
+ from networkx.readwrite import json_graph as _jg
2931
+ from graphify.build import build_from_json as _bfj
2932
+ from graphify.security import check_graph_file_size_cap as _check_cap
2933
+
2934
+ # Solution 3 (#1019): for the HTML view, an oversized graph.json should
2935
+ # not be a hard error. Detect the over-cap condition here and fall back
2936
+ # to the community-aggregation view (node_limit=5000) below instead of
2937
+ # exiting 1. All other subcommands keep the hard cap.
2938
+ _over_cap = False
2939
+ try:
2940
+ _check_cap(graph_path)
2941
+ except ValueError as _cap_err:
2942
+ if subcmd == "html":
2943
+ _over_cap = True
2944
+ try:
2945
+ _over_cap_bytes = graph_path.stat().st_size
2946
+ except OSError:
2947
+ _over_cap_bytes = -1
2948
+ print(
2949
+ f"warning: graph.json exceeds cap ({_over_cap_bytes} bytes); "
2950
+ f"falling back to community-aggregation view (node_limit=5000)",
2951
+ file=sys.stderr,
2952
+ )
2953
+ else:
2954
+ print(f"error: {_cap_err}", file=sys.stderr)
2955
+ sys.exit(1)
2956
+ _raw = json.loads(graph_path.read_text(encoding="utf-8"))
2957
+ if "links" not in _raw and "edges" in _raw:
2958
+ _raw = dict(_raw, links=_raw["edges"])
2959
+ try:
2960
+ G = _jg.node_link_graph(_raw, edges="links")
2961
+ except TypeError:
2962
+ G = _jg.node_link_graph(_raw)
2963
+ if isinstance(_raw.get("hyperedges"), list):
2964
+ G.graph["hyperedges"] = _raw["hyperedges"]
2965
+
2966
+ # Load optional analysis/labels
2967
+ communities: dict[int, list[str]] = {}
2968
+ if analysis_path.exists():
2969
+ _an = json.loads(analysis_path.read_text(encoding="utf-8"))
2970
+ communities = {int(k): v for k, v in _an.get("communities", {}).items()}
2971
+ cohesion: dict[int, float] = {int(k): v for k, v in _an.get("cohesion", {}).items()}
2972
+ gods_data = _an.get("gods", [])
2973
+ else:
2974
+ cohesion = {}
2975
+ gods_data = []
2976
+
2977
+ # Fallback: graph.json carries the per-node community as a node attribute
2978
+ # (`to_json` writes it on every node). The analysis sidecar is the
2979
+ # canonical source — but the post-commit / watch rebuild path doesn't
2980
+ # regenerate it, and `extract` may have its temp files cleaned up. When
2981
+ # that happens, `graphify export html` previously bailed with
2982
+ # "Single community - aggregated view not useful." even though the
2983
+ # per-node attribute had the right data all along. Reconstruct from
2984
+ # the graph itself so downstream subcommands (html, obsidian, wiki,
2985
+ # svg, graphml, neo4j) don't silently produce a degraded artifact.
2986
+ if not communities:
2987
+ reconstructed: dict[int, list[str]] = {}
2988
+ for node_id, data in G.nodes(data=True):
2989
+ cid_raw = data.get("community")
2990
+ if cid_raw is None:
2991
+ continue
2992
+ try:
2993
+ cid = int(cid_raw)
2994
+ except (TypeError, ValueError):
2995
+ continue
2996
+ reconstructed.setdefault(cid, []).append(str(node_id))
2997
+ if reconstructed:
2998
+ communities = reconstructed
2999
+
3000
+ labels: dict[int, str] = {}
3001
+ if labels_path.exists():
3002
+ labels = {int(k): v for k, v in json.loads(labels_path.read_text(encoding="utf-8")).items()}
3003
+
3004
+ out_dir = graph_path.parent
3005
+
3006
+ if subcmd == "html":
3007
+ from graphify.export import to_html as _to_html
3008
+ if no_viz:
3009
+ html_target = out_dir / "graph.html"
3010
+ if html_target.exists():
3011
+ html_target.unlink()
3012
+ print("--no-viz: skipped graph.html")
3013
+ else:
3014
+ # Over-cap fallback (#1019): force the community-aggregation
3015
+ # path so the oversized graph still renders a usable artifact.
3016
+ _effective_node_limit = 5000 if _over_cap else node_limit
3017
+ _to_html(G, communities, str(out_dir / "graph.html"),
3018
+ community_labels=labels or None, node_limit=_effective_node_limit)
3019
+ if G.number_of_nodes() <= _effective_node_limit:
3020
+ print(f"graph.html written - open in any browser, no server needed")
3021
+ if _over_cap:
3022
+ sys.exit(0)
3023
+
3024
+ elif subcmd == "obsidian":
3025
+ from graphify.export import to_obsidian as _to_obsidian, to_canvas as _to_canvas
3026
+ n = _to_obsidian(G, communities, str(obsidian_dir),
3027
+ community_labels=labels or None, cohesion=cohesion or None)
3028
+ print(f"Obsidian vault: {n} notes in {obsidian_dir}/")
3029
+ _to_canvas(G, communities, str(obsidian_dir / "graph.canvas"),
3030
+ community_labels=labels or None)
3031
+ print(f"Canvas: {obsidian_dir}/graph.canvas")
3032
+ print(f"Open {obsidian_dir}/ as a vault in Obsidian.")
3033
+
3034
+ elif subcmd == "wiki":
3035
+ from graphify.wiki import to_wiki as _to_wiki
3036
+ from graphify.analyze import god_nodes as _god_nodes
3037
+ if not communities:
3038
+ print(
3039
+ "error: .graphify_analysis.json is missing or empty — refusing to export wiki to prevent data loss.\n"
3040
+ "Run `graphify extract .` (or `graphify cluster-only .`) to regenerate community data first.",
3041
+ file=sys.stderr,
3042
+ )
3043
+ sys.exit(1)
3044
+ if not gods_data:
3045
+ gods_data = _god_nodes(G)
3046
+ n = _to_wiki(G, communities, str(out_dir / "wiki"),
3047
+ community_labels=labels or None, cohesion=cohesion or None,
3048
+ god_nodes_data=gods_data)
3049
+ print(f"Wiki: {n} articles written to {out_dir}/wiki/")
3050
+ print(f" {out_dir}/wiki/index.md -> agent entry point")
3051
+
3052
+ elif subcmd == "svg":
3053
+ from graphify.export import to_svg as _to_svg
3054
+ _to_svg(G, communities, str(out_dir / "graph.svg"),
3055
+ community_labels=labels or None)
3056
+ print(f"graph.svg written - embeds in Obsidian, Notion, GitHub READMEs")
3057
+
3058
+ elif subcmd == "graphml":
3059
+ from graphify.export import to_graphml as _to_graphml
3060
+ _to_graphml(G, communities, str(out_dir / "graph.graphml"))
3061
+ print(f"graph.graphml written - open in Gephi, yEd, or any GraphML tool")
3062
+
3063
+ elif subcmd == "neo4j":
3064
+ if push_uri:
3065
+ from graphify.export import push_to_neo4j as _push
3066
+ if push_password is None:
3067
+ print("error: --password required for --push", file=sys.stderr)
3068
+ sys.exit(1)
3069
+ result = _push(G, uri=push_uri, user=push_user,
3070
+ password=push_password, communities=communities)
3071
+ print(f"Pushed to Neo4j: {result['nodes']} nodes, {result['edges']} edges")
3072
+ else:
3073
+ from graphify.export import to_cypher as _to_cypher
3074
+ _to_cypher(G, str(out_dir / "cypher.txt"))
3075
+ print(f"cypher.txt written - import with: cypher-shell < {out_dir}/cypher.txt")
3076
+
3077
+ elif subcmd == "falkordb":
3078
+ if push_uri:
3079
+ from graphify.export import push_to_falkordb as _push
3080
+ result = _push(G, uri=push_uri, user=push_user,
3081
+ password=push_password, communities=communities)
3082
+ print(f"Pushed to FalkorDB: {result['nodes']} nodes, {result['edges']} edges")
3083
+ else:
3084
+ from graphify.export import to_cypher as _to_cypher
3085
+ _to_cypher(G, str(out_dir / "cypher.txt"))
3086
+ print(f"cypher.txt written ({out_dir}/cypher.txt) - statements are OpenCypher. "
3087
+ f"FalkorDB's GRAPH.QUERY runs one statement at a time (no bulk script "
3088
+ f"import), so load a graph with: graphify export falkordb --push "
3089
+ f"falkordb://localhost:6379")
3090
+
3091
+ elif cmd == "benchmark":
3092
+ from graphify.benchmark import run_benchmark, print_benchmark
3093
+
3094
+ graph_path = sys.argv[2] if len(sys.argv) > 2 else _default_graph_path()
3095
+ _enforce_graph_size_cap_or_exit(Path(graph_path))
3096
+ # Try to load corpus_words from detect output
3097
+ corpus_words = None
3098
+ detect_path = Path(".graphify_detect.json")
3099
+ if detect_path.exists():
3100
+ try:
3101
+ detect_data = json.loads(detect_path.read_text(encoding="utf-8"))
3102
+ corpus_words = detect_data.get("total_words")
3103
+ except Exception:
3104
+ pass
3105
+ result = run_benchmark(graph_path, corpus_words=corpus_words)
3106
+ print_benchmark(result)
3107
+
3108
+ elif cmd == "global":
3109
+ subcmd = sys.argv[2] if len(sys.argv) > 2 else ""
3110
+ from graphify.global_graph import (
3111
+ global_add as _global_add,
3112
+ global_remove as _global_remove,
3113
+ global_list as _global_list,
3114
+ global_path as _global_path,
3115
+ )
3116
+ if subcmd == "add":
3117
+ # graphify global add <graph.json> [--as <tag>]
3118
+ args = sys.argv[3:]
3119
+ source = None
3120
+ tag = None
3121
+ i = 0
3122
+ while i < len(args):
3123
+ if args[i] == "--as" and i + 1 < len(args):
3124
+ tag = args[i + 1]; i += 2
3125
+ elif not source:
3126
+ source = Path(args[i]); i += 1
3127
+ else:
3128
+ i += 1
3129
+ if not source:
3130
+ print("Usage: graphify global add <graph.json> [--as <repo-tag>]", file=sys.stderr)
3131
+ sys.exit(1)
3132
+ if not tag:
3133
+ # Inferred through merge-graphs' own helper, which degrades to "repo"
3134
+ # instead of "": an empty tag prunes by "" and registers a manifest
3135
+ # entry no later add can address.
3136
+ from graphify.build import distinct_repo_tags
3137
+ tag = distinct_repo_tags([source.absolute()])[0]
3138
+ try:
3139
+ result = _global_add(source, tag)
3140
+ if result["skipped"]:
3141
+ print(f"'{tag}' unchanged since last add - global graph not modified.")
3142
+ else:
3143
+ print(f"Added '{tag}' to global graph: +{result['nodes_added']} nodes, "
3144
+ f"-{result['nodes_removed']} pruned. Global: {_global_path()}")
3145
+ if result.get("cross_repo_calls"):
3146
+ print(f" resolved {result['cross_repo_calls']} "
3147
+ f"member call(s) across repos")
3148
+ except Exception as exc:
3149
+ print(f"error: {exc}", file=sys.stderr); sys.exit(1)
3150
+ elif subcmd == "remove":
3151
+ # An omitted tag is a usage error; an explicitly empty one still has to be
3152
+ # addressable, since earlier versions could register a repo under "".
3153
+ if len(sys.argv) <= 3:
3154
+ print("Usage: graphify global remove <repo-tag>", file=sys.stderr); sys.exit(1)
3155
+ tag = sys.argv[3]
3156
+ try:
3157
+ removed = _global_remove(tag)
3158
+ print(f"Removed '{tag}' from global graph ({removed} nodes pruned).")
3159
+ except KeyError as exc:
3160
+ print(f"error: {exc}", file=sys.stderr); sys.exit(1)
3161
+ elif subcmd == "list":
3162
+ repos = _global_list()
3163
+ if not repos:
3164
+ print("Global graph is empty. Use 'graphify global add' to add a project.")
3165
+ else:
3166
+ print(f"Global graph: {_global_path()}")
3167
+ for tag, info in repos.items():
3168
+ print(f" {tag}: {info.get('node_count', '?')} nodes, added {info.get('added_at', '?')[:10]}")
3169
+ elif subcmd == "path":
3170
+ print(_global_path())
3171
+ else:
3172
+ print("Usage: graphify global [add|remove|list|path]", file=sys.stderr); sys.exit(1)
3173
+
3174
+ elif cmd == "extract":
3175
+ # Headless full-pipeline extraction for CI / scripts (#698).
3176
+ # Runs detect -> AST extraction on code -> semantic LLM extraction on
3177
+ # docs/papers/images -> merge -> build -> cluster -> write outputs.
3178
+ # Unlike the skill.md path (which runs through Claude Code subagents),
3179
+ # this calls extract_corpus_parallel directly using whichever backend
3180
+ # has an API key set.
3181
+ if len(sys.argv) < 3:
3182
+ print(
3183
+ "Usage: graphify extract <path> [--backend gemini|kimi|claude|openai|deepseek|ollama] "
3184
+ "[--model M] [--mode deep] [--out DIR|--output DIR] [--google-workspace] [--no-cluster] "
3185
+ "[--no-gitignore] [--code-only] [--no-dedup] "
3186
+ "[--max-workers N] [--token-budget N] [--max-concurrency N] "
3187
+ "[--api-timeout S] [--postgres DSN] [--cargo] [--allow-partial] [--timing]",
3188
+ file=sys.stderr,
3189
+ )
3190
+ sys.exit(1)
3191
+
3192
+ has_path = True
3193
+ if sys.argv[2].startswith("-"):
3194
+ has_path = False
3195
+ target = Path(".").resolve()
3196
+ else:
3197
+ target = Path(sys.argv[2]).resolve()
3198
+ if not target.exists():
3199
+ print(f"error: path not found: {target}", file=sys.stderr)
3200
+ sys.exit(1)
3201
+
3202
+ backend: str | None = None
3203
+ model: str | None = None
3204
+ extract_mode: str | None = None
3205
+ out_dir: Path | None = None
3206
+ cli_postgres_dsn: str | None = None
3207
+ cli_cargo: bool = False
3208
+ cli_allow_partial: bool = False
3209
+ no_cluster = False
3210
+ dedup_llm = False
3211
+ # --no-dedup: skip entity deduplication entirely. On an incremental
3212
+ # merge the fuzzy pass runs over the COMBINED node set (existing graph +
3213
+ # new chunk), so a small diff merged into a large graph can collapse
3214
+ # pre-existing nodes from files the diff never touched. Turning dedup
3215
+ # off also arms build_merge's #479 shrink guard, which is disabled while
3216
+ # dedup is on because fuzzy merging shrinks the graph legitimately (#2881).
3217
+ no_dedup = False
3218
+ google_workspace = False
3219
+ global_merge = False
3220
+ code_only = False
3221
+ no_gitignore = False
3222
+ global_repo_tag: str | None = None
3223
+ # Performance/tuning knobs (issue #792). None means "use library default".
3224
+ cli_max_workers: int | None = None
3225
+ cli_token_budget: int | None = None
3226
+ cli_max_concurrency: int | None = None
3227
+ cli_api_timeout: float | None = None
3228
+ # Clustering tuning knobs
3229
+ cli_resolution: float = 1.0
3230
+ cli_exclude_hubs: float | None = None
3231
+ cli_excludes: list[str] = []
3232
+ cli_timing: bool = False
3233
+ # --force parity with `graphify update`: the flag or GRAPHIFY_FORCE=1
3234
+ # disables the incremental gate and skips semantic-cache reads (#1894).
3235
+ force = os.environ.get("GRAPHIFY_FORCE", "").lower() in ("1", "true", "yes")
3236
+
3237
+ def _parse_int(name: str, raw: str) -> int:
3238
+ try:
3239
+ v = int(raw)
3240
+ except ValueError:
3241
+ print(f"error: {name} must be a positive integer (got {raw!r})", file=sys.stderr)
3242
+ sys.exit(2)
3243
+ if v <= 0:
3244
+ print(f"error: {name} must be > 0 (got {v})", file=sys.stderr)
3245
+ sys.exit(2)
3246
+ return v
3247
+
3248
+ def _parse_float(name: str, raw: str) -> float:
3249
+ try:
3250
+ v = float(raw)
3251
+ except ValueError:
3252
+ print(f"error: {name} must be a positive number (got {raw!r})", file=sys.stderr)
3253
+ sys.exit(2)
3254
+ if v <= 0:
3255
+ print(f"error: {name} must be > 0 (got {v})", file=sys.stderr)
3256
+ sys.exit(2)
3257
+ return v
3258
+
3259
+ args = sys.argv[3:] if has_path else sys.argv[2:]
3260
+ i = 0
3261
+ while i < len(args):
3262
+ a = args[i]
3263
+ if a == "--backend" and i + 1 < len(args):
3264
+ backend = args[i + 1]; i += 2
3265
+ elif a.startswith("--backend="):
3266
+ backend = a.split("=", 1)[1]; i += 1
3267
+ elif a == "--model" and i + 1 < len(args):
3268
+ model = args[i + 1]; i += 2
3269
+ elif a.startswith("--model="):
3270
+ model = a.split("=", 1)[1]; i += 1
3271
+ elif a == "--mode" and i + 1 < len(args):
3272
+ extract_mode = args[i + 1]; i += 2
3273
+ elif a.startswith("--mode="):
3274
+ extract_mode = a.split("=", 1)[1]; i += 1
3275
+ elif a in ("--out", "--output") and i + 1 < len(args):
3276
+ # --output is an alias of --out (#2004): it was silently dropped
3277
+ # before, and `graphify tree` already documents --output, so the
3278
+ # mistake is natural. (--output= does not startswith --out=.)
3279
+ out_dir = Path(args[i + 1]); i += 2
3280
+ elif a.startswith(("--out=", "--output=")):
3281
+ out_dir = Path(a.split("=", 1)[1]); i += 1
3282
+ elif a == "--no-cluster":
3283
+ no_cluster = True; i += 1
3284
+ elif a == "--dedup-llm":
3285
+ dedup_llm = True; i += 1
3286
+ elif a == "--no-dedup":
3287
+ no_dedup = True; i += 1
3288
+ elif a == "--code-only":
3289
+ code_only = True; i += 1
3290
+ elif a == "--google-workspace":
3291
+ google_workspace = True; i += 1
3292
+ elif a == "--no-gitignore":
3293
+ no_gitignore = True; i += 1
3294
+ elif a == "--global":
3295
+ global_merge = True; i += 1
3296
+ elif a == "--as" and i + 1 < len(args):
3297
+ global_repo_tag = args[i + 1]; i += 2
3298
+ elif a == "--max-workers" and i + 1 < len(args):
3299
+ cli_max_workers = _parse_int("--max-workers", args[i + 1]); i += 2
3300
+ elif a.startswith("--max-workers="):
3301
+ cli_max_workers = _parse_int("--max-workers", a.split("=", 1)[1]); i += 1
3302
+ elif a == "--token-budget" and i + 1 < len(args):
3303
+ cli_token_budget = _parse_int("--token-budget", args[i + 1]); i += 2
3304
+ elif a.startswith("--token-budget="):
3305
+ cli_token_budget = _parse_int("--token-budget", a.split("=", 1)[1]); i += 1
3306
+ elif a == "--max-concurrency" and i + 1 < len(args):
3307
+ cli_max_concurrency = _parse_int("--max-concurrency", args[i + 1]); i += 2
3308
+ elif a.startswith("--max-concurrency="):
3309
+ cli_max_concurrency = _parse_int("--max-concurrency", a.split("=", 1)[1]); i += 1
3310
+ elif a == "--api-timeout" and i + 1 < len(args):
3311
+ cli_api_timeout = _parse_float("--api-timeout", args[i + 1]); i += 2
3312
+ elif a.startswith("--api-timeout="):
3313
+ cli_api_timeout = _parse_float("--api-timeout", a.split("=", 1)[1]); i += 1
3314
+ elif a == "--resolution" and i + 1 < len(args):
3315
+ cli_resolution = _parse_float("--resolution", args[i + 1]); i += 2
3316
+ elif a.startswith("--resolution="):
3317
+ cli_resolution = _parse_float("--resolution", a.split("=", 1)[1]); i += 1
3318
+ elif a == "--exclude-hubs" and i + 1 < len(args):
3319
+ cli_exclude_hubs = float(args[i + 1]); i += 2
3320
+ elif a.startswith("--exclude-hubs="):
3321
+ cli_exclude_hubs = float(a.split("=", 1)[1]); i += 1
3322
+ elif a == "--exclude" and i + 1 < len(args):
3323
+ cli_excludes.append(args[i + 1]); i += 2
3324
+ elif a.startswith("--exclude="):
3325
+ cli_excludes.append(a.split("=", 1)[1]); i += 1
3326
+ elif a == "--postgres" and i + 1 < len(args):
3327
+ cli_postgres_dsn = args[i + 1]; i += 2
3328
+ elif a.startswith("--postgres="):
3329
+ cli_postgres_dsn = a.split("=", 1)[1]; i += 1
3330
+ elif a == "--cargo":
3331
+ cli_cargo = True
3332
+ i += 1
3333
+ elif a == "--force":
3334
+ force = True; i += 1
3335
+ elif a == "--allow-partial":
3336
+ cli_allow_partial = True; i += 1
3337
+ elif a == "--timing":
3338
+ cli_timing = True; i += 1
3339
+ else:
3340
+ i += 1
3341
+
3342
+ if not has_path and cli_postgres_dsn is None:
3343
+ print("error: must specify a path to scan or a --postgres DSN", file=sys.stderr)
3344
+ sys.exit(1)
3345
+
3346
+ if no_dedup and dedup_llm:
3347
+ # --dedup-llm is pass 3 of the dedup pipeline, so with dedup off it
3348
+ # would be a silent no-op that still demands an API key.
3349
+ print(
3350
+ "error: --no-dedup and --dedup-llm are mutually exclusive "
3351
+ "(--dedup-llm is a tiebreaker inside the dedup pass)",
3352
+ file=sys.stderr,
3353
+ )
3354
+ sys.exit(2)
3355
+
3356
+ _VALID_MODES = {"deep"}
3357
+ if extract_mode is not None and extract_mode not in _VALID_MODES:
3358
+ print(
3359
+ f"error: unknown --mode '{extract_mode}'. "
3360
+ f"Available: {', '.join(sorted(_VALID_MODES))}",
3361
+ file=sys.stderr,
3362
+ )
3363
+ sys.exit(2)
3364
+ deep_mode = extract_mode == "deep"
3365
+ if deep_mode:
3366
+ print("[graphify extract] deep mode enabled: richer semantic extraction")
3367
+
3368
+ # CLI flag wins over env var. Setting GRAPHIFY_API_TIMEOUT here so
3369
+ # _call_openai_compat picks it up without needing a new kwarg path.
3370
+ if cli_api_timeout is not None:
3371
+ os.environ["GRAPHIFY_API_TIMEOUT"] = str(cli_api_timeout)
3372
+ if cli_max_workers is not None:
3373
+ os.environ["GRAPHIFY_MAX_WORKERS"] = str(cli_max_workers)
3374
+
3375
+ # Resolve output dir. The user-facing contract is "<out>/graphify-out/"
3376
+ # so a fresh checkout writes graphify-out/ at the project root, matching
3377
+ # the skill.md pipeline.
3378
+ out_root = (out_dir.resolve() if out_dir else target)
3379
+ graphify_out = out_root / _GRAPHIFY_OUT
3380
+ graphify_out.mkdir(parents=True, exist_ok=True)
3381
+ # Persist corpus-shaping options so later update/watch/hook rebuilds
3382
+ # use the same file set as the initial extraction (#1886).
3383
+ from graphify.watch import (
3384
+ _write_build_config as _write_build_cfg,
3385
+ _read_build_excludes as _read_build_ex,
3386
+ _read_build_gitignore as _read_build_gi,
3387
+ )
3388
+ # #1971 persistence: an explicit --no-gitignore persists False; a later
3389
+ # flag-less `graphify extract` must NOT clobber it back to True, which
3390
+ # would make the git-ignored code silently disappear again (the exact
3391
+ # complaint #1971 is about). Honor the persisted value for THIS run when
3392
+ # the flag is absent (read before the write below), and write False only
3393
+ # when the flag is set — None leaves the setting as-is, mirroring how
3394
+ # #1886 persists --exclude.
3395
+ _effective_gitignore = False if no_gitignore else _read_build_gi(graphify_out)
3396
+ # An explicit list replaces the persisted one; omission reuses it.
3397
+ _effective_excludes = cli_excludes or _read_build_ex(graphify_out)
3398
+ _write_build_cfg(
3399
+ graphify_out,
3400
+ excludes=cli_excludes or None,
3401
+ gitignore=False if no_gitignore else None,
3402
+ )
3403
+
3404
+ stages = _StageTimer(cli_timing)
3405
+
3406
+ from graphify.detect import (
3407
+ detect as _detect,
3408
+ detect_incremental as _detect_incremental,
3409
+ save_manifest as _save_manifest,
3410
+ )
3411
+ manifest_path = graphify_out / "manifest.json"
3412
+ existing_graph_path = graphify_out / "graph.json"
3413
+ # #1925: a missing manifest.json must not degrade to a full scan that
3414
+ # discards the existing graph's semantic layer. An existing graph.json
3415
+ # is a sufficient incremental baseline: detect_incremental treats an
3416
+ # absent manifest as "everything is new" (re-extract all, nothing
3417
+ # deleted), and build_merge + _stale_graph_sources reconcile replaced
3418
+ # and genuinely-deleted sources against the current corpus, so doc/
3419
+ # paper/image nodes survive a --code-only rebuild instead of being
3420
+ # dropped with the rest of the committed graph.
3421
+ incremental_mode = existing_graph_path.exists() if has_path else False
3422
+ # --force: full scan, not the manifest-gated incremental diff — a warm
3423
+ # unchanged tree would otherwise dispatch zero files (#1894).
3424
+ incremental_mode = incremental_mode and not force
3425
+ # #2923/#3125: --force --code-only must NOT drop the existing semantic layer.
3426
+ # The AST pass is fully replaced (full code re-scan, AST extraction on all
3427
+ # code files), but the semantic pass is skipped, so doc/paper/image nodes
3428
+ # from the existing graph carry forward via the merge (build_merge /
3429
+ # merge_raw_extraction keep them because no new semantic-tier sources
3430
+ # are dispatched). We do NOT set incremental_mode = True here because that
3431
+ # would run _detect_incremental and drop unchanged code files from the AST
3432
+ # pass; instead we keep incremental_mode = False so all code files are
3433
+ # scanned, while merge_existing_graph below ensures build_merge still runs.
3434
+ merge_existing_graph = incremental_mode or (code_only and existing_graph_path.exists())
3435
+ if force and code_only and existing_graph_path.exists():
3436
+ print(
3437
+ "[graphify extract] --force --code-only: full AST re-scan, "
3438
+ "existing semantic layer preserved (no semantic pass this run)"
3439
+ )
3440
+ if force:
3441
+ print("[graphify extract] --force: full re-scan, semantic cache reads skipped")
3442
+ elif incremental_mode and not manifest_path.exists():
3443
+ print(
3444
+ "[graphify extract] manifest.json missing; using existing "
3445
+ "graph.json as the incremental baseline (all files re-checked; "
3446
+ "nodes for files outside this run's scope are preserved)"
3447
+ )
3448
+
3449
+ if not has_path:
3450
+ detection = {}
3451
+ code_files = []
3452
+ doc_files = []
3453
+ paper_files = []
3454
+ image_files = []
3455
+ deleted_files = []
3456
+ excluded_files = []
3457
+ graph_stale_sources = []
3458
+ unchanged_total = 0
3459
+ files_by_type = {}
3460
+ elif incremental_mode:
3461
+ print(f"[graphify extract] incremental scan of {target}")
3462
+ detection = _detect_incremental(
3463
+ target,
3464
+ manifest_path=str(manifest_path),
3465
+ google_workspace=google_workspace or None,
3466
+ extra_excludes=_effective_excludes or None,
3467
+ gitignore=_effective_gitignore,
3468
+ )
3469
+ files_by_type = detection.get("files", {})
3470
+ new_by_type = detection.get("new_files", {})
3471
+ code_files = [Path(p) for p in new_by_type.get("code", [])]
3472
+ doc_files = [Path(p) for p in new_by_type.get("document", [])]
3473
+ paper_files = [Path(p) for p in new_by_type.get("paper", [])]
3474
+ image_files = [Path(p) for p in new_by_type.get("image", [])]
3475
+ deleted_files = list(detection.get("deleted_files", []))
3476
+ excluded_files = list(detection.get("excluded_files", []))
3477
+ unchanged_total = sum(len(v) for v in detection.get("unchanged_files", {}).values())
3478
+ # #1909: derive the prune set from the existing graph itself, not
3479
+ # just the manifest. A file that became excluded without ever
3480
+ # being manifest-listed (every pre-#1897 graph is in this state)
3481
+ # still has stale nodes carried forward by build_merge unless the
3482
+ # graph's own sources are reconciled against the current corpus.
3483
+ _seen_files = {f for _fl in files_by_type.values() for f in _fl}
3484
+ _seen_files.update(detection.get("unclassified", []))
3485
+ graph_stale_sources = _stale_graph_sources(
3486
+ existing_graph_path, target, _seen_files, detection=detection
3487
+ )
3488
+ # #2543 heal: manifests poisoned BEFORE failed-source unstamping
3489
+ # existed carry live hashes for code files whose extraction failed
3490
+ # (missing extra, crash) — stamped up-to-date yet absent from
3491
+ # graph.json, so the incremental gate skips them forever. Treat
3492
+ # such a file as changed and re-queue it; if it fails again this
3493
+ # run it is now left unstamped, so this cannot wedge.
3494
+ _healed_sources = _zero_node_stamped_code_sources(
3495
+ existing_graph_path,
3496
+ target,
3497
+ detection.get("unchanged_files", {}).get("code", []),
3498
+ )
3499
+ if _healed_sources:
3500
+ print(
3501
+ f"[graphify extract] re-queuing {len(_healed_sources)} "
3502
+ f"manifest-stamped code file(s) with no nodes in graph.json "
3503
+ f"(prior failed extraction, #2543)"
3504
+ )
3505
+ code_files.extend(Path(p) for p in _healed_sources)
3506
+ # #2927 heal: manifests poisoned BEFORE zero-node semantic cache rejection
3507
+ # existed carry live hashes for semantic files (doc/paper/image) whose
3508
+ # extraction produced zero nodes and zero hyperedges (e.g. edge-only).
3509
+ # Re-queue any such file so it is re-dispatched and self-heals.
3510
+ _unchanged_sem: list[str] = []
3511
+ for _k in ("document", "paper", "image"):
3512
+ _unchanged_sem.extend(detection.get("unchanged_files", {}).get(_k, []))
3513
+ _healed_sem_sources = _zero_node_stamped_semantic_sources(
3514
+ existing_graph_path,
3515
+ target,
3516
+ _unchanged_sem,
3517
+ )
3518
+ if _healed_sem_sources:
3519
+ print(
3520
+ f"[graphify extract] re-queuing {len(_healed_sem_sources)} "
3521
+ f"manifest-stamped semantic file(s) with no nodes or hyperedges in graph.json "
3522
+ f"(prior empty/edge-only extraction, #2927)"
3523
+ )
3524
+ _healed_sem_set = set(_healed_sem_sources)
3525
+ for _p in detection.get("unchanged_files", {}).get("document", []):
3526
+ if _p in _healed_sem_set:
3527
+ doc_files.append(Path(_p))
3528
+ for _p in detection.get("unchanged_files", {}).get("paper", []):
3529
+ if _p in _healed_sem_set:
3530
+ paper_files.append(Path(_p))
3531
+ for _p in detection.get("unchanged_files", {}).get("image", []):
3532
+ if _p in _healed_sem_set:
3533
+ image_files.append(Path(_p))
3534
+ else:
3535
+ print(f"[graphify extract] scanning {target}")
3536
+ detection = _detect(
3537
+ target,
3538
+ google_workspace=google_workspace or None,
3539
+ extra_excludes=_effective_excludes or None,
3540
+ cache_root=out_root,
3541
+ gitignore=_effective_gitignore,
3542
+ )
3543
+ files_by_type = detection.get("files", {})
3544
+ code_files = [Path(p) for p in files_by_type.get("code", [])]
3545
+ doc_files = [Path(p) for p in files_by_type.get("document", [])]
3546
+ paper_files = [Path(p) for p in files_by_type.get("paper", [])]
3547
+ image_files = [Path(p) for p in files_by_type.get("image", [])]
3548
+ deleted_files = []
3549
+ excluded_files = []
3550
+ graph_stale_sources = []
3551
+ unchanged_total = 0
3552
+ if existing_graph_path.exists():
3553
+ _seen_files = {f for _fl in files_by_type.values() for f in _fl}
3554
+ _seen_files.update(detection.get("unclassified", []))
3555
+ graph_stale_sources = _stale_graph_sources(
3556
+ existing_graph_path, target, _seen_files, detection=detection
3557
+ )
3558
+
3559
+ semantic_files = doc_files + paper_files + image_files
3560
+ # --code-only: index code (pure local AST, no key) and skip the semantic
3561
+ # (doc/paper/image) pass entirely, so a mixed repo doesn't hard-fail when no
3562
+ # LLM backend is configured (#1734). Report what was skipped rather than
3563
+ # silently dropping it.
3564
+ if code_only and semantic_files:
3565
+ print(
3566
+ f"[graphify extract] --code-only: skipping {len(semantic_files)} "
3567
+ f"non-code file(s) ({len(doc_files)} docs, {len(paper_files)} papers, "
3568
+ f"{len(image_files)} images) — no LLM extraction"
3569
+ )
3570
+ semantic_files = []
3571
+ doc_files = []
3572
+ paper_files = []
3573
+ image_files = []
3574
+ if deep_mode and incremental_mode and not code_only:
3575
+ # Deep mode reads/writes its own cache namespace
3576
+ # (cache/semantic-deep/), so the manifest's changed-file gate is
3577
+ # not a valid proxy for deep coverage: over a warm unchanged tree
3578
+ # it dispatches zero files and `--mode deep` silently no-ops
3579
+ # (#1894). Widen the semantic pass to the FULL live
3580
+ # doc/paper/image set (``files_by_type`` from detect_incremental,
3581
+ # which already excludes excluded files) and let the
3582
+ # mode-namespaced cache decide hits/misses — the first deep run
3583
+ # re-dispatches everything (deep namespace cold), later deep runs
3584
+ # hit the deep cache.
3585
+ _deep_all = [
3586
+ Path(p)
3587
+ for _ftype in ("document", "paper", "image")
3588
+ for p in files_by_type.get(_ftype, [])
3589
+ ]
3590
+ if len(_deep_all) != len(semantic_files):
3591
+ print(
3592
+ f"[graphify extract] deep mode: widening semantic pass from "
3593
+ f"{len(semantic_files)} changed to {len(_deep_all)} live "
3594
+ f"doc/paper/image file(s); the deep semantic cache decides "
3595
+ f"what is re-extracted"
3596
+ )
3597
+ semantic_files = _deep_all
3598
+ if incremental_mode:
3599
+ # Excluded-but-alive files are reported separately from deletions
3600
+ # (#1908): they still exist on disk, the scan just stopped
3601
+ # covering them (ignore rules / --exclude changed).
3602
+ _excl_note = f"; {len(excluded_files)} excluded" if excluded_files else ""
3603
+ print(
3604
+ f"[graphify extract] {len(code_files)} code, {len(doc_files)} docs, "
3605
+ f"{len(paper_files)} papers, {len(image_files)} images changed; "
3606
+ f"{unchanged_total} unchanged; {len(deleted_files)} deleted"
3607
+ f"{_excl_note}"
3608
+ )
3609
+ else:
3610
+ print(
3611
+ f"[graphify extract] found {len(code_files)} code, "
3612
+ f"{len(doc_files)} docs, {len(paper_files)} papers, "
3613
+ f"{len(image_files)} images"
3614
+ )
3615
+ # Surface files that were seen but not classified (extensionless non-shebang
3616
+ # project files like Dockerfile/Makefile, or unsupported extensions), so they
3617
+ # are no longer invisible in graphify's own output (#1692).
3618
+ _unclassified = detection.get("unclassified", []) if isinstance(detection, dict) else []
3619
+ if _unclassified:
3620
+ _names = ", ".join(sorted({Path(p).name for p in _unclassified})[:6])
3621
+ _more = f" (+{len(_unclassified) - 6} more)" if len(_unclassified) > 6 else ""
3622
+ print(
3623
+ f"[graphify extract] {len(_unclassified)} file(s) not classified "
3624
+ f"(no supported extension or shebang), skipped: {_names}{_more}"
3625
+ )
3626
+ # Name the files dropped by the sensitive-file filter so a wrongly-flagged
3627
+ # source/doc is visible, not just a count (#2106). Operational skips
3628
+ # (symlink/office/Workspace) carry a " [reason]" suffix; exclude those here
3629
+ # so this line reports only the security-heuristic drops.
3630
+ _sensitive = detection.get("skipped_sensitive", []) if isinstance(detection, dict) else []
3631
+ _sec = [s for s in _sensitive if " [" not in s]
3632
+ if _sec:
3633
+ _snames = ", ".join(sorted({Path(p).name for p in _sec})[:6])
3634
+ _smore = f" (+{len(_sec) - 6} more)" if len(_sec) > 6 else ""
3635
+ print(
3636
+ f"[graphify extract] {len(_sec)} file(s) skipped as potentially sensitive "
3637
+ f"(rename or move if wrongly flagged): {_snames}{_smore}"
3638
+ )
3639
+ stages.mark("detect")
3640
+
3641
+ # Resolve the LLM backend only now that we know whether the corpus
3642
+ # needs one. A code-only corpus is pure local AST and must not require
3643
+ # an API key; the key is enforced below only when there's LLM work.
3644
+ from graphify.llm import (
3645
+ BACKENDS as _BACKENDS,
3646
+ detect_backend as _detect_backend,
3647
+ estimate_cost as _estimate_cost,
3648
+ extract_corpus_parallel as _extract_corpus_parallel,
3649
+ _format_backend_env_keys,
3650
+ _get_backend_api_key,
3651
+ )
3652
+ needs_llm = bool(semantic_files) or dedup_llm
3653
+ if backend is None and needs_llm:
3654
+ backend = _detect_backend()
3655
+ if backend is not None and backend not in _BACKENDS:
3656
+ print(
3657
+ f"error: unknown backend '{backend}'. "
3658
+ f"Available: {', '.join(sorted(_BACKENDS))}",
3659
+ file=sys.stderr,
3660
+ )
3661
+ sys.exit(1)
3662
+ if needs_llm:
3663
+ if backend is None:
3664
+ reasons = []
3665
+ if semantic_files:
3666
+ reasons.append(
3667
+ f"{len(semantic_files)} doc/paper/image file(s) need semantic extraction"
3668
+ )
3669
+ if dedup_llm:
3670
+ reasons.append("--dedup-llm was passed")
3671
+ hint = ""
3672
+ if semantic_files:
3673
+ hint = (" Or pass --code-only to index just the code "
3674
+ "(local AST, no key) and skip the non-code files.")
3675
+ print(
3676
+ "error: no LLM API key found (" + "; ".join(reasons) + "). "
3677
+ "Set GEMINI_API_KEY or GOOGLE_API_KEY (gemini), MOONSHOT_API_KEY "
3678
+ "(kimi), ANTHROPIC_API_KEY (claude), OPENAI_API_KEY (openai), "
3679
+ "DEEPSEEK_API_KEY (deepseek), or pass --backend. A code-only "
3680
+ "corpus needs no key." + hint,
3681
+ file=sys.stderr,
3682
+ )
3683
+ sys.exit(1)
3684
+ if backend == "ollama":
3685
+ from graphify.llm import _validate_ollama_base_url
3686
+ _oll_url = os.environ.get("OLLAMA_BASE_URL", _BACKENDS["ollama"].get("base_url", ""))
3687
+ try:
3688
+ _validate_ollama_base_url(_oll_url, warn=False)
3689
+ except ValueError as exc:
3690
+ print(f"error: {exc}", file=sys.stderr)
3691
+ sys.exit(2)
3692
+ if not _get_backend_api_key(backend):
3693
+ allow_no_key = False
3694
+ if backend == "ollama":
3695
+ from urllib.parse import urlparse
3696
+ ollama_url = os.environ.get(
3697
+ "OLLAMA_BASE_URL",
3698
+ _BACKENDS["ollama"].get("base_url", ""),
3699
+ )
3700
+ try:
3701
+ host = (urlparse(ollama_url).hostname or "").lower()
3702
+ except Exception:
3703
+ host = ""
3704
+ allow_no_key = (
3705
+ host in ("localhost", "127.0.0.1", "::1")
3706
+ or host.startswith("127.")
3707
+ )
3708
+ elif backend == "bedrock":
3709
+ allow_no_key = bool(
3710
+ os.environ.get("AWS_PROFILE")
3711
+ or os.environ.get("AWS_REGION")
3712
+ or os.environ.get("AWS_DEFAULT_REGION")
3713
+ or os.environ.get("AWS_ACCESS_KEY_ID")
3714
+ )
3715
+ elif backend == "claude-cli":
3716
+ import shutil as _shutil
3717
+ allow_no_key = _shutil.which("claude") is not None
3718
+ if not allow_no_key:
3719
+ print(
3720
+ "error: backend 'claude-cli' requires the `claude` CLI on $PATH "
3721
+ "(install Claude Code and run `claude` once to authenticate).",
3722
+ file=sys.stderr,
3723
+ )
3724
+ sys.exit(1)
3725
+ if not allow_no_key:
3726
+ print(
3727
+ f"error: backend '{backend}' requires {_format_backend_env_keys(backend)} to be set.",
3728
+ file=sys.stderr,
3729
+ )
3730
+ sys.exit(1)
3731
+
3732
+ # Track whether this run's extraction was incomplete (a whole extractor
3733
+ # pass crashed, or some semantic chunks failed). A partial result must not
3734
+ # be force-written over a good complete graph — the final write falls back
3735
+ # to the #479 shrink guard unless --allow-partial is set.
3736
+ _extraction_incomplete = False
3737
+ # A walk that couldn't fully enumerate the corpus (permission-denied
3738
+ # subtree, I/O error) yields a legitimately smaller graph that must not
3739
+ # be force-written over a complete one — same failure class as a crashed
3740
+ # pass. detect()/detect_incremental() already record these; consume them.
3741
+ if detection.get("walk_errors"):
3742
+ _extraction_incomplete = True
3743
+
3744
+ # AST extraction on code files. Empty code list (docs-only corpus) is
3745
+ # the issue #698 case — skip cleanly instead of crashing inside extract().
3746
+ ast_result: dict = {"nodes": [], "edges": [], "input_tokens": 0, "output_tokens": 0}
3747
+ if code_files:
3748
+ from graphify.extract import extract as _ast_extract
3749
+ # Anchor the cache at the output root, not the scanned project:
3750
+ # with --out, a <target>/graphify-out/cache/ would leak a
3751
+ # graphify-out/ dir into a project that asked for external output.
3752
+ # `root` stays the scanned project so source_file/ids relativize
3753
+ # against it; conflating the two basenamed every node (#1941).
3754
+ ast_kwargs: dict = {"cache_root": out_root, "root": target}
3755
+ if cli_max_workers is not None:
3756
+ ast_kwargs["max_workers"] = cli_max_workers
3757
+ # #2437/#2438 (the `graphify update` twin of watch's #2406 fix): an
3758
+ # incremental re-scan extracts only the changed code files, so the
3759
+ # cross-file resolvers cannot see a callee living in an unchanged
3760
+ # file and every changed->unchanged call edge silently vanished on
3761
+ # merge. Hand extract() read-only resolution context from the
3762
+ # persisted graph: its AST-tier nodes (with their `_callable`/
3763
+ # `_callable_class` markers, #2438) plus the contains/method edges
3764
+ # the member-call resolvers walk (#2437), scoped to the UNCHANGED
3765
+ # live corpus — never a re-extracted, deleted, or excluded file, so
3766
+ # stale symbols cannot resurrect. Fails open (changed-batch-only
3767
+ # resolution, the pre-fix behavior) on an unreadable graph.
3768
+ if incremental_mode and existing_graph_path.exists():
3769
+ _ctx_nodes: list[dict] = []
3770
+ _ctx_edges: list[dict] = []
3771
+ try:
3772
+ from graphify.build import _is_ast_tier as _ctx_is_ast_tier
3773
+ from graphify.security import (
3774
+ check_graph_file_size_cap as _ctx_size_cap,
3775
+ )
3776
+ _ctx_size_cap(existing_graph_path)
3777
+ _ctx_graph = json.loads(
3778
+ existing_graph_path.read_text(encoding="utf-8")
3779
+ )
3780
+ _ctx_root = Path(os.path.abspath(target))
3781
+
3782
+ def _ctx_identity(source_file) -> str | None:
3783
+ # graph.json source_file values are relative to the
3784
+ # scanned root (`root=target` above); detect's
3785
+ # unchanged_files keep their scan-time form. Compare
3786
+ # both as absolute posix paths.
3787
+ if not source_file:
3788
+ return None
3789
+ _p = Path(str(source_file))
3790
+ if not _p.is_absolute():
3791
+ _p = _ctx_root / _p
3792
+ return Path(os.path.abspath(_p)).as_posix()
3793
+
3794
+ _ctx_live = {
3795
+ _ctx_identity(f)
3796
+ for _flist in detection.get("unchanged_files", {}).values()
3797
+ for f in _flist
3798
+ }
3799
+ _ctx_live.discard(None)
3800
+ for _node in _ctx_graph.get("nodes", []):
3801
+ if not _node.get("id") or not _ctx_is_ast_tier(_node):
3802
+ continue
3803
+ _sf = _node.get("source_file")
3804
+ if not _sf or _ctx_identity(_sf) not in _ctx_live:
3805
+ continue
3806
+ _ctx_node = {
3807
+ "id": _node["id"],
3808
+ "label": _node.get("label"),
3809
+ "source_file": _sf,
3810
+ "file_type": _node.get("file_type"),
3811
+ "type": _node.get("type"),
3812
+ }
3813
+ for _marker in ("_callable", "_callable_class"):
3814
+ if _node.get(_marker):
3815
+ _ctx_node[_marker] = _node[_marker]
3816
+ _ctx_nodes.append(_ctx_node)
3817
+ for _edge in _ctx_graph.get(
3818
+ "links", _ctx_graph.get("edges", [])
3819
+ ):
3820
+ if _edge.get("relation") not in ("contains", "method"):
3821
+ continue
3822
+ if not _ctx_is_ast_tier(_edge):
3823
+ continue
3824
+ _sf = _edge.get("source_file")
3825
+ if not _sf or _ctx_identity(_sf) not in _ctx_live:
3826
+ continue
3827
+ _ctx_edges.append({
3828
+ "source": _edge.get("source"),
3829
+ "target": _edge.get("target"),
3830
+ "relation": _edge.get("relation"),
3831
+ "source_file": _sf,
3832
+ })
3833
+ except Exception:
3834
+ _ctx_nodes, _ctx_edges = [], []
3835
+ if _ctx_nodes:
3836
+ ast_kwargs["resolution_context_nodes"] = _ctx_nodes
3837
+ if _ctx_edges:
3838
+ ast_kwargs["resolution_context_edges"] = _ctx_edges
3839
+ print(f"[graphify extract] AST extraction on {len(code_files)} code files...")
3840
+ try:
3841
+ ast_result = _ast_extract(code_files, **ast_kwargs)
3842
+ except Exception as exc:
3843
+ print(f"[graphify extract] AST extraction failed: {exc}", file=sys.stderr)
3844
+ # #2445: losing the whole AST pass is fatal by default. The
3845
+ # empty stand-in only reaches the shrink guard when an existing
3846
+ # graph is larger — on a fresh build it used to be written as a
3847
+ # 0-node graph with exit 0, indistinguishable from success.
3848
+ # --allow-partial opts back into the best-effort continuation.
3849
+ if not cli_allow_partial:
3850
+ sys.exit(1)
3851
+ ast_result = {"nodes": [], "edges": [], "input_tokens": 0, "output_tokens": 0}
3852
+ _extraction_incomplete = True # the whole AST pass was lost
3853
+ stages.mark("AST extract")
3854
+
3855
+ # Semantic extraction on docs/papers/images. Check cache first.
3856
+ from graphify.cache import (
3857
+ check_semantic_cache as _check_semantic_cache,
3858
+ prune_semantic_cache as _prune_semantic_cache,
3859
+ save_semantic_cache as _save_semantic_cache,
3860
+ scope_semantic_result as _scope_semantic_result,
3861
+ )
3862
+ sem_result: dict = {
3863
+ "nodes": [], "edges": [], "hyperedges": [],
3864
+ "input_tokens": 0, "output_tokens": 0,
3865
+ }
3866
+ # Semantic files whose extraction truncated this run. They are left
3867
+ # unstamped in the manifest so detect_incremental re-queues them next run
3868
+ # (mirrors the #933 failed-chunk handling); captured below before the
3869
+ # _partial markers are stripped from the corpus.
3870
+ _partial_semantic_files: set[str] = set()
3871
+ sem_cache_hits = 0
3872
+ sem_cache_misses = 0
3873
+ # Deep mode uses its own namespace (cache/semantic-deep/) so deep and
3874
+ # standard results for the same content never shadow each other (#1894).
3875
+ sem_cache_mode = "deep" if deep_mode else None
3876
+ # Entries are attributed to the extraction prompt that produced them, so
3877
+ # a release that changes the prompt re-extracts rather than replaying the
3878
+ # older vintage alongside the new one (#1939). Read and write must pass
3879
+ # the same prompt, or the write lands where the next read won't look.
3880
+ from graphify.llm import _extraction_system as _sem_prompt_for
3881
+ sem_prompt = _sem_prompt_for(deep=deep_mode)
3882
+ if semantic_files:
3883
+ sem_paths_str = [str(p) for p in semantic_files]
3884
+ if force:
3885
+ # --force: skip the cache READ so every semantic file is
3886
+ # re-dispatched; the save below still runs so the fresh
3887
+ # results replace the stale entries.
3888
+ cached_nodes, cached_edges, cached_hyperedges = [], [], []
3889
+ uncached_paths = list(sem_paths_str)
3890
+ else:
3891
+ cached_nodes, cached_edges, cached_hyperedges, uncached_paths = (
3892
+ _check_semantic_cache(sem_paths_str, root=target, cache_root=out_root,
3893
+ mode=sem_cache_mode, prompt=sem_prompt)
3894
+ )
3895
+ sem_cache_hits = len(semantic_files) - len(uncached_paths)
3896
+ sem_cache_misses = len(uncached_paths)
3897
+ sem_result["nodes"].extend(cached_nodes)
3898
+ sem_result["edges"].extend(cached_edges)
3899
+ sem_result["hyperedges"].extend(cached_hyperedges)
3900
+ if sem_cache_hits:
3901
+ print(f"[graphify extract] semantic cache: {sem_cache_hits} hit / {sem_cache_misses} miss")
3902
+
3903
+ if uncached_paths:
3904
+ print(f"[graphify extract] semantic extraction on {len(uncached_paths)} files via {backend}...")
3905
+ corpus_kwargs: dict = {
3906
+ "backend": backend,
3907
+ "model": model,
3908
+ "root": target,
3909
+ "cache_root": out_root,
3910
+ }
3911
+ if deep_mode:
3912
+ corpus_kwargs["deep_mode"] = True
3913
+ if cli_token_budget is not None:
3914
+ corpus_kwargs["token_budget"] = cli_token_budget
3915
+ if cli_max_concurrency is not None:
3916
+ corpus_kwargs["max_concurrency"] = cli_max_concurrency
3917
+
3918
+ # Minimal progress callback so the CLI is no longer silent
3919
+ # during long local-inference runs (issue #792 addendum).
3920
+ # Also track per-chunk success so we can fail loudly when
3921
+ # every chunk errors (e.g. missing backend SDK package).
3922
+ _chunk_stats = {"total": 0, "succeeded": 0}
3923
+ def _progress(idx: int, total: int, _result: dict) -> None:
3924
+ _chunk_stats["total"] = total
3925
+ _chunk_stats["succeeded"] += 1
3926
+ print(
3927
+ f"[graphify extract] chunk {idx + 1}/{total} done",
3928
+ flush=True,
3929
+ )
3930
+ corpus_kwargs["on_chunk_done"] = _progress
3931
+
3932
+ try:
3933
+ fresh = _extract_corpus_parallel(
3934
+ [Path(p) for p in uncached_paths],
3935
+ **corpus_kwargs,
3936
+ )
3937
+ except ImportError as exc:
3938
+ print(f"error: {exc}", file=sys.stderr)
3939
+ sys.exit(1)
3940
+ except Exception as exc:
3941
+ print(
3942
+ f"[graphify extract] semantic extraction failed: {exc}",
3943
+ file=sys.stderr,
3944
+ )
3945
+ fresh = {"nodes": [], "edges": [], "hyperedges": [], "input_tokens": 0, "output_tokens": 0}
3946
+ _extraction_incomplete = True # the semantic pass crashed
3947
+
3948
+ # on_chunk_done only fires after a chunk succeeds. If fresh
3949
+ # semantic extraction was requested and no chunks completed,
3950
+ # fail instead of writing an AST-only graph with exit 0.
3951
+ if uncached_paths and _chunk_stats["succeeded"] == 0:
3952
+ print(
3953
+ f"[graphify extract] error: all semantic chunks failed "
3954
+ f"for backend '{backend}' ({len(uncached_paths)} uncached files) - "
3955
+ f"see per-chunk errors above. If you see 'requires the X package', "
3956
+ f"run `pip install X` and retry.",
3957
+ file=sys.stderr,
3958
+ )
3959
+ sys.exit(1)
3960
+ # Some (but not all) chunks failed — the graph is missing nodes
3961
+ # from the failed chunks, so it must not clobber a larger complete
3962
+ # graph without an explicit --allow-partial override.
3963
+ if _chunk_stats["total"] and _chunk_stats["succeeded"] < _chunk_stats["total"]:
3964
+ _extraction_incomplete = True
3965
+ # #2926: scope the fresh result to the files actually dispatched,
3966
+ # mirroring the allowed_source_files guard the cache write below
3967
+ # applies. A model can attribute stray nodes/edges to a corpus
3968
+ # file that was not dispatched this run; build_merge() derives
3969
+ # its replace-set from the source_files present in new chunks,
3970
+ # so such a fragment would REPLACE that file's entire prior
3971
+ # contribution in graph.json while its manifest entry still says
3972
+ # unchanged — no later incremental run re-dispatches it and the
3973
+ # loss is permanent until a full rebuild.
3974
+ _dropped_files, _dropped_items = _scope_semantic_result(
3975
+ fresh, target, uncached_paths,
3976
+ )
3977
+ if _dropped_files:
3978
+ print(
3979
+ f"[graphify extract] dropped {_dropped_items} out-of-scope "
3980
+ f"item(s) attributed to {len(_dropped_files)} file(s) not "
3981
+ f"dispatched this run: {', '.join(sorted(_dropped_files))}"
3982
+ )
3983
+ # Which files truncated this run (item markers + the empty-parse
3984
+ # _partial_files set). Computed BEFORE the save so it can be passed
3985
+ # as partial_source_files: without it, a file whose only truncated
3986
+ # chunk parsed empty (so it has no item markers here) would be
3987
+ # written as a complete cache entry, re-promoting it (#1950).
3988
+ from graphify.llm import (
3989
+ _partial_source_files as _partial_sf,
3990
+ _strip_partial_markers as _strip_partial,
3991
+ )
3992
+ _partial_semantic_files = set(_partial_sf(fresh))
3993
+ # A chunk that came back hollow after every retry, or as
3994
+ # unparseable JSON, or that simply omitted some of its files,
3995
+ # does not raise - it returns fewer nodes - so it counted as a
3996
+ # SUCCEEDED chunk above and the run read as complete, force=True
3997
+ # bypassed the shrink guard, and a 570-node graph was overwritten
3998
+ # with 111 nodes without a word (#3105). With an LLM backend
3999
+ # that is the normal way an extraction silently produces a
4000
+ # fraction of the graph, so it must arm the guard exactly like a
4001
+ # crashed chunk does. --allow-partial still overrides.
4002
+ _omitted_files = list(fresh.get("uncovered_files") or [])
4003
+ if _omitted_files or _partial_semantic_files:
4004
+ _extraction_incomplete = True
4005
+ print(
4006
+ f"[graphify extract] semantic extraction is incomplete: "
4007
+ f"{len(_omitted_files)} dispatched file(s) produced no nodes and "
4008
+ f"{len(_partial_semantic_files)} came back truncated or hollow. "
4009
+ f"The shrink guard stays armed for this write; pass "
4010
+ f"--allow-partial to overwrite a larger existing graph anyway.",
4011
+ file=sys.stderr,
4012
+ )
4013
+ try:
4014
+ _save_semantic_cache(
4015
+ fresh.get("nodes", []),
4016
+ fresh.get("edges", []),
4017
+ fresh.get("hyperedges", []),
4018
+ root=target,
4019
+ cache_root=out_root,
4020
+ allowed_source_files=uncached_paths,
4021
+ mode=sem_cache_mode,
4022
+ prompt=sem_prompt,
4023
+ partial_source_files=_partial_semantic_files or None,
4024
+ )
4025
+ except Exception as exc:
4026
+ print(f"[graphify extract] warning: could not write semantic cache: {exc}", file=sys.stderr)
4027
+ # Strip the markers before the corpus feeds the graph so the
4028
+ # internal flag never leaks into graph.json.
4029
+ _strip_partial(fresh)
4030
+ sem_result["nodes"].extend(fresh.get("nodes", []))
4031
+ sem_result["edges"].extend(fresh.get("edges", []))
4032
+ sem_result["hyperedges"].extend(fresh.get("hyperedges", []))
4033
+ sem_result["input_tokens"] += fresh.get("input_tokens", 0)
4034
+ sem_result["output_tokens"] += fresh.get("output_tokens", 0)
4035
+
4036
+ # Prune orphaned semantic cache entries. The semantic cache is
4037
+ # content-hash-keyed and unversioned, so it is never swept by the AST
4038
+ # version-cleanup: every content change or file deletion leaves a
4039
+ # permanent orphan that accumulates unbounded (#1527). Sweep it against
4040
+ # the FULL live document set (``files_by_type`` — present in both the
4041
+ # incremental and full branches), NOT the incremental ``semantic_files``
4042
+ # changed-subset, which would delete every unchanged doc's valid entry.
4043
+ # Best-effort: a prune failure must never break extraction.
4044
+ # Hash keys are anchored to the corpus (``target``) — the same anchor
4045
+ # the cache read/write above use — while the stat-index artifact
4046
+ # follows the cache location (``out_root``). Anchoring these hashes to
4047
+ # ``out_root`` instead would mismatch every key under ``--out`` and
4048
+ # sweep the entire fresh cache as orphaned (#1990/#1991).
4049
+ try:
4050
+ from graphify.cache import file_hash as _file_hash
4051
+ _live_hashes: set[str] = set()
4052
+ for _kind in ("document", "paper", "image"):
4053
+ for _fp in files_by_type.get(_kind, []):
4054
+ _abs = Path(_fp)
4055
+ if not _abs.is_absolute():
4056
+ _abs = Path(target) / _abs
4057
+ if not _abs.is_file():
4058
+ continue # deleted/missing — leave out so its entry is pruned
4059
+ try:
4060
+ _live_hashes.add(_file_hash(_abs, target, cache_root=out_root))
4061
+ except OSError:
4062
+ pass
4063
+ # A pathless database extraction has no filesystem corpus to sweep.
4064
+ if has_path:
4065
+ _prune_semantic_cache(out_root, _live_hashes)
4066
+ except Exception as exc:
4067
+ print(f"[graphify extract] warning: could not prune semantic cache: {exc}", file=sys.stderr)
4068
+ stages.mark("semantic extract")
4069
+
4070
+ pg_result: dict = {"nodes": [], "edges": []}
4071
+ if cli_postgres_dsn is not None:
4072
+ from graphify.pg_introspect import introspect_postgres
4073
+ print(f"[graphify extract] introspecting PostgreSQL schema...")
4074
+ try:
4075
+ pg_result = introspect_postgres(cli_postgres_dsn)
4076
+ except (ConnectionError, ImportError) as exc:
4077
+ print(f"error: {exc}", file=sys.stderr)
4078
+ sys.exit(1)
4079
+ print(f"[graphify extract] PostgreSQL: {len(pg_result['nodes'])} nodes, "
4080
+ f"{len(pg_result['edges'])} edges")
4081
+
4082
+ cargo_result: dict = {"nodes": [], "edges": []}
4083
+ if cli_cargo:
4084
+ from graphify.cargo_introspect import introspect_cargo
4085
+ print("[graphify extract] introspecting Cargo workspace...")
4086
+ try:
4087
+ cargo_result = introspect_cargo(target)
4088
+ except (ConnectionError, ImportError, OSError) as exc:
4089
+ print(f"error: {exc}", file=sys.stderr)
4090
+ sys.exit(1)
4091
+ print(f"[graphify extract] Cargo: {len(cargo_result['nodes'])} nodes, "
4092
+ f"{len(cargo_result['edges'])} edges")
4093
+
4094
+ # Merge AST + semantic + pg_result + cargo_result. Order matters for deduplication: passing AST
4095
+ # first means semantic node attributes win on collision (richer labels
4096
+ # for symbols also referenced in docs). Hyperedges only come from the
4097
+ # semantic side.
4098
+ merged: dict = {
4099
+ "nodes": list(ast_result.get("nodes", [])) + list(sem_result.get("nodes", [])) + list(pg_result.get("nodes", [])) + list(cargo_result.get("nodes", [])),
4100
+ "edges": list(ast_result.get("edges", [])) + list(sem_result.get("edges", [])) + list(pg_result.get("edges", [])) + list(cargo_result.get("edges", [])),
4101
+ "hyperedges": list(sem_result.get("hyperedges", [])),
4102
+ "input_tokens": ast_result.get("input_tokens", 0) + sem_result.get("input_tokens", 0),
4103
+ "output_tokens": ast_result.get("output_tokens", 0) + sem_result.get("output_tokens", 0),
4104
+ "extracted_sources": list(ast_result.get("extracted_sources", [])),
4105
+ }
4106
+
4107
+ graph_json_path = graphify_out / "graph.json"
4108
+ analysis_path = graphify_out / ".graphify_analysis.json"
4109
+
4110
+ # Build a manifest-safe files dict: only stamp semantic_hash for files
4111
+ # that actually produced output (cache hit or fresh extraction). Files
4112
+ # whose chunk failed have no source_file entry in sem_result — leaving
4113
+ # their semantic_hash empty so detect_incremental re-queues them (#933).
4114
+ # Path normalization against the scan root happens inside the helper
4115
+ # (#1897) so fresh root-relative source_files match detect()'s
4116
+ # absolute file lists.
4117
+ # #2543: also drop AST sources that failed (missing optional extra /
4118
+ # zero-node anomaly) so they are not frozen as up-to-date.
4119
+ _failed_ast_sources = list(ast_result.get("failed_sources") or [])
4120
+ _manifest_files = _stamped_manifest_files(
4121
+ files_by_type,
4122
+ sem_result,
4123
+ target,
4124
+ partial_source_files=_partial_semantic_files,
4125
+ failed_ast_sources=_failed_ast_sources,
4126
+ )
4127
+
4128
+ # Files dispatched this run but dropped by _stamped_manifest_files
4129
+ # above (failed chunk, LLM omission, or any future exclusion) still
4130
+ # carry a stale semantic_hash from a prior successful run in the
4131
+ # on-disk manifest; save_manifest's seed loop would otherwise copy it
4132
+ # verbatim and mask the omission (#1948). Derived from semantic_files
4133
+ # — what was actually SENT to the backend this run (narrowed by the
4134
+ # incremental gate and --code-only, widened by deep mode) — NOT from
4135
+ # files_by_type: the full live corpus includes untouched files that
4136
+ # were never dispatched, and clearing those would blank the whole
4137
+ # manifest on every partial incremental run, forcing a full-corpus
4138
+ # re-extraction on the next one.
4139
+ _stamped_semantic = {
4140
+ f for _flist in _manifest_files.values() for f in _flist
4141
+ }
4142
+ _cleared_semantic = {str(p) for p in semantic_files} - _stamped_semantic
4143
+ # #2543: AST failures need both hashes blanked (clear_ast), not just
4144
+ # semantic_hash — otherwise a prior bad stamp keeps the file "unchanged".
4145
+ _cleared_ast = set(_failed_ast_sources)
4146
+
4147
+ # Full-scan manifest saves prune rows for in-root files that left the
4148
+ # scan corpus but still exist on disk (#1908). The corpus must be the
4149
+ # RAW detect output (files_by_type), NOT the #933-stamp-filtered
4150
+ # _manifest_files above — pruning to the filtered set would erase
4151
+ # failed-chunk/omitted-doc rows and every doc row on --code-only runs.
4152
+ _scan_corpus = (
4153
+ {f for _fl in files_by_type.values() for f in _fl}
4154
+ if has_path else None
4155
+ )
4156
+
4157
+ def _invalidate_file_manifest_for_db_graph() -> None:
4158
+ if has_path:
4159
+ return
4160
+ try:
4161
+ manifest_path.unlink(missing_ok=True)
4162
+ except OSError as exc:
4163
+ print(f"error: could not invalidate file manifest: {exc}", file=sys.stderr)
4164
+ sys.exit(1)
4165
+
4166
+ if no_cluster:
4167
+ # --no-cluster: dump the raw merged extraction as graph.json.
4168
+ # No NetworkX, no community detection, no analysis sidecar.
4169
+ # Dedupe nodes (by id) and parallel edges so the raw output matches the
4170
+ # clustered path (whose DiGraph collapses both) and stays deterministic
4171
+ # across modes (#1317; node dedup also collapses shared Swift module
4172
+ # anchors emitted per importing file, #1327).
4173
+ from graphify.build import dedupe_edges as _dedupe_edges, dedupe_nodes as _dedupe_nodes
4174
+ from graphify.export import (
4175
+ backup_if_protected as _backup,
4176
+ existing_graph_node_count as _existing_graph_node_count,
4177
+ )
4178
+ if (
4179
+ incremental_mode
4180
+ and not code_files
4181
+ and not semantic_files
4182
+ and not deleted_files
4183
+ and not pg_result.get("nodes")
4184
+ and not pg_result.get("edges")
4185
+ and not cargo_result.get("nodes")
4186
+ and not cargo_result.get("edges")
4187
+ ):
4188
+ # An exclusion-only change reaches this gate (excluded files
4189
+ # are deliberately NOT in deleted_files, #1908) but must still
4190
+ # scrub the newly-excluded sources from the raw graph (#1909).
4191
+ # This path never runs build_merge, so prune in place.
4192
+ if graph_stale_sources:
4193
+ _n_pruned = _prune_graph_json_sources(
4194
+ existing_graph_path, graph_stale_sources
4195
+ )
4196
+ if _n_pruned:
4197
+ print(
4198
+ f"[graphify extract] pruned {_n_pruned} node(s) from "
4199
+ f"{len(graph_stale_sources)} source file(s) no longer "
4200
+ "in the scan (deleted or excluded)."
4201
+ )
4202
+ print(
4203
+ "[graphify extract] no incremental changes detected "
4204
+ "(--no-cluster); outputs left untouched."
4205
+ )
4206
+ try:
4207
+ _save_manifest(_manifest_files, manifest_path=str(manifest_path), kind="both", root=target, scan_corpus=_scan_corpus, clear_semantic=_cleared_semantic, clear_ast=_cleared_ast or None)
4208
+ except Exception as exc:
4209
+ print(f"[graphify extract] warning: could not write manifest: {exc}", file=sys.stderr)
4210
+ stages.total()
4211
+ sys.exit(0)
4212
+
4213
+ if merge_existing_graph:
4214
+ # #2169: this raw path used to write ONLY this run's extraction
4215
+ # over graph.json — on an incremental run that is just the
4216
+ # changed files, silently dropping every node/edge owned by an
4217
+ # unchanged file. Merge the existing graph forward first, with
4218
+ # the same replace/prune semantics as the clustered path's
4219
+ # build_merge: re-extracted sources replaced, deleted +
4220
+ # excluded + graph-stale sources pruned, everything else
4221
+ # carried. Survivors are prepended, so the dedupe below keeps
4222
+ # this run's fresh attributes for re-extracted nodes.
4223
+ from graphify.build import merge_raw_extraction as _merge_raw_extraction
4224
+ _raw_prune_sources: list[str] = list(deleted_files)
4225
+ for _src in list(excluded_files) + graph_stale_sources:
4226
+ if _src not in _raw_prune_sources:
4227
+ _raw_prune_sources.append(_src)
4228
+ try:
4229
+ merged = _merge_raw_extraction(
4230
+ merged,
4231
+ graph_path=existing_graph_path,
4232
+ prune_sources=_raw_prune_sources or None,
4233
+ root=target,
4234
+ )
4235
+ except RuntimeError as exc:
4236
+ # Existing graph present but unparseable: refuse to
4237
+ # raw-dump this run's partial extraction over it.
4238
+ print(f"error: {exc}", file=sys.stderr)
4239
+ sys.exit(1)
4240
+ _shrink = _handle_unverified_semantic_shrink(
4241
+ merged.get("_unverified_semantic_shrink"),
4242
+ cli_allow_partial=cli_allow_partial,
4243
+ files_by_type=files_by_type,
4244
+ sem_result=sem_result,
4245
+ target=target,
4246
+ partial_semantic_files=_partial_semantic_files,
4247
+ failed_ast_sources=_failed_ast_sources,
4248
+ semantic_files=semantic_files,
4249
+ )
4250
+ if _shrink is not None and _shrink[0]:
4251
+ _extraction_incomplete = True
4252
+ _manifest_files = _shrink[1]
4253
+ _stamped_semantic = {
4254
+ f for _flist in _manifest_files.values() for f in _flist
4255
+ }
4256
+ _cleared_semantic = _shrink[2]
4257
+ merged["nodes"] = _dedupe_nodes(merged["nodes"])
4258
+ merged["edges"] = _dedupe_edges(merged["edges"])
4259
+ # Disambiguate colliding-basename file-node labels (#2032). This raw
4260
+ # --no-cluster path bypasses build_from_json (where the clustered path
4261
+ # gets this), so apply it directly on the merged node list.
4262
+ from graphify.build import disambiguate_file_labels_in_nodes as _disamb_labels
4263
+ _disamb_labels(merged["nodes"])
4264
+ # Backfill source_file from endpoint nodes — this raw path bypasses
4265
+ # build_from_json's backfill, and semantic edges sometimes omit it (#1279).
4266
+ _node_sf = {n.get("id"): n.get("source_file") for n in merged["nodes"]}
4267
+ for _e in merged["edges"]:
4268
+ if not _e.get("source_file"):
4269
+ _e["source_file"] = (
4270
+ _node_sf.get(_e.get("source")) or _node_sf.get(_e.get("target")) or ""
4271
+ )
4272
+ # RT-parity for the raw path: an incomplete build must not force a
4273
+ # partial graph over a larger complete one here either. The clustered
4274
+ # path gets this from to_json's #479 guard; this path never calls
4275
+ # to_json, so replicate the shrink check against the existing file and
4276
+ # exit before the write/manifest unless --allow-partial is set.
4277
+ if _extraction_incomplete and not cli_allow_partial:
4278
+ from graphify.export import MALFORMED_GRAPH as _MALFORMED_GRAPH
4279
+ _existing_n = _existing_graph_node_count(graph_json_path)
4280
+ _malformed = _existing_n is _MALFORMED_GRAPH
4281
+ _shrinks = isinstance(_existing_n, int) and len(merged["nodes"]) < _existing_n
4282
+ if _malformed or _shrinks:
4283
+ _detail = (
4284
+ f"the existing {graph_json_path} is present but unparseable "
4285
+ "(corrupt or a mid-write), so a shrink cannot be ruled out"
4286
+ if _malformed
4287
+ else f"smaller than the existing {graph_json_path} "
4288
+ f"({len(merged['nodes'])} < {_existing_n} nodes)"
4289
+ )
4290
+ print(
4291
+ "[graphify extract] error: extraction was incomplete (an AST/"
4292
+ f"semantic pass failed) and the resulting --no-cluster graph is {_detail}. "
4293
+ "Refusing to overwrite a complete graph with a partial one. Re-run after "
4294
+ "fixing the failures, or pass --allow-partial to overwrite anyway.",
4295
+ file=sys.stderr,
4296
+ )
4297
+ sys.exit(1)
4298
+ _backup(graphify_out)
4299
+ _invalidate_file_manifest_for_db_graph()
4300
+ from graphify.paths import write_json_atomic as _write_json_atomic
4301
+ _write_json_atomic(graph_json_path, merged, indent=2)
4302
+ try:
4303
+ # Record the scan root so a later build_merge / update runbook can
4304
+ # relativize deleted-file paths correctly even for a custom --out
4305
+ # (its grandparent-of-graph.json fallback points at the wrong dir
4306
+ # otherwise, and deleted files never prune — #2012/#1571).
4307
+ (graphify_out / ".graphify_root").write_text(
4308
+ str(Path(target).resolve()), encoding="utf-8"
4309
+ )
4310
+ except OSError:
4311
+ pass
4312
+ stages.mark("write")
4313
+ cost = _estimate_cost(
4314
+ backend, merged["input_tokens"], merged["output_tokens"]
4315
+ )
4316
+ print(
4317
+ f"[graphify extract] wrote {graph_json_path} — "
4318
+ f"{len(merged['nodes'])} nodes, {len(merged['edges'])} edges "
4319
+ f"(no clustering)"
4320
+ )
4321
+ if merged["input_tokens"] or merged["output_tokens"]:
4322
+ print(
4323
+ f"[graphify extract] tokens: "
4324
+ f"{merged['input_tokens']:,} in / "
4325
+ f"{merged['output_tokens']:,} out, "
4326
+ f"est. cost: ${cost:.4f}"
4327
+ )
4328
+ try:
4329
+ if has_path:
4330
+ _save_manifest(_manifest_files, manifest_path=str(manifest_path), kind="both", root=target, scan_corpus=_scan_corpus, clear_semantic=_cleared_semantic, clear_ast=_cleared_ast or None)
4331
+ except Exception as exc:
4332
+ print(f"[graphify extract] warning: could not write manifest: {exc}", file=sys.stderr)
4333
+ if global_merge:
4334
+ from graphify.global_graph import global_add as _global_add
4335
+ _tag = global_repo_tag or target.name
4336
+ try:
4337
+ result = _global_add(graphify_out / "graph.json", _tag)
4338
+ if result["skipped"]:
4339
+ print(f"[graphify global] '{_tag}' unchanged since last add - skipped.")
4340
+ else:
4341
+ print(f"[graphify global] '{_tag}' merged into global graph "
4342
+ f"(+{result['nodes_added']} nodes, -{result['nodes_removed']} pruned).")
4343
+ except Exception as exc:
4344
+ print(f"[graphify global] warning: failed to merge into global graph: {exc}", file=sys.stderr)
4345
+ stages.total()
4346
+ sys.exit(0)
4347
+
4348
+ # Build graph + cluster + score + write.
4349
+ from graphify.build import (
4350
+ build as _build,
4351
+ build_from_json as _build_from_json,
4352
+ build_merge as _build_merge,
4353
+ )
4354
+ from graphify.cluster import cluster as _cluster, score_all as _score_all
4355
+ from graphify.export import to_json as _to_json
4356
+ from graphify.analyze import god_nodes as _god_nodes, surprising_connections as _surprising
4357
+ dedup_backend = backend if dedup_llm else None
4358
+ if merge_existing_graph:
4359
+ # Prune everything the current scan no longer covers: genuinely
4360
+ # deleted manifest rows, excluded-but-alive manifest rows (#1908),
4361
+ # and the graph's own stale sources — which catches files that
4362
+ # became excluded without ever being manifest-listed (#1909).
4363
+ _prune_sources: list[str] = list(deleted_files)
4364
+ for _src in list(excluded_files) + graph_stale_sources:
4365
+ if _src not in _prune_sources:
4366
+ _prune_sources.append(_src)
4367
+ try:
4368
+ G = _build_merge(
4369
+ [merged],
4370
+ graph_path=existing_graph_path,
4371
+ prune_sources=_prune_sources or None,
4372
+ dedup=not no_dedup,
4373
+ dedup_llm_backend=dedup_backend,
4374
+ root=target,
4375
+ )
4376
+ _shrink = _handle_unverified_semantic_shrink(
4377
+ G.graph.get("_unverified_semantic_shrink") if hasattr(G, "graph") else None,
4378
+ cli_allow_partial=cli_allow_partial,
4379
+ files_by_type=files_by_type,
4380
+ sem_result=sem_result,
4381
+ target=target,
4382
+ partial_semantic_files=_partial_semantic_files,
4383
+ failed_ast_sources=_failed_ast_sources,
4384
+ semantic_files=semantic_files,
4385
+ )
4386
+ if _shrink is not None and _shrink[0]:
4387
+ _extraction_incomplete = True
4388
+ _manifest_files = _shrink[1]
4389
+ _stamped_semantic = {
4390
+ f for _flist in _manifest_files.values() for f in _flist
4391
+ }
4392
+ _cleared_semantic = _shrink[2]
4393
+ except ValueError as exc:
4394
+ # --no-dedup arms build_merge's #479 shrink guard, which refuses
4395
+ # to drop nodes belonging to files this run neither re-extracted
4396
+ # nor pruned. Report the refusal instead of a traceback (#2881):
4397
+ # graph.json on disk is untouched, so the old graph is intact.
4398
+ print(f"[graphify extract] {exc}", file=sys.stderr)
4399
+ sys.exit(1)
4400
+ else:
4401
+ G = _build([merged], dedup=not no_dedup, dedup_llm_backend=dedup_backend, root=target)
4402
+ stages.mark("build")
4403
+ if G.number_of_nodes() == 0:
4404
+ print(
4405
+ "[graphify extract] graph is empty — extraction produced no nodes. "
4406
+ "Possible causes: all files skipped, binary-only corpus, or LLM "
4407
+ "returned no edges.",
4408
+ file=sys.stderr,
4409
+ )
4410
+ sys.exit(1)
4411
+
4412
+ communities = _cluster(G, resolution=cli_resolution, exclude_hubs_percentile=cli_exclude_hubs)
4413
+ stages.mark("cluster")
4414
+ cohesion = _score_all(G, communities)
4415
+ try:
4416
+ # The percentile that suppressed hubs in cluster() above suppresses
4417
+ # them in the ranking too (#3205).
4418
+ gods = _god_nodes(G, exclude_hubs_percentile=cli_exclude_hubs)
4419
+ except Exception:
4420
+ gods = []
4421
+ try:
4422
+ surprises = _surprising(G, communities)
4423
+ except Exception:
4424
+ surprises = []
4425
+ stages.mark("analyze")
4426
+
4427
+ from graphify.export import backup_if_protected as _backup
4428
+ _backup(graphify_out)
4429
+ _invalidate_file_manifest_for_db_graph()
4430
+ # force=True bypasses the #479 shrink guard entirely. A full build
4431
+ # legitimately shrinks (fuzzy dedup collapse, deleted code) so it keeps
4432
+ # force=True — EXCEPT when this run's extraction was incomplete (an
4433
+ # extractor pass crashed or some semantic chunks failed). Then a partial
4434
+ # graph could silently overwrite a good complete one, so fall back to the
4435
+ # shrink guard (force=False) unless the user opts in with --allow-partial.
4436
+ #
4437
+ # Both write paths are guarded: the clustered path here via to_json's
4438
+ # #479 check, and the `--no-cluster` raw-dump path above via the same
4439
+ # shrink check against the existing file (existing_graph_node_count).
4440
+ #
4441
+ # Trade-off: this reuses to_json's coarse node-count guard, not the
4442
+ # source-aware _check_shrink that watch/update use. On an incremental run
4443
+ # a legitimate deletion that coincides with an unrelated transient chunk
4444
+ # failure can therefore be refused here — recoverable by re-running or
4445
+ # passing --allow-partial (the good graph is preserved and the manifest
4446
+ # is not stamped, so the retry re-extracts).
4447
+ _force_write = cli_allow_partial or not _extraction_incomplete
4448
+ # Stamp provenance from the ANALYSED repo, not the shell's cwd: without
4449
+ # this, to_json's fallback asks `git rev-parse HEAD` in whatever repo the
4450
+ # command was invoked from, so `graphify extract <target>` run from
4451
+ # another repo's root stamped the invoker's commit into the target's
4452
+ # graph.json — and cluster then propagates that stamp into
4453
+ # GRAPH_REPORT.md (#2534 keeps the extract-time stamp by design). Same
4454
+ # cwd-anchoring mistake #2316 fixed for watch/update, surviving in the
4455
+ # extract path.
4456
+ from graphify.watch import _git_head as _gh_target
4457
+ _wrote = _to_json(G, communities, str(graph_json_path), force=_force_write,
4458
+ built_at_commit=_gh_target(cwd=Path(target).resolve()))
4459
+ if not _wrote:
4460
+ # The shrink guard refused: this partial build is smaller than the
4461
+ # existing graph. Exit before writing the manifest/marker below, which
4462
+ # would otherwise stamp these files as done and make the next
4463
+ # incremental run skip re-extracting them (poisoning the manifest
4464
+ # against the graph we declined to write). Exit non-zero so a retry
4465
+ # re-attempts.
4466
+ print(
4467
+ "[graphify extract] error: extraction was incomplete (an AST/semantic "
4468
+ f"pass failed) and the resulting graph is smaller than the existing "
4469
+ f"{graph_json_path}. Refusing to overwrite a complete graph with a "
4470
+ "partial one. Re-run after fixing the failures, or pass --allow-partial "
4471
+ "to overwrite anyway.",
4472
+ file=sys.stderr,
4473
+ )
4474
+ sys.exit(1)
4475
+ try:
4476
+ # See the --no-cluster path above: persist the scan root so build_merge
4477
+ # can relativize deleted-file paths under a custom --out (#2012/#1571).
4478
+ (graphify_out / ".graphify_root").write_text(
4479
+ str(Path(target).resolve()), encoding="utf-8"
4480
+ )
4481
+ except OSError:
4482
+ pass
4483
+ stages.mark("export")
4484
+ if merged.get("output_tokens", 0) > 0:
4485
+ (graphify_out / ".graphify_semantic_marker").write_text(
4486
+ json.dumps({"output_tokens": merged["output_tokens"]}), encoding="utf-8"
4487
+ )
4488
+ if global_merge:
4489
+ from graphify.global_graph import global_add as _global_add
4490
+ _tag = global_repo_tag or target.name
4491
+ try:
4492
+ result = _global_add(graphify_out / "graph.json", _tag)
4493
+ if result["skipped"]:
4494
+ print(f"[graphify global] '{_tag}' unchanged since last add - skipped.")
4495
+ else:
4496
+ print(f"[graphify global] '{_tag}' merged into global graph "
4497
+ f"(+{result['nodes_added']} nodes, -{result['nodes_removed']} pruned).")
4498
+ except Exception as exc:
4499
+ print(f"[graphify global] warning: failed to merge into global graph: {exc}", file=sys.stderr)
4500
+ analysis = {
4501
+ "communities": {str(k): v for k, v in communities.items()},
4502
+ "cohesion": {str(k): v for k, v in cohesion.items()},
4503
+ "gods": gods,
4504
+ "surprises": surprises,
4505
+ "tokens": {
4506
+ "input": merged["input_tokens"],
4507
+ "output": merged["output_tokens"],
4508
+ },
4509
+ }
4510
+ from graphify.paths import write_json_atomic as _wja
4511
+ _wja(analysis_path, analysis, indent=2)
4512
+ try:
4513
+ if has_path:
4514
+ _save_manifest(_manifest_files, manifest_path=str(manifest_path), kind="both", root=target, scan_corpus=_scan_corpus, clear_semantic=_cleared_semantic, clear_ast=_cleared_ast or None)
4515
+ except Exception as exc:
4516
+ print(f"[graphify extract] warning: could not write manifest: {exc}", file=sys.stderr)
4517
+
4518
+ cost = _estimate_cost(backend, merged["input_tokens"], merged["output_tokens"])
4519
+ print(
4520
+ f"[graphify extract] wrote {graph_json_path}: "
4521
+ f"{G.number_of_nodes()} nodes, {G.number_of_edges()} edges, "
4522
+ f"{len(communities)} communities"
4523
+ )
4524
+ print(f"[graphify extract] wrote {analysis_path}")
4525
+ if incremental_mode:
4526
+ _excl_note = f", {len(excluded_files)} excluded" if excluded_files else ""
4527
+ print(
4528
+ f"[graphify extract] incremental summary: "
4529
+ f"{sem_cache_hits + unchanged_total} files cached/unchanged, "
4530
+ f"{len(code_files) + sem_cache_misses} re-extracted, "
4531
+ f"{len(deleted_files)} deleted{_excl_note}"
4532
+ )
4533
+ elif sem_cache_hits:
4534
+ print(f"[graphify extract] semantic cache: {sem_cache_hits} cached, {sem_cache_misses} re-extracted")
4535
+ if merged["input_tokens"] or merged["output_tokens"]:
4536
+ print(
4537
+ f"[graphify extract] tokens: "
4538
+ f"{merged['input_tokens']:,} in / "
4539
+ f"{merged['output_tokens']:,} out, "
4540
+ f"est. cost (~{backend}): ${cost:.4f}"
4541
+ )
4542
+ # extract intentionally stops at graph.json + analysis; the report and
4543
+ # community labels are produced by `cluster-only` (or an agent's Step 5).
4544
+ # Point standalone users at it so communities get named (#1097).
4545
+ print(
4546
+ "[graphify extract] next: run "
4547
+ f"`graphify cluster-only {graphify_out.parent}` "
4548
+ "to generate GRAPH_REPORT.md and name communities"
4549
+ )
4550
+ stages.total()
4551
+
4552
+ elif cmd == "cache-check":
4553
+ # graphify cache-check <files_from> [--root <dir>] [--mode <m> | --deep]
4554
+ # [--prompt-file <path>]
4555
+ # Reads file paths (one per line) from <files_from>, checks semantic cache.
4556
+ # --mode deep (or --deep) checks the cache/semantic-deep/ namespace
4557
+ # written by `extract --mode deep` instead of cache/semantic/ (#1894).
4558
+ # --prompt-file names the extraction prompt the caller will use (an agent's
4559
+ # references/extraction-spec.md), restricting hits to entries produced by
4560
+ # that same prompt (#1939). Omitting it reads the unattributed layout, which
4561
+ # cannot see entries a fingerprinted run wrote.
4562
+ # Writes:
4563
+ # graphify-out/.graphify_cached.json — already-cached nodes/edges/hyperedges
4564
+ # graphify-out/.graphify_uncached.txt — paths that need extraction
4565
+ # Stdout: "Cache: N hit, M miss"
4566
+ from graphify.cache import check_semantic_cache
4567
+ if len(sys.argv) < 3:
4568
+ print("Usage: graphify cache-check <files_from> [--root <dir>] "
4569
+ "[--mode <m> | --deep] [--prompt-file <path>]", file=sys.stderr)
4570
+ sys.exit(1)
4571
+ files_from = Path(sys.argv[2])
4572
+ root = Path(".")
4573
+ cache_mode: str | None = None
4574
+ prompt_file: str | None = None
4575
+ i = 3
4576
+ while i < len(sys.argv):
4577
+ if sys.argv[i] == "--root" and i + 1 < len(sys.argv):
4578
+ root = Path(sys.argv[i + 1])
4579
+ i += 2
4580
+ elif sys.argv[i] == "--mode" and i + 1 < len(sys.argv):
4581
+ cache_mode = sys.argv[i + 1]
4582
+ i += 2
4583
+ elif sys.argv[i].startswith("--mode="):
4584
+ cache_mode = sys.argv[i].split("=", 1)[1]
4585
+ i += 1
4586
+ elif sys.argv[i] == "--deep":
4587
+ cache_mode = "deep"
4588
+ i += 1
4589
+ elif sys.argv[i] == "--prompt-file" and i + 1 < len(sys.argv):
4590
+ prompt_file = sys.argv[i + 1]
4591
+ i += 2
4592
+ elif sys.argv[i].startswith("--prompt-file="):
4593
+ prompt_file = sys.argv[i].split("=", 1)[1]
4594
+ i += 1
4595
+ else:
4596
+ i += 1
4597
+ files = [f for f in files_from.read_text(encoding="utf-8").splitlines() if f.strip()]
4598
+ cached_nodes, cached_edges, cached_hyperedges, uncached = check_semantic_cache(
4599
+ files, root, mode=cache_mode, prompt_file=prompt_file
4600
+ )
4601
+ out = root / _GRAPHIFY_OUT
4602
+ out.mkdir(parents=True, exist_ok=True)
4603
+ if cached_nodes or cached_edges or cached_hyperedges:
4604
+ (out / ".graphify_cached.json").write_text(
4605
+ json.dumps({"nodes": cached_nodes, "edges": cached_edges, "hyperedges": cached_hyperedges},
4606
+ ensure_ascii=False),
4607
+ encoding="utf-8",
4608
+ )
4609
+ (out / ".graphify_uncached.txt").write_text("\n".join(uncached), encoding="utf-8")
4610
+ print(f"Cache: {len(files) - len(uncached)} hit, {len(uncached)} miss")
4611
+
4612
+ elif cmd == "merge-chunks":
4613
+ # graphify merge-chunks <chunk_glob_or_files...> --out <path>
4614
+ # Concatenates .graphify_chunk_*.json files written by semantic subagents.
4615
+ # Deduplicates nodes by id (first writer wins). Sums token counts.
4616
+ import glob as _glob
4617
+ if len(sys.argv) < 3:
4618
+ print("Usage: graphify merge-chunks <chunk_files...> --out <path>", file=sys.stderr)
4619
+ sys.exit(1)
4620
+ out_path: Path | None = None
4621
+ chunk_args: list[str] = []
4622
+ i = 2
4623
+ while i < len(sys.argv):
4624
+ if sys.argv[i] == "--out" and i + 1 < len(sys.argv):
4625
+ out_path = Path(sys.argv[i + 1])
4626
+ i += 2
4627
+ else:
4628
+ chunk_args.append(sys.argv[i])
4629
+ i += 1
4630
+ if not out_path:
4631
+ print("error: --out <path> required", file=sys.stderr)
4632
+ sys.exit(1)
4633
+ chunk_files: list[str] = []
4634
+ for arg in chunk_args:
4635
+ expanded = _glob.glob(arg)
4636
+ chunk_files.extend(sorted(expanded) if expanded else [arg])
4637
+ merged: dict = {"nodes": [], "edges": [], "hyperedges": [], "input_tokens": 0, "output_tokens": 0}
4638
+ seen_ids: set[str] = set()
4639
+ valid_chunks = 0
4640
+ # These chunk files are untrusted subagent output. load_validated_...
4641
+ # stats the file size BEFORE reading it (so a multi-GB chunk can't blow up
4642
+ # memory), parses the JSON, and validates the security caps + the node/
4643
+ # edge id charset that blocks path traversal (#825) — the same enforcement
4644
+ # the skill merge path applies. A bad chunk is skipped with a warning
4645
+ # while valid siblings still merge; if every chunk is invalid, fail
4646
+ # closed instead of reporting success and replacing --out with an empty
4647
+ # semantic layer. Deliberately NOT wired into
4648
+ # build_from_json/load_graph_json, which must keep loading valid
4649
+ # pre-existing graphs. file_type is left to build's coercion (#840).
4650
+ from graphify.semantic_cleanup import load_validated_semantic_fragment
4651
+ for cf in chunk_files:
4652
+ chunk, _chunk_errs = load_validated_semantic_fragment(Path(cf))
4653
+ if _chunk_errs:
4654
+ print(
4655
+ f"[graphify merge-chunks] warning: skipping invalid chunk {cf}: "
4656
+ f"{'; '.join(_chunk_errs[:3])}",
4657
+ file=sys.stderr,
4658
+ )
4659
+ continue
4660
+ valid_chunks += 1
4661
+ for n in chunk.get("nodes", []):
4662
+ if n.get("id") not in seen_ids:
4663
+ seen_ids.add(n["id"])
4664
+ merged["nodes"].append(n)
4665
+ merged["edges"].extend(chunk.get("edges", []))
4666
+ merged["hyperedges"].extend(chunk.get("hyperedges", []))
4667
+ # Coerce token counts: a chunk is untrusted, so a non-numeric
4668
+ # input_tokens/output_tokens must not abort the whole merge with a
4669
+ # TypeError after other chunks already merged.
4670
+ for _tok in ("input_tokens", "output_tokens"):
4671
+ _v = chunk.get(_tok, 0)
4672
+ merged[_tok] += _v if isinstance(_v, (int, float)) else 0
4673
+ if not valid_chunks:
4674
+ print(
4675
+ f"[graphify merge-chunks] error: no valid chunks to merge; "
4676
+ f"refusing to write {out_path}",
4677
+ file=sys.stderr,
4678
+ )
4679
+ sys.exit(1)
4680
+ out_path.parent.mkdir(parents=True, exist_ok=True)
4681
+ from graphify.paths import write_json_atomic as _wja
4682
+ _wja(out_path, merged, ensure_ascii=False)
4683
+ chunk_summary = (
4684
+ f"{valid_chunks} chunks"
4685
+ if valid_chunks == len(chunk_files)
4686
+ else f"{valid_chunks} of {len(chunk_files)} chunks"
4687
+ )
4688
+ print(
4689
+ f"Merged {chunk_summary}: {len(merged['nodes'])} nodes, {len(merged['edges'])} edges, "
4690
+ f"{merged['input_tokens']:,} in / {merged['output_tokens']:,} out tokens"
4691
+ )
4692
+
4693
+ elif cmd == "merge-semantic":
4694
+ # graphify merge-semantic --cached <path> --new <path> --out <path>
4695
+ # Merges cached semantic results with freshly-extracted chunk results.
4696
+ # Deduplicates nodes by id (cached entries take priority over new ones).
4697
+ if len(sys.argv) < 3:
4698
+ print("Usage: graphify merge-semantic --cached <path> --new <path> --out <path>", file=sys.stderr)
4699
+ sys.exit(1)
4700
+ cached_path: Path | None = None
4701
+ new_path: Path | None = None
4702
+ out_path2: Path | None = None
4703
+ i = 2
4704
+ while i < len(sys.argv):
4705
+ if sys.argv[i] == "--cached" and i + 1 < len(sys.argv):
4706
+ cached_path = Path(sys.argv[i + 1]); i += 2
4707
+ elif sys.argv[i] == "--new" and i + 1 < len(sys.argv):
4708
+ new_path = Path(sys.argv[i + 1]); i += 2
4709
+ elif sys.argv[i] == "--out" and i + 1 < len(sys.argv):
4710
+ out_path2 = Path(sys.argv[i + 1]); i += 2
4711
+ else:
4712
+ i += 1
4713
+ if not out_path2:
4714
+ print("error: --out <path> required", file=sys.stderr)
4715
+ sys.exit(1)
4716
+ empty: dict = {"nodes": [], "edges": [], "hyperedges": []}
4717
+ cached_data = json.loads(cached_path.read_text(encoding="utf-8")) if cached_path and cached_path.exists() else empty
4718
+ new_data = json.loads(new_path.read_text(encoding="utf-8")) if new_path and new_path.exists() else empty
4719
+ seen_ids2: set[str] = set()
4720
+ all_nodes: list[dict] = []
4721
+ for n in cached_data.get("nodes", []) + new_data.get("nodes", []):
4722
+ if n.get("id") not in seen_ids2:
4723
+ seen_ids2.add(n["id"])
4724
+ all_nodes.append(n)
4725
+ merged2 = {
4726
+ "nodes": all_nodes,
4727
+ "edges": cached_data.get("edges", []) + new_data.get("edges", []),
4728
+ "hyperedges": cached_data.get("hyperedges", []) + new_data.get("hyperedges", []),
4729
+ }
4730
+ out_path2.parent.mkdir(parents=True, exist_ok=True)
4731
+ from graphify.paths import write_json_atomic as _wja
4732
+ _wja(out_path2, merged2, ensure_ascii=False)
4733
+ print(f"Merged: {len(merged2['nodes'])} nodes, {len(merged2['edges'])} edges")
4734
+
4735
+ elif Path(cmd).exists() or cmd in (".", "..") or cmd.startswith(("./", "../", "/", "~")):
4736
+ # User ran `graphify <path>` directly — treat as `graphify extract <path>`.
4737
+ # Common when following the PowerShell note in README (`graphify .`) or
4738
+ # copy-pasting skill invocations without the leading slash.
4739
+ sys.argv.insert(2, sys.argv[1])
4740
+ sys.argv[1] = "extract"
4741
+ _reenter_main()
4742
+ else:
4743
+ print(f"error: unknown command '{cmd}'", file=sys.stderr)
4744
+ print("Run 'graphify --help' for usage.", file=sys.stderr)
4745
+ sys.exit(1)