graphitect 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (336) hide show
  1. graphify/__init__.py +30 -0
  2. graphify/__main__.py +757 -0
  3. graphify/_minhash.py +107 -0
  4. graphify/affected.py +318 -0
  5. graphify/always_on/agents-md.md +12 -0
  6. graphify/always_on/antigravity-rules.md +14 -0
  7. graphify/always_on/claude-md.md +9 -0
  8. graphify/always_on/gemini-md.md +9 -0
  9. graphify/always_on/kiro-steering.md +5 -0
  10. graphify/always_on/vscode-instructions.md +17 -0
  11. graphify/analyze.py +769 -0
  12. graphify/benchmark.py +152 -0
  13. graphify/build.py +2300 -0
  14. graphify/cache.py +1746 -0
  15. graphify/callflow_html.py +2051 -0
  16. graphify/cargo_introspect.py +109 -0
  17. graphify/cli.py +4745 -0
  18. graphify/cluster.py +409 -0
  19. graphify/command-kilo.md +15 -0
  20. graphify/cross_repo_calls.py +216 -0
  21. graphify/cross_repo_types.py +75 -0
  22. graphify/csharp_dispatch.py +154 -0
  23. graphify/dedup.py +1213 -0
  24. graphify/detect.py +2566 -0
  25. graphify/diagnostics.py +406 -0
  26. graphify/export.py +1349 -0
  27. graphify/exporters/__init__.py +1 -0
  28. graphify/exporters/base.py +14 -0
  29. graphify/exporters/graphdb.py +173 -0
  30. graphify/exporters/html.py +637 -0
  31. graphify/extract.py +7856 -0
  32. graphify/extractors/MIGRATION.md +107 -0
  33. graphify/extractors/__init__.py +66 -0
  34. graphify/extractors/apex.py +215 -0
  35. graphify/extractors/base.py +85 -0
  36. graphify/extractors/bash.py +579 -0
  37. graphify/extractors/blade.py +53 -0
  38. graphify/extractors/commonlisp.py +540 -0
  39. graphify/extractors/csharp.py +448 -0
  40. graphify/extractors/dart.py +564 -0
  41. graphify/extractors/dm.py +494 -0
  42. graphify/extractors/elixir.py +241 -0
  43. graphify/extractors/engine.py +6509 -0
  44. graphify/extractors/fortran.py +311 -0
  45. graphify/extractors/go.py +527 -0
  46. graphify/extractors/json_config.py +240 -0
  47. graphify/extractors/julia.py +289 -0
  48. graphify/extractors/markdown.py +408 -0
  49. graphify/extractors/models.py +131 -0
  50. graphify/extractors/objc.py +566 -0
  51. graphify/extractors/ocaml.py +289 -0
  52. graphify/extractors/pascal.py +688 -0
  53. graphify/extractors/pascal_forms.py +196 -0
  54. graphify/extractors/powershell.py +522 -0
  55. graphify/extractors/razor.py +192 -0
  56. graphify/extractors/resolution.py +3584 -0
  57. graphify/extractors/robot.py +296 -0
  58. graphify/extractors/rust.py +470 -0
  59. graphify/extractors/sln.py +92 -0
  60. graphify/extractors/sql.py +720 -0
  61. graphify/extractors/terraform.py +181 -0
  62. graphify/extractors/verilog.py +329 -0
  63. graphify/extractors/zig.py +181 -0
  64. graphify/file_slice.py +246 -0
  65. graphify/global_graph.py +194 -0
  66. graphify/google_workspace.py +237 -0
  67. graphify/hooks.py +933 -0
  68. graphify/ids.py +93 -0
  69. graphify/ingest.py +358 -0
  70. graphify/install.py +2366 -0
  71. graphify/llm.py +3544 -0
  72. graphify/manifest.py +4 -0
  73. graphify/manifest_ingest.py +311 -0
  74. graphify/mcp_ingest.py +386 -0
  75. graphify/multigraph_compat.py +212 -0
  76. graphify/pascal_resolution.py +129 -0
  77. graphify/paths.py +436 -0
  78. graphify/pg_introspect.py +165 -0
  79. graphify/prs.py +770 -0
  80. graphify/querylog.py +80 -0
  81. graphify/reflect.py +882 -0
  82. graphify/report.py +346 -0
  83. graphify/resolver_registry.py +85 -0
  84. graphify/ruby_resolution.py +242 -0
  85. graphify/scip_ingest.py +363 -0
  86. graphify/security.py +460 -0
  87. graphify/semantic_cleanup.py +336 -0
  88. graphify/serve.py +2608 -0
  89. graphify/skill-agents.md +710 -0
  90. graphify/skill-aider.md +1283 -0
  91. graphify/skill-amp.md +710 -0
  92. graphify/skill-claw.md +713 -0
  93. graphify/skill-codex.md +710 -0
  94. graphify/skill-copilot.md +713 -0
  95. graphify/skill-devin.md +1410 -0
  96. graphify/skill-droid.md +710 -0
  97. graphify/skill-kilo.md +722 -0
  98. graphify/skill-kiro.md +713 -0
  99. graphify/skill-opencode.md +705 -0
  100. graphify/skill-pi.md +713 -0
  101. graphify/skill-trae.md +711 -0
  102. graphify/skill-vscode.md +709 -0
  103. graphify/skill-windows.md +755 -0
  104. graphify/skill.md +713 -0
  105. graphify/skills/agents/references/add-watch.md +56 -0
  106. graphify/skills/agents/references/exports.md +87 -0
  107. graphify/skills/agents/references/extraction-spec.md +70 -0
  108. graphify/skills/agents/references/github-and-merge.md +46 -0
  109. graphify/skills/agents/references/hooks.md +33 -0
  110. graphify/skills/agents/references/query.md +311 -0
  111. graphify/skills/agents/references/transcribe.md +52 -0
  112. graphify/skills/agents/references/update.md +210 -0
  113. graphify/skills/amp/references/add-watch.md +56 -0
  114. graphify/skills/amp/references/exports.md +87 -0
  115. graphify/skills/amp/references/extraction-spec.md +70 -0
  116. graphify/skills/amp/references/github-and-merge.md +46 -0
  117. graphify/skills/amp/references/hooks.md +33 -0
  118. graphify/skills/amp/references/query.md +311 -0
  119. graphify/skills/amp/references/transcribe.md +52 -0
  120. graphify/skills/amp/references/update.md +210 -0
  121. graphify/skills/claude/references/add-watch.md +56 -0
  122. graphify/skills/claude/references/exports.md +87 -0
  123. graphify/skills/claude/references/extraction-spec.md +70 -0
  124. graphify/skills/claude/references/github-and-merge.md +46 -0
  125. graphify/skills/claude/references/hooks.md +33 -0
  126. graphify/skills/claude/references/query.md +311 -0
  127. graphify/skills/claude/references/transcribe.md +52 -0
  128. graphify/skills/claude/references/update.md +210 -0
  129. graphify/skills/claw/references/add-watch.md +56 -0
  130. graphify/skills/claw/references/exports.md +87 -0
  131. graphify/skills/claw/references/extraction-spec.md +31 -0
  132. graphify/skills/claw/references/github-and-merge.md +46 -0
  133. graphify/skills/claw/references/hooks.md +33 -0
  134. graphify/skills/claw/references/query.md +311 -0
  135. graphify/skills/claw/references/transcribe.md +52 -0
  136. graphify/skills/claw/references/update.md +210 -0
  137. graphify/skills/codex/references/add-watch.md +56 -0
  138. graphify/skills/codex/references/exports.md +87 -0
  139. graphify/skills/codex/references/extraction-spec.md +31 -0
  140. graphify/skills/codex/references/github-and-merge.md +46 -0
  141. graphify/skills/codex/references/hooks.md +33 -0
  142. graphify/skills/codex/references/query.md +311 -0
  143. graphify/skills/codex/references/transcribe.md +52 -0
  144. graphify/skills/codex/references/update.md +210 -0
  145. graphify/skills/copilot/references/add-watch.md +56 -0
  146. graphify/skills/copilot/references/exports.md +87 -0
  147. graphify/skills/copilot/references/extraction-spec.md +70 -0
  148. graphify/skills/copilot/references/github-and-merge.md +46 -0
  149. graphify/skills/copilot/references/hooks.md +33 -0
  150. graphify/skills/copilot/references/query.md +311 -0
  151. graphify/skills/copilot/references/transcribe.md +52 -0
  152. graphify/skills/copilot/references/update.md +210 -0
  153. graphify/skills/droid/references/add-watch.md +56 -0
  154. graphify/skills/droid/references/exports.md +87 -0
  155. graphify/skills/droid/references/extraction-spec.md +70 -0
  156. graphify/skills/droid/references/github-and-merge.md +46 -0
  157. graphify/skills/droid/references/hooks.md +33 -0
  158. graphify/skills/droid/references/query.md +311 -0
  159. graphify/skills/droid/references/transcribe.md +52 -0
  160. graphify/skills/droid/references/update.md +210 -0
  161. graphify/skills/kilo/references/add-watch.md +56 -0
  162. graphify/skills/kilo/references/exports.md +87 -0
  163. graphify/skills/kilo/references/extraction-spec.md +70 -0
  164. graphify/skills/kilo/references/github-and-merge.md +46 -0
  165. graphify/skills/kilo/references/hooks.md +33 -0
  166. graphify/skills/kilo/references/query.md +311 -0
  167. graphify/skills/kilo/references/transcribe.md +52 -0
  168. graphify/skills/kilo/references/update.md +210 -0
  169. graphify/skills/kiro/references/add-watch.md +56 -0
  170. graphify/skills/kiro/references/exports.md +87 -0
  171. graphify/skills/kiro/references/extraction-spec.md +31 -0
  172. graphify/skills/kiro/references/github-and-merge.md +46 -0
  173. graphify/skills/kiro/references/hooks.md +33 -0
  174. graphify/skills/kiro/references/query.md +311 -0
  175. graphify/skills/kiro/references/transcribe.md +52 -0
  176. graphify/skills/kiro/references/update.md +210 -0
  177. graphify/skills/opencode/references/add-watch.md +56 -0
  178. graphify/skills/opencode/references/exports.md +87 -0
  179. graphify/skills/opencode/references/extraction-spec.md +70 -0
  180. graphify/skills/opencode/references/github-and-merge.md +46 -0
  181. graphify/skills/opencode/references/hooks.md +33 -0
  182. graphify/skills/opencode/references/query.md +311 -0
  183. graphify/skills/opencode/references/transcribe.md +52 -0
  184. graphify/skills/opencode/references/update.md +210 -0
  185. graphify/skills/pi/references/add-watch.md +56 -0
  186. graphify/skills/pi/references/exports.md +87 -0
  187. graphify/skills/pi/references/extraction-spec.md +31 -0
  188. graphify/skills/pi/references/github-and-merge.md +46 -0
  189. graphify/skills/pi/references/hooks.md +33 -0
  190. graphify/skills/pi/references/query.md +311 -0
  191. graphify/skills/pi/references/transcribe.md +52 -0
  192. graphify/skills/pi/references/update.md +210 -0
  193. graphify/skills/trae/references/add-watch.md +56 -0
  194. graphify/skills/trae/references/exports.md +87 -0
  195. graphify/skills/trae/references/extraction-spec.md +70 -0
  196. graphify/skills/trae/references/github-and-merge.md +46 -0
  197. graphify/skills/trae/references/hooks.md +35 -0
  198. graphify/skills/trae/references/query.md +311 -0
  199. graphify/skills/trae/references/transcribe.md +52 -0
  200. graphify/skills/trae/references/update.md +210 -0
  201. graphify/skills/vscode/references/add-watch.md +56 -0
  202. graphify/skills/vscode/references/exports.md +87 -0
  203. graphify/skills/vscode/references/extraction-spec.md +70 -0
  204. graphify/skills/vscode/references/github-and-merge.md +46 -0
  205. graphify/skills/vscode/references/hooks.md +33 -0
  206. graphify/skills/vscode/references/query.md +311 -0
  207. graphify/skills/vscode/references/transcribe.md +52 -0
  208. graphify/skills/vscode/references/update.md +210 -0
  209. graphify/skills/windows/references/add-watch.md +56 -0
  210. graphify/skills/windows/references/exports.md +87 -0
  211. graphify/skills/windows/references/extraction-spec.md +70 -0
  212. graphify/skills/windows/references/github-and-merge.md +46 -0
  213. graphify/skills/windows/references/hooks.md +33 -0
  214. graphify/skills/windows/references/query.md +311 -0
  215. graphify/skills/windows/references/transcribe.md +52 -0
  216. graphify/skills/windows/references/update.md +210 -0
  217. graphify/symbol_resolution.py +556 -0
  218. graphify/transcribe.py +186 -0
  219. graphify/tree_html.py +603 -0
  220. graphify/validate.py +95 -0
  221. graphify/watch.py +2280 -0
  222. graphify/wiki.py +405 -0
  223. graphitect/__init__.py +28 -0
  224. graphitect/__main__.py +4 -0
  225. graphitect/_vendor/__init__.py +2 -0
  226. graphitect/_vendor/archify/LICENSE +22 -0
  227. graphitect/_vendor/archify/SKILL.md +137 -0
  228. graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
  229. graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
  230. graphitect/_vendor/archify/assets/template.html +14935 -0
  231. graphitect/_vendor/archify/bin/archify.mjs +2091 -0
  232. graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
  233. graphitect/_vendor/archify/bin/preview.mjs +653 -0
  234. graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
  235. graphitect/_vendor/archify/brand-marks/README.md +31 -0
  236. graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
  237. graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
  238. graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
  239. graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
  240. graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
  241. graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
  242. graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
  243. graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
  244. graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
  245. graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
  246. graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
  247. graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
  248. graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
  249. graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
  250. graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
  251. graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
  252. graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
  253. graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
  254. graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
  255. graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
  256. graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
  257. graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
  258. graphitect/_vendor/archify/package-lock.json +149 -0
  259. graphitect/_vendor/archify/package.json +39 -0
  260. graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
  261. graphitect/_vendor/archify/references/authoring-contract.md +243 -0
  262. graphitect/_vendor/archify/references/brand-marks.md +65 -0
  263. graphitect/_vendor/archify/references/delivery-contract.md +120 -0
  264. graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
  265. graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
  266. graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
  267. graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
  268. graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
  269. graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
  270. graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
  271. graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
  272. graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
  273. graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
  274. graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
  275. graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
  276. graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
  277. graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
  278. graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
  279. graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
  280. graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
  281. graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
  282. graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
  283. graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
  284. graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
  285. graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
  286. graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
  287. graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
  288. graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
  289. graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
  290. graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
  291. graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
  292. graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
  293. graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
  294. graphitect/_vendor/archify/schemas/README.md +211 -0
  295. graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
  296. graphitect/_vendor/archify/schemas/common.schema.json +115 -0
  297. graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
  298. graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
  299. graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
  300. graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
  301. graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
  302. graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
  303. graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
  304. graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
  305. graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
  306. graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
  307. graphitect/_vendor/archify/skill-release.json +10 -0
  308. graphitect/cli.py +981 -0
  309. graphitect/deliver/__init__.py +5 -0
  310. graphitect/deliver/archify_adapter.py +1877 -0
  311. graphitect/deliver/archify_ir.py +160 -0
  312. graphitect/deliver/archify_repair.py +135 -0
  313. graphitect/deliver/doc_compiler.py +916 -0
  314. graphitect/ground/__init__.py +5 -0
  315. graphitect/ground/describe_source.py +27 -0
  316. graphitect/ground/fullread_source.py +56 -0
  317. graphitect/ground/graphify_source.py +107 -0
  318. graphitect/models.py +118 -0
  319. graphitect/skill/SKILL.md +80 -0
  320. graphitect/skill/agents/openai.yaml +4 -0
  321. graphitect/synthesize/__init__.py +5 -0
  322. graphitect/synthesize/engine.py +281 -0
  323. graphitect/synthesize/llm_backend.py +331 -0
  324. graphitect/synthesize/questions.py +139 -0
  325. graphitect/synthesize/rubric.py +104 -0
  326. graphitect-0.2.0.dist-info/METADATA +284 -0
  327. graphitect-0.2.0.dist-info/RECORD +336 -0
  328. graphitect-0.2.0.dist-info/WHEEL +5 -0
  329. graphitect-0.2.0.dist-info/entry_points.txt +2 -0
  330. graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
  331. graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
  332. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
  333. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
  334. graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
  335. graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
  336. graphitect-0.2.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,311 @@
1
+ # graphify reference: query, path, explain
2
+
3
+ Load this when the user asks a question against an existing graph, or runs `/graphify path` or `/graphify explain`. The core's query stub points here for the full traversal flow. These flows use the `graphify query` CLI when it is available and fall back to an inline NetworkX traversal otherwise.
4
+
5
+ Two traversal modes - choose based on the question:
6
+
7
+ | Mode | Flag | Best for |
8
+ |------|------|----------|
9
+ | BFS (default) | _(none)_ | "What is X connected to?" - broad context, nearest neighbors first |
10
+ | DFS | `--dfs` | "How does X reach Y?" - trace a specific chain or dependency path |
11
+
12
+ First check the graph exists:
13
+ ```bash
14
+ $(cat graphify-out/.graphify_python) -c "
15
+ from pathlib import Path
16
+ if not Path('graphify-out/graph.json').exists():
17
+ print('ERROR: No graph found. Run /graphify <path> first to build the graph.')
18
+ raise SystemExit(1)
19
+ "
20
+ ```
21
+ If it fails, stop and tell the user to run `/graphify <path>` first.
22
+
23
+ ### Step 0 — Constrained query expansion (REQUIRED before traversal)
24
+
25
+ graphify's `query` CLI matches nodes via case-folded substring + IDF — there is **no stemming, no synonyms, no cross-language match** inside the binary, and the inline fallback below matches the same way. If the user's question uses different language or different domain vocabulary than the graph's labels (user says "обработчик" / graph says "handler"; user says "authentication" / graph says "Guardian"), the literal matcher returns 0 hits and the answer collapses to noise.
26
+
27
+ Fix this **without inventing tokens** by expanding the query against the actual graph vocabulary first:
28
+
29
+ 1. Extract the token vocabulary from node labels:
30
+ ```bash
31
+ $(cat graphify-out/.graphify_python) -c "
32
+ import json, re
33
+ from pathlib import Path
34
+ data = json.loads(Path('graphify-out/graph.json').read_text(encoding='utf-8'))
35
+ vocab = set()
36
+ for n in data['nodes']:
37
+ for c in re.findall(r'[^\W\d_]+', n.get('label','') or '', re.UNICODE):
38
+ parts = re.findall(r'[A-Z]+(?=[A-Z][a-z])|[A-Z]?[a-z]+|[A-Z]+', c) or [c]
39
+ for p in parts:
40
+ t = p.lower()
41
+ if 3 <= len(t) <= 30:
42
+ vocab.add(t)
43
+ Path('graphify-out/.vocab.txt').write_text('\n'.join(sorted(vocab)), encoding='utf-8')
44
+ print(f'vocab: {len(vocab)} tokens')
45
+ "
46
+ ```
47
+
48
+ 2. Read `graphify-out/.vocab.txt`. Then for the user's question, select **up to 12 tokens from this exact list** that semantically match the query intent. Hard constraints:
49
+ - You MUST pick only tokens present in the vocabulary file. Do NOT invent tokens.
50
+ - If a query concept has no plausible token in the vocab, skip it — do not substitute a near-synonym from training memory.
51
+ - If **no** vocab tokens match the query at all, output an empty list and tell the user the corpus has no relevant vocabulary for this question. Do not fabricate a search.
52
+ - Translate cross-language: Russian "аутентификация" → look for `auth`, `credential`, `token`, `security` IFF present in vocab.
53
+ - Morphology: "handlers" maps to `handler` IFF present; "todos" maps to `todo` IFF present.
54
+
55
+ 3. Print the selection explicitly to the user before running the query, so the expansion is auditable:
56
+ ```
57
+ Query expanded to (from graph vocab, N tokens): [token1, token2, ...]
58
+ ```
59
+ If the list is empty, say so plainly and stop — do not proceed to traversal.
60
+
61
+ ### Step 1 — Traversal
62
+
63
+ Build the **expanded query string** by joining the selected tokens with spaces. Use this string as `QUESTION` below — NOT the original user question. (The original question is preserved only for `save-result` at the end.)
64
+
65
+ Prefer the CLI when it is installed:
66
+ ```bash
67
+ graphify query "QUESTION"
68
+ # or: graphify query "QUESTION" --dfs --budget 3000
69
+ ```
70
+
71
+ If the CLI is unavailable, load `graphify-out/graph.json` and run the traversal inline:
72
+
73
+ 1. Find the 1-3 nodes whose label best matches the expanded tokens.
74
+ 2. Run the appropriate traversal from each starting node.
75
+ 3. Read the subgraph - node labels, edge relations, confidence tags, source locations.
76
+ 4. Answer using **only** what the graph contains. Quote `source_location` when citing a specific fact.
77
+ 5. If the graph lacks enough information, say so - do not hallucinate edges.
78
+
79
+ ```bash
80
+ $(cat graphify-out/.graphify_python) -c "
81
+ import sys, json
82
+ from networkx.readwrite import json_graph
83
+ import networkx as nx
84
+ from pathlib import Path
85
+
86
+ data = json.loads(Path('graphify-out/graph.json').read_text(encoding='utf-8'))
87
+ G = json_graph.node_link_graph(data, edges='links')
88
+
89
+ question = 'QUESTION'
90
+ mode = 'MODE' # 'bfs' or 'dfs'
91
+ terms = [t.lower() for t in question.split() if len(t) >= 3] # match the vocab threshold; keeps api/jwt/ios (#1392)
92
+
93
+ # Find best-matching start nodes
94
+ scored = []
95
+ for nid, ndata in G.nodes(data=True):
96
+ label = ndata.get('label', '').lower()
97
+ score = sum(1 for t in terms if t in label)
98
+ if score > 0:
99
+ scored.append((score, nid))
100
+ scored.sort(reverse=True)
101
+ start_nodes = [nid for _, nid in scored[:3]]
102
+
103
+ if not start_nodes:
104
+ print('No matching nodes found for query terms:', terms)
105
+ sys.exit(0)
106
+
107
+ subgraph_nodes = set()
108
+ subgraph_edges = []
109
+
110
+ if mode == 'dfs':
111
+ # DFS: follow one path as deep as possible before backtracking.
112
+ # Depth-limited to 6 to avoid traversing the whole graph.
113
+ visited = set()
114
+ stack = [(n, 0) for n in reversed(start_nodes)]
115
+ while stack:
116
+ node, depth = stack.pop()
117
+ if node in visited or depth > 6:
118
+ continue
119
+ visited.add(node)
120
+ subgraph_nodes.add(node)
121
+ for neighbor in G.neighbors(node):
122
+ if neighbor not in visited:
123
+ stack.append((neighbor, depth + 1))
124
+ subgraph_edges.append((node, neighbor))
125
+ else:
126
+ # BFS: explore all neighbors layer by layer up to depth 3.
127
+ frontier = set(start_nodes)
128
+ subgraph_nodes = set(start_nodes)
129
+ for _ in range(3):
130
+ next_frontier = set()
131
+ for n in frontier:
132
+ for neighbor in G.neighbors(n):
133
+ if neighbor not in subgraph_nodes:
134
+ next_frontier.add(neighbor)
135
+ subgraph_edges.append((n, neighbor))
136
+ subgraph_nodes.update(next_frontier)
137
+ frontier = next_frontier
138
+
139
+ # Token-budget aware output: rank by relevance, cut at budget (~4 chars/token)
140
+ token_budget = BUDGET # default 2000
141
+ char_budget = token_budget * 4
142
+
143
+ # Score each node by term overlap for ranked output
144
+ def relevance(nid):
145
+ label = G.nodes[nid].get('label', '').lower()
146
+ return sum(1 for t in terms if t in label)
147
+
148
+ ranked_nodes = sorted(subgraph_nodes, key=relevance, reverse=True)
149
+
150
+ lines = [f'Traversal: {mode.upper()} | Start: {[G.nodes[n].get(\"label\",n) for n in start_nodes]} | {len(subgraph_nodes)} nodes']
151
+ for nid in ranked_nodes:
152
+ d = G.nodes[nid]
153
+ lines.append(f' NODE {d.get(\"label\", nid)} [src={d.get(\"source_file\",\"\")} loc={d.get(\"source_location\",\"\")}]')
154
+ for u, v in subgraph_edges:
155
+ if u in subgraph_nodes and v in subgraph_nodes:
156
+ _raw = G[u][v]; d = next(iter(_raw.values()), {}) if isinstance(G, nx.MultiGraph) else _raw
157
+ lines.append(f' EDGE {G.nodes[u].get(\"label\",u)} --{d.get(\"relation\",\"\")} [{d.get(\"confidence\",\"\")}]--> {G.nodes[v].get(\"label\",v)}')
158
+
159
+ output = '\n'.join(lines)
160
+ if len(output) > char_budget:
161
+ output = output[:char_budget] + f'\n... (truncated at ~{token_budget} token budget - use --budget N for more)'
162
+ print(output)
163
+ "
164
+ ```
165
+
166
+ Replace `QUESTION` with the **expanded** query string, `MODE` with `bfs` or `dfs`, and `BUDGET` with the token budget (default `2000`, or whatever `--budget N` specifies). Then answer based on the subgraph output above, using only what the graph contains.
167
+
168
+ After writing the answer, save it back into the graph so it improves future queries. Include the expanded tokens inside the `--answer` text (e.g. `"Expanded from original query via vocab: [tokens]. Then traversed..."`) so the next `--update` extracts the expansion history as a graph node:
169
+
170
+ ```bash
171
+ $(cat graphify-out/.graphify_python) -m graphify save-result --question "ORIGINAL_QUESTION" --answer "ANSWER" --type query --nodes NODE1 NODE2
172
+ ```
173
+
174
+ Replace `ORIGINAL_QUESTION` with the user's verbatim question, `ANSWER` with your full answer text (containing the expanded-token trace), `NODE1 NODE2` with the list of node labels you cited. This closes the feedback loop: the next `--update` will extract this Q&A as a node in the graph.
175
+
176
+ **Work memory (self-improving loop).** Add an `--outcome` so future sessions learn from this one — append `--outcome useful|dead_end|corrected` to the `save-result` command (and `--correction "the right answer"` when correcting):
177
+
178
+ - `useful` — the cited nodes answered the question well (they become *preferred sources*).
179
+ - `dead_end` — the question/path led nowhere; don't re-derive it next time.
180
+ - `corrected` — the saved answer was wrong; `--correction` records what was right.
181
+
182
+ At the **start** of graph work, refresh and read the lessons: run `graphify reflect --if-stale` (cheap, deterministic, no LLM; `--if-stale` makes it a no-op when `LESSONS.md` is already newer than every input, e.g. when the git hook just refreshed it), then read `graphify-out/reflections/LESSONS.md`. It lists **preferred sources** (start there), **known dead ends** (skip them), and prior **corrections**. Running `reflect` yourself keeps the lessons current even without the git hook installed; if the post-commit hook *is* installed, `--if-stale` means your session-start run costs almost nothing.
183
+
184
+ ---
185
+
186
+ ## For /graphify path
187
+
188
+ Find the shortest path between two named concepts in the graph. Prefer the CLI when installed:
189
+
190
+ ```bash
191
+ graphify path "NODE_A" "NODE_B"
192
+ ```
193
+
194
+ If the CLI is unavailable, run it inline:
195
+
196
+ ```bash
197
+ $(cat graphify-out/.graphify_python) -c "
198
+ import json, sys
199
+ import networkx as nx
200
+ from networkx.readwrite import json_graph
201
+ from pathlib import Path
202
+
203
+ data = json.loads(Path('graphify-out/graph.json').read_text(encoding='utf-8'))
204
+ G = json_graph.node_link_graph(data, edges='links')
205
+
206
+ a_term = 'NODE_A'
207
+ b_term = 'NODE_B'
208
+
209
+ def find_node(term):
210
+ term = term.lower()
211
+ scored = sorted(
212
+ [(sum(1 for w in term.split() if w in G.nodes[n].get('label','').lower()), n)
213
+ for n in G.nodes()],
214
+ reverse=True
215
+ )
216
+ return scored[0][1] if scored and scored[0][0] > 0 else None
217
+
218
+ src = find_node(a_term)
219
+ tgt = find_node(b_term)
220
+
221
+ if not src or not tgt:
222
+ print(f'Could not find nodes matching: {a_term!r} or {b_term!r}')
223
+ sys.exit(0)
224
+
225
+ try:
226
+ path = nx.shortest_path(G, src, tgt)
227
+ print(f'Shortest path ({len(path)-1} hops):')
228
+ for i, nid in enumerate(path):
229
+ label = G.nodes[nid].get('label', nid)
230
+ if i < len(path) - 1:
231
+ _raw = G[nid][path[i+1]]; edge = next(iter(_raw.values()), {}) if isinstance(G, nx.MultiGraph) else _raw
232
+ rel = edge.get('relation', '')
233
+ conf = edge.get('confidence', '')
234
+ print(f' {label} --{rel}--> [{conf}]')
235
+ else:
236
+ print(f' {label}')
237
+ except nx.NetworkXNoPath:
238
+ print(f'No path found between {a_term!r} and {b_term!r}')
239
+ except nx.NodeNotFound as e:
240
+ print(f'Node not found: {e}')
241
+ "
242
+ ```
243
+
244
+ Replace `NODE_A` and `NODE_B` with the actual concept names from the user. Then explain the path in plain language - what each hop means, why it's significant.
245
+
246
+ After writing the explanation, save it back:
247
+
248
+ ```bash
249
+ $(cat graphify-out/.graphify_python) -m graphify save-result --question "Path from NODE_A to NODE_B" --answer "ANSWER" --type path_query --nodes NODE_A NODE_B
250
+ ```
251
+
252
+ ---
253
+
254
+ ## For /graphify explain
255
+
256
+ Give a plain-language explanation of a single node - everything connected to it. Prefer the CLI when installed:
257
+
258
+ ```bash
259
+ graphify explain "NODE_NAME"
260
+ ```
261
+
262
+ If the CLI is unavailable, run it inline:
263
+
264
+ ```bash
265
+ $(cat graphify-out/.graphify_python) -c "
266
+ import json, sys
267
+ import networkx as nx
268
+ from networkx.readwrite import json_graph
269
+ from pathlib import Path
270
+
271
+ data = json.loads(Path('graphify-out/graph.json').read_text(encoding='utf-8'))
272
+ G = json_graph.node_link_graph(data, edges='links')
273
+
274
+ term = 'NODE_NAME'
275
+ term_lower = term.lower()
276
+
277
+ # Find best matching node
278
+ scored = sorted(
279
+ [(sum(1 for w in term_lower.split() if w in G.nodes[n].get('label','').lower()), n)
280
+ for n in G.nodes()],
281
+ reverse=True
282
+ )
283
+ if not scored or scored[0][0] == 0:
284
+ print(f'No node matching {term!r}')
285
+ sys.exit(0)
286
+
287
+ nid = scored[0][1]
288
+ data_n = G.nodes[nid]
289
+ print(f'NODE: {data_n.get(\"label\", nid)}')
290
+ print(f' source: {data_n.get(\"source_file\",\"unknown\")}')
291
+ print(f' type: {data_n.get(\"file_type\",\"unknown\")}')
292
+ print(f' degree: {G.degree(nid)}')
293
+ print()
294
+ print('CONNECTIONS:')
295
+ for neighbor in G.neighbors(nid):
296
+ _raw = G[nid][neighbor]; edge = next(iter(_raw.values()), {}) if isinstance(G, nx.MultiGraph) else _raw
297
+ nlabel = G.nodes[neighbor].get('label', neighbor)
298
+ rel = edge.get('relation', '')
299
+ conf = edge.get('confidence', '')
300
+ src_file = G.nodes[neighbor].get('source_file', '')
301
+ print(f' --{rel}--> {nlabel} [{conf}] ({src_file})')
302
+ "
303
+ ```
304
+
305
+ Replace `NODE_NAME` with the concept the user asked about. Then write a 3-5 sentence explanation of what this node is, what it connects to, and why those connections are significant. Use the source locations as citations.
306
+
307
+ After writing the explanation, save it back:
308
+
309
+ ```bash
310
+ $(cat graphify-out/.graphify_python) -m graphify save-result --question "Explain NODE_NAME" --answer "ANSWER" --type explain --nodes NODE_NAME
311
+ ```
@@ -0,0 +1,52 @@
1
+ # graphify reference: transcribe video and audio
2
+
3
+ Load this only when `detect` reported one or more `video` files. A corpus with no video never reads this.
4
+
5
+ ### Step 2.5 - Transcribe video / audio files (only if video files detected)
6
+
7
+ Skip this step entirely if `detect` returned zero `video` files.
8
+
9
+ Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
10
+
11
+ **Strategy:** Read the god nodes from `graphify-out/.graphify_detect.json` (or the analysis file if it exists from a previous run). You are already a language model — write a one-sentence domain hint yourself from those labels. Then pass it to Whisper as the initial prompt. No separate API call needed.
12
+
13
+ **However**, if the corpus has *only* video files and no other docs/code, use the generic fallback prompt: `"Use proper punctuation and paragraph breaks."`
14
+
15
+ **Step 1 - Write the Whisper prompt yourself.**
16
+
17
+ Read the top god node labels from detect output or analysis, then compose a short domain hint sentence, for example:
18
+
19
+ - Labels: `transformer, attention, encoder, decoder` → `"Machine learning research on transformer architectures and attention mechanisms. Use proper punctuation and paragraph breaks."`
20
+ - Labels: `kubernetes, deployment, pod, helm` → `"DevOps discussion about Kubernetes deployments and Helm charts. Use proper punctuation and paragraph breaks."`
21
+
22
+ **Export** it as `GRAPHIFY_WHISPER_PROMPT` (the exact name the transcriber reads — and it must be `export`ed so the child Python process sees it) for the next command.
23
+
24
+ **Step 2 - Transcribe:**
25
+
26
+ ```bash
27
+ export GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed (must be exported)
28
+ export GRAPHIFY_WHISPER_PROMPT="<the one-sentence domain hint you composed in Step 1>"
29
+ $(cat graphify-out/.graphify_python) -c "
30
+ import json, os, sys
31
+ from pathlib import Path
32
+ from graphify.transcribe import transcribe_all
33
+
34
+ detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text(encoding=\"utf-8\"))
35
+ video_files = detect.get('files', {}).get('video', [])
36
+ prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.')
37
+
38
+ transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
39
+ # Write the JSON from Python (NOT a shell '>' redirect): transcribe_all/Whisper
40
+ # print progress to stdout, which would otherwise corrupt the JSON file (#1392).
41
+ Path('graphify-out/.graphify_transcripts.json').write_text(json.dumps(transcript_paths, ensure_ascii=False), encoding=\"utf-8\")
42
+ print(f'Transcribed {len(transcript_paths)} file(s)', file=sys.stderr)
43
+ "
44
+ ```
45
+
46
+ After transcription:
47
+ - Read the transcript paths from `graphify-out/.graphify_transcripts.json`
48
+ - Add them to the docs list before dispatching semantic subagents in Step 3B
49
+ - Print how many transcripts were created: `Transcribed N video file(s) -> treating as docs`
50
+ - If transcription fails for a file, print a warning and continue with the rest
51
+
52
+ **Whisper model:** Default is `base`. If the user passed `--whisper-model <name>`, `export GRAPHIFY_WHISPER_MODEL=<name>` (it must be exported, not just assigned) before running the command above.
@@ -0,0 +1,210 @@
1
+ # graphify reference: incremental update and cluster-only
2
+
3
+ Load this only when the user passed `--update` or `--cluster-only`. A first-time full build never reads this file.
4
+
5
+ ## For --update (incremental re-extraction)
6
+
7
+ Use when you've added or modified files since the last run. Only re-extracts changed files - saves tokens and time.
8
+
9
+ ```bash
10
+ $(cat graphify-out/.graphify_python) -c "
11
+ import sys, json
12
+ from graphify.detect import detect_incremental, save_manifest
13
+ from pathlib import Path
14
+
15
+ result = detect_incremental(Path('INPUT_PATH'))
16
+ new_total = result.get('new_total', 0)
17
+ print(json.dumps(result, indent=2, ensure_ascii=False))
18
+ Path('graphify-out/.graphify_incremental.json').write_text(json.dumps(result, ensure_ascii=False), encoding=\"utf-8\")
19
+ deleted = list(result.get('deleted_files', []))
20
+ if new_total == 0 and not deleted:
21
+ print('No files changed since last run. Nothing to update.')
22
+ raise SystemExit(0)
23
+ if deleted:
24
+ print(f'{len(deleted)} deleted file(s) to prune.')
25
+ if new_total > 0:
26
+ print(f'{new_total} new/changed file(s) to re-extract.')
27
+ "
28
+ ```
29
+
30
+ Then populate `.graphify_detect.json` so Steps 3A–6 (which read it unconditionally) see the right state for an incremental run. `files` carries the changed subset (drives Step 3A AST + Step 3B0 cache check on only what changed); `all_files` carries the full corpus for any step that needs corpus-wide context:
31
+
32
+ ```bash
33
+ $(cat graphify-out/.graphify_python) -c "
34
+ import json
35
+ from pathlib import Path
36
+ r = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\"))
37
+ Path('graphify-out/.graphify_detect.json').write_text(json.dumps({
38
+ 'files': r.get('new_files', {}),
39
+ 'all_files': r.get('files', {}),
40
+ 'total_files': r.get('new_total', 0),
41
+ 'total_words': r.get('total_words', 0),
42
+ 'skipped_sensitive': r.get('skipped_sensitive', []),
43
+ 'needs_graph': True,
44
+ }, ensure_ascii=False), encoding=\"utf-8\")
45
+ "
46
+ ```
47
+
48
+ If new files exist, first check whether all changed files are code files:
49
+
50
+ ```bash
51
+ $(cat graphify-out/.graphify_python) -c "
52
+ import json
53
+ from pathlib import Path
54
+
55
+ result = json.loads(open('graphify-out/.graphify_incremental.json', encoding='utf-8').read()) if Path('graphify-out/.graphify_incremental.json').exists() else {}
56
+ code_exts = {'.py','.ts','.js','.go','.rs','.java','.cpp','.c','.rb','.swift','.kt','.cs','.scala','.php','.cc','.cxx','.hpp','.h','.kts','.lua','.toc','.f','.F','.f90','.F90','.f95','.F95','.f03','.F03','.f08','.F08'}
57
+ new_files = result.get('new_files', {})
58
+ all_changed = [f for files in new_files.values() for f in files]
59
+ code_only = all(Path(f).suffix.lower() in code_exts for f in all_changed)
60
+ print('code_only:', code_only)
61
+ "
62
+ ```
63
+
64
+ If `code_only` is True: print `[graphify update] Code-only changes detected - skipping semantic extraction (no LLM needed)`, run only Step 3A (AST) on the changed files, skip Step 3B entirely (no subagents), then go straight to merge and Steps 4–8.
65
+
66
+ If `code_only` is False (any changed file is a doc/paper/image/video): **first, if any changed file is in `new_files['video']`, run `references/transcribe.md` (Step 2.5) on those files, then rewrite `.graphify_detect.json` to move the resulting transcript paths into `files['document']` and drop `files['video']`** — otherwise raw `.mp4/.mp3` paths are fed to semantic subagents as unreadable media (#1392). Then run the full Steps 3A–3C pipeline as normal.
67
+
68
+
69
+ If no new files exist (only deletions), create an empty extraction so the merge step can prune:
70
+
71
+ ```bash
72
+ if [ ! -f graphify-out/.graphify_extract.json ]; then
73
+ echo '[graphify update] Only deletions -- creating empty extraction for merge.'
74
+ $(cat graphify-out/.graphify_python) -c "
75
+ import json
76
+ from pathlib import Path
77
+ Path('graphify-out/.graphify_extract.json').write_text(json.dumps({'nodes':[],'edges':[],'hyperedges':[],'input_tokens':0,'output_tokens':0}), encoding='utf-8')
78
+ "
79
+ fi
80
+ ```
81
+
82
+
83
+ Then:
84
+
85
+ ```bash
86
+ $(cat graphify-out/.graphify_python) -c "
87
+ import json
88
+ from pathlib import Path
89
+ from graphify.build import build_merge
90
+ from graphify.detect import save_manifest
91
+
92
+ # Load new extraction and incremental state
93
+ new_extraction = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
94
+ incremental = json.loads(Path('graphify-out/.graphify_incremental.json').read_text(encoding=\"utf-8\"))
95
+ deleted = list(incremental.get('deleted_files', []))
96
+ # prune_sources is ONLY for genuinely DELETED files. Changed/re-extracted files are
97
+ # handled by build_merge's replace-on-re-extract (#1344): every source_file in
98
+ # new_chunks is dropped from the base before merge, so old/stale nodes don't survive.
99
+ # Do NOT add `changed` here: with root= passed, prune_set relativizes to the same base
100
+ # as the freshly merged nodes and would DELETE the re-extracted content (#1178 is moot
101
+ # now that replace — not the dedup pass — reconciles changed files).
102
+ prune = list(deleted) or None
103
+
104
+ # Use build_merge() — reads graph.json directly without NetworkX round-trip
105
+ # so edge direction (calls, implements, imports) is always preserved (#801).
106
+ # Pass root= so prune_sources (absolute paths from detect_incremental) are
107
+ # relativized to match the graph's relative source_file values; without it
108
+ # nothing is pruned and stale nodes accumulate on every update (#1361).
109
+ # directed=IS_DIRECTED: replace IS_DIRECTED with True if --directed was given, else
110
+ # False. Without it a --directed --update silently rebuilds undirected and collapses
111
+ # reciprocal A<->B edges (#1392).
112
+ G = build_merge(
113
+ [new_extraction],
114
+ graph_path='graphify-out/graph.json',
115
+ prune_sources=prune,
116
+ root='INPUT_PATH',
117
+ directed=IS_DIRECTED,
118
+ )
119
+ print(f'[graphify update] Merged: {G.number_of_nodes()} nodes, {G.number_of_edges()} edges')
120
+
121
+ # Write merged result back to .graphify_extract.json so Step 4 sees the full graph
122
+ merged_out = {
123
+ 'nodes': [{'id': n, **d} for n, d in G.nodes(data=True)],
124
+ 'edges': [
125
+ # Explicit source/target last so they win over any stale attrs in d.
126
+ {**{k: val for k, val in d.items() if k not in ('_src', '_tgt', 'source', 'target')},
127
+ 'source': d.get('_src', u), 'target': d.get('_tgt', v)}
128
+ for u, v, d in G.edges(data=True)
129
+ ],
130
+ # G.graph["hyperedges"] holds hyperedges from both existing graph.json
131
+ # and new_extraction (build_merge combines them). Falling back to
132
+ # new_extraction only would silently drop prior-run hyperedges (#801).
133
+ 'hyperedges': list(G.graph.get('hyperedges', [])),
134
+ 'input_tokens': new_extraction.get('input_tokens', 0),
135
+ 'output_tokens': new_extraction.get('output_tokens', 0),
136
+ }
137
+ Path('graphify-out/.graphify_extract.json').write_text(json.dumps(merged_out, ensure_ascii=False), encoding=\"utf-8\")
138
+ print(f'[graphify update] Merged extraction written ({len(merged_out[\"nodes\"])} nodes, {len(merged_out[\"edges\"])} edges)')
139
+
140
+ # Save manifest so next --update diffs against today's state, not the
141
+ # prior run's baseline (prevents ghost-node reports on subsequent updates).
142
+ # root= matches the build_merge call above so the manifest keys stay relative to
143
+ # the scan root — portable across clones/machines, so --update keeps matching
144
+ # cached files instead of missing every one after a move (#1417).
145
+ #
146
+ # Only stamp semantic files (docs/papers/images) that ACTUALLY produced output
147
+ # THIS run (new_extraction is this run's fresh extraction, read above before the
148
+ # merge overwrote the file): a changed doc whose chunk failed must stay unstamped
149
+ # so the next --update re-queues it, otherwise it is marked done and its content
150
+ # is lost forever (#2015). Mirrors the library extract path
151
+ # (cli._stamped_manifest_files + clear_semantic + scan_corpus).
152
+ from graphify.cli import _stamped_manifest_files
153
+ _manifest_files = _stamped_manifest_files(incremental['files'], new_extraction, Path('INPUT_PATH'))
154
+ # Changed semantic files dispatched this run but NOT stamped had their chunk fail
155
+ # or be omitted; clear any stale semantic_hash so they are re-queued (#1948).
156
+ _sem_types = ('document', 'paper', 'image')
157
+ _dispatched = {f for t, fl in incremental.get('new_files', {}).items() if t in _sem_types for f in fl}
158
+ _stamped = {f for fl in _manifest_files.values() for f in fl}
159
+ _cleared = _dispatched - _stamped
160
+ # scan_corpus = the RAW full corpus so in-root files newly excluded since last run
161
+ # are dropped rather than masquerading as deletions; untouched rows preserved (#1908).
162
+ _scan = {f for fl in incremental['files'].values() for f in fl}
163
+ save_manifest(_manifest_files, root='INPUT_PATH', scan_corpus=_scan, clear_semantic=_cleared or None)
164
+ print('[graphify update] Manifest saved.')
165
+ "
166
+ ```
167
+
168
+ Then run Steps 4–8 on the merged graph as normal.
169
+
170
+ After Step 4, show the graph diff:
171
+
172
+ ```bash
173
+ $(cat graphify-out/.graphify_python) -c "
174
+ import json
175
+ from graphify.analyze import graph_diff
176
+ from graphify.build import build_from_json
177
+ from networkx.readwrite import json_graph
178
+ import networkx as nx
179
+ from pathlib import Path
180
+
181
+ # Load old graph (before update) from backup written before merge
182
+ old_data = json.loads(Path('graphify-out/.graphify_old.json').read_text(encoding=\"utf-8\")) if Path('graphify-out/.graphify_old.json').exists() else None
183
+ new_extract = json.loads(Path('graphify-out/.graphify_extract.json').read_text(encoding=\"utf-8\"))
184
+ G_new = build_from_json(new_extract, directed=IS_DIRECTED)
185
+
186
+ if old_data:
187
+ G_old = json_graph.node_link_graph(old_data, edges='links')
188
+ diff = graph_diff(G_old, G_new)
189
+ print(diff['summary'])
190
+ if diff['new_nodes']:
191
+ print('New nodes:', ', '.join(n['label'] for n in diff['new_nodes'][:5]))
192
+ if diff['new_edges']:
193
+ print('New edges:', len(diff['new_edges']))
194
+ "
195
+ ```
196
+
197
+ Before the merge step, save the old graph: `cp graphify-out/graph.json graphify-out/.graphify_old.json`
198
+ Clean up after: `rm -f graphify-out/.graphify_old.json`
199
+
200
+ ---
201
+
202
+ ## For --cluster-only
203
+
204
+ Skip Steps 1–3. Re-run clustering on the existing graph:
205
+
206
+ ```bash
207
+ graphify cluster-only .
208
+ ```
209
+
210
+ `graphify cluster-only .` is **self-contained**: it re-clusters, names communities, and regenerates `GRAPH_REPORT.md`, `graph.json`, and `graph.html` from the existing graph. **Do not re-run Steps 5–9** — they read intermediate files (`.graphify_extract.json`, `.graphify_detect.json`, `.graphify_analysis.json`) that a prior build's cleanup (Step 9) already deleted, so they raise `FileNotFoundError` (#1392). When it finishes, present the refreshed `GRAPH_REPORT.md` summary as usual.
@@ -0,0 +1,56 @@
1
+ # graphify reference: add a URL and watch a folder
2
+
3
+ Load this when the user ran `/graphify add <url>` or passed `--watch`. Neither is part of the default build.
4
+
5
+ ## For /graphify add
6
+
7
+ Fetch a URL and add it to the corpus, then update the graph.
8
+
9
+ ```bash
10
+ $(cat graphify-out/.graphify_python) -c "
11
+ import sys
12
+ from graphify.ingest import ingest
13
+ from pathlib import Path
14
+
15
+ try:
16
+ out = ingest('URL', Path('./raw'), author='AUTHOR', contributor='CONTRIBUTOR')
17
+ print(f'Saved to {out}')
18
+ except ValueError as e:
19
+ print(f'error: {e}', file=sys.stderr)
20
+ sys.exit(1)
21
+ except RuntimeError as e:
22
+ print(f'error: {e}', file=sys.stderr)
23
+ sys.exit(1)
24
+ "
25
+ ```
26
+
27
+ Replace `URL` with the actual URL, `AUTHOR` with the user's name if provided, `CONTRIBUTOR` likewise. If the command exits with an error, tell the user what went wrong - do not silently continue. After a successful save, automatically run the `--update` pipeline on `./raw` to merge the new file into the existing graph.
28
+
29
+ Supported URL types (auto-detected):
30
+ - YouTube / any video URL → audio downloaded via yt-dlp, transcribed to `.txt` on next run (requires `pip install 'graphifyy[video]'`)
31
+ - Twitter/X → fetched via oEmbed, saved as `.md` with tweet text and author
32
+ - arXiv → abstract + metadata saved as `.md`
33
+ - PDF → downloaded as `.pdf`
34
+ - Images (.png/.jpg/.webp) → downloaded, Claude vision extracts on next run
35
+ - Any webpage → converted to markdown via html2text
36
+
37
+ ---
38
+
39
+ ## For --watch
40
+
41
+ Start a background watcher that monitors a folder and auto-updates the graph when files change.
42
+
43
+ ```bash
44
+ $(cat graphify-out/.graphify_python) -m graphify.watch INPUT_PATH --debounce 3
45
+ ```
46
+
47
+ Replace INPUT_PATH with the folder to watch. Behavior depends on what changed:
48
+
49
+ - **Code files only (.py, .ts, .go, etc.):** re-runs AST extraction + rebuild + cluster immediately, no LLM needed. `graph.json` and `GRAPH_REPORT.md` are updated automatically.
50
+ - **Docs, papers, or images:** writes a `graphify-out/needs_update` flag and prints a notification to run `/graphify --update` (LLM semantic re-extraction required).
51
+
52
+ Debounce (default 3s): waits until file activity stops before triggering, so a wave of parallel agent writes doesn't trigger a rebuild per file.
53
+
54
+ Press Ctrl+C to stop.
55
+
56
+ For agentic workflows: run `--watch` in a background terminal. Code changes from agent waves are picked up automatically between waves. If agents are also writing docs or notes, you'll need a manual `/graphify --update` after those waves.