graphitect 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (336) hide show
  1. graphify/__init__.py +30 -0
  2. graphify/__main__.py +757 -0
  3. graphify/_minhash.py +107 -0
  4. graphify/affected.py +318 -0
  5. graphify/always_on/agents-md.md +12 -0
  6. graphify/always_on/antigravity-rules.md +14 -0
  7. graphify/always_on/claude-md.md +9 -0
  8. graphify/always_on/gemini-md.md +9 -0
  9. graphify/always_on/kiro-steering.md +5 -0
  10. graphify/always_on/vscode-instructions.md +17 -0
  11. graphify/analyze.py +769 -0
  12. graphify/benchmark.py +152 -0
  13. graphify/build.py +2300 -0
  14. graphify/cache.py +1746 -0
  15. graphify/callflow_html.py +2051 -0
  16. graphify/cargo_introspect.py +109 -0
  17. graphify/cli.py +4745 -0
  18. graphify/cluster.py +409 -0
  19. graphify/command-kilo.md +15 -0
  20. graphify/cross_repo_calls.py +216 -0
  21. graphify/cross_repo_types.py +75 -0
  22. graphify/csharp_dispatch.py +154 -0
  23. graphify/dedup.py +1213 -0
  24. graphify/detect.py +2566 -0
  25. graphify/diagnostics.py +406 -0
  26. graphify/export.py +1349 -0
  27. graphify/exporters/__init__.py +1 -0
  28. graphify/exporters/base.py +14 -0
  29. graphify/exporters/graphdb.py +173 -0
  30. graphify/exporters/html.py +637 -0
  31. graphify/extract.py +7856 -0
  32. graphify/extractors/MIGRATION.md +107 -0
  33. graphify/extractors/__init__.py +66 -0
  34. graphify/extractors/apex.py +215 -0
  35. graphify/extractors/base.py +85 -0
  36. graphify/extractors/bash.py +579 -0
  37. graphify/extractors/blade.py +53 -0
  38. graphify/extractors/commonlisp.py +540 -0
  39. graphify/extractors/csharp.py +448 -0
  40. graphify/extractors/dart.py +564 -0
  41. graphify/extractors/dm.py +494 -0
  42. graphify/extractors/elixir.py +241 -0
  43. graphify/extractors/engine.py +6509 -0
  44. graphify/extractors/fortran.py +311 -0
  45. graphify/extractors/go.py +527 -0
  46. graphify/extractors/json_config.py +240 -0
  47. graphify/extractors/julia.py +289 -0
  48. graphify/extractors/markdown.py +408 -0
  49. graphify/extractors/models.py +131 -0
  50. graphify/extractors/objc.py +566 -0
  51. graphify/extractors/ocaml.py +289 -0
  52. graphify/extractors/pascal.py +688 -0
  53. graphify/extractors/pascal_forms.py +196 -0
  54. graphify/extractors/powershell.py +522 -0
  55. graphify/extractors/razor.py +192 -0
  56. graphify/extractors/resolution.py +3584 -0
  57. graphify/extractors/robot.py +296 -0
  58. graphify/extractors/rust.py +470 -0
  59. graphify/extractors/sln.py +92 -0
  60. graphify/extractors/sql.py +720 -0
  61. graphify/extractors/terraform.py +181 -0
  62. graphify/extractors/verilog.py +329 -0
  63. graphify/extractors/zig.py +181 -0
  64. graphify/file_slice.py +246 -0
  65. graphify/global_graph.py +194 -0
  66. graphify/google_workspace.py +237 -0
  67. graphify/hooks.py +933 -0
  68. graphify/ids.py +93 -0
  69. graphify/ingest.py +358 -0
  70. graphify/install.py +2366 -0
  71. graphify/llm.py +3544 -0
  72. graphify/manifest.py +4 -0
  73. graphify/manifest_ingest.py +311 -0
  74. graphify/mcp_ingest.py +386 -0
  75. graphify/multigraph_compat.py +212 -0
  76. graphify/pascal_resolution.py +129 -0
  77. graphify/paths.py +436 -0
  78. graphify/pg_introspect.py +165 -0
  79. graphify/prs.py +770 -0
  80. graphify/querylog.py +80 -0
  81. graphify/reflect.py +882 -0
  82. graphify/report.py +346 -0
  83. graphify/resolver_registry.py +85 -0
  84. graphify/ruby_resolution.py +242 -0
  85. graphify/scip_ingest.py +363 -0
  86. graphify/security.py +460 -0
  87. graphify/semantic_cleanup.py +336 -0
  88. graphify/serve.py +2608 -0
  89. graphify/skill-agents.md +710 -0
  90. graphify/skill-aider.md +1283 -0
  91. graphify/skill-amp.md +710 -0
  92. graphify/skill-claw.md +713 -0
  93. graphify/skill-codex.md +710 -0
  94. graphify/skill-copilot.md +713 -0
  95. graphify/skill-devin.md +1410 -0
  96. graphify/skill-droid.md +710 -0
  97. graphify/skill-kilo.md +722 -0
  98. graphify/skill-kiro.md +713 -0
  99. graphify/skill-opencode.md +705 -0
  100. graphify/skill-pi.md +713 -0
  101. graphify/skill-trae.md +711 -0
  102. graphify/skill-vscode.md +709 -0
  103. graphify/skill-windows.md +755 -0
  104. graphify/skill.md +713 -0
  105. graphify/skills/agents/references/add-watch.md +56 -0
  106. graphify/skills/agents/references/exports.md +87 -0
  107. graphify/skills/agents/references/extraction-spec.md +70 -0
  108. graphify/skills/agents/references/github-and-merge.md +46 -0
  109. graphify/skills/agents/references/hooks.md +33 -0
  110. graphify/skills/agents/references/query.md +311 -0
  111. graphify/skills/agents/references/transcribe.md +52 -0
  112. graphify/skills/agents/references/update.md +210 -0
  113. graphify/skills/amp/references/add-watch.md +56 -0
  114. graphify/skills/amp/references/exports.md +87 -0
  115. graphify/skills/amp/references/extraction-spec.md +70 -0
  116. graphify/skills/amp/references/github-and-merge.md +46 -0
  117. graphify/skills/amp/references/hooks.md +33 -0
  118. graphify/skills/amp/references/query.md +311 -0
  119. graphify/skills/amp/references/transcribe.md +52 -0
  120. graphify/skills/amp/references/update.md +210 -0
  121. graphify/skills/claude/references/add-watch.md +56 -0
  122. graphify/skills/claude/references/exports.md +87 -0
  123. graphify/skills/claude/references/extraction-spec.md +70 -0
  124. graphify/skills/claude/references/github-and-merge.md +46 -0
  125. graphify/skills/claude/references/hooks.md +33 -0
  126. graphify/skills/claude/references/query.md +311 -0
  127. graphify/skills/claude/references/transcribe.md +52 -0
  128. graphify/skills/claude/references/update.md +210 -0
  129. graphify/skills/claw/references/add-watch.md +56 -0
  130. graphify/skills/claw/references/exports.md +87 -0
  131. graphify/skills/claw/references/extraction-spec.md +31 -0
  132. graphify/skills/claw/references/github-and-merge.md +46 -0
  133. graphify/skills/claw/references/hooks.md +33 -0
  134. graphify/skills/claw/references/query.md +311 -0
  135. graphify/skills/claw/references/transcribe.md +52 -0
  136. graphify/skills/claw/references/update.md +210 -0
  137. graphify/skills/codex/references/add-watch.md +56 -0
  138. graphify/skills/codex/references/exports.md +87 -0
  139. graphify/skills/codex/references/extraction-spec.md +31 -0
  140. graphify/skills/codex/references/github-and-merge.md +46 -0
  141. graphify/skills/codex/references/hooks.md +33 -0
  142. graphify/skills/codex/references/query.md +311 -0
  143. graphify/skills/codex/references/transcribe.md +52 -0
  144. graphify/skills/codex/references/update.md +210 -0
  145. graphify/skills/copilot/references/add-watch.md +56 -0
  146. graphify/skills/copilot/references/exports.md +87 -0
  147. graphify/skills/copilot/references/extraction-spec.md +70 -0
  148. graphify/skills/copilot/references/github-and-merge.md +46 -0
  149. graphify/skills/copilot/references/hooks.md +33 -0
  150. graphify/skills/copilot/references/query.md +311 -0
  151. graphify/skills/copilot/references/transcribe.md +52 -0
  152. graphify/skills/copilot/references/update.md +210 -0
  153. graphify/skills/droid/references/add-watch.md +56 -0
  154. graphify/skills/droid/references/exports.md +87 -0
  155. graphify/skills/droid/references/extraction-spec.md +70 -0
  156. graphify/skills/droid/references/github-and-merge.md +46 -0
  157. graphify/skills/droid/references/hooks.md +33 -0
  158. graphify/skills/droid/references/query.md +311 -0
  159. graphify/skills/droid/references/transcribe.md +52 -0
  160. graphify/skills/droid/references/update.md +210 -0
  161. graphify/skills/kilo/references/add-watch.md +56 -0
  162. graphify/skills/kilo/references/exports.md +87 -0
  163. graphify/skills/kilo/references/extraction-spec.md +70 -0
  164. graphify/skills/kilo/references/github-and-merge.md +46 -0
  165. graphify/skills/kilo/references/hooks.md +33 -0
  166. graphify/skills/kilo/references/query.md +311 -0
  167. graphify/skills/kilo/references/transcribe.md +52 -0
  168. graphify/skills/kilo/references/update.md +210 -0
  169. graphify/skills/kiro/references/add-watch.md +56 -0
  170. graphify/skills/kiro/references/exports.md +87 -0
  171. graphify/skills/kiro/references/extraction-spec.md +31 -0
  172. graphify/skills/kiro/references/github-and-merge.md +46 -0
  173. graphify/skills/kiro/references/hooks.md +33 -0
  174. graphify/skills/kiro/references/query.md +311 -0
  175. graphify/skills/kiro/references/transcribe.md +52 -0
  176. graphify/skills/kiro/references/update.md +210 -0
  177. graphify/skills/opencode/references/add-watch.md +56 -0
  178. graphify/skills/opencode/references/exports.md +87 -0
  179. graphify/skills/opencode/references/extraction-spec.md +70 -0
  180. graphify/skills/opencode/references/github-and-merge.md +46 -0
  181. graphify/skills/opencode/references/hooks.md +33 -0
  182. graphify/skills/opencode/references/query.md +311 -0
  183. graphify/skills/opencode/references/transcribe.md +52 -0
  184. graphify/skills/opencode/references/update.md +210 -0
  185. graphify/skills/pi/references/add-watch.md +56 -0
  186. graphify/skills/pi/references/exports.md +87 -0
  187. graphify/skills/pi/references/extraction-spec.md +31 -0
  188. graphify/skills/pi/references/github-and-merge.md +46 -0
  189. graphify/skills/pi/references/hooks.md +33 -0
  190. graphify/skills/pi/references/query.md +311 -0
  191. graphify/skills/pi/references/transcribe.md +52 -0
  192. graphify/skills/pi/references/update.md +210 -0
  193. graphify/skills/trae/references/add-watch.md +56 -0
  194. graphify/skills/trae/references/exports.md +87 -0
  195. graphify/skills/trae/references/extraction-spec.md +70 -0
  196. graphify/skills/trae/references/github-and-merge.md +46 -0
  197. graphify/skills/trae/references/hooks.md +35 -0
  198. graphify/skills/trae/references/query.md +311 -0
  199. graphify/skills/trae/references/transcribe.md +52 -0
  200. graphify/skills/trae/references/update.md +210 -0
  201. graphify/skills/vscode/references/add-watch.md +56 -0
  202. graphify/skills/vscode/references/exports.md +87 -0
  203. graphify/skills/vscode/references/extraction-spec.md +70 -0
  204. graphify/skills/vscode/references/github-and-merge.md +46 -0
  205. graphify/skills/vscode/references/hooks.md +33 -0
  206. graphify/skills/vscode/references/query.md +311 -0
  207. graphify/skills/vscode/references/transcribe.md +52 -0
  208. graphify/skills/vscode/references/update.md +210 -0
  209. graphify/skills/windows/references/add-watch.md +56 -0
  210. graphify/skills/windows/references/exports.md +87 -0
  211. graphify/skills/windows/references/extraction-spec.md +70 -0
  212. graphify/skills/windows/references/github-and-merge.md +46 -0
  213. graphify/skills/windows/references/hooks.md +33 -0
  214. graphify/skills/windows/references/query.md +311 -0
  215. graphify/skills/windows/references/transcribe.md +52 -0
  216. graphify/skills/windows/references/update.md +210 -0
  217. graphify/symbol_resolution.py +556 -0
  218. graphify/transcribe.py +186 -0
  219. graphify/tree_html.py +603 -0
  220. graphify/validate.py +95 -0
  221. graphify/watch.py +2280 -0
  222. graphify/wiki.py +405 -0
  223. graphitect/__init__.py +28 -0
  224. graphitect/__main__.py +4 -0
  225. graphitect/_vendor/__init__.py +2 -0
  226. graphitect/_vendor/archify/LICENSE +22 -0
  227. graphitect/_vendor/archify/SKILL.md +137 -0
  228. graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
  229. graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
  230. graphitect/_vendor/archify/assets/template.html +14935 -0
  231. graphitect/_vendor/archify/bin/archify.mjs +2091 -0
  232. graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
  233. graphitect/_vendor/archify/bin/preview.mjs +653 -0
  234. graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
  235. graphitect/_vendor/archify/brand-marks/README.md +31 -0
  236. graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
  237. graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
  238. graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
  239. graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
  240. graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
  241. graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
  242. graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
  243. graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
  244. graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
  245. graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
  246. graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
  247. graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
  248. graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
  249. graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
  250. graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
  251. graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
  252. graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
  253. graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
  254. graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
  255. graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
  256. graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
  257. graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
  258. graphitect/_vendor/archify/package-lock.json +149 -0
  259. graphitect/_vendor/archify/package.json +39 -0
  260. graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
  261. graphitect/_vendor/archify/references/authoring-contract.md +243 -0
  262. graphitect/_vendor/archify/references/brand-marks.md +65 -0
  263. graphitect/_vendor/archify/references/delivery-contract.md +120 -0
  264. graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
  265. graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
  266. graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
  267. graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
  268. graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
  269. graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
  270. graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
  271. graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
  272. graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
  273. graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
  274. graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
  275. graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
  276. graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
  277. graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
  278. graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
  279. graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
  280. graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
  281. graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
  282. graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
  283. graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
  284. graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
  285. graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
  286. graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
  287. graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
  288. graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
  289. graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
  290. graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
  291. graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
  292. graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
  293. graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
  294. graphitect/_vendor/archify/schemas/README.md +211 -0
  295. graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
  296. graphitect/_vendor/archify/schemas/common.schema.json +115 -0
  297. graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
  298. graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
  299. graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
  300. graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
  301. graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
  302. graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
  303. graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
  304. graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
  305. graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
  306. graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
  307. graphitect/_vendor/archify/skill-release.json +10 -0
  308. graphitect/cli.py +981 -0
  309. graphitect/deliver/__init__.py +5 -0
  310. graphitect/deliver/archify_adapter.py +1877 -0
  311. graphitect/deliver/archify_ir.py +160 -0
  312. graphitect/deliver/archify_repair.py +135 -0
  313. graphitect/deliver/doc_compiler.py +916 -0
  314. graphitect/ground/__init__.py +5 -0
  315. graphitect/ground/describe_source.py +27 -0
  316. graphitect/ground/fullread_source.py +56 -0
  317. graphitect/ground/graphify_source.py +107 -0
  318. graphitect/models.py +118 -0
  319. graphitect/skill/SKILL.md +80 -0
  320. graphitect/skill/agents/openai.yaml +4 -0
  321. graphitect/synthesize/__init__.py +5 -0
  322. graphitect/synthesize/engine.py +281 -0
  323. graphitect/synthesize/llm_backend.py +331 -0
  324. graphitect/synthesize/questions.py +139 -0
  325. graphitect/synthesize/rubric.py +104 -0
  326. graphitect-0.2.0.dist-info/METADATA +284 -0
  327. graphitect-0.2.0.dist-info/RECORD +336 -0
  328. graphitect-0.2.0.dist-info/WHEEL +5 -0
  329. graphitect-0.2.0.dist-info/entry_points.txt +2 -0
  330. graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
  331. graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
  332. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
  333. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
  334. graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
  335. graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
  336. graphitect-0.2.0.dist-info/top_level.txt +2 -0
graphify/export.py ADDED
@@ -0,0 +1,1349 @@
1
+ # write graph to HTML, JSON, SVG, GraphML, Obsidian vault, and Neo4j Cypher
2
+ from __future__ import annotations
3
+ import hashlib
4
+ import html as _html
5
+ import json
6
+ import math
7
+ import os
8
+ import re
9
+ import shutil
10
+ import sys
11
+ from collections import Counter
12
+ from datetime import date
13
+ from pathlib import Path
14
+ import networkx as nx
15
+ from networkx.readwrite import json_graph
16
+ from graphify.security import sanitize_label
17
+ from graphify.analyze import _node_community_map
18
+ from graphify.build import edge_data
19
+ from graphify.paths import stem_filename_budget, write_json_atomic, write_text_atomic
20
+
21
+ from graphify.exporters.graphdb import push_to_falkordb, push_to_neo4j # noqa: E402,F401
22
+
23
+
24
+ # Artifacts worth preserving across rebuilds (non-regenerable without LLM or curation).
25
+ _BACKUP_ARTIFACTS = [
26
+ "graph.json",
27
+ "GRAPH_REPORT.md",
28
+ ".graphify_labels.json",
29
+ ".graphify_analysis.json",
30
+ "manifest.json",
31
+ ".graphify_semantic_marker",
32
+ "cost.json",
33
+ ]
34
+
35
+
36
+ def backup_if_protected(out_dir: Path) -> "Path | None":
37
+ """Snapshot graph artifacts to a dated subfolder before an overwrite.
38
+
39
+ Triggers when graph.json exists AND either:
40
+ - .graphify_semantic_marker is present (graph cost real LLM tokens), or
41
+ - .graphify_labels.json contains at least one non-default community label
42
+ (graph has been curated by a human or skill).
43
+
44
+ Returns the backup folder path, or None if no backup was taken.
45
+ Never raises — backup failure prints a warning but never blocks the write.
46
+ Set GRAPHIFY_NO_BACKUP=1 to disable.
47
+ """
48
+ if os.environ.get("GRAPHIFY_NO_BACKUP"):
49
+ return None
50
+ out = Path(out_dir)
51
+ if not (out / "graph.json").exists():
52
+ return None
53
+
54
+ is_semantic = (out / ".graphify_semantic_marker").exists()
55
+ is_curated = False
56
+ labels_file = out / ".graphify_labels.json"
57
+ if labels_file.exists():
58
+ try:
59
+ labels = json.loads(labels_file.read_text(encoding="utf-8"))
60
+ is_curated = any(v != f"Community {k}" for k, v in labels.items())
61
+ except Exception:
62
+ pass
63
+
64
+ if not is_semantic and not is_curated:
65
+ return None
66
+
67
+ reason = "+".join(filter(None, ["semantic" if is_semantic else "", "curated" if is_curated else ""]))
68
+ today = date.today().isoformat()
69
+ backup_dir = out / today
70
+ graph_src = out / "graph.json"
71
+
72
+ # Skip re-copying if today's backup already has identical graph.json content.
73
+ # If content differs (graph changed since the last backup today), overwrite
74
+ # the backup in place — one folder per day, always the latest pre-overwrite state.
75
+ if backup_dir.exists() and (backup_dir / "graph.json").exists():
76
+ src_hash = hashlib.sha256(graph_src.read_bytes()).hexdigest()
77
+ bak_hash = hashlib.sha256((backup_dir / "graph.json").read_bytes()).hexdigest()
78
+ if src_hash == bak_hash:
79
+ return backup_dir # identical content, nothing to do
80
+
81
+ try:
82
+ backup_dir.mkdir(parents=True, exist_ok=True)
83
+ copied = 0
84
+ for name in _BACKUP_ARTIFACTS:
85
+ src = out / name
86
+ if src.exists():
87
+ try:
88
+ shutil.copy2(src, backup_dir / name)
89
+ copied += 1
90
+ except Exception:
91
+ pass
92
+ if copied:
93
+ print(f"[graphify] backed up {reason} graph ({copied} files) -> {backup_dir.name}/")
94
+ return backup_dir
95
+ except Exception as exc:
96
+ import sys
97
+ print(f"[graphify] warning: backup failed ({exc}) - continuing with overwrite", file=sys.stderr)
98
+ return None
99
+
100
+ def _obsidian_tag(name: str) -> str:
101
+ r"""Sanitize a community name for use as an Obsidian tag.
102
+
103
+ Obsidian tags accept letters from any language plus digits, hyphens,
104
+ underscores and slashes; spaces and most punctuation are not allowed, and a
105
+ tag cannot be digits-only. ``\w`` is Unicode-aware in Python 3, so Hangul,
106
+ CJK, Cyrillic and accented Latin survive instead of being stripped (#2862):
107
+ an ASCII-only filter collapsed every non-Latin community label to
108
+ underscores, so every note in that community carried the same tag.
109
+ """
110
+ tag = re.sub(r"[^\w\-/]", "", name.replace(" ", "_"))
111
+ if not tag.strip("_-/"):
112
+ return "unnamed" # label was punctuation only
113
+ if tag.isdigit():
114
+ return f"c{tag}" # Obsidian ignores digits-only tags
115
+ return tag
116
+
117
+
118
+ def _strip_diacritics(text: str | None) -> str:
119
+ import unicodedata
120
+ if not isinstance(text, str):
121
+ text = "" if text is None else str(text)
122
+ nfkd = unicodedata.normalize("NFKD", text)
123
+ return "".join(c for c in nfkd if not unicodedata.combining(c))
124
+
125
+
126
+ def _yaml_str(s: str) -> str:
127
+ """Escape a value for safe embedding in a YAML double-quoted scalar (F-009).
128
+
129
+ See `graphify.ingest._yaml_str` for the full rationale; duplicated here to
130
+ avoid pulling the URL-fetching `ingest` module into export's dependency
131
+ graph. Handles backslash, double-quote, all line breaks (\\n, \\r,
132
+ U+2028, U+2029), tab, NUL, and other C0/DEL control characters that
133
+ would otherwise let a hostile `source_file` / `community` / etc. break
134
+ out of the YAML scalar and inject sibling keys.
135
+ """
136
+ if s is None:
137
+ return ""
138
+ out: list[str] = []
139
+ for ch in str(s):
140
+ cp = ord(ch)
141
+ if ch == "\\":
142
+ out.append("\\\\")
143
+ elif ch == '"':
144
+ out.append('\\"')
145
+ elif ch == "\n":
146
+ out.append("\\n")
147
+ elif ch == "\r":
148
+ out.append("\\r")
149
+ elif ch == "\t":
150
+ out.append("\\t")
151
+ elif ch == "\0":
152
+ out.append("\\0")
153
+ elif cp == 0x2028:
154
+ out.append("\\L")
155
+ elif cp == 0x2029:
156
+ out.append("\\P")
157
+ elif cp < 0x20 or cp == 0x7F:
158
+ out.append(f"\\x{cp:02x}")
159
+ else:
160
+ out.append(ch)
161
+ return "".join(out)
162
+
163
+
164
+ from graphify.exporters.base import COMMUNITY_COLORS # noqa: E402,F401
165
+
166
+ from graphify.exporters.html import to_html # noqa: E402,F401
167
+
168
+
169
+ # Fallback scores for an edge that carries a confidence tier but no
170
+ # confidence_score. The INFERRED default was 0.5, which references/extraction-spec.md
171
+ # rules out in as many words — "never omit it, never use 0.5 as a default" — and
172
+ # which is not in the discrete INFERRED set {0.55, 0.65, 0.75, 0.85, 0.95} either.
173
+ # It is now the bottom of that set: a missing score is an absence of evidence
174
+ # about strength, so the honest fallback is the weakest value the rubric allows,
175
+ # not a midpoint that reads as a coin flip (#2813). Every AST emission site now
176
+ # supplies its own score, so this is a backstop rather than a routine path.
177
+ _CONFIDENCE_SCORE_DEFAULTS = {"EXTRACTED": 1.0, "INFERRED": 0.55, "AMBIGUOUS": 0.2}
178
+
179
+
180
+ def attach_hyperedges(G: nx.Graph, hyperedges: list) -> None:
181
+ """Store hyperedges in the graph's metadata dict."""
182
+ existing = G.graph.get("hyperedges", [])
183
+ # Skip id-less persisted entries when seeding the dedup set (#2775): the
184
+ # semantic extractor emits hyperedges with no `id` and build.py persists them
185
+ # verbatim, so a prior graph.json can contain id-less hyperedges. A hard
186
+ # `h["id"]` here raised `KeyError: 'id'` on every incremental re-extract,
187
+ # symmetric with the `.get("id")` guard the loop below already applies to the
188
+ # incoming set.
189
+ seen_ids = {h["id"] for h in existing if h.get("id")}
190
+ for h in hyperedges:
191
+ if h.get("id") and h["id"] not in seen_ids:
192
+ existing.append(h)
193
+ seen_ids.add(h["id"])
194
+ G.graph["hyperedges"] = existing
195
+
196
+
197
+ def _git_head(cwd: "str | Path | None" = None) -> str | None:
198
+ """Return git HEAD for the repo containing ``cwd``, or None outside a repo.
199
+
200
+ ``cwd`` selects the repository to ask, exactly as in watch._git_head
201
+ (#2316). Without it the command inherits the caller's working directory,
202
+ which stamps the *invoking* repo's commit when the graph being written
203
+ describes a different repo — provenance must come from the repo the graph
204
+ describes, so callers pass the graph's own location.
205
+ """
206
+ import subprocess as _sp
207
+ try:
208
+ r = _sp.run(
209
+ ["git", "rev-parse", "HEAD"], capture_output=True, text=True, timeout=3,
210
+ cwd=str(cwd) if cwd is not None else None,
211
+ )
212
+ return r.stdout.strip() if r.returncode == 0 else None
213
+ except Exception:
214
+ return None
215
+
216
+
217
+ # Sentinel: an existing graph.json is present and non-empty but cannot be parsed
218
+ # into a node count (corrupt, mid-write, or structurally wrong). The caller must
219
+ # fail CLOSED on this — the same way to_json's #479 guard refuses to overwrite
220
+ # such a file — because we cannot prove the new graph isn't a silent shrink.
221
+ MALFORMED_GRAPH = object()
222
+
223
+
224
+ def existing_graph_node_count(path: "str | Path"):
225
+ """Node count of an existing graph.json.
226
+
227
+ Returns:
228
+ - an ``int`` node count when the file parses;
229
+ - ``None`` when there is verifiably nothing to protect — absent, empty, or
230
+ over the size cap (matching how :func:`to_json` lets the new graph
231
+ replace an empty/oversized file);
232
+ - :data:`MALFORMED_GRAPH` when the file is present and non-empty but
233
+ unparseable — the caller must treat this as fail-closed (refuse to
234
+ overwrite), mirroring to_json's #479 handling of a corrupt/mid-write file.
235
+
236
+ The raw ``--no-cluster`` write path uses this to apply the same #479 shrink
237
+ guard that :func:`to_json` applies inline for the clustered path.
238
+ """
239
+ p = Path(path)
240
+ if not p.exists():
241
+ return None
242
+ from graphify.security import check_graph_file_size_cap
243
+ try:
244
+ check_graph_file_size_cap(p)
245
+ except Exception:
246
+ # Oversized: reading it to compare would be the DoS the cap guards against.
247
+ return None
248
+ try:
249
+ raw = p.read_text(encoding="utf-8")
250
+ except Exception:
251
+ # Present but unreadable: fail closed if it has bytes, else nothing to lose.
252
+ try:
253
+ return MALFORMED_GRAPH if p.stat().st_size > 0 else None
254
+ except Exception:
255
+ return None
256
+ if not raw.strip():
257
+ return None
258
+ try:
259
+ data = json.loads(raw)
260
+ except Exception:
261
+ return MALFORMED_GRAPH
262
+ nodes = data.get("nodes") if isinstance(data, dict) else None
263
+ return len(nodes) if isinstance(nodes, list) else MALFORMED_GRAPH
264
+
265
+
266
+ def to_json(G: nx.Graph, communities: dict[int, list[str]], output_path: str, *, force: bool = False, built_at_commit: str | None = None, community_labels: dict[int, str] | None = None) -> bool:
267
+ # Safety check: refuse to silently shrink an existing graph (#479)
268
+ existing_path = Path(output_path)
269
+ if not force and existing_path.exists():
270
+ from graphify.security import check_graph_file_size_cap
271
+ try:
272
+ check_graph_file_size_cap(existing_path)
273
+ except Exception:
274
+ # Existing graph.json trips the size cap; reading it to compare would
275
+ # be the very DoS the cap guards against. Can't verify — let the new
276
+ # graph replace the oversized file.
277
+ oversized = True
278
+ else:
279
+ oversized = False
280
+ if not oversized:
281
+ try:
282
+ raw = existing_path.read_text(encoding="utf-8")
283
+ except Exception:
284
+ raw = ""
285
+ if not raw.strip():
286
+ # Empty/whitespace existing file (e.g. a freshly touched path):
287
+ # no nodes to lose, so any new graph is a growth — proceed.
288
+ existing_n = 0
289
+ else:
290
+ try:
291
+ existing_data = json.loads(raw)
292
+ existing_n = len(existing_data.get("nodes", []))
293
+ except Exception as exc:
294
+ # Non-empty but unparseable existing graph (corrupt or a
295
+ # mid-write): we cannot verify the new graph is not a silent
296
+ # shrink. Fail SAFE — refuse rather than overwrite. A
297
+ # fail-OPEN here (the prior behavior) is the silent data-loss
298
+ # path #479 exists to prevent: a transiently unreadable
299
+ # graph.json would let a partial rebuild clobber a good one.
300
+ import sys as _sys
301
+ print(
302
+ f"[graphify] WARNING: existing {existing_path} could not be "
303
+ f"read to verify the new graph is not smaller ({exc}). "
304
+ f"Refusing to overwrite; pass force=True to override.",
305
+ file=_sys.stderr,
306
+ )
307
+ return False
308
+ new_n = G.number_of_nodes()
309
+ if new_n < existing_n:
310
+ import sys as _sys
311
+ print(
312
+ f"[graphify] WARNING: new graph has {new_n} nodes but existing "
313
+ f"graph.json has {existing_n} (net -{existing_n - new_n}). "
314
+ f"Refusing to overwrite. Possible causes: missing chunk files from "
315
+ f"a previous session, or fuzzy dedup collapsed same-named symbols "
316
+ f"across files during an --update on an already-current graph. "
317
+ f"Run a full rebuild (/graphify .) to be safe, or pass force=True "
318
+ f"only if you have verified the reduction is legitimate.",
319
+ file=_sys.stderr,
320
+ )
321
+ return False
322
+
323
+ node_community = _node_community_map(communities)
324
+ _labels: dict[int, str] = {int(k): v for k, v in (community_labels or {}).items()}
325
+ try:
326
+ data = json_graph.node_link_data(G, edges="links")
327
+ except TypeError:
328
+ data = json_graph.node_link_data(G)
329
+
330
+ def _json_sort_key(item: dict) -> str:
331
+ return json.dumps(item, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
332
+
333
+ for node in data["nodes"]:
334
+ cid = node_community.get(node["id"])
335
+ node["community"] = cid
336
+ if cid is not None and _labels:
337
+ node["community_name"] = _labels.get(cid, f"Community {cid}")
338
+ node["norm_label"] = _strip_diacritics(node.get("label", "")).lower()
339
+ for link in data["links"]:
340
+ if "confidence_score" not in link:
341
+ conf = link.get("confidence", "EXTRACTED")
342
+ link["confidence_score"] = _CONFIDENCE_SCORE_DEFAULTS.get(conf, 1.0)
343
+ # Restore original edge direction. Undirected NetworkX storage may
344
+ # canonicalize endpoint order, flipping `calls` and other directional
345
+ # edges in graph.json. The build path stashes the true endpoints in
346
+ # _src/_tgt for exactly this purpose (#563).
347
+ true_src = link.pop("_src", None)
348
+ true_tgt = link.pop("_tgt", None)
349
+ if true_src is not None and true_tgt is not None:
350
+ link["source"] = true_src
351
+ link["target"] = true_tgt
352
+ # Canonicalize the key order WITHIN each node/link dict. node_link_data always
353
+ # appends the node key (`id`) at the end, so a node whose `id` was an inline
354
+ # attribute on a cold build (position varies) lands last after a read-rebuild
355
+ # (build_from_json consumes `id` as the pure node key). The values are
356
+ # identical either way, but the field order churns, so a byte-diff of two
357
+ # equivalent graph.json files is noisy and any position-sensitive consumer
358
+ # sees a spurious change on every round-trip. Emit a stable order — the
359
+ # identity keys first, then the remaining keys sorted — so the serialized
360
+ # form is invariant regardless of how the attribute was stored in memory.
361
+ def _canonical(item: dict, lead: tuple[str, ...]) -> dict:
362
+ leading = [k for k in lead if k in item]
363
+ rest = sorted(k for k in item if k not in leading)
364
+ return {k: item[k] for k in (*leading, *rest)}
365
+
366
+ data["nodes"] = [_canonical(n, ("id", "label")) for n in data["nodes"]]
367
+ data["links"] = [_canonical(link, ("source", "target", "relation")) for link in data["links"]]
368
+ data["nodes"].sort(key=_json_sort_key)
369
+ data["links"].sort(key=_json_sort_key)
370
+ if "hyperedges" not in getattr(G, "graph", {}):
371
+ # Hardening (#2485): a graph with NO hyperedges key at all was built by
372
+ # a path that never engaged hyperedge metadata — distinct from an
373
+ # intentional empty set ([], which build_from_json now stores
374
+ # explicitly after a full-wipeout revalidation). If the file on disk
375
+ # already holds a non-empty set, emptying it without a trace is silent
376
+ # data loss; warn loudly so the wipeout is attributable. We still write
377
+ # the graph's truth rather than preserving the stale set — resurrecting
378
+ # hyperedges whose members may no longer exist would reintroduce the
379
+ # dangling-member shape #1916 removed.
380
+ _prev_hyperedges = None
381
+ try:
382
+ if existing_path.exists():
383
+ from graphify.security import check_graph_file_size_cap
384
+ check_graph_file_size_cap(existing_path)
385
+ _prev = json.loads(existing_path.read_text(encoding="utf-8"))
386
+ if isinstance(_prev, dict):
387
+ _prev_hyperedges = _prev.get("hyperedges")
388
+ except Exception:
389
+ _prev_hyperedges = None
390
+ if _prev_hyperedges:
391
+ print(
392
+ f"[graphify] WARNING: graph carries no hyperedge metadata but "
393
+ f"{existing_path} already holds {len(_prev_hyperedges)} "
394
+ f"hyperedge(s); writing an empty set. Rebuild from the original "
395
+ f"extraction if this is unexpected.",
396
+ file=sys.stderr,
397
+ )
398
+ hyperedges = sorted(getattr(G, "graph", {}).get("hyperedges", []), key=_json_sort_key)
399
+ if isinstance(data.get("graph"), dict) and "hyperedges" in data["graph"]:
400
+ data["graph"]["hyperedges"] = hyperedges
401
+ data["hyperedges"] = hyperedges
402
+ # Fallback provenance comes from the repo the graph is being written INTO
403
+ # (output_path lives in <target>/graphify-out/), never the shell's cwd —
404
+ # the same cwd-anchoring mistake #2316 fixed for `update`.
405
+ commit = built_at_commit if built_at_commit is not None else _git_head(Path(output_path).resolve().parent)
406
+ if commit:
407
+ data["built_at_commit"] = commit
408
+ from graphify.paths import write_json_atomic
409
+ # Atomic write: a crash/ENOSPC mid-write must not truncate a good graph.json.
410
+ write_json_atomic(output_path, data, indent=2)
411
+ return True
412
+
413
+
414
+ def prune_dangling_edges(graph_data: dict) -> tuple[dict, int]:
415
+ """Remove edges whose source or target node is not in the node set.
416
+
417
+ Returns the cleaned graph_data dict and the number of pruned edges.
418
+ """
419
+ node_ids = {n["id"] for n in graph_data["nodes"]}
420
+ links_key = "links" if "links" in graph_data else "edges"
421
+ before = len(graph_data[links_key])
422
+ graph_data[links_key] = [
423
+ e for e in graph_data[links_key]
424
+ if e["source"] in node_ids and e["target"] in node_ids
425
+ ]
426
+ return graph_data, before - len(graph_data[links_key])
427
+
428
+
429
+ def _cypher_escape(s: str) -> str:
430
+ """Escape a string for safe embedding in a Cypher single-quoted literal.
431
+
432
+ Handles all characters that could prematurely terminate the literal or
433
+ inject control sequences:
434
+ - `\\` and `'` (literal terminators)
435
+ - newlines/CRs (would break the per-line statement framing)
436
+ - NUL/control bytes (defensive — Neo4j errors on raw NULs)
437
+
438
+ Also strips any leading/trailing whitespace that would let an attacker
439
+ break the `;`-terminated statement boundary used by `cypher-shell`.
440
+ Closing `}` and `)` are NOT special inside a single-quoted Cypher string,
441
+ so escaping the quote and backslash correctly is sufficient (a `}` inside
442
+ a properly-closed `'...'` literal is just a character) — but we previously
443
+ missed `\\n` / `\\r` which DO let a payload break out of the statement
444
+ line and inject a fresh MATCH/DELETE on the following line. See F-008.
445
+ """
446
+ # First normalise: drop NUL and other C0 control chars except tab.
447
+ s = "".join(ch for ch in s if ch >= " " or ch == "\t")
448
+ return (
449
+ s.replace("\\", "\\\\")
450
+ .replace("'", "\\'")
451
+ .replace("\n", "\\n")
452
+ .replace("\r", "\\r")
453
+ )
454
+
455
+
456
+ # Restrict identifier-position values (labels and relationship types are NOT
457
+ # quoted in Cypher and so cannot be safely escaped — they must be allowlisted).
458
+ _CYPHER_IDENT_RE = re.compile(r"[^A-Za-z0-9_]")
459
+
460
+
461
+ def _cypher_label(raw: str, fallback: str) -> str:
462
+ """Sanitise a value used in identifier position (node label / rel type).
463
+
464
+ Cypher does not provide a way to escape `:Foo` label syntax, so we must
465
+ strip everything except `[A-Za-z0-9_]` and require the result to start
466
+ with a letter; otherwise we fall back to a safe constant.
467
+ """
468
+ cleaned = _CYPHER_IDENT_RE.sub("", raw or "")
469
+ if not cleaned or not cleaned[0].isalpha():
470
+ return fallback
471
+ return cleaned
472
+
473
+
474
+ def to_cypher(G: nx.Graph, output_path: str) -> None:
475
+ lines = ["// Neo4j Cypher import - generated by /graphify", ""]
476
+ for node_id, data in G.nodes(data=True):
477
+ label = _cypher_escape(data.get("label", node_id))
478
+ node_id_esc = _cypher_escape(node_id)
479
+ ftype = _cypher_label(
480
+ (data.get("file_type", "unknown") or "unknown").capitalize(),
481
+ "Entity",
482
+ )
483
+ lines.append(f"MERGE (n:{ftype} {{id: '{node_id_esc}', label: '{label}'}});")
484
+ lines.append("")
485
+ for u, v, data in G.edges(data=True):
486
+ rel = _cypher_label(
487
+ (data.get("relation", "RELATES_TO") or "RELATES_TO").upper(),
488
+ "RELATES_TO",
489
+ )
490
+ conf = _cypher_escape(data.get("confidence", "EXTRACTED"))
491
+ u_esc = _cypher_escape(u)
492
+ v_esc = _cypher_escape(v)
493
+ lines.append(
494
+ f"MATCH (a {{id: '{u_esc}'}}), (b {{id: '{v_esc}'}}) "
495
+ f"MERGE (a)-[:{rel} {{confidence: '{conf}'}}]->(b);"
496
+ )
497
+ with open(output_path, "w", encoding="utf-8") as f: # nosec
498
+ f.write("\n".join(lines))
499
+
500
+
501
+ # Keep backward-compatible alias - skill.md calls generate_html
502
+ generate_html = to_html
503
+
504
+
505
+ # Characters XML 1.0 cannot carry: the C0 controls except tab, LF and CR.
506
+ _XML_ILLEGAL_RE = re.compile("[\x00-\x08\x0b\x0c\x0e-\x1f]")
507
+
508
+
509
+ def _strip_xml_illegal(s: str) -> str:
510
+ """Drop characters XML 1.0 cannot represent, leaving tab/LF/CR intact.
511
+
512
+ ``nx.write_graphml`` raises ``ValueError("All strings must be XML
513
+ compatible: Unicode or ASCII, no NULL bytes or control characters")`` on any
514
+ of them and aborts the whole export over a single label. Labels arrive
515
+ unfiltered from the corpus, so this is ordinary content rather than hostile
516
+ input: an ANSI escape in a markdown heading pasted from a terminal capture,
517
+ or the form feed some Python/Emacs sources use as a section separator
518
+ (#2897).
519
+ """
520
+ return _XML_ILLEGAL_RE.sub("", s)
521
+
522
+
523
+ # C0 controls and DEL, folded to a space when building a filename stem. Windows
524
+ # rejects them in a path outright with OSError EINVAL, so one of them in a label
525
+ # aborted a whole Obsidian vault export; POSIX would accept the name but leave a
526
+ # note nothing can comfortably open (#2897).
527
+ _CONTROL_TO_SPACE_RE = re.compile("[\x00-\x1f\x7f]")
528
+
529
+
530
+ def _cap_filename(s: str, limit: int = 200) -> str:
531
+ """Cap a filename stem to ``limit`` UTF-8 bytes so it stays under the 255-byte
532
+ filesystem limit even after the ``.md`` extension and dedup suffix are added
533
+ (#1094). The cap is on BYTES, not chars, because a label of multibyte
534
+ characters (CJK, accented) can exceed 255 bytes well under 255 chars. When
535
+ truncation happens, an 8-char hash of the full label is appended so two
536
+ distinct labels sharing a long prefix produce distinct, deterministic
537
+ filenames instead of colliding."""
538
+ b = s.encode("utf-8")
539
+ if len(b) <= limit:
540
+ return s
541
+ digest = hashlib.sha1(s.encode("utf-8")).hexdigest()[:8] # nosec - not security
542
+ keep = limit - 9 # "_" + 8 hex chars
543
+ truncated = b[:keep].decode("utf-8", "ignore") # "ignore" drops a split trailing char
544
+ return f"{truncated}_{digest}"
545
+
546
+
547
+ # A frontmatter tag entry in graphify's own namespace, e.g. " - graphify/document".
548
+ _GRAPHIFY_TAG_RE = re.compile(r"^\s*-\s+graphify/\S")
549
+
550
+ # Frontmatter sits at the very top of a note; reading this much is enough to see
551
+ # the whole block without pulling a large note into memory.
552
+ _NOTE_FRONTMATTER_PROBE_BYTES = 4096
553
+
554
+ # Community notes carry no frontmatter; graphify identifies its own by the
555
+ # Dataview query it writes into every one of them.
556
+ _COMMUNITY_QUERY_MARKER = "FROM #community/"
557
+
558
+
559
+ def _is_graphify_note(path: Path) -> bool:
560
+ """Whether a vault note carries graphify's own frontmatter signature.
561
+
562
+ Every note graphify writes opens with a YAML frontmatter block tagging it in
563
+ the ``graphify/`` namespace::
564
+
565
+ ---
566
+ source_file: "d0.md"
567
+ tags:
568
+ - graphify/document
569
+ - graphify/EXTRACTED
570
+ ---
571
+
572
+ Only that block is inspected, and only a tag entry inside it counts — a
573
+ user's note that merely mentions graphify in its prose is not adopted.
574
+
575
+ Community overview notes are recognised separately: they carry no
576
+ frontmatter at all, so they are identified by graphify's own filename prefix
577
+ together with the Dataview query it writes into the body. Requiring both
578
+ keeps a user's own ``_COMMUNITY_*.md`` from being adopted on the name alone.
579
+ """
580
+ try:
581
+ with path.open("r", encoding="utf-8", errors="replace") as fh:
582
+ head = fh.read(_NOTE_FRONTMATTER_PROBE_BYTES)
583
+ except OSError:
584
+ return False
585
+ if path.name.startswith(_COMMUNITY_PREFIX) and _COMMUNITY_QUERY_MARKER in head:
586
+ return True
587
+ if not head.startswith("---"):
588
+ return False
589
+ for line in head.splitlines()[1:]:
590
+ if line.strip() == "---":
591
+ return False # frontmatter closed without a graphify tag
592
+ if _GRAPHIFY_TAG_RE.match(line):
593
+ return True
594
+ return False
595
+
596
+
597
+ def _adopt_pre_manifest_notes(out: Path) -> set[str]:
598
+ """Names of notes in *out* that graphify itself wrote before manifests existed.
599
+
600
+ Deliberately limited to top-level ``*.md``: those are the only files graphify
601
+ can identify as its own from their content. ``.obsidian/graph.json`` is NOT
602
+ adopted — graphify writes one, but so does Obsidian, and with no manifest
603
+ there is no way to tell whose it is. Leaving it unowned keeps the
604
+ conservative behaviour for the one file where guessing wrong would cost the
605
+ user their own vault configuration.
606
+ """
607
+ try:
608
+ candidates = sorted(out.glob("*.md"))
609
+ except OSError:
610
+ return set()
611
+ return {p.name for p in candidates if _is_graphify_note(p)}
612
+
613
+
614
+ def _obsidian_safe_stem(label: str, limit: int = 200) -> str:
615
+ """Filename stem for an Obsidian note / canvas card from a node label.
616
+
617
+ Strips filesystem-unsafe characters, a trailing ``.md``-family extension
618
+ (so ``CLAUDE.md`` does not become ``CLAUDE.md.md``), and a leading ``.`` —
619
+ Obsidian hides every note whose name starts with a dot, so ``.env.md``
620
+ would be written but invisible in the UI (#2205). The ``dot-`` prefix keeps
621
+ the name recognizable; H1 / frontmatter still carry the true label.
622
+ """
623
+ cleaned = re.sub(
624
+ r'[\\/*?:"<>|#^[\]]',
625
+ "",
626
+ # CR/LF were already folded to spaces here; every other C0 control now
627
+ # goes the same way. They are not merely awkward in a filename — Windows
628
+ # rejects them outright, so a single one aborted the whole vault export
629
+ # rather than spoiling one note (#2897).
630
+ _CONTROL_TO_SPACE_RE.sub(" ", label),
631
+ ).strip()
632
+ cleaned = re.sub(r"\.(md|mdx|qmd|markdown)$", "", cleaned, flags=re.IGNORECASE)
633
+ # Obsidian treats a leading-dot filename as a hidden file (#2205). Only
634
+ # prefix when something nameable remains after the dots: an all-dots label
635
+ # like "..." would otherwise become the meaningless stem "dot-" instead of
636
+ # falling through to the "unnamed" guard below (#1409).
637
+ if cleaned.startswith(".") and re.search(r"\w", cleaned.lstrip("."), flags=re.UNICODE):
638
+ cleaned = "dot-" + cleaned.lstrip(".")
639
+ # A stem of only punctuation (e.g. "@", "*", "#") survives the unsafe-char
640
+ # strip above but is empty once a downstream tool re-slugs on word chars
641
+ # (e.g. qmd's handelize() reduces "@" -> "" and raises, aborting the whole
642
+ # `qmd update`). Require at least one word char; else fall back so we never
643
+ # emit a "@.md"-style filename. (#1409)
644
+ if not re.search(r"\w", cleaned, flags=re.UNICODE):
645
+ return "unnamed"
646
+ return _cap_filename(cleaned, limit)
647
+
648
+
649
+ # Room _dedup_node_filenames / the community loop need for a collision suffix
650
+ # ("_1" … "_9999") appended AFTER the stem was capped. The suffix is technically
651
+ # unbounded, but 5 chars ("_" + 4 digits) covers ~10k identical stems, far past
652
+ # anything real; sizing it to 3 digits let a 1000th collision overrun MAX_PATH.
653
+ _DEDUP_SUFFIX_RESERVE = 5
654
+
655
+ # Prefix the community overview notes carry ("_COMMUNITY_Backend.md").
656
+ _COMMUNITY_PREFIX = "_COMMUNITY_"
657
+
658
+
659
+ def _dedup_node_filenames(G: nx.Graph, safe_name) -> dict[str, str]:
660
+ """Map each node_id to a unique note filename, appending a numeric suffix on
661
+ collision. The collision set is keyed on the lowercased name so two labels
662
+ differing only by case (e.g. "References" vs "references") still get distinct
663
+ filenames - on case-insensitive filesystems (macOS/APFS, Windows/NTFS) they
664
+ would otherwise resolve to one path and silently overwrite each other on disk.
665
+ The suffixed candidate is itself re-checked, so a generated "base_1" never
666
+ silently overwrites a node whose literal label is already "base_1"."""
667
+ node_filenames: dict[str, str] = {}
668
+ used: set[str] = set()
669
+ for node_id, data in G.nodes(data=True):
670
+ base = safe_name(data.get("label", node_id))
671
+ candidate = base
672
+ n = 1
673
+ while candidate.lower() in used:
674
+ candidate = f"{base}_{n}"
675
+ n += 1
676
+ used.add(candidate.lower())
677
+ node_filenames[node_id] = candidate
678
+ return node_filenames
679
+
680
+
681
+ def to_obsidian(
682
+ G: nx.Graph,
683
+ communities: dict[int, list[str]],
684
+ output_dir: str,
685
+ community_labels: dict[int, str] | None = None,
686
+ cohesion: dict[int, float] | None = None,
687
+ ) -> int:
688
+ """Export graph as an Obsidian vault - one .md file per node with [[wikilinks]],
689
+ plus one _COMMUNITY_name.md overview note per community (sorted to top by underscore prefix).
690
+
691
+ Open the output directory as a vault in Obsidian to get an interactive
692
+ graph view with community colors and full-text search over node metadata.
693
+
694
+ Returns the number of node notes + community notes written.
695
+ """
696
+ out = Path(output_dir)
697
+ out.mkdir(parents=True, exist_ok=True)
698
+
699
+ # #1506: when the export target is an existing Obsidian vault (a user pointed
700
+ # --obsidian-dir at one), we must not clobber the user's own notes or their
701
+ # .obsidian/ config. Track the files graphify owns in a manifest; a pre-existing
702
+ # file NOT in the manifest is the user's and is never overwritten.
703
+ _manifest_path = out / ".graphify_obsidian_manifest.json"
704
+ try:
705
+ _owned: set[str] = set(json.loads(_manifest_path.read_text(encoding="utf-8")).get("files", []))
706
+ _manifest_existed = True
707
+ except (OSError, ValueError):
708
+ _owned = set()
709
+ _manifest_existed = False
710
+ if not _manifest_existed:
711
+ # A vault written before the manifest existed has no record of what
712
+ # graphify owns, so every note it wrote last time reads as the user's and
713
+ # is skipped. The re-export then writes fresh notes BESIDE the stale ones
714
+ # and the vault carries two generations, with a warning claiming graphify
715
+ # "did not create" files it did (#2863). Adopt the notes that carry
716
+ # graphify's own frontmatter, once, so the manifest starts out honest.
717
+ _owned |= _adopt_pre_manifest_notes(out)
718
+ _written: list[str] = []
719
+ _skipped: list[str] = []
720
+
721
+ def _owned_write(rel_name: str, content: str) -> bool:
722
+ """Write a graphify-owned file, refusing to overwrite a pre-existing file
723
+ graphify didn't create. Returns True if written."""
724
+ target = out / rel_name
725
+ if target.exists() and rel_name not in _owned:
726
+ _skipped.append(rel_name)
727
+ return False
728
+ target.parent.mkdir(parents=True, exist_ok=True)
729
+ write_text_atomic(target, content)
730
+ _written.append(rel_name)
731
+ return True
732
+
733
+ node_community = _node_community_map(communities)
734
+
735
+ # Cap stems against THIS vault's path, not just NAME_MAX: on Windows the
736
+ # 200-byte default plus an ordinary vault directory overruns MAX_PATH and
737
+ # every note write raises FileNotFoundError (#2655). No-op on POSIX.
738
+ _stem_limit = stem_filename_budget(out, reserve=_DEDUP_SUFFIX_RESERVE)
739
+
740
+ # Map node_id → safe filename so wikilinks stay consistent.
741
+ # Deduplicate: if two nodes produce the same filename, append a numeric suffix.
742
+ node_filename = _dedup_node_filenames(
743
+ G, lambda label: _obsidian_safe_stem(label, _stem_limit)
744
+ )
745
+
746
+ # Helper: compute dominant confidence for a node across all its edges
747
+ def _dominant_confidence(node_id: str) -> str:
748
+ confs = []
749
+ for u, v, edata in G.edges(node_id, data=True):
750
+ confs.append(edata.get("confidence", "EXTRACTED"))
751
+ if not confs:
752
+ return "EXTRACTED"
753
+ return Counter(confs).most_common(1)[0][0]
754
+
755
+ # Map file_type → graphify tag
756
+ _FTYPE_TAG = {
757
+ "code": "graphify/code",
758
+ "document": "graphify/document",
759
+ "paper": "graphify/paper",
760
+ "image": "graphify/image",
761
+ }
762
+
763
+ # Write one .md file per node
764
+ node_notes_written = 0
765
+ for node_id, data in G.nodes(data=True):
766
+ label = data.get("label", node_id)
767
+ cid = node_community.get(node_id)
768
+ community_name = (
769
+ community_labels.get(cid, f"Community {cid}")
770
+ if community_labels and cid is not None
771
+ else f"Community {cid}"
772
+ )
773
+
774
+ # Build tags for this node
775
+ ftype = data.get("file_type", "")
776
+ ftype_tag = _FTYPE_TAG.get(ftype, f"graphify/{ftype}" if ftype else "graphify/document")
777
+ dom_conf = _dominant_confidence(node_id)
778
+ conf_tag = f"graphify/{dom_conf}"
779
+ comm_tag = f"community/{_obsidian_tag(community_name)}"
780
+ node_tags = [ftype_tag, conf_tag, comm_tag]
781
+
782
+ lines: list[str] = []
783
+
784
+ # YAML frontmatter - readable in Obsidian's properties panel.
785
+ # All scalars pass through _yaml_str so a hostile source_file or
786
+ # community label cannot break out and inject sibling keys (F-009).
787
+ lines += [
788
+ "---",
789
+ f'source_file: "{_yaml_str(data.get("source_file", ""))}"',
790
+ f'type: "{_yaml_str(ftype)}"',
791
+ f'community: "{_yaml_str(community_name)}"',
792
+ ]
793
+ if data.get("source_location"):
794
+ lines.append(f'location: "{_yaml_str(str(data["source_location"]))}"')
795
+ # Add tags list to frontmatter
796
+ lines.append("tags:")
797
+ for tag in node_tags:
798
+ lines.append(f" - {tag}")
799
+ lines += ["---", "", f"# {label}", ""]
800
+
801
+ # Outgoing edges as wikilinks
802
+ neighbors = list(G.neighbors(node_id))
803
+ if neighbors:
804
+ lines.append("## Connections")
805
+ for neighbor in sorted(neighbors, key=lambda n: G.nodes[n].get("label", n)):
806
+ edata = edge_data(G, node_id, neighbor)
807
+ neighbor_label = node_filename[neighbor]
808
+ relation = edata.get("relation", "")
809
+ confidence = edata.get("confidence", "EXTRACTED")
810
+ lines.append(f"- [[{neighbor_label}]] - `{relation}` [{confidence}]")
811
+ lines.append("")
812
+
813
+ # Inline tags at bottom of note body (for Obsidian tag panel)
814
+ inline_tags = " ".join(f"#{t}" for t in node_tags)
815
+ lines.append(inline_tags)
816
+
817
+ fname = node_filename[node_id] + ".md"
818
+ if _owned_write(fname, "\n".join(lines)):
819
+ node_notes_written += 1
820
+
821
+ # Write one _COMMUNITY_name.md overview note per community
822
+ # Build inter-community edge counts for "Connections to other communities"
823
+ inter_community_edges: dict[int, dict[int, int]] = {}
824
+ for cid in communities:
825
+ inter_community_edges[cid] = {}
826
+ for u, v in G.edges():
827
+ cu = node_community.get(u)
828
+ cv = node_community.get(v)
829
+ if cu is not None and cv is not None and cu != cv:
830
+ inter_community_edges.setdefault(cu, {})
831
+ inter_community_edges.setdefault(cv, {})
832
+ inter_community_edges[cu][cv] = inter_community_edges[cu].get(cv, 0) + 1
833
+ inter_community_edges[cv][cu] = inter_community_edges[cv].get(cu, 0) + 1
834
+
835
+ # Precompute per-node community reach (number of distinct communities a node connects to)
836
+ def _community_reach(node_id: str) -> int:
837
+ neighbor_cids = {
838
+ node_community[nb]
839
+ for nb in G.neighbors(node_id)
840
+ if nb in node_community and node_community[nb] != node_community.get(node_id)
841
+ }
842
+ return len(neighbor_cids)
843
+
844
+ def _community_name(cid) -> str:
845
+ return (
846
+ community_labels.get(cid, f"Community {cid}")
847
+ if community_labels and cid is not None
848
+ else f"Community {cid}"
849
+ )
850
+
851
+ # One case-folded-deduped filename per community, computed once so the note we
852
+ # write and every [[_COMMUNITY_...]] cross-reference resolve to the same file.
853
+ # Two community labels differing only by case (e.g. LLM labels "API" vs "Api")
854
+ # would otherwise overwrite each other on case-insensitive filesystems - and
855
+ # this path had no dedup at all, so even same-case duplicate labels collided.
856
+ community_filename: dict = {}
857
+ used_community: set[str] = set()
858
+ # The community stem carries the "_COMMUNITY_" prefix on top of the dedup
859
+ # suffix, so it gets that much less of the MAX_PATH window (#2655).
860
+ _community_stem_limit = stem_filename_budget(
861
+ out, reserve=_DEDUP_SUFFIX_RESERVE + len(_COMMUNITY_PREFIX)
862
+ )
863
+ for cid in communities:
864
+ base = f"{_COMMUNITY_PREFIX}{_obsidian_safe_stem(_community_name(cid), _community_stem_limit)}"
865
+ candidate = base
866
+ n = 1
867
+ while candidate.lower() in used_community:
868
+ candidate = f"{base}_{n}"
869
+ n += 1
870
+ used_community.add(candidate.lower())
871
+ community_filename[cid] = candidate
872
+
873
+ community_notes_written = 0
874
+ for cid, all_members in communities.items():
875
+ community_name = _community_name(cid)
876
+ # A community's member list can contain ids with no backing node in G
877
+ # (e.g. pruned nodes, stale community assignments from a prior run, or
878
+ # synthesized/merge-artifact ids). Dereferencing those via G.nodes[n] or
879
+ # node_filename[n] raises KeyError and aborts the whole vault export, so
880
+ # skip dangling members rather than crashing (issue #1236).
881
+ members = [m for m in all_members if m in G and m in node_filename]
882
+ n_members = len(members)
883
+ coh_value = cohesion.get(cid) if cohesion else None
884
+
885
+ lines: list[str] = []
886
+
887
+ # YAML frontmatter
888
+ lines.append("---")
889
+ lines.append("type: community")
890
+ if coh_value is not None:
891
+ lines.append(f"cohesion: {coh_value:.2f}")
892
+ lines.append(f"members: {n_members}")
893
+ lines.append("---")
894
+ lines.append("")
895
+ lines.append(f"# {community_name}")
896
+ lines.append("")
897
+
898
+ # Cohesion + member count summary
899
+ if coh_value is not None:
900
+ cohesion_desc = (
901
+ "tightly connected" if coh_value >= 0.7
902
+ else "moderately connected" if coh_value >= 0.4
903
+ else "loosely connected"
904
+ )
905
+ lines.append(f"**Cohesion:** {coh_value:.2f} - {cohesion_desc}")
906
+ lines.append(f"**Members:** {n_members} nodes")
907
+ lines.append("")
908
+
909
+ # Members section
910
+ lines.append("## Members")
911
+ for node_id in sorted(members, key=lambda n: G.nodes[n].get("label", n)):
912
+ data = G.nodes[node_id]
913
+ node_label = node_filename[node_id]
914
+ ftype = data.get("file_type", "")
915
+ source = data.get("source_file", "")
916
+ entry = f"- [[{node_label}]]"
917
+ if ftype:
918
+ entry += f" - {ftype}"
919
+ if source:
920
+ entry += f" - {source}"
921
+ lines.append(entry)
922
+ lines.append("")
923
+
924
+ # Dataview live query (improvement 2)
925
+ comm_tag_name = _obsidian_tag(community_name)
926
+ lines.append("## Live Query (requires Dataview plugin)")
927
+ lines.append("")
928
+ lines.append("```dataview")
929
+ lines.append(f"TABLE source_file, type FROM #community/{comm_tag_name}")
930
+ lines.append("SORT file.name ASC")
931
+ lines.append("```")
932
+ lines.append("")
933
+
934
+ # Connections to other communities
935
+ cross = inter_community_edges.get(cid, {})
936
+ if cross:
937
+ lines.append("## Connections to other communities")
938
+ for other_cid, edge_count in sorted(cross.items(), key=lambda x: -x[1]):
939
+ other_fname = community_filename.get(other_cid) or (
940
+ f"{_COMMUNITY_PREFIX}"
941
+ f"{_obsidian_safe_stem(_community_name(other_cid), _community_stem_limit)}"
942
+ )
943
+ lines.append(f"- {edge_count} edge{'s' if edge_count != 1 else ''} to [[{other_fname}]]")
944
+ lines.append("")
945
+
946
+ # Top bridge nodes - highest degree nodes that connect to other communities
947
+ bridge_nodes = [
948
+ (node_id, G.degree(node_id), _community_reach(node_id))
949
+ for node_id in members
950
+ if _community_reach(node_id) > 0
951
+ ]
952
+ bridge_nodes.sort(key=lambda x: (-x[2], -x[1]))
953
+ top_bridges = bridge_nodes[:5]
954
+ if top_bridges:
955
+ lines.append("## Top bridge nodes")
956
+ for node_id, degree, reach in top_bridges:
957
+ node_label = node_filename[node_id]
958
+ lines.append(
959
+ f"- [[{node_label}]] - degree {degree}, connects to {reach} "
960
+ f"{'community' if reach == 1 else 'communities'}"
961
+ )
962
+
963
+ fname = community_filename[cid] + ".md"
964
+ if _owned_write(fname, "\n".join(lines)):
965
+ community_notes_written += 1
966
+
967
+ # Improvement 4: write .obsidian/graph.json to color nodes by community in graph
968
+ # view — but never clobber an existing .obsidian/graph.json graphify doesn't own
969
+ # (the user's graph-view settings live there). _owned_write handles that and
970
+ # creates the .obsidian/ dir only when it actually writes.
971
+ graph_config = {
972
+ "colorGroups": [
973
+ {
974
+ # Same sanitizer as the note tags (#2862): built from the raw
975
+ # label, the canvas colour group queried a tag that no note
976
+ # carries whenever the label held non-ASCII or punctuation.
977
+ "query": f"tag:#community/{_obsidian_tag(label)}",
978
+ "color": {"a": 1, "rgb": int(COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)].lstrip('#'), 16)}
979
+ }
980
+ for cid, label in sorted((community_labels or {}).items())
981
+ ]
982
+ }
983
+ _owned_write(".obsidian/graph.json", json.dumps(graph_config, indent=2))
984
+
985
+ # #1896: prune notes for nodes that dropped out of the graph. Only files the
986
+ # manifest says graphify owns are candidates, and anything written or skipped
987
+ # this run is excluded — so a user's own note is never touched (foreign files
988
+ # land in _skipped, never _owned). Guard each path to stay inside the vault in
989
+ # case a corrupt/hostile manifest contains `../` entries.
990
+ stale = _owned - set(_written) - set(_skipped)
991
+ pruned = 0
992
+ for rel_name in sorted(stale):
993
+ target = (out / rel_name).resolve()
994
+ if out.resolve() not in target.parents:
995
+ continue
996
+ try:
997
+ target.unlink(missing_ok=True)
998
+ pruned += 1
999
+ except OSError:
1000
+ pass
1001
+ if pruned:
1002
+ print(
1003
+ f"[graphify] pruned {pruned} note(s) for nodes no longer in the graph",
1004
+ file=sys.stderr,
1005
+ )
1006
+
1007
+ # Persist the manifest of files graphify owns, so a re-run can safely update its
1008
+ # own notes while still refusing to touch the user's. Warn (once, aggregated)
1009
+ # about anything skipped to avoid clobbering a pre-existing file.
1010
+ try:
1011
+ _manifest_path.write_text(json.dumps({"files": sorted(set(_written))}, indent=2), encoding="utf-8")
1012
+ except OSError:
1013
+ pass
1014
+ if _skipped:
1015
+ shown = ", ".join(_skipped[:5]) + (f" (+{len(_skipped) - 5} more)" if len(_skipped) > 5 else "")
1016
+ print(
1017
+ f"[graphify] WARNING: skipped {len(_skipped)} pre-existing file(s) graphify "
1018
+ f"did not create, to avoid overwriting your notes: {shown}. "
1019
+ f"Export into an empty directory (or the default graphify-out/obsidian) "
1020
+ f"to get the full vault.",
1021
+ file=sys.stderr,
1022
+ )
1023
+
1024
+ return node_notes_written + community_notes_written
1025
+
1026
+
1027
+ def to_canvas(
1028
+ G: nx.Graph,
1029
+ communities: dict[int, list[str]],
1030
+ output_path: str,
1031
+ community_labels: dict[int, str] | None = None,
1032
+ node_filenames: dict[str, str] | None = None,
1033
+ ) -> None:
1034
+ """Export graph as an Obsidian Canvas file - communities as groups, nodes as cards.
1035
+
1036
+ Generates a structured layout: communities arranged in a grid, nodes within
1037
+ each community arranged in rows. Edges shown between connected nodes.
1038
+ Opens in Obsidian as an infinite canvas with community groupings visible.
1039
+ """
1040
+ # Obsidian canvas color codes (cycle through for communities)
1041
+ CANVAS_COLORS = ["1", "2", "3", "4", "5", "6"] # red, orange, yellow, green, cyan, purple
1042
+
1043
+ # Build node_filenames if not provided (same dedup logic as to_obsidian).
1044
+ # The CLI calls to_canvas without passing the map, so it must derive the
1045
+ # SAME stem budget to keep card links pointing at the notes to_obsidian
1046
+ # wrote — hence budgeting against the canvas's own directory, which is the
1047
+ # vault directory (#2655).
1048
+ _stem_limit = stem_filename_budget(
1049
+ Path(output_path).parent, reserve=_DEDUP_SUFFIX_RESERVE
1050
+ )
1051
+ if node_filenames is None:
1052
+ node_filenames = _dedup_node_filenames(
1053
+ G, lambda label: _obsidian_safe_stem(label, _stem_limit)
1054
+ )
1055
+
1056
+ # Fallback: with no community data (e.g. --no-cluster builds or a missing
1057
+ # analysis sidecar) the grid below produces nothing and the canvas is written
1058
+ # as an empty 32-byte shell on an otherwise populated graph. Emit every node
1059
+ # into one synthetic community so the canvas always reflects the graph (#1324).
1060
+ if not communities and G.number_of_nodes() > 0:
1061
+ communities = {0: [str(n) for n in G.nodes()]}
1062
+
1063
+ num_communities = len(communities)
1064
+ cols = math.ceil(math.sqrt(num_communities)) if num_communities > 0 else 1
1065
+ rows = math.ceil(num_communities / cols) if num_communities > 0 else 1
1066
+
1067
+ canvas_nodes: list[dict] = []
1068
+ canvas_edges: list[dict] = []
1069
+
1070
+ # Lay out communities in a grid
1071
+ gap = 80
1072
+ group_x_offsets: list[int] = []
1073
+ group_y_offsets: list[int] = []
1074
+
1075
+ # Precompute group sizes so we can calculate offsets.
1076
+ # inner_cols is the per-community grid width; the box dimensions AND the node
1077
+ # placement loop below both derive from it, so the cards always fill the box
1078
+ # instead of wrapping into a narrow strip inside an oversized box.
1079
+ sorted_cids = sorted(communities.keys())
1080
+ group_sizes: dict[int, tuple[int, int]] = {}
1081
+ group_cols: dict[int, int] = {}
1082
+ for cid in sorted_cids:
1083
+ # Skip dangling community members with no backing node / filename, so box
1084
+ # sizing matches the cards actually laid out and `G.nodes[m]` never
1085
+ # KeyErrors below — mirrors the to_obsidian guard (#1236).
1086
+ members = [m for m in communities[cid] if m in G and m in node_filenames]
1087
+ n = len(members)
1088
+ inner_cols = max(1, math.ceil(math.sqrt(n)))
1089
+ w = max(600, 220 * inner_cols)
1090
+ h = max(400, 100 * math.ceil(n / inner_cols) + 120)
1091
+ group_sizes[cid] = (w, h)
1092
+ group_cols[cid] = inner_cols
1093
+
1094
+ # Compute cumulative row heights and col widths for grid placement
1095
+ # Each grid cell uses the max width/height in its col/row
1096
+ col_widths: list[int] = []
1097
+ row_heights: list[int] = []
1098
+ for col_idx in range(cols):
1099
+ max_w = 0
1100
+ for row_idx in range(rows):
1101
+ linear = row_idx * cols + col_idx
1102
+ if linear < len(sorted_cids):
1103
+ cid = sorted_cids[linear]
1104
+ w, _ = group_sizes[cid]
1105
+ max_w = max(max_w, w)
1106
+ col_widths.append(max_w)
1107
+
1108
+ for row_idx in range(rows):
1109
+ max_h = 0
1110
+ for col_idx in range(cols):
1111
+ linear = row_idx * cols + col_idx
1112
+ if linear < len(sorted_cids):
1113
+ cid = sorted_cids[linear]
1114
+ _, h = group_sizes[cid]
1115
+ max_h = max(max_h, h)
1116
+ row_heights.append(max_h)
1117
+
1118
+ # Map from cid → (group_x, group_y, group_w, group_h)
1119
+ group_layout: dict[int, tuple[int, int, int, int]] = {}
1120
+ for idx, cid in enumerate(sorted_cids):
1121
+ col_idx = idx % cols
1122
+ row_idx = idx // cols
1123
+ gx = sum(col_widths[:col_idx]) + col_idx * gap
1124
+ gy = sum(row_heights[:row_idx]) + row_idx * gap
1125
+ gw, gh = group_sizes[cid]
1126
+ group_layout[cid] = (gx, gy, gw, gh)
1127
+
1128
+ # Build set of all node_ids in canvas for edge filtering
1129
+ all_canvas_nodes: set[str] = set()
1130
+ for members in communities.values():
1131
+ all_canvas_nodes.update(members)
1132
+
1133
+ # Generate group and node canvas entries
1134
+ for idx, cid in enumerate(sorted_cids):
1135
+ members = communities[cid]
1136
+ community_name = (
1137
+ community_labels.get(cid, f"Community {cid}")
1138
+ if community_labels and cid is not None
1139
+ else f"Community {cid}"
1140
+ )
1141
+ gx, gy, gw, gh = group_layout[cid]
1142
+ canvas_color = CANVAS_COLORS[idx % len(CANVAS_COLORS)]
1143
+
1144
+ # Group node
1145
+ canvas_nodes.append({
1146
+ "id": f"g{cid}",
1147
+ "type": "group",
1148
+ "label": community_name,
1149
+ "x": gx,
1150
+ "y": gy,
1151
+ "width": gw,
1152
+ "height": gh,
1153
+ "color": canvas_color,
1154
+ })
1155
+
1156
+ # Node cards inside the group - laid out in the same ceil(sqrt(n))-column
1157
+ # grid the box was sized for (group_cols[cid]), so cards fill the box.
1158
+ inner_cols = group_cols[cid]
1159
+ # Same dangling-member guard as the sizing loop and to_obsidian (#1236):
1160
+ # a community id absent from G / node_filenames would KeyError the sort.
1161
+ members = [m for m in members if m in G and m in node_filenames]
1162
+ sorted_members = sorted(members, key=lambda n: G.nodes[n].get("label", n))
1163
+ for m_idx, node_id in enumerate(sorted_members):
1164
+ col = m_idx % inner_cols
1165
+ row = m_idx // inner_cols
1166
+ nx_x = gx + 20 + col * (180 + 20)
1167
+ nx_y = gy + 80 + row * (60 + 20)
1168
+ fname = node_filenames.get(
1169
+ node_id,
1170
+ _obsidian_safe_stem(G.nodes[node_id].get("label", node_id), _stem_limit),
1171
+ )
1172
+ canvas_nodes.append({
1173
+ "id": f"n_{node_id}",
1174
+ "type": "file",
1175
+ "file": f"{fname}.md",
1176
+ "x": nx_x,
1177
+ "y": nx_y,
1178
+ "width": 180,
1179
+ "height": 60,
1180
+ })
1181
+
1182
+ # Generate edges - only between nodes both in canvas, cap at 200 highest-weight
1183
+ all_edges_weighted: list[tuple[float, str, str, str]] = []
1184
+ for u, v, edata in G.edges(data=True):
1185
+ if u in all_canvas_nodes and v in all_canvas_nodes:
1186
+ weight = edata.get("weight", 1.0)
1187
+ relation = edata.get("relation", "")
1188
+ conf = edata.get("confidence", "EXTRACTED")
1189
+ label = f"{relation} [{conf}]" if relation else f"[{conf}]"
1190
+ all_edges_weighted.append((weight, u, v, label))
1191
+
1192
+ all_edges_weighted.sort(key=lambda x: -x[0])
1193
+ for weight, u, v, label in all_edges_weighted[:200]:
1194
+ canvas_edges.append({
1195
+ "id": f"e_{u}_{v}",
1196
+ "fromNode": f"n_{u}",
1197
+ "toNode": f"n_{v}",
1198
+ "label": label,
1199
+ })
1200
+
1201
+ canvas_data = {"nodes": canvas_nodes, "edges": canvas_edges}
1202
+ write_json_atomic(output_path, canvas_data, indent=2)
1203
+
1204
+
1205
+ def to_graphml(
1206
+ G: nx.Graph,
1207
+ communities: dict[int, list[str]],
1208
+ output_path: str,
1209
+ ) -> None:
1210
+ """Export graph as GraphML - opens in Gephi, yEd, and any GraphML-compatible tool.
1211
+
1212
+ Community IDs are written as a node attribute so Gephi can colour by community.
1213
+ Edge confidence (EXTRACTED/INFERRED/AMBIGUOUS) is preserved as an edge attribute.
1214
+ """
1215
+ H = G.copy()
1216
+ node_community = _node_community_map(communities)
1217
+ for node_id in H.nodes():
1218
+ H.nodes[node_id]["community"] = node_community.get(node_id, -1)
1219
+ # Drop internal markers (e.g. the AST-provenance "_origin" tag, #1116, and
1220
+ # the "_src"/"_tgt" direction markers) — they are persistence/runtime details,
1221
+ # not graph data, and should not leak into the exported file.
1222
+ for _, attrs in H.nodes(data=True):
1223
+ for k in [k for k in attrs if k.startswith("_")]:
1224
+ del attrs[k]
1225
+ for _, _, attrs in H.edges(data=True):
1226
+ for k in [k for k in attrs if k.startswith("_")]:
1227
+ del attrs[k]
1228
+ # nx.write_graphml only accepts scalar attribute values: None raises, and a
1229
+ # dict/list value (e.g. a per-node `metadata` dict, or the graph-level
1230
+ # `hyperedges` list set by attach_hyperedges()) raises
1231
+ # "GraphML does not support type <class 'dict'/'list'> as data values" (#1831).
1232
+ # Coerce None -> "" and non-scalars -> a JSON string, across all three scopes.
1233
+ def _graphml_safe(val):
1234
+ if val is None:
1235
+ return ""
1236
+ if isinstance(val, bool) or isinstance(val, (int, float)):
1237
+ return val # GraphML-native scalars pass through unchanged
1238
+ if isinstance(val, str):
1239
+ # Scalar, but still has to be XML-representable — see
1240
+ # _strip_xml_illegal. This is the line that turns "one label carried
1241
+ # an ANSI escape" from a lost export into a lost escape character.
1242
+ return _strip_xml_illegal(val)
1243
+ try:
1244
+ return _strip_xml_illegal(json.dumps(val, default=str, sort_keys=True))
1245
+ except (TypeError, ValueError):
1246
+ return _strip_xml_illegal(str(val))
1247
+
1248
+ # Node IDs become the `id` attribute of every <node> and edge endpoint, so
1249
+ # they must be XML-representable too. Normalised ids never carry a control
1250
+ # character, but a caller can hand us a hand-built graph, and a crash here
1251
+ # loses the export just as completely as one in the values.
1252
+ _id_remap = {n: _strip_xml_illegal(n) for n in H.nodes if isinstance(n, str)}
1253
+ _id_remap = {k: v for k, v in _id_remap.items() if k != v}
1254
+ if _id_remap:
1255
+ H = nx.relabel_nodes(H, _id_remap, copy=True)
1256
+
1257
+ for key, val in list(H.graph.items()):
1258
+ H.graph[key] = _graphml_safe(val)
1259
+ for node_id in H.nodes():
1260
+ for key, val in list(H.nodes[node_id].items()):
1261
+ H.nodes[node_id][key] = _graphml_safe(val)
1262
+ for u, v in H.edges():
1263
+ for key, val in list(H.edges[u, v].items()):
1264
+ H.edges[u, v][key] = _graphml_safe(val)
1265
+
1266
+ # Write atomically: a mid-serialization error otherwise leaves a 0-byte
1267
+ # .graphml on disk that downstream tooling mistakes for a completed export
1268
+ # (#1831). Write to a sibling temp file, then replace on success.
1269
+ out = Path(output_path)
1270
+ tmp = out.with_name(out.name + ".tmp")
1271
+ try:
1272
+ nx.write_graphml(H, str(tmp))
1273
+ os.replace(str(tmp), str(out))
1274
+ finally:
1275
+ if tmp.exists():
1276
+ try:
1277
+ tmp.unlink()
1278
+ except OSError:
1279
+ pass
1280
+
1281
+
1282
+ def to_svg(
1283
+ G: nx.Graph,
1284
+ communities: dict[int, list[str]],
1285
+ output_path: str,
1286
+ community_labels: dict[int, str] | None = None,
1287
+ figsize: tuple[int, int] = (20, 14),
1288
+ ) -> None:
1289
+ """Export graph as an SVG file using matplotlib + spring layout.
1290
+
1291
+ Lightweight and embeddable - works in Obsidian notes, Notion, GitHub READMEs,
1292
+ and any markdown renderer. No JavaScript required.
1293
+
1294
+ Node size scales with degree. Community colors match the HTML output.
1295
+ """
1296
+ try:
1297
+ import matplotlib
1298
+ matplotlib.use("Agg")
1299
+ import matplotlib.pyplot as plt
1300
+ import matplotlib.patches as mpatches
1301
+ except ImportError as e:
1302
+ raise ImportError("matplotlib not installed. Run: pip install matplotlib") from e
1303
+
1304
+ node_community = _node_community_map(communities)
1305
+
1306
+ fig, ax = plt.subplots(figsize=figsize, facecolor="#1a1a2e")
1307
+ ax.set_facecolor("#1a1a2e")
1308
+ ax.axis("off")
1309
+
1310
+ pos = nx.spring_layout(G, seed=42, k=2.0 / (G.number_of_nodes() ** 0.5 + 1))
1311
+
1312
+ degree = dict(G.degree())
1313
+ max_deg = max(degree.values(), default=1) or 1
1314
+
1315
+ node_colors = [COMMUNITY_COLORS[node_community.get(n, 0) % len(COMMUNITY_COLORS)] for n in G.nodes()]
1316
+ node_sizes = [300 + 1200 * (degree.get(n, 1) / max_deg) for n in G.nodes()]
1317
+
1318
+ # Draw edges - dashed for non-EXTRACTED
1319
+ for u, v, data in G.edges(data=True):
1320
+ conf = data.get("confidence", "EXTRACTED")
1321
+ style = "solid" if conf == "EXTRACTED" else "dashed"
1322
+ alpha = 0.6 if conf == "EXTRACTED" else 0.3
1323
+ x0, y0 = pos[u]
1324
+ x1, y1 = pos[v]
1325
+ ax.plot([x0, x1], [y0, y1], color="#aaaaaa", linewidth=0.8,
1326
+ linestyle=style, alpha=alpha, zorder=1)
1327
+
1328
+ nx.draw_networkx_nodes(G, pos, ax=ax, node_color=node_colors,
1329
+ node_size=node_sizes, alpha=0.9)
1330
+ nx.draw_networkx_labels(G, pos, ax=ax,
1331
+ labels={n: G.nodes[n].get("label", n) for n in G.nodes()},
1332
+ font_size=7, font_color="white")
1333
+
1334
+ # Legend
1335
+ if community_labels:
1336
+ patches = [
1337
+ mpatches.Patch(
1338
+ color=COMMUNITY_COLORS[cid % len(COMMUNITY_COLORS)],
1339
+ label=f"{label} ({len(communities.get(cid, []))})",
1340
+ )
1341
+ for cid, label in sorted(community_labels.items())
1342
+ ]
1343
+ ax.legend(handles=patches, loc="upper left", framealpha=0.7,
1344
+ facecolor="#2a2a4e", labelcolor="white", fontsize=8)
1345
+
1346
+ plt.tight_layout()
1347
+ plt.savefig(output_path, format="svg", bbox_inches="tight",
1348
+ facecolor=fig.get_facecolor())
1349
+ plt.close(fig)