graphitect 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (336) hide show
  1. graphify/__init__.py +30 -0
  2. graphify/__main__.py +757 -0
  3. graphify/_minhash.py +107 -0
  4. graphify/affected.py +318 -0
  5. graphify/always_on/agents-md.md +12 -0
  6. graphify/always_on/antigravity-rules.md +14 -0
  7. graphify/always_on/claude-md.md +9 -0
  8. graphify/always_on/gemini-md.md +9 -0
  9. graphify/always_on/kiro-steering.md +5 -0
  10. graphify/always_on/vscode-instructions.md +17 -0
  11. graphify/analyze.py +769 -0
  12. graphify/benchmark.py +152 -0
  13. graphify/build.py +2300 -0
  14. graphify/cache.py +1746 -0
  15. graphify/callflow_html.py +2051 -0
  16. graphify/cargo_introspect.py +109 -0
  17. graphify/cli.py +4745 -0
  18. graphify/cluster.py +409 -0
  19. graphify/command-kilo.md +15 -0
  20. graphify/cross_repo_calls.py +216 -0
  21. graphify/cross_repo_types.py +75 -0
  22. graphify/csharp_dispatch.py +154 -0
  23. graphify/dedup.py +1213 -0
  24. graphify/detect.py +2566 -0
  25. graphify/diagnostics.py +406 -0
  26. graphify/export.py +1349 -0
  27. graphify/exporters/__init__.py +1 -0
  28. graphify/exporters/base.py +14 -0
  29. graphify/exporters/graphdb.py +173 -0
  30. graphify/exporters/html.py +637 -0
  31. graphify/extract.py +7856 -0
  32. graphify/extractors/MIGRATION.md +107 -0
  33. graphify/extractors/__init__.py +66 -0
  34. graphify/extractors/apex.py +215 -0
  35. graphify/extractors/base.py +85 -0
  36. graphify/extractors/bash.py +579 -0
  37. graphify/extractors/blade.py +53 -0
  38. graphify/extractors/commonlisp.py +540 -0
  39. graphify/extractors/csharp.py +448 -0
  40. graphify/extractors/dart.py +564 -0
  41. graphify/extractors/dm.py +494 -0
  42. graphify/extractors/elixir.py +241 -0
  43. graphify/extractors/engine.py +6509 -0
  44. graphify/extractors/fortran.py +311 -0
  45. graphify/extractors/go.py +527 -0
  46. graphify/extractors/json_config.py +240 -0
  47. graphify/extractors/julia.py +289 -0
  48. graphify/extractors/markdown.py +408 -0
  49. graphify/extractors/models.py +131 -0
  50. graphify/extractors/objc.py +566 -0
  51. graphify/extractors/ocaml.py +289 -0
  52. graphify/extractors/pascal.py +688 -0
  53. graphify/extractors/pascal_forms.py +196 -0
  54. graphify/extractors/powershell.py +522 -0
  55. graphify/extractors/razor.py +192 -0
  56. graphify/extractors/resolution.py +3584 -0
  57. graphify/extractors/robot.py +296 -0
  58. graphify/extractors/rust.py +470 -0
  59. graphify/extractors/sln.py +92 -0
  60. graphify/extractors/sql.py +720 -0
  61. graphify/extractors/terraform.py +181 -0
  62. graphify/extractors/verilog.py +329 -0
  63. graphify/extractors/zig.py +181 -0
  64. graphify/file_slice.py +246 -0
  65. graphify/global_graph.py +194 -0
  66. graphify/google_workspace.py +237 -0
  67. graphify/hooks.py +933 -0
  68. graphify/ids.py +93 -0
  69. graphify/ingest.py +358 -0
  70. graphify/install.py +2366 -0
  71. graphify/llm.py +3544 -0
  72. graphify/manifest.py +4 -0
  73. graphify/manifest_ingest.py +311 -0
  74. graphify/mcp_ingest.py +386 -0
  75. graphify/multigraph_compat.py +212 -0
  76. graphify/pascal_resolution.py +129 -0
  77. graphify/paths.py +436 -0
  78. graphify/pg_introspect.py +165 -0
  79. graphify/prs.py +770 -0
  80. graphify/querylog.py +80 -0
  81. graphify/reflect.py +882 -0
  82. graphify/report.py +346 -0
  83. graphify/resolver_registry.py +85 -0
  84. graphify/ruby_resolution.py +242 -0
  85. graphify/scip_ingest.py +363 -0
  86. graphify/security.py +460 -0
  87. graphify/semantic_cleanup.py +336 -0
  88. graphify/serve.py +2608 -0
  89. graphify/skill-agents.md +710 -0
  90. graphify/skill-aider.md +1283 -0
  91. graphify/skill-amp.md +710 -0
  92. graphify/skill-claw.md +713 -0
  93. graphify/skill-codex.md +710 -0
  94. graphify/skill-copilot.md +713 -0
  95. graphify/skill-devin.md +1410 -0
  96. graphify/skill-droid.md +710 -0
  97. graphify/skill-kilo.md +722 -0
  98. graphify/skill-kiro.md +713 -0
  99. graphify/skill-opencode.md +705 -0
  100. graphify/skill-pi.md +713 -0
  101. graphify/skill-trae.md +711 -0
  102. graphify/skill-vscode.md +709 -0
  103. graphify/skill-windows.md +755 -0
  104. graphify/skill.md +713 -0
  105. graphify/skills/agents/references/add-watch.md +56 -0
  106. graphify/skills/agents/references/exports.md +87 -0
  107. graphify/skills/agents/references/extraction-spec.md +70 -0
  108. graphify/skills/agents/references/github-and-merge.md +46 -0
  109. graphify/skills/agents/references/hooks.md +33 -0
  110. graphify/skills/agents/references/query.md +311 -0
  111. graphify/skills/agents/references/transcribe.md +52 -0
  112. graphify/skills/agents/references/update.md +210 -0
  113. graphify/skills/amp/references/add-watch.md +56 -0
  114. graphify/skills/amp/references/exports.md +87 -0
  115. graphify/skills/amp/references/extraction-spec.md +70 -0
  116. graphify/skills/amp/references/github-and-merge.md +46 -0
  117. graphify/skills/amp/references/hooks.md +33 -0
  118. graphify/skills/amp/references/query.md +311 -0
  119. graphify/skills/amp/references/transcribe.md +52 -0
  120. graphify/skills/amp/references/update.md +210 -0
  121. graphify/skills/claude/references/add-watch.md +56 -0
  122. graphify/skills/claude/references/exports.md +87 -0
  123. graphify/skills/claude/references/extraction-spec.md +70 -0
  124. graphify/skills/claude/references/github-and-merge.md +46 -0
  125. graphify/skills/claude/references/hooks.md +33 -0
  126. graphify/skills/claude/references/query.md +311 -0
  127. graphify/skills/claude/references/transcribe.md +52 -0
  128. graphify/skills/claude/references/update.md +210 -0
  129. graphify/skills/claw/references/add-watch.md +56 -0
  130. graphify/skills/claw/references/exports.md +87 -0
  131. graphify/skills/claw/references/extraction-spec.md +31 -0
  132. graphify/skills/claw/references/github-and-merge.md +46 -0
  133. graphify/skills/claw/references/hooks.md +33 -0
  134. graphify/skills/claw/references/query.md +311 -0
  135. graphify/skills/claw/references/transcribe.md +52 -0
  136. graphify/skills/claw/references/update.md +210 -0
  137. graphify/skills/codex/references/add-watch.md +56 -0
  138. graphify/skills/codex/references/exports.md +87 -0
  139. graphify/skills/codex/references/extraction-spec.md +31 -0
  140. graphify/skills/codex/references/github-and-merge.md +46 -0
  141. graphify/skills/codex/references/hooks.md +33 -0
  142. graphify/skills/codex/references/query.md +311 -0
  143. graphify/skills/codex/references/transcribe.md +52 -0
  144. graphify/skills/codex/references/update.md +210 -0
  145. graphify/skills/copilot/references/add-watch.md +56 -0
  146. graphify/skills/copilot/references/exports.md +87 -0
  147. graphify/skills/copilot/references/extraction-spec.md +70 -0
  148. graphify/skills/copilot/references/github-and-merge.md +46 -0
  149. graphify/skills/copilot/references/hooks.md +33 -0
  150. graphify/skills/copilot/references/query.md +311 -0
  151. graphify/skills/copilot/references/transcribe.md +52 -0
  152. graphify/skills/copilot/references/update.md +210 -0
  153. graphify/skills/droid/references/add-watch.md +56 -0
  154. graphify/skills/droid/references/exports.md +87 -0
  155. graphify/skills/droid/references/extraction-spec.md +70 -0
  156. graphify/skills/droid/references/github-and-merge.md +46 -0
  157. graphify/skills/droid/references/hooks.md +33 -0
  158. graphify/skills/droid/references/query.md +311 -0
  159. graphify/skills/droid/references/transcribe.md +52 -0
  160. graphify/skills/droid/references/update.md +210 -0
  161. graphify/skills/kilo/references/add-watch.md +56 -0
  162. graphify/skills/kilo/references/exports.md +87 -0
  163. graphify/skills/kilo/references/extraction-spec.md +70 -0
  164. graphify/skills/kilo/references/github-and-merge.md +46 -0
  165. graphify/skills/kilo/references/hooks.md +33 -0
  166. graphify/skills/kilo/references/query.md +311 -0
  167. graphify/skills/kilo/references/transcribe.md +52 -0
  168. graphify/skills/kilo/references/update.md +210 -0
  169. graphify/skills/kiro/references/add-watch.md +56 -0
  170. graphify/skills/kiro/references/exports.md +87 -0
  171. graphify/skills/kiro/references/extraction-spec.md +31 -0
  172. graphify/skills/kiro/references/github-and-merge.md +46 -0
  173. graphify/skills/kiro/references/hooks.md +33 -0
  174. graphify/skills/kiro/references/query.md +311 -0
  175. graphify/skills/kiro/references/transcribe.md +52 -0
  176. graphify/skills/kiro/references/update.md +210 -0
  177. graphify/skills/opencode/references/add-watch.md +56 -0
  178. graphify/skills/opencode/references/exports.md +87 -0
  179. graphify/skills/opencode/references/extraction-spec.md +70 -0
  180. graphify/skills/opencode/references/github-and-merge.md +46 -0
  181. graphify/skills/opencode/references/hooks.md +33 -0
  182. graphify/skills/opencode/references/query.md +311 -0
  183. graphify/skills/opencode/references/transcribe.md +52 -0
  184. graphify/skills/opencode/references/update.md +210 -0
  185. graphify/skills/pi/references/add-watch.md +56 -0
  186. graphify/skills/pi/references/exports.md +87 -0
  187. graphify/skills/pi/references/extraction-spec.md +31 -0
  188. graphify/skills/pi/references/github-and-merge.md +46 -0
  189. graphify/skills/pi/references/hooks.md +33 -0
  190. graphify/skills/pi/references/query.md +311 -0
  191. graphify/skills/pi/references/transcribe.md +52 -0
  192. graphify/skills/pi/references/update.md +210 -0
  193. graphify/skills/trae/references/add-watch.md +56 -0
  194. graphify/skills/trae/references/exports.md +87 -0
  195. graphify/skills/trae/references/extraction-spec.md +70 -0
  196. graphify/skills/trae/references/github-and-merge.md +46 -0
  197. graphify/skills/trae/references/hooks.md +35 -0
  198. graphify/skills/trae/references/query.md +311 -0
  199. graphify/skills/trae/references/transcribe.md +52 -0
  200. graphify/skills/trae/references/update.md +210 -0
  201. graphify/skills/vscode/references/add-watch.md +56 -0
  202. graphify/skills/vscode/references/exports.md +87 -0
  203. graphify/skills/vscode/references/extraction-spec.md +70 -0
  204. graphify/skills/vscode/references/github-and-merge.md +46 -0
  205. graphify/skills/vscode/references/hooks.md +33 -0
  206. graphify/skills/vscode/references/query.md +311 -0
  207. graphify/skills/vscode/references/transcribe.md +52 -0
  208. graphify/skills/vscode/references/update.md +210 -0
  209. graphify/skills/windows/references/add-watch.md +56 -0
  210. graphify/skills/windows/references/exports.md +87 -0
  211. graphify/skills/windows/references/extraction-spec.md +70 -0
  212. graphify/skills/windows/references/github-and-merge.md +46 -0
  213. graphify/skills/windows/references/hooks.md +33 -0
  214. graphify/skills/windows/references/query.md +311 -0
  215. graphify/skills/windows/references/transcribe.md +52 -0
  216. graphify/skills/windows/references/update.md +210 -0
  217. graphify/symbol_resolution.py +556 -0
  218. graphify/transcribe.py +186 -0
  219. graphify/tree_html.py +603 -0
  220. graphify/validate.py +95 -0
  221. graphify/watch.py +2280 -0
  222. graphify/wiki.py +405 -0
  223. graphitect/__init__.py +28 -0
  224. graphitect/__main__.py +4 -0
  225. graphitect/_vendor/__init__.py +2 -0
  226. graphitect/_vendor/archify/LICENSE +22 -0
  227. graphitect/_vendor/archify/SKILL.md +137 -0
  228. graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
  229. graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
  230. graphitect/_vendor/archify/assets/template.html +14935 -0
  231. graphitect/_vendor/archify/bin/archify.mjs +2091 -0
  232. graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
  233. graphitect/_vendor/archify/bin/preview.mjs +653 -0
  234. graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
  235. graphitect/_vendor/archify/brand-marks/README.md +31 -0
  236. graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
  237. graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
  238. graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
  239. graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
  240. graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
  241. graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
  242. graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
  243. graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
  244. graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
  245. graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
  246. graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
  247. graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
  248. graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
  249. graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
  250. graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
  251. graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
  252. graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
  253. graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
  254. graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
  255. graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
  256. graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
  257. graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
  258. graphitect/_vendor/archify/package-lock.json +149 -0
  259. graphitect/_vendor/archify/package.json +39 -0
  260. graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
  261. graphitect/_vendor/archify/references/authoring-contract.md +243 -0
  262. graphitect/_vendor/archify/references/brand-marks.md +65 -0
  263. graphitect/_vendor/archify/references/delivery-contract.md +120 -0
  264. graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
  265. graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
  266. graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
  267. graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
  268. graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
  269. graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
  270. graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
  271. graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
  272. graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
  273. graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
  274. graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
  275. graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
  276. graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
  277. graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
  278. graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
  279. graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
  280. graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
  281. graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
  282. graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
  283. graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
  284. graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
  285. graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
  286. graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
  287. graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
  288. graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
  289. graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
  290. graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
  291. graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
  292. graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
  293. graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
  294. graphitect/_vendor/archify/schemas/README.md +211 -0
  295. graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
  296. graphitect/_vendor/archify/schemas/common.schema.json +115 -0
  297. graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
  298. graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
  299. graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
  300. graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
  301. graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
  302. graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
  303. graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
  304. graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
  305. graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
  306. graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
  307. graphitect/_vendor/archify/skill-release.json +10 -0
  308. graphitect/cli.py +981 -0
  309. graphitect/deliver/__init__.py +5 -0
  310. graphitect/deliver/archify_adapter.py +1877 -0
  311. graphitect/deliver/archify_ir.py +160 -0
  312. graphitect/deliver/archify_repair.py +135 -0
  313. graphitect/deliver/doc_compiler.py +916 -0
  314. graphitect/ground/__init__.py +5 -0
  315. graphitect/ground/describe_source.py +27 -0
  316. graphitect/ground/fullread_source.py +56 -0
  317. graphitect/ground/graphify_source.py +107 -0
  318. graphitect/models.py +118 -0
  319. graphitect/skill/SKILL.md +80 -0
  320. graphitect/skill/agents/openai.yaml +4 -0
  321. graphitect/synthesize/__init__.py +5 -0
  322. graphitect/synthesize/engine.py +281 -0
  323. graphitect/synthesize/llm_backend.py +331 -0
  324. graphitect/synthesize/questions.py +139 -0
  325. graphitect/synthesize/rubric.py +104 -0
  326. graphitect-0.2.0.dist-info/METADATA +284 -0
  327. graphitect-0.2.0.dist-info/RECORD +336 -0
  328. graphitect-0.2.0.dist-info/WHEEL +5 -0
  329. graphitect-0.2.0.dist-info/entry_points.txt +2 -0
  330. graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
  331. graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
  332. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
  333. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
  334. graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
  335. graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
  336. graphitect-0.2.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,1877 @@
1
+ """Maps a GroundedUnderstanding into archify's real architecture IR, then
2
+ shells out to the `archify` CLI to validate and render it.
3
+
4
+ The componentType classification and grid row/col assignment are the two
5
+ things the original plan sketch didn't account for (plan.md §04). Both are
6
+ heuristics here, deliberately simple - a real implementation should let
7
+ Synthesize's own LLM pass override _classify_component with better judgment
8
+ when grounding evidence (e.g. Graphify's file_type/source_file) disagrees.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ import re
15
+ import shlex
16
+ import shutil
17
+ import subprocess
18
+ import warnings
19
+ from collections import Counter
20
+ from pathlib import Path
21
+ from typing import Literal
22
+
23
+ from ..models import GroundedUnderstanding
24
+ from ..synthesize.llm_backend import LLMBackend
25
+ from . import archify_repair
26
+ from .archify_ir import ArchitectureIR, Component, Connection, GridLayout, GuidedView, Meta
27
+
28
+ # Archify is part of the Graphitect wheel. It intentionally has no npm runtime
29
+ # dependencies, so the bundled sources run directly with Node 18+ and never
30
+ # trigger a network install.
31
+ _BUNDLED_ARCHIFY_PATH = (
32
+ Path(__file__).resolve().parents[1] / "_vendor" / "archify" / "bin" / "archify.mjs"
33
+ )
34
+
35
+ _TYPE_KEYWORDS: dict[str, tuple[str, ...]] = {
36
+ "frontend": ("component", "page", "view", ".tsx", ".jsx", "ui/", "frontend/"),
37
+ "database": ("table", "repository", "db", "postgres", "sqlite", "supabase", ".sql"),
38
+ "security": ("auth", "jwt", "oauth", "gate", "verifier"),
39
+ "messagebus": ("queue", "sqs", "kafka", "pubsub", "event bus"),
40
+ "cloud": ("s3", "cdn", "vercel", "cloudfront", "lambda", "cloud run", "k3s", "ec2"),
41
+ "external": ("hook", "human", "external", "third-party", "llm_client", "portfolio_site"),
42
+ }
43
+
44
+
45
+ def _classify_component(node: dict) -> str:
46
+ """First-pass heuristic classifier - see module docstring."""
47
+ haystack = " ".join(
48
+ str(node.get(k, "")) for k in ("id", "label", "source_file", "file_type")
49
+ ).lower()
50
+ for archify_type, keywords in _TYPE_KEYWORDS.items():
51
+ if any(kw in haystack for kw in keywords):
52
+ return archify_type
53
+ return "backend" # default: most graph nodes in a service repo are backend logic
54
+
55
+
56
+ def _assign_grid(nodes: list[dict], edges: list[dict], cols: int = 5) -> dict[str, tuple[int, int]]:
57
+ """Row = longest-path depth from a source node (topological layering).
58
+ Col = stable order within a row, by first-seen index. Deterministic, not
59
+ pretty - a real layout pass would minimize edge crossings within a row.
60
+ """
61
+ node_ids = [n["id"] for n in nodes]
62
+ incoming: dict[str, set[str]] = {nid: set() for nid in node_ids}
63
+ for e in edges:
64
+ src, dst = e.get("from") or e.get("source"), e.get("to") or e.get("target")
65
+ if src in incoming and dst in incoming:
66
+ incoming[dst].add(src)
67
+
68
+ depth: dict[str, int] = {}
69
+
70
+ def _depth_of(nid: str, _seen: frozenset[str] = frozenset()) -> int:
71
+ if nid in depth:
72
+ return depth[nid]
73
+ if nid in _seen: # cycle guard
74
+ return 0
75
+ parents = incoming.get(nid, set())
76
+ d = 0 if not parents else 1 + max(_depth_of(p, _seen | {nid}) for p in parents)
77
+ depth[nid] = d
78
+ return d
79
+
80
+ for nid in node_ids:
81
+ _depth_of(nid)
82
+
83
+ by_depth: dict[int, list[str]] = {}
84
+ for nid in node_ids:
85
+ by_depth.setdefault(depth[nid], []).append(nid)
86
+
87
+ # A depth with more than `cols` members used to wrap via `col % cols`,
88
+ # which silently placed multiple nodes in the exact same (row, col) cell
89
+ # once a depth exceeded `cols` nodes - confirmed live against FundersAI's
90
+ # real graph, where depth-0 alone had more than 5 nodes and archify's
91
+ # layout validator rejected the resulting "c0"/"c10" overlap. Each depth
92
+ # now claims as many grid rows as it needs (ceil(count / cols)), and the
93
+ # next depth's rows start after all of those - no two nodes ever share
94
+ # a cell.
95
+ positions: dict[str, tuple[int, int]] = {}
96
+ next_row = 0
97
+ for d in sorted(by_depth):
98
+ ids_at_depth = by_depth[d]
99
+ rows_needed = -(-len(ids_at_depth) // cols) # ceil division
100
+ for i, nid in enumerate(ids_at_depth):
101
+ positions[nid] = (next_row + i // cols, i % cols)
102
+ next_row += rows_needed
103
+ return positions
104
+
105
+
106
+ def _estimate_box_size(labels: list[str]) -> tuple[int, int]:
107
+ """A uniform box size wide enough to fit the longest label, rather than
108
+ archify's own 120px grid default which is only right for short labels.
109
+
110
+ ~7px/char is a rough estimate for archify's default UI font at its
111
+ default size - not measured against the actual rendered font metrics
112
+ (that would need a real text-measurement pass, out of scope here), but
113
+ confirmed live to fix the specific "label wider than component" failures
114
+ this heuristic was built to address, on a real 12-node graph with labels
115
+ up to "Portfolio Description Agent" (~178px measured by archify itself).
116
+ Uniform rather than per-component sizing: keeps every box in a shared
117
+ grid column the same width, avoiding the inter-column overlaps a
118
+ variable-width box would otherwise risk.
119
+ """
120
+ longest = max((len(label) for label in labels), default=0)
121
+ width = max(120, longest * 7 + 40)
122
+ return width, 60
123
+
124
+
125
+ # archify's own authoring-contract.md recommends "6-12 primary components" -
126
+ # above that, a diagram either fails archify's showcase layout checks or is
127
+ # just unreadable regardless. Mirrors Graphify's own graph.json/export html
128
+ # fallback to an aggregated community view once a raw graph gets too big to
129
+ # render meaningfully (confirmed live during Phase 1 against a 7,885-node
130
+ # real graph), except the trigger here is "readable diagram" (~15 boxes),
131
+ # not "would this crash a force-directed graph renderer" (Graphify's own
132
+ # threshold there is 5,000 nodes - a completely different concern).
133
+ _MAX_RAW_COMPONENTS = 15
134
+
135
+ # Keep the default report readable while retaining a separate complete
136
+ # relationship explorer. The cap is deliberately above the 6-12 primary
137
+ # components Archify recommends, so the overview can still give every
138
+ # connected component one representative relationship on a 15-box graph.
139
+ _OVERVIEW_MAX_CONNECTIONS = 18
140
+
141
+ # Purely mechanical spacing retries render() falls back to when there's no
142
+ # LLM repair_backend - confirmed live (12 Sep 2026) that a genuinely small
143
+ # 5-node graph can still fail archify's validator on tight label spacing
144
+ # alone. Confirmed NOT to help on its own, though: archify's default
145
+ # connector-label placement turned out to sit at a FIXED offset from the
146
+ # FROM component regardless of how much room widening gapY actually opens
147
+ # up (the exact same overlapping label rect at every multiplier tried) - so
148
+ # this is combined with _apply_suggested_label_fixes below, not relied on
149
+ # alone.
150
+ _SPACING_RETRY_MULTIPLIERS = (1.0, 1.75, 2.5)
151
+
152
+ # How many times render()'s no-repair-backend path retries the mechanical
153
+ # label-fix at a single spacing level before giving up and escalating.
154
+ # Confirmed live (12 Sep 2026) on a real dense large graph (FundersAI,
155
+ # aggregated to 15 boxes / ~35 connections): fixing one label overlap
156
+ # reveals or shifts another - archify's validator doesn't necessarily
157
+ # surface every issue in one pass - so this can take several rounds to
158
+ # fully settle. Each fix only ever pins a connection's labelAt to a fixed
159
+ # point (never undoes an earlier fix), so this is guaranteed to converge
160
+ # within at most one round per connection that ever needs fixing, not
161
+ # oscillate - and each round is a cheap local subprocess call, no LLM
162
+ # involved, so a generous budget costs nothing but a little wall time.
163
+ _MAX_LABEL_FIX_ROUNDS = 20
164
+
165
+ # archify's `layout/constraint` (label-overlap) diagnostics carry no
166
+ # structured `subject`/`evidence` identifying which connection is at fault
167
+ # (confirmed live, 12 Sep 2026 - unlike `clean-flow/edge-through-node`,
168
+ # which does) - only the free-text message names the label text and the
169
+ # component it overlaps, and often a concrete "Suggested fix: labelAt
170
+ # [x, y]". These parse that message well enough to apply the fix directly,
171
+ # no LLM needed - archify's own validator already computed the exact
172
+ # answer, this just has to read it.
173
+ _LABEL_OVERLAP_RE = re.compile(r'Label "([^"]+)" overlaps component "([^"]+)"')
174
+ _LABEL_AT_RE = re.compile(r"labelAt \[(-?[\d.]+),\s*(-?[\d.]+)\]")
175
+ _GENERIC_COMMUNITY_RE = re.compile(
176
+ r"^(?:community|cluster)\s*[-_#]?\s*\d+$", re.IGNORECASE
177
+ )
178
+ _TEST_PATH_PARTS = {"test", "tests", "__tests__", "spec", "specs", "fixtures"}
179
+
180
+
181
+ def _normalized_source_file(node: dict) -> str:
182
+ return str(node.get("source_file") or "").replace("\\", "/").strip("/")
183
+
184
+
185
+ def _is_test_node(node: dict) -> bool:
186
+ path = _normalized_source_file(node).lower()
187
+ if not path:
188
+ return False
189
+ parts = path.split("/")
190
+ filename = parts[-1]
191
+ return any(part in _TEST_PATH_PARTS for part in parts) or filename.startswith(
192
+ ("test_", "test-", ".test.", ".spec.")
193
+ )
194
+
195
+
196
+ def _dominant_source_file(nodes: list[dict]) -> str:
197
+ production = [node for node in nodes if not _is_test_node(node)]
198
+ evidence = production or nodes
199
+ paths = [_normalized_source_file(node) for node in evidence]
200
+ paths = [path for path in paths if path]
201
+ return Counter(paths).most_common(1)[0][0] if paths else ""
202
+
203
+
204
+ def _humanize_identifier(value: str) -> str:
205
+ value = re.sub(r"\.(?:py|tsx?|jsx?|json|ya?ml)$", "", value, flags=re.IGNORECASE)
206
+ words = re.sub(r"[_-]+", " ", value).strip().split()
207
+ acronyms = {
208
+ "adk": "ADK",
209
+ "ai": "AI",
210
+ "api": "API",
211
+ "ats": "ATS",
212
+ "db": "DB",
213
+ "http": "HTTP",
214
+ "llm": "LLM",
215
+ "ui": "UI",
216
+ }
217
+ return " ".join(acronyms.get(word.lower(), word.capitalize()) for word in words)
218
+
219
+
220
+ def _has_meaningful_community_label(label: str | None, key: str) -> bool:
221
+ if not label:
222
+ return False
223
+ cleaned = str(label).strip()
224
+ return bool(cleaned) and cleaned.lower() != key.lower() and not _GENERIC_COMMUNITY_RE.fullmatch(
225
+ cleaned
226
+ )
227
+
228
+
229
+ def _label_suffix(nodes: list[dict]) -> str:
230
+ """A concise, source-grounded tie-breaker for duplicate labels."""
231
+ filename = _dominant_source_file(nodes).rsplit("/", 1)[-1]
232
+ stem = re.sub(r"\.[^.]+$", "", filename).lower()
233
+ if stem not in {"", "__init__", "app", "index", "main", "page", "route"}:
234
+ return _humanize_identifier(stem)
235
+
236
+ for node in nodes:
237
+ symbol = re.sub(r"\(.*", "", str(node.get("label") or "")).strip()
238
+ if symbol and len(symbol) <= 40:
239
+ return _humanize_identifier(symbol)
240
+ return "component"
241
+
242
+
243
+ def _disambiguate_component_labels(
244
+ label_of: dict[str, str], members_by_key: dict[str, list[dict]], keys: list[str]
245
+ ) -> None:
246
+ """Keep rendered component labels unique without inventing an LLM name."""
247
+ keys_by_label: dict[str, list[str]] = {}
248
+ for key in keys:
249
+ keys_by_label.setdefault(label_of[key], []).append(key)
250
+
251
+ existing = set(label_of.values())
252
+ for label, duplicate_keys in keys_by_label.items():
253
+ if len(duplicate_keys) < 2:
254
+ continue
255
+
256
+ suffixes = [_label_suffix(members_by_key.get(key, [])) for key in duplicate_keys]
257
+ duplicate_suffixes = Counter(suffixes)
258
+ for index, (key, suffix) in enumerate(zip(duplicate_keys, suffixes), start=1):
259
+ if duplicate_suffixes[suffix] > 1:
260
+ suffix = f"component {index}"
261
+ candidate = f"{label} ({suffix})"
262
+ while candidate in existing:
263
+ index += 1
264
+ candidate = f"{label} (component {index})"
265
+ label_of[key] = candidate
266
+ existing.add(candidate)
267
+
268
+
269
+ def _derive_community_label(nodes: list[dict], key: str) -> str:
270
+ """Name a Graphify community from repository evidence, without an LLM."""
271
+ production = [node for node in nodes if not _is_test_node(node)]
272
+ evidence = production or nodes
273
+ paths = [_normalized_source_file(node) for node in evidence]
274
+ paths = [path for path in paths if path]
275
+ dominant_path = _dominant_source_file(evidence)
276
+ filename = dominant_path.rsplit("/", 1)[-1] if dominant_path else ""
277
+ stem = re.sub(r"\.[^.]+$", "", filename).lower()
278
+ corpus = " ".join(
279
+ str(node.get(field) or "") for node in evidence for field in ("id", "label")
280
+ ).lower()
281
+
282
+ if stem == "freelance_graph" or "talentos // studio" in corpus:
283
+ return "Studio Pipeline"
284
+ if stem == "graph" and ("pipeline" in corpus or "talentos // careers" in corpus):
285
+ return "Careers Pipeline"
286
+
287
+ if "firestore" in dominant_path.lower() or stem == "firestore_store":
288
+ budget_score = sum(
289
+ corpus.count(term)
290
+ for term in ("budget", "run_slot", "claim_run", "reserve", "settle", "release")
291
+ )
292
+ record_score = sum(
293
+ corpus.count(term)
294
+ for term in ("application", "materials", "job", "lead", "client", "profile", "listing")
295
+ )
296
+ if budget_score >= 2 and budget_score > record_score:
297
+ return "Firestore Budgets & Runs"
298
+ if record_score:
299
+ return "Firestore Records"
300
+ return "Firestore"
301
+
302
+ if stem in {"notify", "notification", "notifications"}:
303
+ return "Notifications"
304
+
305
+ frontend_paths = [path.lower() for path in paths if "/frontend/" in f"/{path.lower()}/"]
306
+ if paths and len(frontend_paths) * 2 >= len(paths):
307
+ if stem == "package":
308
+ return "Frontend Dependencies"
309
+ if any("/app/api/" in f"/{path}" for path in frontend_paths) or stem in {
310
+ "cloud-run",
311
+ "auth-server",
312
+ }:
313
+ return "Frontend API"
314
+ # A community may merely import authentication helpers. Prefer the
315
+ # dominant file's role so a generic UI component does not become auth.
316
+ if "auth" in stem or "firebase" in stem:
317
+ return "Frontend Authentication"
318
+ if stem == "types":
319
+ return "Frontend Domain Models"
320
+ if "dashboard" in stem:
321
+ return _humanize_identifier(stem)
322
+ return "Frontend UI"
323
+
324
+ source_paths = [path for path in paths if "/sources/" in f"/{path.lower()}/"]
325
+ if source_paths and len(source_paths) * 2 >= len(paths):
326
+ source_names = {
327
+ "aggregators": "Job Aggregators",
328
+ "ats_boards": "ATS Sources",
329
+ "company_portals": "Company Sources",
330
+ "freelance_boards": "Freelance Sources",
331
+ "profile_sources": "Profile Sources",
332
+ "text_utils": "Source Utilities",
333
+ }
334
+ return source_names.get(stem, "Sources")
335
+
336
+ named_files = {
337
+ "agent": "AI Agents",
338
+ "board_scout": "Board Discovery",
339
+ "matching": "Job Matching",
340
+ "models": "Domain Models",
341
+ "pipeline": "Evaluation Pipeline",
342
+ "resume_render": "Resume Generation",
343
+ "review_groups": "Review Workflows",
344
+ "run_progress": "Run Progress",
345
+ "schemas": "Data Contracts",
346
+ "telemetry": "Observability",
347
+ }
348
+ if stem in named_files:
349
+ return named_files[stem]
350
+ if stem not in {"", "__init__", "app", "index", "main", "page", "route"}:
351
+ return _humanize_identifier(stem)
352
+
353
+ labels = [str(node.get("label") or "").strip() for node in evidence]
354
+ labels = [label for label in labels if label and len(label) <= 48 and not label.startswith(".")]
355
+ return _humanize_identifier(labels[0]) if labels else key
356
+
357
+
358
+ def _apply_suggested_label_fixes(ir: ArchitectureIR, diagnostics: list[dict]) -> ArchitectureIR | None:
359
+ """Best-effort: returns a copy of `ir` with archify's own suggested
360
+ `labelAt` applied to whichever connection each parseable
361
+ `layout/constraint` diagnostic names, or None if no diagnostic could be
362
+ matched to a real connection (nothing to retry with). A diagnostic that
363
+ doesn't parse, or whose named label doesn't match any real connection,
364
+ is skipped rather than treated as fatal - the caller's own retry loop
365
+ still has other spacing attempts left either way.
366
+
367
+ Matches purely by label text, NOT by requiring the overlapped
368
+ component to be one of the connection's own endpoints - confirmed live
369
+ (12 Sep 2026) that this requirement made the fix silently never apply
370
+ at all on a real large graph: a rerouted connection's label can default
371
+ to a position anywhere along its (now much longer) multi-waypoint path,
372
+ landing on and overlapping a totally unrelated third component that
373
+ isn't its `from`/`to` at all - the earlier version's endpoint check
374
+ rejected exactly the one real match available, and the same diagnostic
375
+ kept recurring identically across every retry round because the actual
376
+ offending connection was never touched.
377
+
378
+ Also skips connections that already have `label_at` set. Label text
379
+ alone isn't a reliable unique key either - `_aggregate_by_community`
380
+ generates generic labels like "3 connections" that multiple different
381
+ community pairs can share verbatim, confirmed live on a real graph
382
+ (three separate connections all labeled "12 connections"). Without this
383
+ check, once the first match got fixed, every later round kept
384
+ "fixing" that SAME already-fixed connection again (its label still
385
+ equalled label_text) instead of moving on to the next real offender
386
+ sharing that label - the retry loop looked like it was making no
387
+ progress at all when it was actually just stuck re-patching one
388
+ connection repeatedly.
389
+ """
390
+ connections = list(ir.connections)
391
+ changed = False
392
+ for diag in diagnostics:
393
+ if diag.get("code") != "layout/constraint":
394
+ continue
395
+ message = diag.get("message", "")
396
+ overlap_match = _LABEL_OVERLAP_RE.search(message)
397
+ labelat_match = _LABEL_AT_RE.search(message)
398
+ if not overlap_match or not labelat_match:
399
+ continue
400
+ label_text, _component_id = overlap_match.groups()
401
+ x, y = (float(v) for v in labelat_match.groups())
402
+ for i, conn in enumerate(connections):
403
+ if conn.label == label_text and conn.label_at is None:
404
+ connections[i] = conn.model_copy(update={"label_at": (x, y)})
405
+ changed = True
406
+ break
407
+ if not changed:
408
+ return None
409
+ return ir.model_copy(update={"connections": connections})
410
+
411
+
412
+ def _group_by_community(
413
+ nodes: list[dict], community_labels: dict[str, str]
414
+ ) -> tuple[dict[str, str], dict[str, str], list[str]]:
415
+ """The grouping + overflow-folding step of _aggregate_by_community,
416
+ extracted so node_id_remap() can compute the exact same raw-node-id ->
417
+ diagram-component-id mapping used for the actual rendered diagram,
418
+ without duplicating (and risking drifting from) the aggregation logic
419
+ itself. See _aggregate_by_community's docstring for the *why*.
420
+
421
+ Returns (community_of, label_of, ordered_component_ids).
422
+ """
423
+ community_of: dict[str, str] = {}
424
+ label_of: dict[str, str] = {}
425
+ member_count: dict[str, int] = {}
426
+ members_by_key: dict[str, list[dict]] = {}
427
+ for n in nodes:
428
+ cid = n.get("community")
429
+ key = f"c{cid}" if cid is not None else f"solo_{n['id']}"
430
+ community_of[n["id"]] = key
431
+ members_by_key.setdefault(key, []).append(n)
432
+ member_count[key] = member_count.get(key, 0) + 1
433
+
434
+ for key, members in members_by_key.items():
435
+ cid = members[0].get("community")
436
+ supplied = community_labels.get(str(cid)) if cid is not None else None
437
+ if _has_meaningful_community_label(supplied, key):
438
+ label_of[key] = str(supplied).strip()
439
+ elif cid is None:
440
+ label_of[key] = members[0].get("label", members[0]["id"])
441
+ else:
442
+ label_of[key] = _derive_community_label(members, key)
443
+
444
+ all_keys = list(dict.fromkeys(community_of.values())) # stable order, de-duplicated
445
+ if len(all_keys) > _MAX_RAW_COMPONENTS:
446
+ # Tests often contain more symbols than the production subsystem they
447
+ # exercise. Rank by production members first so test-only communities
448
+ # cannot displace the architecture users are trying to understand.
449
+ production_count = {
450
+ key: sum(not _is_test_node(node) for node in members_by_key[key])
451
+ for key in all_keys
452
+ }
453
+ ranked = sorted(
454
+ all_keys,
455
+ key=lambda key: (production_count[key], member_count[key]),
456
+ reverse=True,
457
+ )
458
+ kept_order = ranked[: _MAX_RAW_COMPONENTS - 1]
459
+ kept = set(kept_order)
460
+ overflow_count = sum(member_count[k] for k in all_keys if k not in kept)
461
+ for nid, key in community_of.items():
462
+ if key not in kept:
463
+ community_of[nid] = "other_components"
464
+ label_of["other_components"] = f"Other components ({overflow_count})"
465
+ all_keys = kept_order + ["other_components"]
466
+
467
+ _disambiguate_component_labels(label_of, members_by_key, all_keys)
468
+ return community_of, label_of, all_keys
469
+
470
+
471
+ def _aggregate_by_community(
472
+ nodes: list[dict], edges: list[dict], community_labels: dict[str, str]
473
+ ) -> tuple[list[dict], list[dict]]:
474
+ """Collapse to one box per Graphify community instead of one box per
475
+ raw AST node, when there are too many of the latter for a readable
476
+ diagram. Real Graphify communities, not an LLM's invented summary -
477
+ exactly Naman's point (12 Sep 2026): a large codebase's full graph
478
+ either won't render or renders unreadably, so fall back to Graphify's
479
+ own community-detection output, which was built to solve exactly this
480
+ problem, rather than asking a model to guess at a smaller structure.
481
+
482
+ Nodes without a `community` field (e.g. non-Graphify grounding) each
483
+ become their own single-node "community" - this aggregation only
484
+ meaningfully reduces node count when real community data is present.
485
+
486
+ A real codebase can have far more communities than _MAX_RAW_COMPONENTS
487
+ allows for a single diagram - confirmed live against FundersAI's actual
488
+ graph.json: 548 communities, not the 2-3 a small unit-test graph has.
489
+ One box per community isn't enough on its own; if there are still too
490
+ many after that first collapse, keep only the largest (by member count,
491
+ a proxy for architectural significance - the same intuition behind
492
+ Graphify's own "god nodes" ranking) and fold everything else into a
493
+ single "Other components" box, rather than emitting hundreds of boxes
494
+ archify will just reject anyway.
495
+ """
496
+ community_of, label_of, all_keys = _group_by_community(nodes, community_labels)
497
+ members_by_component: dict[str, list[dict]] = {key: [] for key in all_keys}
498
+ for node in nodes:
499
+ component_id = community_of[node["id"]]
500
+ members_by_component.setdefault(component_id, []).append(node)
501
+ agg_nodes = [
502
+ {
503
+ "id": key,
504
+ "label": label_of[key] or key,
505
+ "file_type": "community",
506
+ "source_file": _dominant_source_file(members_by_component[key]),
507
+ }
508
+ for key in all_keys
509
+ ]
510
+
511
+ cross_edge_counts: dict[tuple[str, str], int] = {}
512
+ for e in edges:
513
+ src, dst = e.get("from") or e.get("source"), e.get("to") or e.get("target")
514
+ c_src, c_dst = community_of.get(src), community_of.get(dst)
515
+ if not c_src or not c_dst or c_src == c_dst:
516
+ continue # drop intra-community edges - they're implied by sharing a box
517
+ pair = tuple(sorted((c_src, c_dst)))
518
+ cross_edge_counts[pair] = cross_edge_counts.get(pair, 0) + 1
519
+
520
+ agg_edges = [
521
+ {
522
+ "from": a,
523
+ "to": b,
524
+ "label": f"{count} connection" + ("s" if count != 1 else ""),
525
+ }
526
+ for (a, b), count in cross_edge_counts.items()
527
+ ]
528
+ return agg_nodes, agg_edges
529
+
530
+
531
+ def _edge_endpoints(edge: dict) -> tuple[str, str]:
532
+ return (
533
+ str(edge.get("from") or edge.get("source") or ""),
534
+ str(edge.get("to") or edge.get("target") or ""),
535
+ )
536
+
537
+
538
+ def _edge_strength(edge: dict) -> int:
539
+ """Return a deterministic importance proxy for a structural edge.
540
+
541
+ Community aggregation turns many raw imports into a label such as
542
+ ``"12 connections"``. Prefer that measured count for the overview;
543
+ preserve the original edge order only as a final stable tie-breaker.
544
+ """
545
+ for key in ("weight", "count"):
546
+ try:
547
+ value = int(edge.get(key, 0))
548
+ except (TypeError, ValueError):
549
+ value = 0
550
+ if value > 0:
551
+ return value
552
+ match = re.match(r"\s*(\d+)\s+connections?\b", str(edge.get("label") or ""), re.IGNORECASE)
553
+ return int(match.group(1)) if match else 1
554
+
555
+
556
+ def _select_overview_edges(
557
+ nodes: list[dict], edges: list[dict], *, max_connections: int = _OVERVIEW_MAX_CONNECTIONS
558
+ ) -> list[dict]:
559
+ """Keep a compact, deterministic structural subset for the default view.
560
+
561
+ The complete graph remains available in the Full graph mode. Here, first
562
+ choose each component's strongest incident relationship so no connected
563
+ subsystem disappears, then fill the remaining slots by aggregate edge
564
+ strength. This is evidence-preserving filtering, not an inferred summary.
565
+ """
566
+ if len(edges) <= max_connections:
567
+ return list(edges)
568
+
569
+ def sort_key(index: int) -> tuple[int, str, str, str, int]:
570
+ source, target = _edge_endpoints(edges[index])
571
+ first, second = sorted((source, target))
572
+ return (-_edge_strength(edges[index]), first, second, str(edges[index].get("label") or ""), index)
573
+
574
+ ranked = sorted(range(len(edges)), key=sort_key)
575
+ selected: set[int] = set()
576
+ for node_id in (str(node.get("id") or "") for node in nodes):
577
+ if not node_id or len(selected) >= max_connections:
578
+ break
579
+ candidate = next(
580
+ (
581
+ index
582
+ for index in ranked
583
+ if node_id in _edge_endpoints(edges[index]) and index not in selected
584
+ ),
585
+ None,
586
+ )
587
+ if candidate is not None:
588
+ selected.add(candidate)
589
+
590
+ for index in ranked:
591
+ if len(selected) >= max_connections:
592
+ break
593
+ selected.add(index)
594
+
595
+ return [edges[index] for index in ranked if index in selected]
596
+
597
+
598
+ # archify's own grid math (archify/renderers/architecture/grid.mjs,
599
+ # resolveComponentPos/DEFAULT_GRID - read directly from its source, not
600
+ # inferred from rendered output): pos = origin + [col, row] * (cellSize +
601
+ # gap). graphitect never overrides origin, and never used to set gapX
602
+ # explicitly either (silently inheriting archify's own default) - both are
603
+ # made explicit constants here so _route_around_obstacles's geometry is
604
+ # guaranteed to match what archify will actually render, not an assumed
605
+ # implicit default that could drift.
606
+ _GRID_ORIGIN = (40, 80)
607
+ _GRID_GAP_X = 30
608
+ _GRID_COLUMNS = 5
609
+ _VIEWBOX_MARGIN = 40
610
+ _VIEWBOX_LEGEND_RESERVE = 68 # 40px renderer margin + Archify's 28px legend rail
611
+
612
+ # Workflow views deliberately use a short, direct path. A codebase-wide
613
+ # dependency graph is valuable as the Overview/Full rollup, but it is not a
614
+ # presentation path and becomes jumpy when replayed one node at a time.
615
+ _WORKFLOW_MAX_PATH_NODES = 6
616
+ _WORKFLOW_MAX_BRANCHES = 2
617
+ _WORKFLOW_RELATION_ORDER = {
618
+ "calls": 0,
619
+ "invokes": 0,
620
+ "uses": 1,
621
+ "imports": 2,
622
+ "depends_on": 2,
623
+ "contains": 3,
624
+ "method": 4,
625
+ "inherits": 5,
626
+ }
627
+
628
+ # A sequence diagram has a stricter evidence threshold than the workflow
629
+ # story. It is only useful when Graphify observed an ordered chain of actual
630
+ # calls; imports, generic uses, and community links are not request evidence.
631
+ _SEQUENCE_MAX_PARTICIPANTS = 6
632
+ _SEQUENCE_RELATIONS = frozenset({"calls", "invokes"})
633
+
634
+
635
+ def _route_around_obstacles(
636
+ connections: list[Connection],
637
+ positions: dict[str, tuple[int, int]],
638
+ *,
639
+ box_width: int,
640
+ box_height: int,
641
+ cell_w: int,
642
+ cell_h: int,
643
+ gap_x: int,
644
+ gap_y: int,
645
+ cols: int,
646
+ ) -> list[Connection]:
647
+ """Deterministic obstacle-avoiding routing for connections that would
648
+ otherwise have to cross through an unrelated component's box - the real
649
+ fix for `clean-flow/edge-through-node` on large aggregated diagrams
650
+ without an LLM (Naman, 12 Sep 2026: "at least the graph should work"
651
+ for big, complex codebases with no key configured).
652
+
653
+ The grid layout here has no crossing-avoidance of its own. An earlier
654
+ version of this function decided whether to reroute a connection purely
655
+ by row/column distance ("2+ rows apart always gets a detour") -
656
+ confirmed live to reroute far more connections than actually had
657
+ anything in their way, producing a needlessly busy diagram of long
658
+ side-lane detours where archify's own short default route would have
659
+ been perfectly safe. `is_blocked()` below checks the real geometry
660
+ instead: only reroute when some other component's rectangle actually
661
+ overlaps the bounding box between the two endpoints. Rerouted
662
+ connections travel ONLY through space that is provably always empty,
663
+ regardless of how many components exist:
664
+
665
+ - The horizontal strip in the gap between any two adjacent rows (from
666
+ one row's bottom edge to the next row's top edge) contains no
667
+ component at ANY column, since every component in a row shares that
668
+ row's y-range - a horizontal move confined to that strip can never
669
+ cross a box.
670
+ - A vertical "lane" positioned past the last column contains no
671
+ component at any row, for the same reason - a vertical move confined
672
+ to that lane can never cross one either.
673
+
674
+ So a connection more than one row apart exits its own box vertically
675
+ (safe - it's leaving through its own column, stopping before the next
676
+ row's box starts), travels the empty inter-row strip out to the lane,
677
+ travels down/up the empty lane, then reverses through the target's own
678
+ inter-row strip into its box - never once crossing box-body height in
679
+ another column. A same-row, far-column connection is simpler: dip into
680
+ the empty strip below (or above) that row and back up, without needing
681
+ the lane at all - no other row lies between two components in the same
682
+ row. (An earlier version of this router connected box centers directly
683
+ through box-body height and was confirmed live to still cross other
684
+ components - this waypoint choice is the actual fix, not a refinement
685
+ of it.) Short, local connections (adjacent row and/or column) are left
686
+ on archify's own default routing, already confirmed to work at that
687
+ scale.
688
+ """
689
+ ox, oy = _GRID_ORIGIN
690
+ step_x = cell_w + gap_x
691
+ step_y = cell_h + gap_y
692
+ # Safely past the last *occupied* column, rather than after the configured
693
+ # grid width. A 5-column grid can have only four used columns; using the
694
+ # phantom fifth column made routes needlessly wide and, before meta.viewBox
695
+ # was authored, put them outside Archify's component-derived canvas.
696
+ rightmost_col = max((col for _row, col in positions.values()), default=cols - 1)
697
+ lane_x = ox + (rightmost_col + 1) * step_x + _VIEWBOX_MARGIN
698
+
699
+ def rect_of(row: int, col: int) -> tuple[float, float, float, float]:
700
+ x = ox + col * step_x
701
+ y = oy + row * step_y
702
+ return x, y, x + box_width, y + box_height
703
+
704
+ def gap_below(row: int) -> float:
705
+ return oy + row * step_y + box_height + gap_y / 2
706
+
707
+ def gap_above(row: int) -> float:
708
+ return gap_below(row - 1)
709
+
710
+ def x_mid(col: int) -> float:
711
+ return ox + col * step_x + box_width / 2
712
+
713
+ def is_blocked(from_pos: tuple[int, int], to_pos: tuple[int, int]) -> bool:
714
+ """Conservative check: does any OTHER component's rectangle overlap
715
+ the bounding box spanned by the two endpoints? Confirmed live (12
716
+ Sep 2026) that a pure row-distance heuristic ("2+ rows apart always
717
+ needs a detour") reroutes far more connections than actually have
718
+ anything in their way, producing a needlessly cluttered diagram of
719
+ long side-lane detours where a short default route would have been
720
+ perfectly safe. Nothing else even overlapping the rectangle between
721
+ two components means there is nothing for any reasonable route
722
+ between them to cross - safe to leave on archify's own (simpler,
723
+ shorter-looking) default routing.
724
+ """
725
+ from_rect, to_rect = rect_of(*from_pos), rect_of(*to_pos)
726
+ bx1 = min(from_rect[0], to_rect[0])
727
+ by1 = min(from_rect[1], to_rect[1])
728
+ bx2 = max(from_rect[2], to_rect[2])
729
+ by2 = max(from_rect[3], to_rect[3])
730
+ for pos in positions.values():
731
+ if pos == from_pos or pos == to_pos:
732
+ continue
733
+ rx1, ry1, rx2, ry2 = rect_of(*pos)
734
+ if rx1 < bx2 and rx2 > bx1 and ry1 < by2 and ry2 > by1:
735
+ return True
736
+ return False
737
+
738
+ # Confirmed live (12 Sep 2026), twice: every connection crossing the
739
+ # SAME row-boundary used the exact same y at first (gap_below(row)
740
+ # depends only on `row`, not on which connection) - their lines
741
+ # coincided into what looked like one flat "highway" with labels
742
+ # floating on it with no distinguishable line to trace. A first fix
743
+ # (a fixed per-connection step, applied greedily as each connection was
744
+ # visited) was also confirmed live to fall short on a real dense graph:
745
+ # 8px apart reads as one solid band at normal zoom, and a boundary with
746
+ # more connections than the step's clamp allows starts recolliding
747
+ # anyway. The real fix needs to know, before assigning any offset, how
748
+ # many connections will actually share each boundary - hence two passes:
749
+ # first tally how many connections use each boundary, then space that
750
+ # boundary's connections evenly across the FULL safe band (not a fixed
751
+ # step), so N connections sharing a corridor are always maximally and
752
+ # evenly separated regardless of how large N is.
753
+ boundary_counts: dict[int, int] = {}
754
+
755
+ def count_boundary(key: int) -> None:
756
+ boundary_counts[key] = boundary_counts.get(key, 0) + 1
757
+
758
+ decisions: list[
759
+ tuple[Connection, tuple[int, int] | None, tuple[int, int] | None, tuple[int, ...]]
760
+ ] = []
761
+ for conn in connections:
762
+ from_pos = positions.get(conn.from_)
763
+ to_pos = positions.get(conn.to)
764
+ if from_pos is None or to_pos is None or not is_blocked(from_pos, to_pos):
765
+ decisions.append((conn, from_pos, to_pos, ()))
766
+ continue
767
+
768
+ from_row, _ = from_pos
769
+ to_row, _ = to_pos
770
+ row_gap = to_row - from_row
771
+ if row_gap == 0:
772
+ boundary_keys = (from_row,)
773
+ elif abs(row_gap) == 1:
774
+ boundary_keys = (min(from_row, to_row),)
775
+ elif row_gap > 0:
776
+ boundary_keys = (from_row, to_row - 1) # gap_above(to_row) == gap_below(to_row - 1)
777
+ else:
778
+ boundary_keys = (from_row - 1, to_row)
779
+ for key in boundary_keys:
780
+ count_boundary(key)
781
+ decisions.append((conn, from_pos, to_pos, boundary_keys))
782
+
783
+ stagger_max = max(0.0, gap_y / 2 - 12)
784
+ boundary_seen: dict[int, int] = {}
785
+
786
+ def next_offset(key: int) -> float:
787
+ count = boundary_counts.get(key, 1)
788
+ index = boundary_seen.get(key, 0)
789
+ boundary_seen[key] = index + 1
790
+ if count <= 1:
791
+ return 0.0
792
+ # Evenly spaced from -stagger_max to +stagger_max across all
793
+ # `count` connections sharing this boundary - the more that share
794
+ # it, the closer together they sit, but they never collide and
795
+ # never leave the safe band, however many there are.
796
+ return -stagger_max + index * (2 * stagger_max) / (count - 1)
797
+
798
+ routed: list[Connection] = []
799
+ lane_offset = 0
800
+ for conn, from_pos, to_pos, boundary_keys in decisions:
801
+ if not boundary_keys:
802
+ routed.append(conn) # unknown endpoint, or nothing in the way - archify's default is fine
803
+ continue
804
+
805
+ from_row, from_col = from_pos
806
+ to_row, to_col = to_pos
807
+ row_gap = to_row - from_row
808
+
809
+ if row_gap == 0:
810
+ # Same row: no other ROW lies between them, so a simple dip
811
+ # into this row's own inter-row gap strip and back is enough -
812
+ # the lane isn't needed.
813
+ y = gap_below(from_row) + next_offset(boundary_keys[0])
814
+ routed.append(
815
+ conn.model_copy(
816
+ update={
817
+ "from_side": "bottom",
818
+ "to_side": "bottom",
819
+ "via": [(x_mid(from_col), y), (x_mid(to_col), y)],
820
+ }
821
+ )
822
+ )
823
+ continue
824
+
825
+ if abs(row_gap) == 1:
826
+ # Adjacent rows: exactly one shared gap strip lies directly
827
+ # between them - still no need for the lane, just use it.
828
+ y = gap_below(boundary_keys[0]) + next_offset(boundary_keys[0])
829
+ from_side = "bottom" if row_gap > 0 else "top"
830
+ to_side = "top" if row_gap > 0 else "bottom"
831
+ routed.append(
832
+ conn.model_copy(
833
+ update={
834
+ "from_side": from_side,
835
+ "to_side": to_side,
836
+ "via": [(x_mid(from_col), y), (x_mid(to_col), y)],
837
+ }
838
+ )
839
+ )
840
+ continue
841
+
842
+ # 2+ rows apart: no single gap strip touches both endpoints, so
843
+ # bridge between the two different strips via the empty side lane.
844
+ this_lane_x = lane_x + lane_offset
845
+ # Stagger parallel lane routes so a diagram with many of them
846
+ # doesn't stack every route on the exact same vertical line.
847
+ lane_offset = (lane_offset + 20) % 100
848
+ from_boundary, to_boundary = boundary_keys
849
+
850
+ if row_gap > 0:
851
+ from_side, to_side = "bottom", "top"
852
+ from_gap_y = gap_below(from_row) + next_offset(from_boundary)
853
+ to_gap_y = gap_above(to_row) + next_offset(to_boundary)
854
+ else:
855
+ from_side, to_side = "top", "bottom"
856
+ from_gap_y = gap_above(from_row) + next_offset(from_boundary)
857
+ to_gap_y = gap_below(to_row) + next_offset(to_boundary)
858
+
859
+ routed.append(
860
+ conn.model_copy(
861
+ update={
862
+ "from_side": from_side,
863
+ "to_side": to_side,
864
+ "via": [
865
+ (x_mid(from_col), from_gap_y),
866
+ (this_lane_x, from_gap_y),
867
+ (this_lane_x, to_gap_y),
868
+ (x_mid(to_col), to_gap_y),
869
+ ],
870
+ }
871
+ )
872
+ )
873
+ return routed
874
+
875
+
876
+ def _view_box_for_routes(
877
+ positions: dict[str, tuple[int, int]],
878
+ connections: list[Connection],
879
+ *,
880
+ box_width: int,
881
+ box_height: int,
882
+ cell_w: int,
883
+ cell_h: int,
884
+ gap_x: int,
885
+ gap_y: int,
886
+ ) -> tuple[int, int]:
887
+ """Fit components, explicit detours, connection labels, and the legend.
888
+
889
+ Archify's automatic architecture viewBox intentionally measures component
890
+ and boundary boxes only. Graphitect also authors explicit ``via`` points,
891
+ so its canvas must include those points or a valid side lane is visibly
892
+ cropped. The constants mirror Archify's architecture renderer layout.
893
+ """
894
+ ox, oy = _GRID_ORIGIN
895
+ step_x = cell_w + gap_x
896
+ step_y = cell_h + gap_y
897
+ max_component_x = max(
898
+ (ox + col * step_x + box_width for _row, col in positions.values()), default=ox + box_width
899
+ )
900
+ max_component_y = max(
901
+ (oy + row * step_y + box_height for row, _col in positions.values()), default=oy + box_height
902
+ )
903
+ route_points = [point for connection in connections for point in (connection.via or [])]
904
+ max_route_x = max((point[0] for point in route_points), default=max_component_x)
905
+ max_route_y = max((point[1] for point in route_points), default=max_component_y)
906
+ label_half_width = max(
907
+ (max(30.0, len(connection.label or "") * 4.8 + 10.0) / 2 for connection in connections),
908
+ default=0.0,
909
+ )
910
+ right_padding = max(float(_VIEWBOX_MARGIN), label_half_width + 14.0)
911
+ return (
912
+ max(320, int(max(max_component_x, max_route_x) + right_padding + 0.999)),
913
+ max(240, int(max(max_component_y, max_route_y) + _VIEWBOX_LEGEND_RESERVE + 0.999)),
914
+ )
915
+
916
+
917
+ def node_id_remap(understanding: GroundedUnderstanding) -> dict[str, str]:
918
+ """Maps every raw Graphify node id to the diagram component id it
919
+ actually renders as - itself, unchanged, when node count is at or under
920
+ _MAX_RAW_COMPONENTS (to_architecture_ir never aggregates in that case),
921
+ or its community/"other_components" box id when it does.
922
+
923
+ A doc section's `related_node_ids` are set once during Synthesize using
924
+ real raw node ids - correct for citing evidence, but once aggregation
925
+ kicks in on a large graph those exact ids no longer exist as elements in
926
+ the rendered SVG, so a node-ref pill's hover-highlight silently finds
927
+ nothing (a known gap flagged in plan.md, fixed by having doc_compiler
928
+ look elements up via this remap rather than the raw id directly).
929
+ """
930
+ nodes = understanding.nodes
931
+ if len(nodes) <= _MAX_RAW_COMPONENTS:
932
+ return {n["id"]: n["id"] for n in nodes if "id" in n}
933
+ community_of, _, _ = _group_by_community(nodes, understanding.community_labels)
934
+ return community_of
935
+
936
+
937
+ def _story_view_id(heading: str, index: int) -> str:
938
+ """Stable, schema-safe chapter id derived from an existing doc heading."""
939
+ slug = re.sub(r"[^a-z0-9]+", "-", heading.lower()).strip("-")
940
+ return slug or f"chapter-{index + 1}"
941
+
942
+
943
+ def _component_adjacency(component_ids: set[str], edges: list[dict]) -> dict[str, set[str]]:
944
+ """Undirected adjacency for a walk over exact authored relationships.
945
+
946
+ The story viewer separately identifies each hop as forward or reverse.
947
+ This helper only decides whether two consecutive focus nodes have a real
948
+ relationship at all; it never invents a transitive hop from proximity.
949
+ """
950
+ adjacency = {component_id: set() for component_id in component_ids}
951
+ for edge in edges:
952
+ source, target = _edge_endpoints(edge)
953
+ if source in adjacency and target in adjacency:
954
+ adjacency[source].add(target)
955
+ adjacency[target].add(source)
956
+ return adjacency
957
+
958
+
959
+ def _continuous_story_path(
960
+ adjacency: dict[str, set[str]],
961
+ component_ids: set[str],
962
+ *,
963
+ required_id: str | None = None,
964
+ max_nodes: int = 5,
965
+ ) -> list[str]:
966
+ """Pick a bounded simple path whose every adjacent pair is observed.
967
+
968
+ `meta.views.focus` is also Archify's Story Beat order. Supplying a hub
969
+ followed by unrelated neighbours made the viewer honestly report
970
+ grouped/no-direct-link transitions, which reads as a jump. A path keeps
971
+ every beat on a real connection and leaves the viewer to truthfully mark
972
+ that connection's direction.
973
+ """
974
+ eligible = set(component_ids)
975
+ if len(eligible) > 1:
976
+ eligible.discard("other_components")
977
+ if not eligible:
978
+ return []
979
+ if required_id is not None and required_id not in eligible:
980
+ return [required_id] if required_id in component_ids else []
981
+
982
+ best: tuple[str, ...] = ()
983
+
984
+ def consider(path: list[str]) -> None:
985
+ nonlocal best
986
+ if required_id is not None and required_id not in path:
987
+ return
988
+ candidate = tuple(path)
989
+ if len(candidate) > len(best) or (len(candidate) == len(best) and candidate < best):
990
+ best = candidate
991
+
992
+ def visit(current: str, path: list[str]) -> None:
993
+ consider(path)
994
+ if len(path) == max_nodes:
995
+ return
996
+ for neighbour in sorted(adjacency.get(current, set()) & eligible):
997
+ if neighbour not in path:
998
+ visit(neighbour, [*path, neighbour])
999
+
1000
+ for start in sorted(eligible):
1001
+ visit(start, [start])
1002
+ return list(best)
1003
+
1004
+
1005
+ def _story_views(
1006
+ understanding: GroundedUnderstanding,
1007
+ components: list[Component],
1008
+ all_edges: list[dict],
1009
+ ) -> list[GuidedView]:
1010
+ """Create guided chapters without inferring an execution path.
1011
+
1012
+ LLM-backed reports focus repository components explicitly cited by a doc
1013
+ section. Diagram-only reports use connected walks around prominent
1014
+ components. In both cases, adjacent Story beats share an actual graph
1015
+ relationship rather than jumping among a hub's unrelated neighbours.
1016
+ """
1017
+ component_ids = {component.id for component in components}
1018
+ remap = node_id_remap(understanding)
1019
+ adjacency = _component_adjacency(component_ids, all_edges)
1020
+ views: list[GuidedView] = []
1021
+ seen_paths: set[frozenset[str]] = set()
1022
+
1023
+ for index, section in enumerate(understanding.doc):
1024
+ cited_components = set(
1025
+ dict.fromkeys(
1026
+ remap.get(node_id, node_id)
1027
+ for node_id in section.related_node_ids
1028
+ if remap.get(node_id, node_id) in component_ids
1029
+ )
1030
+ )
1031
+ path = _continuous_story_path(adjacency, cited_components)
1032
+ path_key = frozenset(path)
1033
+ if not path or path_key in seen_paths:
1034
+ continue
1035
+ seen_paths.add(path_key)
1036
+ views.append(
1037
+ GuidedView(
1038
+ id=_story_view_id(section.heading, index),
1039
+ label=section.heading,
1040
+ focus=path,
1041
+ note=f"Follows directly observed relationships cited in the {section.heading} explanation.",
1042
+ )
1043
+ )
1044
+ if len(views) == 5:
1045
+ return views
1046
+
1047
+ if views:
1048
+ return views
1049
+
1050
+ # The aggregation catch-all can touch almost every community. It is useful
1051
+ # in the map but a bad story anchor because its focus would dim nothing.
1052
+ # Prefer named components whenever any are available.
1053
+ anchors = [component for component in components if component.id != "other_components"] or components
1054
+ component_by_id = {component.id: component for component in components}
1055
+ ranked = sorted(
1056
+ anchors,
1057
+ key=lambda component: (-len(adjacency[component.id]), component.label.lower()),
1058
+ )
1059
+ for index, component in enumerate(ranked[:3]):
1060
+ path = _continuous_story_path(
1061
+ adjacency,
1062
+ set(component_by_id),
1063
+ required_id=component.id,
1064
+ )[:5]
1065
+ path_key = frozenset(path)
1066
+ if not path or path_key in seen_paths:
1067
+ continue
1068
+ seen_paths.add(path_key)
1069
+ views.append(
1070
+ GuidedView(
1071
+ id=f"component-{index + 1}",
1072
+ label=component.label,
1073
+ focus=path,
1074
+ note="Follows a connected path through directly observed architecture relationships.",
1075
+ )
1076
+ )
1077
+ return views
1078
+
1079
+
1080
+ def _workflow_relation_key(edge: dict) -> tuple[int, str, str, str]:
1081
+ """Keep workflow selection deterministic while preferring executable links."""
1082
+ source, target = _edge_endpoints(edge)
1083
+ relation = str(edge.get("relation") or edge.get("label") or "relates to").lower()
1084
+ return (_WORKFLOW_RELATION_ORDER.get(relation, 99), relation, source, target)
1085
+
1086
+
1087
+ def _workflow_source_label(node: dict) -> str:
1088
+ """A compact, code-grounded subtitle for a workflow node."""
1089
+ source_file = _normalized_source_file(node)
1090
+ if not source_file:
1091
+ return "Observed source component"
1092
+ return source_file.rsplit("/", 1)[-1]
1093
+
1094
+
1095
+ def _workflow_display_label(node: dict) -> str:
1096
+ """Turn a raw code symbol into a readable, bounded diagram label."""
1097
+ raw = str(node.get("label") or node.get("id") or "")
1098
+ # A source helper named ``function_ollama_*`` is not, by itself, proof
1099
+ # that the deployed workflow uses Ollama. Keep the exact symbol in the
1100
+ # hover card and label the diagram block by its code-level responsibility.
1101
+ if raw.casefold().startswith("function_ollama_"):
1102
+ return "LLM Provider Adapter"
1103
+ raw = re.sub(r"^function_", "", raw, flags=re.IGNORECASE)
1104
+ words = re.sub(r"([a-z0-9])([A-Z])", r"\1 \2", raw)
1105
+ words = re.sub(r"[^A-Za-z0-9]+", " ", words).strip().split()
1106
+ label_words: list[str] = []
1107
+ for word in words:
1108
+ candidate = " ".join([*label_words, word])
1109
+ if label_words and len(candidate) > 20:
1110
+ break
1111
+ label_words.append(word)
1112
+ return " ".join(label_words).title() or "Observed Component"
1113
+
1114
+
1115
+ def workflow_hover_details(understanding: GroundedUnderstanding) -> dict[str, dict[str, str]]:
1116
+ """Return source-grounded hover details for the selected workflow path.
1117
+
1118
+ These details are embedded only in Graphitect's report iframe. The
1119
+ delivered Archify artifact remains byte-for-byte unchanged and validates
1120
+ against Archify's workflow schema on its own.
1121
+ """
1122
+ path, path_edges = _workflow_path(understanding)
1123
+ node_by_id = {str(node["id"]): node for node in understanding.nodes if node.get("id")}
1124
+ details: dict[str, dict[str, str]] = {}
1125
+ for index, node_id in enumerate(path):
1126
+ node = node_by_id[node_id]
1127
+ raw_symbol = str(node.get("label") or node_id)
1128
+ source_file = _normalized_source_file(node)
1129
+ if index == 0:
1130
+ summary = "Entry in the selected direct code path."
1131
+ elif index == len(path) - 1:
1132
+ summary = "Endpoint of the selected direct code path."
1133
+ else:
1134
+ incoming = str(path_edges[index - 1].get("relation") or "relationship")
1135
+ outgoing = str(path_edges[index].get("relation") or "relationship")
1136
+ summary = (
1137
+ f"Reached through a direct {incoming} relationship and continues "
1138
+ f"through a direct {outgoing} relationship."
1139
+ )
1140
+ if raw_symbol.casefold().startswith("function_ollama_"):
1141
+ summary = (
1142
+ "Code-level LLM provider adapter. Its historical source name is not "
1143
+ "evidence that a particular provider is used."
1144
+ )
1145
+ detail = {"symbol": raw_symbol, "summary": summary}
1146
+ if source_file:
1147
+ detail["source"] = source_file
1148
+ details[f"step-{index + 1}"] = detail
1149
+ return details
1150
+
1151
+
1152
+ def _workflow_node_width(node: dict) -> int:
1153
+ """Leave enough room for exact code symbols without making a giant card."""
1154
+ label = _workflow_display_label(node)
1155
+ return min(220, max(132, len(label) * 8 + 32))
1156
+
1157
+
1158
+ def _workflow_seed_ids(understanding: GroundedUnderstanding, node_ids: set[str]) -> list[str]:
1159
+ """Prioritise components the explanation calls a workflow, if available."""
1160
+ workflow_sections = [
1161
+ section for section in understanding.doc if section.heading.lower() == "key workflows"
1162
+ ]
1163
+ ordered_sections = [*workflow_sections, *understanding.doc]
1164
+ seeds = [
1165
+ node_id
1166
+ for section in ordered_sections
1167
+ for node_id in section.related_node_ids
1168
+ if node_id in node_ids
1169
+ ]
1170
+ return list(dict.fromkeys(seeds))
1171
+
1172
+
1173
+ def _choose_workflow_edge(
1174
+ candidates: list[dict], seen_nodes: set[str], *, incoming: bool
1175
+ ) -> dict | None:
1176
+ valid = []
1177
+ for edge in candidates:
1178
+ source, target = _edge_endpoints(edge)
1179
+ next_node = source if incoming else target
1180
+ if next_node and next_node not in seen_nodes:
1181
+ valid.append(edge)
1182
+ return min(valid, key=_workflow_relation_key) if valid else None
1183
+
1184
+
1185
+ def _workflow_path(understanding: GroundedUnderstanding) -> tuple[list[str], list[dict]]:
1186
+ """Find one short connected code path without inferring runtime behavior.
1187
+
1188
+ The path follows only direct Graphify edges. It is intentionally not a
1189
+ global longest path: a six-node slice can be read in presentation mode,
1190
+ while a repository-wide path would be both unstable and misleading.
1191
+ """
1192
+ node_by_id = {str(node.get("id") or ""): node for node in understanding.nodes}
1193
+ node_by_id = {
1194
+ node_id: node for node_id, node in node_by_id.items() if node_id and not _is_test_node(node)
1195
+ }
1196
+ if not node_by_id:
1197
+ raise ValueError("workflow story needs at least one non-test code node")
1198
+
1199
+ incoming: dict[str, list[dict]] = {node_id: [] for node_id in node_by_id}
1200
+ outgoing: dict[str, list[dict]] = {node_id: [] for node_id in node_by_id}
1201
+ for edge in understanding.edges:
1202
+ source, target = _edge_endpoints(edge)
1203
+ if source in node_by_id and target in node_by_id and source != target:
1204
+ outgoing[source].append(edge)
1205
+ incoming[target].append(edge)
1206
+ for edges in [*incoming.values(), *outgoing.values()]:
1207
+ edges.sort(key=_workflow_relation_key)
1208
+
1209
+ cited_seeds = _workflow_seed_ids(understanding, set(node_by_id))
1210
+ ranked_nodes = sorted(
1211
+ node_by_id,
1212
+ key=lambda node_id: (
1213
+ -(len(incoming[node_id]) + len(outgoing[node_id])),
1214
+ node_id,
1215
+ ),
1216
+ )
1217
+ seeds = list(dict.fromkeys([*cited_seeds, *ranked_nodes]))
1218
+ best_path: list[str] = []
1219
+ best_edges: list[dict] = []
1220
+ best_score: tuple[int, int, int] = (-1, -1, -1)
1221
+
1222
+ for seed in seeds:
1223
+ path = [seed]
1224
+ path_edges: list[dict] = []
1225
+ seen = {seed}
1226
+
1227
+ # A caller of a cited component makes a better beginning than the
1228
+ # component itself, when Graphify observed one. Then extend forward.
1229
+ first = _choose_workflow_edge(incoming[seed], seen, incoming=True)
1230
+ if first is not None:
1231
+ source, _ = _edge_endpoints(first)
1232
+ path.insert(0, source)
1233
+ path_edges.insert(0, first)
1234
+ seen.add(source)
1235
+
1236
+ while len(path) < _WORKFLOW_MAX_PATH_NODES:
1237
+ next_edge = _choose_workflow_edge(outgoing[path[-1]], seen, incoming=False)
1238
+ if next_edge is None:
1239
+ break
1240
+ _, target = _edge_endpoints(next_edge)
1241
+ path.append(target)
1242
+ path_edges.append(next_edge)
1243
+ seen.add(target)
1244
+
1245
+ # A sink-only cited component may still have a useful direct caller.
1246
+ if len(path) == 1:
1247
+ next_edge = _choose_workflow_edge(outgoing[seed], seen, incoming=False)
1248
+ if next_edge is not None:
1249
+ _, target = _edge_endpoints(next_edge)
1250
+ path.append(target)
1251
+ path_edges.append(next_edge)
1252
+
1253
+ relation_quality = -sum(_workflow_relation_key(edge)[0] for edge in path_edges)
1254
+ score = (
1255
+ sum(node_id in cited_seeds for node_id in path),
1256
+ len(path),
1257
+ relation_quality,
1258
+ )
1259
+ if score > best_score:
1260
+ best_path, best_edges, best_score = path, path_edges, score
1261
+
1262
+ if len(best_path) >= 2:
1263
+ return best_path, best_edges
1264
+
1265
+ # A graph with no cited anchors or executable chain still deserves a
1266
+ # workflow-like view if Graphify observed any relationship at all.
1267
+ for source in ranked_nodes:
1268
+ if outgoing[source]:
1269
+ edge = outgoing[source][0]
1270
+ _, target = _edge_endpoints(edge)
1271
+ return [source, target], [edge]
1272
+ raise ValueError("workflow story needs at least one direct code relationship")
1273
+
1274
+
1275
+ def to_workflow_spec(understanding: GroundedUnderstanding, title: str) -> dict:
1276
+ """Build a compact Archify workflow artifact from direct graph evidence.
1277
+
1278
+ This is intentionally a separate diagram type from the architecture
1279
+ rollup. Its main path is a continuous, bounded sequence of observed
1280
+ edges; it never turns community aggregates or transitive guesses into a
1281
+ narrated flow.
1282
+ """
1283
+ if understanding.diagram_kind != "architecture":
1284
+ raise ValueError(
1285
+ f"to_workflow_spec only handles diagram_kind='architecture', got "
1286
+ f"{understanding.diagram_kind!r}"
1287
+ )
1288
+
1289
+ path, path_edges = _workflow_path(understanding)
1290
+ node_by_id = {str(node["id"]): node for node in understanding.nodes if node.get("id")}
1291
+ path_set = set(path)
1292
+ node_ids = {node_id: f"step-{index + 1}" for index, node_id in enumerate(path)}
1293
+ last_col = 5 if len(path) > 1 else 0
1294
+
1295
+ def column(index: int) -> int:
1296
+ return round(index * last_col / (len(path) - 1)) if len(path) > 1 else 0
1297
+
1298
+ nodes = [
1299
+ {
1300
+ "id": node_ids[node_id],
1301
+ "lane": "path",
1302
+ "col": column(index),
1303
+ "type": _classify_component(node_by_id[node_id]),
1304
+ "label": _workflow_display_label(node_by_id[node_id]),
1305
+ "sublabel": _workflow_source_label(node_by_id[node_id]),
1306
+ "width": _workflow_node_width(node_by_id[node_id]),
1307
+ }
1308
+ for index, node_id in enumerate(path)
1309
+ ]
1310
+ edges = [
1311
+ {
1312
+ "id": f"path-{index}",
1313
+ "from": node_ids[source],
1314
+ "to": node_ids[target],
1315
+ "label": str(edge.get("relation") or "observed relationship"),
1316
+ "variant": "emphasis" if index == 1 else "default",
1317
+ "role": "main",
1318
+ }
1319
+ for index, (edge, source, target) in enumerate(
1320
+ zip(path_edges, path, path[1:]), start=1
1321
+ )
1322
+ ]
1323
+
1324
+ # Add two genuine, directly linked supporting nodes at most. They give
1325
+ # the workflow a second lane without turning it back into a dense map.
1326
+ branch_count = 0
1327
+ for index, node_id in enumerate(path):
1328
+ if branch_count == _WORKFLOW_MAX_BRANCHES:
1329
+ break
1330
+ for edge in sorted(understanding.edges, key=_workflow_relation_key):
1331
+ source, target = _edge_endpoints(edge)
1332
+ other = target if source == node_id else source if target == node_id else ""
1333
+ if not other or other in path_set or other not in node_by_id:
1334
+ continue
1335
+ if _is_test_node(node_by_id[other]) or not _normalized_source_file(node_by_id[other]):
1336
+ continue
1337
+ branch_count += 1
1338
+ branch_id = f"dependency-{branch_count}"
1339
+ nodes.append(
1340
+ {
1341
+ "id": branch_id,
1342
+ "lane": "dependencies",
1343
+ "col": column(index),
1344
+ "type": _classify_component(node_by_id[other]),
1345
+ "label": _workflow_display_label(node_by_id[other]),
1346
+ "sublabel": _workflow_source_label(node_by_id[other]),
1347
+ "width": _workflow_node_width(node_by_id[other]),
1348
+ }
1349
+ )
1350
+ edges.append(
1351
+ {
1352
+ "id": f"branch-{branch_count}",
1353
+ "from": node_ids[node_id] if source == node_id else branch_id,
1354
+ "to": branch_id if source == node_id else node_ids[node_id],
1355
+ "label": str(edge.get("relation") or "observed relationship"),
1356
+ "variant": "dashed",
1357
+ "role": "branch",
1358
+ }
1359
+ )
1360
+ break
1361
+
1362
+ main_path = [node_ids[node_id] for node_id in path]
1363
+ phases = [
1364
+ {"id": "entry", "label": "Entry", "fromCol": 0, "toCol": min(1, last_col)},
1365
+ {
1366
+ "id": "trace",
1367
+ "label": "Observed code path",
1368
+ "fromCol": min(2, last_col),
1369
+ "toCol": min(3, last_col),
1370
+ "variant": "emphasis",
1371
+ },
1372
+ {
1373
+ "id": "endpoint",
1374
+ "label": "Endpoint",
1375
+ "fromCol": min(4, last_col),
1376
+ "toCol": last_col,
1377
+ "variant": "dashed",
1378
+ },
1379
+ ]
1380
+ return {
1381
+ "schema_version": 2,
1382
+ "diagram_type": "workflow",
1383
+ "meta": {
1384
+ "title": f"{title} · observed code path",
1385
+ "animation": "trace",
1386
+ "visual_preset": "signal-flow",
1387
+ "quality_profile": "showcase",
1388
+ "views": [
1389
+ {
1390
+ "id": "traced-path",
1391
+ "label": "Traced code path",
1392
+ "focus": main_path,
1393
+ "note": "Follow each direct Graphify relationship in order; no transitive links are added.",
1394
+ }
1395
+ ],
1396
+ },
1397
+ "lanes": [
1398
+ {"id": "path", "label": "Observed code path"},
1399
+ {"id": "dependencies", "label": "Direct dependencies"},
1400
+ ],
1401
+ "phases": phases,
1402
+ "mainPath": main_path,
1403
+ "nodes": nodes,
1404
+ "edges": edges,
1405
+ "cards": [
1406
+ {
1407
+ "dot": "cyan",
1408
+ "title": "Evidence boundary",
1409
+ "items": [
1410
+ "Every arrow is a direct Graphify relationship.",
1411
+ "Node subtitles identify the observed source file.",
1412
+ ],
1413
+ }
1414
+ ],
1415
+ }
1416
+
1417
+
1418
+ def render_workflow_story(
1419
+ understanding: GroundedUnderstanding,
1420
+ title: str,
1421
+ out_path: Path,
1422
+ *,
1423
+ archify_bin: str | None = None,
1424
+ auto_install: bool = True,
1425
+ ) -> Path:
1426
+ """Render the compact, presentation-safe workflow with bundled Archify."""
1427
+ bin_cmd = _resolve_archify_bin(archify_bin, auto_install=auto_install)
1428
+ spec_path = out_path.with_suffix(".workflow.json")
1429
+ spec_path.write_text(json.dumps(to_workflow_spec(understanding, title), indent=2), encoding="utf-8")
1430
+ report = archify_repair.run_deliver_json(
1431
+ bin_cmd, spec_path, out_path, diagram_type="workflow"
1432
+ )
1433
+ if report.get("ok"):
1434
+ return out_path
1435
+ raise subprocess.CalledProcessError(
1436
+ 1,
1437
+ [*bin_cmd, "deliver", "workflow", str(spec_path), str(out_path)],
1438
+ output=json.dumps(report),
1439
+ )
1440
+
1441
+
1442
+ def _sequence_relation(edge: dict) -> str:
1443
+ return str(edge.get("relation") or edge.get("label") or "").strip().lower()
1444
+
1445
+
1446
+ def _is_observed_static_call(edge: dict) -> bool:
1447
+ """Require extracted call evidence when Graphify provides confidence.
1448
+
1449
+ Graphify can add inferred symbol-resolution links to its graph. They are
1450
+ useful in the architecture explorer, but a sequence must not turn them
1451
+ into an apparently observed call trace.
1452
+ """
1453
+ if _sequence_relation(edge) not in _SEQUENCE_RELATIONS:
1454
+ return False
1455
+ confidence = str(edge.get("confidence") or "").strip().upper()
1456
+ context = str(edge.get("context") or "").strip().lower()
1457
+ return (not confidence or confidence == "EXTRACTED") and (not context or context == "call")
1458
+
1459
+
1460
+ def _sequence_path(understanding: GroundedUnderstanding) -> tuple[list[str], list[dict]]:
1461
+ """Find one bounded, ordered chain of direct static call edges.
1462
+
1463
+ This deliberately does *not* infer a runtime request path. A sequence is
1464
+ offered only when Graphify found at least two consecutive ``calls`` or
1465
+ ``invokes`` relationships between non-test symbols.
1466
+ """
1467
+ node_by_id = {str(node.get("id") or ""): node for node in understanding.nodes}
1468
+ node_by_id = {
1469
+ node_id: node for node_id, node in node_by_id.items() if node_id and not _is_test_node(node)
1470
+ }
1471
+ if not node_by_id:
1472
+ raise ValueError("sequence trace needs non-test code nodes")
1473
+
1474
+ outgoing: dict[str, list[dict]] = {node_id: [] for node_id in node_by_id}
1475
+ for edge in understanding.edges:
1476
+ source, target = _edge_endpoints(edge)
1477
+ if (
1478
+ source in node_by_id
1479
+ and target in node_by_id
1480
+ and source != target
1481
+ and _is_observed_static_call(edge)
1482
+ ):
1483
+ outgoing[source].append(edge)
1484
+ for edges in outgoing.values():
1485
+ edges.sort(key=_workflow_relation_key)
1486
+
1487
+ cited_seeds = set(_workflow_seed_ids(understanding, set(node_by_id)))
1488
+ starts = sorted(
1489
+ node_by_id,
1490
+ key=lambda node_id: (
1491
+ node_id not in cited_seeds,
1492
+ -(len(outgoing[node_id])),
1493
+ node_id,
1494
+ ),
1495
+ )
1496
+ best_path: list[str] = []
1497
+ best_edges: list[dict] = []
1498
+ best_score = (-1, -1, -1)
1499
+
1500
+ for start in starts:
1501
+ path = [start]
1502
+ path_edges: list[dict] = []
1503
+ seen = {start}
1504
+ while len(path) < _SEQUENCE_MAX_PARTICIPANTS:
1505
+ next_edge = _choose_workflow_edge(outgoing[path[-1]], seen, incoming=False)
1506
+ if next_edge is None:
1507
+ break
1508
+ _, target = _edge_endpoints(next_edge)
1509
+ path.append(target)
1510
+ path_edges.append(next_edge)
1511
+ seen.add(target)
1512
+
1513
+ if len(path_edges) < 2:
1514
+ continue
1515
+ score = (
1516
+ sum(node_id in cited_seeds for node_id in path),
1517
+ len(path),
1518
+ -sum(_workflow_relation_key(edge)[0] for edge in path_edges),
1519
+ )
1520
+ if score > best_score:
1521
+ best_path, best_edges, best_score = path, path_edges, score
1522
+
1523
+ if len(best_edges) < 2:
1524
+ raise ValueError(
1525
+ "sequence trace needs two consecutive direct Graphify calls or invocations"
1526
+ )
1527
+ return best_path, best_edges
1528
+
1529
+
1530
+ def to_sequence_spec(understanding: GroundedUnderstanding, title: str) -> dict:
1531
+ """Build an evidence-gated Archify sequence from direct static calls.
1532
+
1533
+ The explicit subtitle and card prevent readers from mistaking source
1534
+ analysis for request telemetry, timing data, or a captured trace.
1535
+ """
1536
+ if understanding.diagram_kind != "architecture":
1537
+ raise ValueError(
1538
+ f"to_sequence_spec only handles diagram_kind='architecture', got "
1539
+ f"{understanding.diagram_kind!r}"
1540
+ )
1541
+
1542
+ path, path_edges = _sequence_path(understanding)
1543
+ node_by_id = {str(node["id"]): node for node in understanding.nodes if node.get("id")}
1544
+ participant_ids = {node_id: f"step-{index + 1}" for index, node_id in enumerate(path)}
1545
+ message_start_y = 170
1546
+ message_gap = 58
1547
+ view_box_height = max(480, message_start_y + len(path_edges) * message_gap + 100)
1548
+ view_box_width = max(720, len(path) * 180)
1549
+
1550
+ messages = [
1551
+ {
1552
+ "id": f"call-{index}",
1553
+ "from": participant_ids[source],
1554
+ "to": participant_ids[target],
1555
+ "y": message_start_y + (index - 1) * message_gap,
1556
+ "label": _sequence_relation(edge),
1557
+ "variant": "emphasis",
1558
+ }
1559
+ for index, (edge, source, target) in enumerate(
1560
+ zip(path_edges, path, path[1:]), start=1
1561
+ )
1562
+ ]
1563
+ return {
1564
+ "schema_version": 1,
1565
+ "diagram_type": "sequence",
1566
+ "meta": {
1567
+ "title": f"{title} · observed static call sequence",
1568
+ "subtitle": "Direct Graphify calls/invocations; not runtime telemetry.",
1569
+ "viewBox": [view_box_width, view_box_height],
1570
+ "column_fit": "spread",
1571
+ "animation": "trace",
1572
+ "visual_preset": "signal-flow",
1573
+ "quality_profile": "showcase",
1574
+ "legend": {
1575
+ "mode": "auto",
1576
+ "entries": {"emphasis": {"label": "direct static call"}},
1577
+ },
1578
+ "views": [
1579
+ {
1580
+ "id": "observed-static-call-sequence",
1581
+ "label": "Observed call chain",
1582
+ "focus": [participant_ids[node_id] for node_id in path],
1583
+ "note": "Each arrow is a direct extracted Graphify calls/invokes relationship.",
1584
+ }
1585
+ ],
1586
+ },
1587
+ "participants": [
1588
+ {
1589
+ "id": participant_ids[node_id],
1590
+ "type": _classify_component(node_by_id[node_id]),
1591
+ "label": _workflow_display_label(node_by_id[node_id]),
1592
+ }
1593
+ for node_id in path
1594
+ ],
1595
+ "messages": messages,
1596
+ "activations": [
1597
+ {
1598
+ "participant": participant_ids[node_id],
1599
+ "from": message_start_y + index * message_gap - 6,
1600
+ "to": min(
1601
+ view_box_height - 22,
1602
+ message_start_y + (index + 1) * message_gap + 6,
1603
+ ),
1604
+ "type": _classify_component(node_by_id[node_id]),
1605
+ }
1606
+ for index, node_id in enumerate(path[1:])
1607
+ ],
1608
+ "cards": [
1609
+ {
1610
+ "dot": "cyan",
1611
+ "title": "Evidence boundary",
1612
+ "items": [
1613
+ "Every arrow is a direct extracted Graphify calls/invokes edge.",
1614
+ "No runtime request, response, timing, or trace is implied.",
1615
+ ],
1616
+ }
1617
+ ],
1618
+ }
1619
+
1620
+
1621
+ def render_sequence_trace(
1622
+ understanding: GroundedUnderstanding,
1623
+ title: str,
1624
+ out_path: Path,
1625
+ *,
1626
+ archify_bin: str | None = None,
1627
+ auto_install: bool = True,
1628
+ ) -> Path:
1629
+ """Render the optional evidence-gated sequence with bundled Archify."""
1630
+ bin_cmd = _resolve_archify_bin(archify_bin, auto_install=auto_install)
1631
+ spec_path = out_path.with_suffix(".sequence.json")
1632
+ spec_path.write_text(json.dumps(to_sequence_spec(understanding, title), indent=2), encoding="utf-8")
1633
+ report = archify_repair.run_deliver_json(
1634
+ bin_cmd, spec_path, out_path, diagram_type="sequence"
1635
+ )
1636
+ if report.get("ok"):
1637
+ return out_path
1638
+ raise subprocess.CalledProcessError(
1639
+ 1,
1640
+ [*bin_cmd, "deliver", "sequence", str(spec_path), str(out_path)],
1641
+ output=json.dumps(report),
1642
+ )
1643
+
1644
+
1645
+ def to_architecture_ir(
1646
+ understanding: GroundedUnderstanding,
1647
+ title: str,
1648
+ *,
1649
+ spacing_multiplier: float = 1.0,
1650
+ view: Literal["story", "overview", "full"] = "full",
1651
+ ) -> ArchitectureIR:
1652
+ """`spacing_multiplier` widens the vertical gap between grid rows beyond
1653
+ the normal default - the only knob render() has to make the layout more
1654
+ forgiving without an LLM. Confirmed live (12 Sep 2026): even a genuinely
1655
+ small, non-aggregated 5-node graph can fail archify's validator with
1656
+ "label overlaps component" (archify's own default auto-placement for a
1657
+ connector's label sometimes lands inside the FROM box when the row gap
1658
+ is tight) - a `layout/constraint` failure, not the large-graph
1659
+ `clean-flow/edge-through-node` crossing issue the LLM repair loop
1660
+ targets. Since "no LLM key configured" must not mean "no diagram"
1661
+ (Naman, 12 Sep 2026: "the diagram is a must"), render() retries this
1662
+ purely mechanically with progressively more room before giving up.
1663
+ """
1664
+ if understanding.diagram_kind != "architecture":
1665
+ raise ValueError(
1666
+ f"to_architecture_ir only handles diagram_kind='architecture', got "
1667
+ f"{understanding.diagram_kind!r}"
1668
+ )
1669
+
1670
+ if view not in {"story", "overview", "full"}:
1671
+ raise ValueError(f"view must be 'story', 'overview', or 'full', got {view!r}")
1672
+
1673
+ nodes, all_edges = understanding.nodes, understanding.edges
1674
+ if len(nodes) > _MAX_RAW_COMPONENTS:
1675
+ nodes, all_edges = _aggregate_by_community(nodes, all_edges, understanding.community_labels)
1676
+
1677
+ # Both tabs use the same component positions. Filtering overview edges
1678
+ # must not rewrite the dependency layers or make the two maps disagree.
1679
+ positions = _assign_grid(nodes, all_edges, cols=_GRID_COLUMNS)
1680
+ # Story shares Overview's compact structural subset. Its additional value
1681
+ # is guided focus, not a second dense map.
1682
+ edges = _select_overview_edges(nodes, all_edges) if view in {"story", "overview"} else all_edges
1683
+ labels = [n.get("label", n["id"]) for n in nodes]
1684
+ box_width, box_height = _estimate_box_size(labels)
1685
+
1686
+ components = [
1687
+ Component(
1688
+ id=n["id"],
1689
+ type=_classify_component(n),
1690
+ label=n.get("label", n["id"]),
1691
+ sublabel=n.get("sublabel"),
1692
+ row=positions[n["id"]][0],
1693
+ col=positions[n["id"]][1],
1694
+ size=(box_width, box_height),
1695
+ )
1696
+ for n in nodes
1697
+ ]
1698
+ connections = [
1699
+ Connection(
1700
+ id=e.get("id") or f"{e.get('from') or e.get('source')}-{e.get('to') or e.get('target')}",
1701
+ **{"from": e.get("from") or e.get("source")},
1702
+ to=e.get("to") or e.get("target"),
1703
+ label=e.get("label"),
1704
+ )
1705
+ for e in edges
1706
+ ]
1707
+ gap_y = int(max(60, box_height + 20) * spacing_multiplier)
1708
+ connections = _route_around_obstacles(
1709
+ connections,
1710
+ positions,
1711
+ box_width=box_width,
1712
+ box_height=box_height,
1713
+ cell_w=box_width + 30,
1714
+ cell_h=box_height,
1715
+ gap_x=_GRID_GAP_X,
1716
+ gap_y=gap_y,
1717
+ cols=_GRID_COLUMNS,
1718
+ )
1719
+ view_box = _view_box_for_routes(
1720
+ positions,
1721
+ connections,
1722
+ box_width=box_width,
1723
+ box_height=box_height,
1724
+ cell_w=box_width + 30,
1725
+ cell_h=box_height,
1726
+ gap_x=_GRID_GAP_X,
1727
+ gap_y=gap_y,
1728
+ )
1729
+ return ArchitectureIR(
1730
+ meta=Meta(
1731
+ title=title,
1732
+ viewBox=view_box,
1733
+ animation="trace" if view == "story" else "none",
1734
+ visual_preset="signal-flow" if view == "story" else None,
1735
+ views=_story_views(understanding, components, all_edges) if view == "story" else [],
1736
+ ),
1737
+ layout=GridLayout(
1738
+ cols=_GRID_COLUMNS,
1739
+ cellW=box_width + 30,
1740
+ cellH=box_height,
1741
+ gapX=_GRID_GAP_X,
1742
+ # Wider than archify's own 40px default - confirmed live that a
1743
+ # labeled connection between two vertically-adjacent grid rows
1744
+ # otherwise has nowhere to sit without overlapping one of the
1745
+ # two component boxes it connects (multiple "label overlaps
1746
+ # component" failures against a real 12-node graph).
1747
+ gapY=gap_y,
1748
+ ),
1749
+ components=components,
1750
+ connections=connections,
1751
+ )
1752
+
1753
+
1754
+ def _resolve_archify_bin(archify_bin: str | None, *, auto_install: bool) -> list[str]:
1755
+ """Return the explicit override or Graphitect's bundled Archify runtime.
1756
+
1757
+ ``auto_install`` remains in the signature for API compatibility with the
1758
+ earlier bridge, but is intentionally ignored: a Graphitect run must never
1759
+ download another tool at runtime.
1760
+ """
1761
+ if archify_bin:
1762
+ # shlex.split handles both a single executable ("archify", once on
1763
+ # PATH) and a multi-token invocation ("node C:\path\to\archify.mjs").
1764
+ # posix=False is required on Windows: shlex's default POSIX mode
1765
+ # treats backslash as an escape character and silently mangles
1766
+ # Windows paths (confirmed live - "node C:\Users\...\archify.mjs"
1767
+ # became "C:UsersnamanOneDrive..." with every backslash-letter pair
1768
+ # eaten as a fake escape sequence).
1769
+ return shlex.split(archify_bin, posix=False)
1770
+
1771
+ if not _BUNDLED_ARCHIFY_PATH.exists():
1772
+ raise FileNotFoundError(
1773
+ "bundled Archify runtime is missing from this Graphitect installation"
1774
+ )
1775
+
1776
+ node = shutil.which("node")
1777
+ if not node:
1778
+ raise FileNotFoundError(
1779
+ "Graphitect includes Archify, but Node.js 18+ is required to run "
1780
+ "the bundled renderer"
1781
+ )
1782
+ return [node, str(_BUNDLED_ARCHIFY_PATH)]
1783
+
1784
+
1785
+ def render(
1786
+ understanding: GroundedUnderstanding,
1787
+ title: str,
1788
+ out_path: Path,
1789
+ *,
1790
+ archify_bin: str | None = None,
1791
+ auto_install: bool = True,
1792
+ repair_backend: LLMBackend | None = None,
1793
+ max_repair_iterations: int = 3,
1794
+ view: Literal["story", "overview", "full"] = "full",
1795
+ ) -> Path:
1796
+ """Write the archify IR, then shell out to `archify deliver` to render it.
1797
+
1798
+ Archify ships inside Graphitect and is never installed or downloaded at
1799
+ runtime. ``auto_install`` is accepted only for backward compatibility.
1800
+
1801
+ When `repair_backend` is given, a rejected layout (archify's own strict
1802
+ validator - dense real-world graphs routinely fail on edges crossing
1803
+ unrelated components) is handed to archify_repair's LLM loop instead of
1804
+ failing outright: it patches the IR's routing/label-position fields
1805
+ against archify's real structured diagnostics and retries, up to
1806
+ `max_repair_iterations` total attempts. The repair is optional: provider
1807
+ failures and unresolved repair diagnostics fall back to the deterministic
1808
+ renderer, so an LLM quota response can never abort `graphitect build`.
1809
+
1810
+ Without a backend, there's no LLM available to patch anything, but "no
1811
+ key configured" must still not mean "no diagram" (Naman, 12 Sep 2026:
1812
+ "the diagram is a must") - confirmed live that even a small, non-
1813
+ aggregated graph can fail archify's validator on tight label spacing
1814
+ alone (see to_architecture_ir's spacing_multiplier docstring), so this
1815
+ retries the deterministic layout a few times with progressively more
1816
+ room before finally raising CalledProcessError for the caller to handle.
1817
+ """
1818
+ bin_cmd = _resolve_archify_bin(archify_bin, auto_install=auto_install)
1819
+ ir_path = out_path.with_suffix(".architecture.json")
1820
+
1821
+ if repair_backend is not None:
1822
+ ir = to_architecture_ir(understanding, title, view=view)
1823
+ try:
1824
+ report = archify_repair.repair_and_deliver(
1825
+ ir, bin_cmd, ir_path, out_path, repair_backend, max_iterations=max_repair_iterations
1826
+ )
1827
+ except archify_repair.RepairUnavailableError:
1828
+ warnings.warn(
1829
+ "Archify's optional LLM layout repair was unavailable; "
1830
+ "using the deterministic renderer.",
1831
+ RuntimeWarning,
1832
+ stacklevel=2,
1833
+ )
1834
+ else:
1835
+ if report.get("ok"):
1836
+ return out_path
1837
+ warnings.warn(
1838
+ "Archify's optional LLM layout repair did not resolve the layout; "
1839
+ "using the deterministic renderer.",
1840
+ RuntimeWarning,
1841
+ stacklevel=2,
1842
+ )
1843
+
1844
+ last_report: dict | None = None
1845
+ for multiplier in _SPACING_RETRY_MULTIPLIERS:
1846
+ ir = to_architecture_ir(understanding, title, spacing_multiplier=multiplier, view=view)
1847
+ ir_path.write_text(json.dumps(ir.model_dump_archify(), indent=2), encoding="utf-8")
1848
+ report = archify_repair.run_deliver_json(bin_cmd, ir_path, out_path)
1849
+ if report.get("ok"):
1850
+ return out_path
1851
+ last_report = report
1852
+
1853
+ # archify's own validator already computed the exact fix for a
1854
+ # "layout/constraint" label-overlap failure (a concrete suggested
1855
+ # labelAt) - confirmed live (12 Sep 2026) that widening gapY alone
1856
+ # doesn't move a fixed-offset default label at all, so apply the
1857
+ # suggested fix directly and retry, repeating up to
1858
+ # _MAX_LABEL_FIX_ROUNDS times at this same spacing level (fixing one
1859
+ # overlap can reveal or shift another on a real dense graph -
1860
+ # confirmed live that a single attempt wasn't always enough) before
1861
+ # escalating to more room.
1862
+ for _ in range(_MAX_LABEL_FIX_ROUNDS):
1863
+ patched_ir = _apply_suggested_label_fixes(ir, last_report.get("diagnostics") or [])
1864
+ if patched_ir is None:
1865
+ break
1866
+ ir = patched_ir
1867
+ ir_path.write_text(json.dumps(ir.model_dump_archify(), indent=2), encoding="utf-8")
1868
+ report = archify_repair.run_deliver_json(bin_cmd, ir_path, out_path)
1869
+ if report.get("ok"):
1870
+ return out_path
1871
+ last_report = report
1872
+
1873
+ raise subprocess.CalledProcessError(
1874
+ 1,
1875
+ [*bin_cmd, "deliver", "architecture", str(ir_path), str(out_path)],
1876
+ output=json.dumps(last_report),
1877
+ )