graphitect 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (336) hide show
  1. graphify/__init__.py +30 -0
  2. graphify/__main__.py +757 -0
  3. graphify/_minhash.py +107 -0
  4. graphify/affected.py +318 -0
  5. graphify/always_on/agents-md.md +12 -0
  6. graphify/always_on/antigravity-rules.md +14 -0
  7. graphify/always_on/claude-md.md +9 -0
  8. graphify/always_on/gemini-md.md +9 -0
  9. graphify/always_on/kiro-steering.md +5 -0
  10. graphify/always_on/vscode-instructions.md +17 -0
  11. graphify/analyze.py +769 -0
  12. graphify/benchmark.py +152 -0
  13. graphify/build.py +2300 -0
  14. graphify/cache.py +1746 -0
  15. graphify/callflow_html.py +2051 -0
  16. graphify/cargo_introspect.py +109 -0
  17. graphify/cli.py +4745 -0
  18. graphify/cluster.py +409 -0
  19. graphify/command-kilo.md +15 -0
  20. graphify/cross_repo_calls.py +216 -0
  21. graphify/cross_repo_types.py +75 -0
  22. graphify/csharp_dispatch.py +154 -0
  23. graphify/dedup.py +1213 -0
  24. graphify/detect.py +2566 -0
  25. graphify/diagnostics.py +406 -0
  26. graphify/export.py +1349 -0
  27. graphify/exporters/__init__.py +1 -0
  28. graphify/exporters/base.py +14 -0
  29. graphify/exporters/graphdb.py +173 -0
  30. graphify/exporters/html.py +637 -0
  31. graphify/extract.py +7856 -0
  32. graphify/extractors/MIGRATION.md +107 -0
  33. graphify/extractors/__init__.py +66 -0
  34. graphify/extractors/apex.py +215 -0
  35. graphify/extractors/base.py +85 -0
  36. graphify/extractors/bash.py +579 -0
  37. graphify/extractors/blade.py +53 -0
  38. graphify/extractors/commonlisp.py +540 -0
  39. graphify/extractors/csharp.py +448 -0
  40. graphify/extractors/dart.py +564 -0
  41. graphify/extractors/dm.py +494 -0
  42. graphify/extractors/elixir.py +241 -0
  43. graphify/extractors/engine.py +6509 -0
  44. graphify/extractors/fortran.py +311 -0
  45. graphify/extractors/go.py +527 -0
  46. graphify/extractors/json_config.py +240 -0
  47. graphify/extractors/julia.py +289 -0
  48. graphify/extractors/markdown.py +408 -0
  49. graphify/extractors/models.py +131 -0
  50. graphify/extractors/objc.py +566 -0
  51. graphify/extractors/ocaml.py +289 -0
  52. graphify/extractors/pascal.py +688 -0
  53. graphify/extractors/pascal_forms.py +196 -0
  54. graphify/extractors/powershell.py +522 -0
  55. graphify/extractors/razor.py +192 -0
  56. graphify/extractors/resolution.py +3584 -0
  57. graphify/extractors/robot.py +296 -0
  58. graphify/extractors/rust.py +470 -0
  59. graphify/extractors/sln.py +92 -0
  60. graphify/extractors/sql.py +720 -0
  61. graphify/extractors/terraform.py +181 -0
  62. graphify/extractors/verilog.py +329 -0
  63. graphify/extractors/zig.py +181 -0
  64. graphify/file_slice.py +246 -0
  65. graphify/global_graph.py +194 -0
  66. graphify/google_workspace.py +237 -0
  67. graphify/hooks.py +933 -0
  68. graphify/ids.py +93 -0
  69. graphify/ingest.py +358 -0
  70. graphify/install.py +2366 -0
  71. graphify/llm.py +3544 -0
  72. graphify/manifest.py +4 -0
  73. graphify/manifest_ingest.py +311 -0
  74. graphify/mcp_ingest.py +386 -0
  75. graphify/multigraph_compat.py +212 -0
  76. graphify/pascal_resolution.py +129 -0
  77. graphify/paths.py +436 -0
  78. graphify/pg_introspect.py +165 -0
  79. graphify/prs.py +770 -0
  80. graphify/querylog.py +80 -0
  81. graphify/reflect.py +882 -0
  82. graphify/report.py +346 -0
  83. graphify/resolver_registry.py +85 -0
  84. graphify/ruby_resolution.py +242 -0
  85. graphify/scip_ingest.py +363 -0
  86. graphify/security.py +460 -0
  87. graphify/semantic_cleanup.py +336 -0
  88. graphify/serve.py +2608 -0
  89. graphify/skill-agents.md +710 -0
  90. graphify/skill-aider.md +1283 -0
  91. graphify/skill-amp.md +710 -0
  92. graphify/skill-claw.md +713 -0
  93. graphify/skill-codex.md +710 -0
  94. graphify/skill-copilot.md +713 -0
  95. graphify/skill-devin.md +1410 -0
  96. graphify/skill-droid.md +710 -0
  97. graphify/skill-kilo.md +722 -0
  98. graphify/skill-kiro.md +713 -0
  99. graphify/skill-opencode.md +705 -0
  100. graphify/skill-pi.md +713 -0
  101. graphify/skill-trae.md +711 -0
  102. graphify/skill-vscode.md +709 -0
  103. graphify/skill-windows.md +755 -0
  104. graphify/skill.md +713 -0
  105. graphify/skills/agents/references/add-watch.md +56 -0
  106. graphify/skills/agents/references/exports.md +87 -0
  107. graphify/skills/agents/references/extraction-spec.md +70 -0
  108. graphify/skills/agents/references/github-and-merge.md +46 -0
  109. graphify/skills/agents/references/hooks.md +33 -0
  110. graphify/skills/agents/references/query.md +311 -0
  111. graphify/skills/agents/references/transcribe.md +52 -0
  112. graphify/skills/agents/references/update.md +210 -0
  113. graphify/skills/amp/references/add-watch.md +56 -0
  114. graphify/skills/amp/references/exports.md +87 -0
  115. graphify/skills/amp/references/extraction-spec.md +70 -0
  116. graphify/skills/amp/references/github-and-merge.md +46 -0
  117. graphify/skills/amp/references/hooks.md +33 -0
  118. graphify/skills/amp/references/query.md +311 -0
  119. graphify/skills/amp/references/transcribe.md +52 -0
  120. graphify/skills/amp/references/update.md +210 -0
  121. graphify/skills/claude/references/add-watch.md +56 -0
  122. graphify/skills/claude/references/exports.md +87 -0
  123. graphify/skills/claude/references/extraction-spec.md +70 -0
  124. graphify/skills/claude/references/github-and-merge.md +46 -0
  125. graphify/skills/claude/references/hooks.md +33 -0
  126. graphify/skills/claude/references/query.md +311 -0
  127. graphify/skills/claude/references/transcribe.md +52 -0
  128. graphify/skills/claude/references/update.md +210 -0
  129. graphify/skills/claw/references/add-watch.md +56 -0
  130. graphify/skills/claw/references/exports.md +87 -0
  131. graphify/skills/claw/references/extraction-spec.md +31 -0
  132. graphify/skills/claw/references/github-and-merge.md +46 -0
  133. graphify/skills/claw/references/hooks.md +33 -0
  134. graphify/skills/claw/references/query.md +311 -0
  135. graphify/skills/claw/references/transcribe.md +52 -0
  136. graphify/skills/claw/references/update.md +210 -0
  137. graphify/skills/codex/references/add-watch.md +56 -0
  138. graphify/skills/codex/references/exports.md +87 -0
  139. graphify/skills/codex/references/extraction-spec.md +31 -0
  140. graphify/skills/codex/references/github-and-merge.md +46 -0
  141. graphify/skills/codex/references/hooks.md +33 -0
  142. graphify/skills/codex/references/query.md +311 -0
  143. graphify/skills/codex/references/transcribe.md +52 -0
  144. graphify/skills/codex/references/update.md +210 -0
  145. graphify/skills/copilot/references/add-watch.md +56 -0
  146. graphify/skills/copilot/references/exports.md +87 -0
  147. graphify/skills/copilot/references/extraction-spec.md +70 -0
  148. graphify/skills/copilot/references/github-and-merge.md +46 -0
  149. graphify/skills/copilot/references/hooks.md +33 -0
  150. graphify/skills/copilot/references/query.md +311 -0
  151. graphify/skills/copilot/references/transcribe.md +52 -0
  152. graphify/skills/copilot/references/update.md +210 -0
  153. graphify/skills/droid/references/add-watch.md +56 -0
  154. graphify/skills/droid/references/exports.md +87 -0
  155. graphify/skills/droid/references/extraction-spec.md +70 -0
  156. graphify/skills/droid/references/github-and-merge.md +46 -0
  157. graphify/skills/droid/references/hooks.md +33 -0
  158. graphify/skills/droid/references/query.md +311 -0
  159. graphify/skills/droid/references/transcribe.md +52 -0
  160. graphify/skills/droid/references/update.md +210 -0
  161. graphify/skills/kilo/references/add-watch.md +56 -0
  162. graphify/skills/kilo/references/exports.md +87 -0
  163. graphify/skills/kilo/references/extraction-spec.md +70 -0
  164. graphify/skills/kilo/references/github-and-merge.md +46 -0
  165. graphify/skills/kilo/references/hooks.md +33 -0
  166. graphify/skills/kilo/references/query.md +311 -0
  167. graphify/skills/kilo/references/transcribe.md +52 -0
  168. graphify/skills/kilo/references/update.md +210 -0
  169. graphify/skills/kiro/references/add-watch.md +56 -0
  170. graphify/skills/kiro/references/exports.md +87 -0
  171. graphify/skills/kiro/references/extraction-spec.md +31 -0
  172. graphify/skills/kiro/references/github-and-merge.md +46 -0
  173. graphify/skills/kiro/references/hooks.md +33 -0
  174. graphify/skills/kiro/references/query.md +311 -0
  175. graphify/skills/kiro/references/transcribe.md +52 -0
  176. graphify/skills/kiro/references/update.md +210 -0
  177. graphify/skills/opencode/references/add-watch.md +56 -0
  178. graphify/skills/opencode/references/exports.md +87 -0
  179. graphify/skills/opencode/references/extraction-spec.md +70 -0
  180. graphify/skills/opencode/references/github-and-merge.md +46 -0
  181. graphify/skills/opencode/references/hooks.md +33 -0
  182. graphify/skills/opencode/references/query.md +311 -0
  183. graphify/skills/opencode/references/transcribe.md +52 -0
  184. graphify/skills/opencode/references/update.md +210 -0
  185. graphify/skills/pi/references/add-watch.md +56 -0
  186. graphify/skills/pi/references/exports.md +87 -0
  187. graphify/skills/pi/references/extraction-spec.md +31 -0
  188. graphify/skills/pi/references/github-and-merge.md +46 -0
  189. graphify/skills/pi/references/hooks.md +33 -0
  190. graphify/skills/pi/references/query.md +311 -0
  191. graphify/skills/pi/references/transcribe.md +52 -0
  192. graphify/skills/pi/references/update.md +210 -0
  193. graphify/skills/trae/references/add-watch.md +56 -0
  194. graphify/skills/trae/references/exports.md +87 -0
  195. graphify/skills/trae/references/extraction-spec.md +70 -0
  196. graphify/skills/trae/references/github-and-merge.md +46 -0
  197. graphify/skills/trae/references/hooks.md +35 -0
  198. graphify/skills/trae/references/query.md +311 -0
  199. graphify/skills/trae/references/transcribe.md +52 -0
  200. graphify/skills/trae/references/update.md +210 -0
  201. graphify/skills/vscode/references/add-watch.md +56 -0
  202. graphify/skills/vscode/references/exports.md +87 -0
  203. graphify/skills/vscode/references/extraction-spec.md +70 -0
  204. graphify/skills/vscode/references/github-and-merge.md +46 -0
  205. graphify/skills/vscode/references/hooks.md +33 -0
  206. graphify/skills/vscode/references/query.md +311 -0
  207. graphify/skills/vscode/references/transcribe.md +52 -0
  208. graphify/skills/vscode/references/update.md +210 -0
  209. graphify/skills/windows/references/add-watch.md +56 -0
  210. graphify/skills/windows/references/exports.md +87 -0
  211. graphify/skills/windows/references/extraction-spec.md +70 -0
  212. graphify/skills/windows/references/github-and-merge.md +46 -0
  213. graphify/skills/windows/references/hooks.md +33 -0
  214. graphify/skills/windows/references/query.md +311 -0
  215. graphify/skills/windows/references/transcribe.md +52 -0
  216. graphify/skills/windows/references/update.md +210 -0
  217. graphify/symbol_resolution.py +556 -0
  218. graphify/transcribe.py +186 -0
  219. graphify/tree_html.py +603 -0
  220. graphify/validate.py +95 -0
  221. graphify/watch.py +2280 -0
  222. graphify/wiki.py +405 -0
  223. graphitect/__init__.py +28 -0
  224. graphitect/__main__.py +4 -0
  225. graphitect/_vendor/__init__.py +2 -0
  226. graphitect/_vendor/archify/LICENSE +22 -0
  227. graphitect/_vendor/archify/SKILL.md +137 -0
  228. graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
  229. graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
  230. graphitect/_vendor/archify/assets/template.html +14935 -0
  231. graphitect/_vendor/archify/bin/archify.mjs +2091 -0
  232. graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
  233. graphitect/_vendor/archify/bin/preview.mjs +653 -0
  234. graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
  235. graphitect/_vendor/archify/brand-marks/README.md +31 -0
  236. graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
  237. graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
  238. graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
  239. graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
  240. graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
  241. graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
  242. graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
  243. graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
  244. graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
  245. graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
  246. graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
  247. graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
  248. graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
  249. graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
  250. graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
  251. graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
  252. graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
  253. graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
  254. graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
  255. graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
  256. graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
  257. graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
  258. graphitect/_vendor/archify/package-lock.json +149 -0
  259. graphitect/_vendor/archify/package.json +39 -0
  260. graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
  261. graphitect/_vendor/archify/references/authoring-contract.md +243 -0
  262. graphitect/_vendor/archify/references/brand-marks.md +65 -0
  263. graphitect/_vendor/archify/references/delivery-contract.md +120 -0
  264. graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
  265. graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
  266. graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
  267. graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
  268. graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
  269. graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
  270. graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
  271. graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
  272. graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
  273. graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
  274. graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
  275. graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
  276. graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
  277. graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
  278. graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
  279. graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
  280. graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
  281. graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
  282. graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
  283. graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
  284. graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
  285. graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
  286. graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
  287. graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
  288. graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
  289. graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
  290. graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
  291. graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
  292. graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
  293. graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
  294. graphitect/_vendor/archify/schemas/README.md +211 -0
  295. graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
  296. graphitect/_vendor/archify/schemas/common.schema.json +115 -0
  297. graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
  298. graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
  299. graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
  300. graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
  301. graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
  302. graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
  303. graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
  304. graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
  305. graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
  306. graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
  307. graphitect/_vendor/archify/skill-release.json +10 -0
  308. graphitect/cli.py +981 -0
  309. graphitect/deliver/__init__.py +5 -0
  310. graphitect/deliver/archify_adapter.py +1877 -0
  311. graphitect/deliver/archify_ir.py +160 -0
  312. graphitect/deliver/archify_repair.py +135 -0
  313. graphitect/deliver/doc_compiler.py +916 -0
  314. graphitect/ground/__init__.py +5 -0
  315. graphitect/ground/describe_source.py +27 -0
  316. graphitect/ground/fullread_source.py +56 -0
  317. graphitect/ground/graphify_source.py +107 -0
  318. graphitect/models.py +118 -0
  319. graphitect/skill/SKILL.md +80 -0
  320. graphitect/skill/agents/openai.yaml +4 -0
  321. graphitect/synthesize/__init__.py +5 -0
  322. graphitect/synthesize/engine.py +281 -0
  323. graphitect/synthesize/llm_backend.py +331 -0
  324. graphitect/synthesize/questions.py +139 -0
  325. graphitect/synthesize/rubric.py +104 -0
  326. graphitect-0.2.0.dist-info/METADATA +284 -0
  327. graphitect-0.2.0.dist-info/RECORD +336 -0
  328. graphitect-0.2.0.dist-info/WHEEL +5 -0
  329. graphitect-0.2.0.dist-info/entry_points.txt +2 -0
  330. graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
  331. graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
  332. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
  333. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
  334. graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
  335. graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
  336. graphitect-0.2.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,720 @@
1
+ """Sql extractor. Moved verbatim from graphify/extract.py."""
2
+ from __future__ import annotations
3
+
4
+ import re
5
+
6
+ from pathlib import Path
7
+ from graphify.extractors.base import _file_stem, _make_id
8
+
9
+ # Recovers CREATE FUNCTION/PROCEDURE statements the grammar could not parse
10
+ # structurally. Used by BOTH recovery sites — the walk-time ERROR-node scan and
11
+ # the whole-file has_error fallback (#2180). They MUST share one pattern: when
12
+ # they disagreed, the same statement produced two nodes with different names
13
+ # (the ERROR scan captured `dbo.[usp_Mixed]`, the fallback stopped at `dbo`,
14
+ # and _add_node's id-dedupe never fired because the ids differed).
15
+ #
16
+ # Each name part is a bare identifier, a double-quoted (delimited) one, or a
17
+ # T-SQL bracket-delimited one, so CREATE OR REPLACE FUNCTION "public"."fn"(...)
18
+ # and CREATE PROCEDURE [dbo].[usp_Load] ... are both recovered. A bare [\w$.]+
19
+ # stops dead at the leading delimiter, which silently dropped every quoted
20
+ # PL/pgSQL routine (#2180) and every bracket-named T-SQL procedure. T-SQL's
21
+ # AS BEGIN...END body idiom always lands in recovery — the grammar has no
22
+ # create_procedure parse for it — and T-SQL spells re-creation CREATE OR ALTER
23
+ # (it has no OR REPLACE), so accept that form too, mirroring fb_proc_or_trigger.
24
+ # Inside a bracket-delimited part, a literal ] is escaped by doubling
25
+ # ([a]]b] names the identifier a]b), so consume ]] before treating a
26
+ # lone ] as the closing delimiter — stopping at the first ] truncated
27
+ # the name and minted a phantom that could collide with a real [a].
28
+ # PROC is T-SQL's official shorthand for PROCEDURE and equally common in the
29
+ # wild; the optional (?:EDURE)? still requires trailing whitespace, so a word
30
+ # that merely starts with PROC cannot match.
31
+ # \bCREATE: without the boundary, CREATE matched inside a bare word, so
32
+ # 'SELECT AUTOCREATE PROCEDURE x FROM t;' in an error-bearing file minted a
33
+ # phantom routine x() (delimited identifiers are span-skipped at the scan
34
+ # site, but a bare word has no span).
35
+ _ROUTINE_RECOVERY_RX = re.compile(
36
+ r"\bCREATE\s+(?:OR\s+(?:REPLACE|ALTER)\s+)?(?:FUNCTION|PROC(?:EDURE)?)\s+"
37
+ r"(?:IF\s+NOT\s+EXISTS\s+)?"
38
+ r"((?:\"(?:[^\"\n]|\"\")+\"|\[(?:[^\]\n]|\]\])+\]|[\w$]+)"
39
+ r"(?:\s*\.\s*(?:\"(?:[^\"\n]|\"\")+\"|\[(?:[^\]\n]|\]\])+\]|[\w$]+))*)",
40
+ re.IGNORECASE,
41
+ )
42
+
43
+ # _mask_sql_comments is a linear character scanner, not a regex: the four
44
+ # span kinds interact in ways a single pattern cannot express safely —
45
+ # comment-opener parity inside strings, nested block comments, and
46
+ # end-of-line abandonment of unclosed literals. Span handling:
47
+ #
48
+ # - single-quoted strings are BLANKED like comments: routine names never
49
+ # live in single quotes, and dynamic SQL (EXEC(N'CREATE PROC [dbo].[Fake]
50
+ # ...')) would otherwise fabricate a routine node whenever an unrelated
51
+ # parse error arms the whole-file scan. (Dialects where other quoting
52
+ # carries strings — MySQL double quotes, PostgreSQL dollar-quoting — are
53
+ # NOT modelled; DDL inside those still reaches the scan.)
54
+ # - double-quoted and bracket-delimited identifiers are PRESERVED verbatim —
55
+ # they are exactly the delimited names the recovery regex must see ("" and
56
+ # ]] escapes consumed, mirroring _ROUTINE_RECOVERY_RX).
57
+ # - line comments blank to end-of-line; block comments blank to their
58
+ # matching */ with NESTING tracked (SQL Server and PostgreSQL both nest
59
+ # /* */, and a lazy first-*/ match let commented-out DDL inside a nested
60
+ # comment fabricate nodes). MySQL and Oracle do NOT nest — there the
61
+ # depth tracking over-blanks, losing (never fabricating) a routine after
62
+ # an inner */; the primary T-SQL/PostgreSQL targets win that trade. An
63
+ # UNCLOSED block comment blanks to end-of-file, matching SQL semantics.
64
+ #
65
+ # Literals are deliberately line-scoped: this mask only runs on files that
66
+ # already failed to parse, where an unclosed delimiter is likely, and a
67
+ # multi-line literal span would let one unclosed quote swallow real DDL
68
+ # below it. The cost is asymmetric by kind. A single-quoted string that
69
+ # continues past its line is blanked only up to the newline, so a comment
70
+ # opener inside it cannot fire on that line — but its CONTINUATION lines are
71
+ # scanned as code, and DDL there fabricates: multi-line dynamic SQL
72
+ # (SET @sql = N\'\n CREATE PROC ...\') is a KNOWN HOLE, alongside the
73
+ # unmodelled quoting dialects above; only same-line dynamic SQL is blanked.
74
+ # Closing it would need a real string heuristic (e.g. an end-of-line opening
75
+ # quote), judged not worth the swallow risk in a recovery-only path.
76
+
77
+
78
+ def _scan_sql(text: str) -> tuple[str, list[tuple[int, int]]]:
79
+ """Blank comment and string-literal spans, preserving every offset.
80
+
81
+ One output character per input character: non-newline characters inside
82
+ a blanked span become spaces and newlines are kept, so positions and
83
+ line numbers computed against the masked text are valid against the
84
+ original. Double-quoted and bracket-delimited identifiers are preserved
85
+ verbatim (they carry recoverable routine names); single-quoted strings,
86
+ line comments, and (nesting-aware) block comments are blanked. Used by
87
+ the routine-recovery scan so CREATE PROCEDURE/FUNCTION DDL reachable
88
+ only through a comment or a single-quoted string cannot fabricate a
89
+ routine node when an unrelated parse error arms recovery.
90
+
91
+ Returns (masked_text, ident_spans) where ident_spans holds the [start,
92
+ end) of every PRESERVED delimited identifier: preserved spans keep their
93
+ text verbatim, so DDL keywords inside one are still visible in the
94
+ masked text, and the recovery scan must skip a match that starts there
95
+ (identifier data, not DDL — 'SELECT 1 AS [CREATE PROCEDURE x pending]'
96
+ must not mint a routine).
97
+ """
98
+ ident_spans: list[tuple[int, int]] = []
99
+ out: list[str] = []
100
+ i, n = 0, len(text)
101
+
102
+ def _blank(upto: int) -> int:
103
+ """Blank [i, upto), keeping newlines; return upto."""
104
+ for c in text[i:upto]:
105
+ out.append("\n" if c == "\n" else " ")
106
+ return upto
107
+
108
+ def _blank_tail_and_carry(start: int) -> int:
109
+ """Blank from start to end-of-line, carrying comment state forward.
110
+
111
+ The union-of-readings blank for an ambiguous stretch: the rest of the
112
+ line is blanked outright; if the raw text of that stretch leaves a /*
113
+ unclosed on its own line (the maximum comment depth any reading could
114
+ be left holding), blanking continues, nesting-aware, to the closing
115
+ */ or EOF. Where a carry closes MID-line the same rule applies to the
116
+ remainder of that line — under the reading where the carry never
117
+ opened, that whole line may be a comment or a string, so emitting the
118
+ post-*/ text verbatim exposed it (found by differential fuzzing).
119
+ Repeats until a line ends with no carry pending. Appends one output
120
+ character per input character; returns the resume index.
121
+ """
122
+ k = start
123
+ while True:
124
+ eol = text.find("\n", k)
125
+ eol = n if eol == -1 else eol
126
+ depth = 0
127
+ m2 = k
128
+ while m2 < eol:
129
+ if text.startswith("/*", m2):
130
+ depth += 1
131
+ m2 += 2
132
+ elif text.startswith("*/", m2):
133
+ if depth:
134
+ depth -= 1
135
+ m2 += 2
136
+ else:
137
+ m2 += 1
138
+ for ch in text[k:eol]:
139
+ out.append("\n" if ch == "\n" else " ")
140
+ k = eol
141
+ if not depth:
142
+ return k
143
+ j2 = k
144
+ while j2 < n and depth:
145
+ if text.startswith("/*", j2):
146
+ depth += 1
147
+ j2 += 2
148
+ elif text.startswith("*/", j2):
149
+ depth -= 1
150
+ j2 += 2
151
+ else:
152
+ j2 += 1
153
+ for ch in text[k:j2]:
154
+ out.append("\n" if ch == "\n" else " ")
155
+ k = j2
156
+ if k >= n:
157
+ return k
158
+ # the carry closed mid-line: the remainder of THIS line is the
159
+ # same ambiguous stretch — loop and blank it too
160
+
161
+ while i < n:
162
+ c = text[i]
163
+ if c == "'":
164
+ # Single-quoted string: blank it. '' is an escaped quote; a
165
+ # newline abandons the literal (see the comment above).
166
+ j = i + 1
167
+ while j < n and text[j] != "\n":
168
+ if text[j] == "'":
169
+ if j + 1 < n and text[j + 1] == "'":
170
+ j += 2
171
+ continue
172
+ j += 1
173
+ break
174
+ j += 1
175
+ i = _blank(j)
176
+ elif c == '"' or c == "[":
177
+ # Delimited identifier: preserve verbatim. Doubled closers are
178
+ # escapes. A span is DISTRUSTED when it is unterminated (no
179
+ # closer before the newline) or would swallow a comment opener
180
+ # on its way to the closer ('SELECT [Col FROM t -- CREATE PROC
181
+ # [dbo]' closes on [dbo]'s bracket) — a stray delimiter is
182
+ # ordinary in exactly the broken files this mask runs on.
183
+ #
184
+ # A distrusted span is irreducibly ambiguous (identifier data vs
185
+ # stray delimiter before real comments/strings), and any attempt
186
+ # to pick one reading exposed text the other reading blanks —
187
+ # re-emitting the delimiter and rescanning even re-paired later
188
+ # single quotes and uncovered dynamic SQL. So blank the UNION of
189
+ # every reading: the rest of the line is blanked outright, and
190
+ # any raw /* on it with no later */ on the same line carries
191
+ # forward as (nesting-aware) comment state, since some reading
192
+ # may have left it open — and where that carry closes mid-line,
193
+ # the remainder of THAT line gets the same treatment, repeated
194
+ # until a line ends carry-free (under the no-carry reading the
195
+ # close line may itself be all comment or string, so emitting its
196
+ # post-*/ tail verbatim was an exposure). Over-blanking loses at
197
+ # most routines on lines already entangled with the broken one (a
198
+ # conservative false negative; a routine named like [a--b] is
199
+ # inside that loss); under-blanking is what fabricates, and every
200
+ # reading's blank set stays a subset of this one — except where a
201
+ # */ + * versus * + /* token split makes two readings consume the
202
+ # same /, an irreducible divergence whose only closure would be
203
+ # blanking to EOF on every */* sequence (accepted, documented
204
+ # limitation; the token sequence appears in no dialect's idiom).
205
+ closer = '"' if c == '"' else "]"
206
+ j = i + 1
207
+ closed = False
208
+ while j < n and text[j] != "\n":
209
+ if text[j] == closer:
210
+ if j + 1 < n and text[j + 1] == closer:
211
+ j += 2
212
+ continue
213
+ j += 1
214
+ closed = True
215
+ break
216
+ j += 1
217
+ span = text[i:j]
218
+ if closed and "--" not in span and "/*" not in span:
219
+ ident_spans.append((i, j))
220
+ out.append(span)
221
+ i = j
222
+ else:
223
+ i = _blank_tail_and_carry(i)
224
+ elif c == "-" and i + 1 < n and text[i + 1] == "-":
225
+ j = i
226
+ while j < n and text[j] != "\n":
227
+ j += 1
228
+ i = _blank(j)
229
+ elif c == "/" and i + 1 < n and text[i + 1] == "*":
230
+ depth, j = 1, i + 2
231
+ while j < n and depth:
232
+ if text[j] == "/" and j + 1 < n and text[j + 1] == "*":
233
+ depth += 1
234
+ j += 2
235
+ elif text[j] == "*" and j + 1 < n and text[j + 1] == "/":
236
+ depth -= 1
237
+ j += 2
238
+ else:
239
+ j += 1
240
+ i = _blank(j) # unclosed comment: j == n, blanks to end-of-file
241
+ else:
242
+ out.append(c)
243
+ i += 1
244
+ return "".join(out), ident_spans
245
+
246
+
247
+ def _mask_sql_comments(text: str) -> str:
248
+ """Masked text only — see _scan_sql for the span-reporting form."""
249
+ return _scan_sql(text)[0]
250
+
251
+
252
+ def _norm_ident(name: str) -> str:
253
+ """Normalize a SQL identifier for name-based reference resolution.
254
+
255
+ Splits on `.`, strips one pair of surrounding delimiters from each part
256
+ (double quotes for Postgres/ANSI, backticks for MySQL, brackets for
257
+ T-SQL), lowercases, and rejoins. So `"public"."users"`, `public.users`,
258
+ and `PUBLIC.USERS` all normalize to `public.users`. Used ONLY for
259
+ `table_nids` keys and lookups — node ids and display labels keep the
260
+ original text.
261
+ """
262
+ parts = []
263
+ for part in name.split("."):
264
+ p = part.strip()
265
+ if len(p) >= 2 and ((p[0] == p[-1] and p[0] in ('"', "`"))
266
+ or (p[0] == "[" and p[-1] == "]")):
267
+ p = p[1:-1]
268
+ parts.append(p.lower())
269
+ return ".".join(parts)
270
+
271
+
272
+ def extract_sql(path: Path, content: str | bytes | None = None) -> dict:
273
+ """Extract tables, views, functions, and relationships from .sql files via tree-sitter."""
274
+ try:
275
+ import tree_sitter_sql as tssql
276
+ from tree_sitter import Language, Parser
277
+ except ImportError as e:
278
+ import importlib.util
279
+ # An installed-but-broken grammar (e.g. a C extension built for a
280
+ # different Python ABI, #2602) raises ImportError here too. Reporting
281
+ # that as "not installed" sends the user to a no-op `pip install`, so
282
+ # distinguish a genuinely-absent module from one that failed to load
283
+ # and surface the real exception in the latter case.
284
+ if importlib.util.find_spec("tree_sitter_sql") is None:
285
+ return {"nodes": [], "edges": [],
286
+ "error": "tree_sitter_sql not installed. Run: pip install tree-sitter-sql"}
287
+ return {"nodes": [], "edges": [],
288
+ "error": f"tree_sitter_sql is installed but failed to load: {e}"}
289
+
290
+ try:
291
+ language = Language(tssql.language())
292
+ parser = Parser(language)
293
+ source = (
294
+ content.encode("utf-8") if isinstance(content, str)
295
+ else content if content is not None
296
+ else path.read_bytes()
297
+ )
298
+ tree = parser.parse(source)
299
+ root = tree.root_node
300
+ except Exception as e:
301
+ return {"nodes": [], "edges": [], "error": str(e)}
302
+
303
+
304
+ stem = _file_stem(path)
305
+ str_path = str(path)
306
+ file_nid = _make_id(str_path)
307
+ nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
308
+ "source_file": str_path, "source_location": None}]
309
+ edges: list[dict] = []
310
+ seen_ids: set[str] = {file_nid}
311
+ table_nids: dict[str, str] = {} # name → nid for reference resolution
312
+
313
+ def _read(n) -> str:
314
+ return source[n.start_byte:n.end_byte].decode("utf-8", errors="replace")
315
+
316
+ def _obj_name(n) -> str | None:
317
+ for c in n.children:
318
+ if c.type == "object_reference":
319
+ return _read(c)
320
+ return None
321
+
322
+ def _add_node(nid: str, label: str, line: int) -> None:
323
+ if nid not in seen_ids:
324
+ seen_ids.add(nid)
325
+ nodes.append({"id": nid, "label": label, "file_type": "code",
326
+ "source_file": str_path, "source_location": f"L{line}"})
327
+ edges.append({"source": file_nid, "target": nid, "relation": "contains",
328
+ "confidence": "EXTRACTED", "source_file": str_path,
329
+ "source_location": f"L{line}", "weight": 1.0})
330
+
331
+ def _add_edge(src: str, tgt: str, relation: str, line: int) -> None:
332
+ edges.append({"source": src, "target": tgt, "relation": relation,
333
+ "confidence": "EXTRACTED", "source_file": str_path,
334
+ "source_location": f"L{line}", "weight": 1.0})
335
+
336
+ def _ref_stub(name: str) -> str:
337
+ """Sourceless bare-name stub for a table referenced but not defined here.
338
+
339
+ SQL references are NAME-based, so a table defined in another file (e.g.
340
+ prisma migration m2 referencing a table created in m1) can only resolve
341
+ at the corpus level. Minting `_make_id(stem, name)` under THIS file's
342
+ stem fabricated a node-less compound id — an absolute-path slug when the
343
+ input path was absolute — that could never match the real definition
344
+ (#2324). Instead emit a SOURCELESS stub, mirroring the Go extractor's
345
+ cross-file pattern (#1402): `_rewire_unique_stub_nodes` collapses it
346
+ onto the unique real table definition, and an unresolvable name survives
347
+ as a portable name-only node instead of dangling. No contains edge: a
348
+ sourced/contained stub would get the referencing file's path baked into
349
+ its id by disambiguation, blocking the rewire.
350
+ """
351
+ nid = _make_id(name)
352
+ if nid not in seen_ids:
353
+ seen_ids.add(nid)
354
+ nodes.append({"id": nid, "label": name, "file_type": "code",
355
+ "source_file": "", "source_location": "",
356
+ "origin_file": str_path})
357
+ return nid
358
+
359
+ def walk(node) -> None:
360
+ t = node.type
361
+ line = node.start_point[0] + 1
362
+
363
+ if t == "create_table":
364
+ name = _obj_name(node)
365
+ if name:
366
+ nid = _make_id(stem, name)
367
+ _add_node(nid, name, line)
368
+ table_nids[_norm_ident(name)] = nid
369
+ # Foreign key REFERENCES
370
+ for col in node.children:
371
+ if col.type == "column_definitions":
372
+ has_error = any(cd.type == "ERROR" for cd in col.children)
373
+ seen_refs: set[str] = set()
374
+ for cd in col.children:
375
+ if cd.type == "column_definition":
376
+ # Inline column-level REFERENCES
377
+ ref_name: str | None = None
378
+ found_ref = False
379
+ for cc in cd.children:
380
+ if cc.type == "keyword_references":
381
+ found_ref = True
382
+ elif found_ref and cc.type == "object_reference":
383
+ ref_name = _read(cc)
384
+ break
385
+ if ref_name:
386
+ ref_nid = table_nids.get(_norm_ident(ref_name)) or _ref_stub(ref_name)
387
+ _add_edge(nid, ref_nid, "references", line)
388
+ seen_refs.add(_norm_ident(ref_name))
389
+ elif cd.type == "constraints":
390
+ # Table-level FOREIGN KEY ... REFERENCES ... constraints
391
+ for constraint in cd.children:
392
+ if constraint.type != "constraint":
393
+ continue
394
+ ref_name = None
395
+ found_ref = False
396
+ for cc in constraint.children:
397
+ if cc.type == "keyword_references":
398
+ found_ref = True
399
+ elif found_ref and cc.type == "object_reference":
400
+ ref_name = _read(cc)
401
+ break
402
+ if ref_name:
403
+ ref_nid = table_nids.get(_norm_ident(ref_name)) or _ref_stub(ref_name)
404
+ _add_edge(nid, ref_nid, "references", line)
405
+ seen_refs.add(_norm_ident(ref_name))
406
+ if has_error:
407
+ # Dialect-specific syntax (e.g. Firebird COMPUTED BY) causes ERROR
408
+ # nodes that make the parser drop the trailing constraints block.
409
+ # Regex-scan the raw column_definitions text as fallback.
410
+ col_text = _read(col)
411
+ for rm in re.finditer(r"\bREFERENCES\s+([\w$]+)", col_text, re.IGNORECASE):
412
+ ref_name = rm.group(1)
413
+ if _norm_ident(ref_name) not in seen_refs:
414
+ ref_nid = table_nids.get(_norm_ident(ref_name)) or _ref_stub(ref_name)
415
+ _add_edge(nid, ref_nid, "references", line)
416
+ seen_refs.add(_norm_ident(ref_name))
417
+
418
+ elif t == "create_view":
419
+ name = _obj_name(node)
420
+ if name:
421
+ nid = _make_id(stem, name)
422
+ _add_node(nid, name, line)
423
+ table_nids[_norm_ident(name)] = nid
424
+ # FROM/JOIN table references inside view body
425
+ _walk_from_refs(node, nid, line)
426
+
427
+ elif t == "create_function":
428
+ name = _obj_name(node)
429
+ if name:
430
+ nid = _make_id(stem, name)
431
+ _add_node(nid, f"{name}()", line)
432
+ _walk_from_refs(node, nid, line)
433
+
434
+ elif t == "create_procedure":
435
+ name = _obj_name(node)
436
+ if name:
437
+ nid = _make_id(stem, name)
438
+ _add_node(nid, f"{name}()", line)
439
+ _walk_from_refs(node, nid, line)
440
+
441
+ elif t == "alter_table":
442
+ name = _obj_name(node)
443
+ if name:
444
+ src_nid = table_nids.get(_norm_ident(name))
445
+ if not src_nid:
446
+ # Subject table not defined in this file: sourceless stub,
447
+ # not a sourced wrong-stem node (#2324).
448
+ src_nid = _ref_stub(name)
449
+ table_nids[_norm_ident(name)] = src_nid
450
+ for child in node.children:
451
+ if child.type == "add_constraint":
452
+ for cc in child.children:
453
+ if cc.type != "constraint":
454
+ continue
455
+ found_ref = False
456
+ ref_name: str | None = None
457
+ for ccc in cc.children:
458
+ if ccc.type == "keyword_references":
459
+ found_ref = True
460
+ elif found_ref and ccc.type == "object_reference":
461
+ ref_name = _read(ccc)
462
+ break
463
+ if ref_name:
464
+ ref_nid = (table_nids.get(_norm_ident(ref_name))
465
+ or _ref_stub(ref_name))
466
+ _add_edge(src_nid, ref_nid, "references", line)
467
+
468
+ elif t == "create_trigger":
469
+ trig_name: str | None = None
470
+ tbl_name: str | None = None
471
+ after_trigger = False
472
+ after_for = False
473
+ for c in node.children:
474
+ if c.type == "keyword_trigger":
475
+ after_trigger = True
476
+ elif after_trigger and not trig_name and c.type == "object_reference":
477
+ trig_name = _read(c)
478
+ elif c.type == "keyword_for":
479
+ after_for = True
480
+ elif after_for and not tbl_name and c.type == "object_reference":
481
+ tbl_name = _read(c)
482
+ if trig_name:
483
+ trig_nid = _make_id(stem, trig_name)
484
+ _add_node(trig_nid, trig_name, line)
485
+ if tbl_name:
486
+ tbl_nid = table_nids.get(_norm_ident(tbl_name)) or _ref_stub(tbl_name)
487
+ _add_edge(trig_nid, tbl_nid, "triggers", line)
488
+
489
+ elif t == "create_index":
490
+ # CREATE [UNIQUE] INDEX [CONCURRENTLY] [IF NOT EXISTS] <name>
491
+ # ON <table> (...). Unlike CREATE POLICY (#3401) the grammar
492
+ # parses this statement fine; the walk simply never dispatched on
493
+ # it, so every index was silently dropped (#3467). The name is the
494
+ # identifier (or a quoted literal) before ON; the table is the
495
+ # object_reference after it. An unnamed index (`CREATE INDEX ON
496
+ # t (c)`) has nothing to name a node after and is skipped.
497
+ index_name: str | None = None
498
+ index_table: str | None = None
499
+ after_on = False
500
+ for c in node.children:
501
+ if c.type == "keyword_on":
502
+ after_on = True
503
+ elif not after_on and index_name is None and c.type in ("identifier", "literal"):
504
+ index_name = _read(c).strip('"`')
505
+ elif after_on and index_table is None and c.type == "object_reference":
506
+ index_table = _read(c)
507
+ if index_name:
508
+ index_nid = _make_id(stem, index_name)
509
+ _add_node(index_nid, index_name, line)
510
+ if index_table:
511
+ index_tbl_nid = (table_nids.get(_norm_ident(index_table))
512
+ or _ref_stub(index_table))
513
+ _add_edge(index_nid, index_tbl_nid, "indexes", line)
514
+
515
+ # NOTE: there is deliberately NO recovery scan on individual ERROR
516
+ # nodes. Any ERROR node anywhere makes root.has_error true, so the
517
+ # whole-file masked scan below this walk already recovers everything
518
+ # a per-node scan could — from the SAME shared _ROUTINE_RECOVERY_RX,
519
+ # with _add_node deduping by id. A per-node scan is not just
520
+ # redundant, it is unsound: _mask_sql_comments needs file-level
521
+ # context, and an ERROR fragment can begin MID-comment (tree-sitter's
522
+ # lexer does not nest /* */, so the text after an inner */ parses as
523
+ # code and lands in an ERROR blob with no comment opener in sight),
524
+ # which let commented-out DDL fabricate routine nodes.
525
+
526
+ elif t == "fb_proc_or_trigger":
527
+ text = _read(node)
528
+ m = re.match(
529
+ r"CREATE\s+(?:OR\s+(?:REPLACE|ALTER)\s+)?"
530
+ r"(PROCEDURE|TRIGGER|FUNCTION)\s+([\w$]+)",
531
+ text, re.IGNORECASE,
532
+ )
533
+ if m:
534
+ obj_type = m.group(1).upper()
535
+ obj_name = m.group(2)
536
+ obj_nid = _make_id(stem, obj_name)
537
+ label = obj_name if obj_type == "TRIGGER" else f"{obj_name}()"
538
+ _add_node(obj_nid, label, line)
539
+ if obj_type == "TRIGGER":
540
+ fm = re.search(r"\bFOR\s+([\w$]+)", text, re.IGNORECASE)
541
+ if fm:
542
+ tbl = fm.group(1)
543
+ tbl_nid = table_nids.get(_norm_ident(tbl)) or _ref_stub(tbl)
544
+ _add_edge(obj_nid, tbl_nid, "triggers", line)
545
+ _NON_TABLES = {
546
+ "select", "where", "set", "dual", "null", "true", "false",
547
+ "first", "skip", "rows", "next", "only", "lateral",
548
+ }
549
+ # Same CTE-blindness as the AST path (#2577): a `WITH <name> AS (`
550
+ # binding is statement-local, not a table, so its name must not
551
+ # become a reads_from stub. The regex has no scope tree, so the
552
+ # skip is body-wide — the right trade for a recovery path.
553
+ for cm in re.finditer(
554
+ r"(?:\bWITH\s+(?:RECURSIVE\s+)?|,\s*)([\w$]+)\s*(?:\([^()]*\))?\s+AS\s*\(",
555
+ text, re.IGNORECASE,
556
+ ):
557
+ _NON_TABLES.add(_norm_ident(cm.group(1)))
558
+ seen_tbls: set[str] = set()
559
+ for rm in re.finditer(r"\b(?:FROM|JOIN|INTO)\s+([\w$]+)", text, re.IGNORECASE):
560
+ tbl = rm.group(1)
561
+ if _norm_ident(tbl) not in _NON_TABLES and _norm_ident(tbl) not in seen_tbls:
562
+ seen_tbls.add(_norm_ident(tbl))
563
+ tbl_nid = table_nids.get(_norm_ident(tbl)) or _ref_stub(tbl)
564
+ _add_edge(obj_nid, tbl_nid, "reads_from", line)
565
+ for rm in re.finditer(r"\bUPDATE\s+([\w$]+)", text, re.IGNORECASE):
566
+ tbl = rm.group(1)
567
+ if _norm_ident(tbl) not in _NON_TABLES and _norm_ident(tbl) not in seen_tbls:
568
+ seen_tbls.add(_norm_ident(tbl))
569
+ tbl_nid = table_nids.get(_norm_ident(tbl)) or _ref_stub(tbl)
570
+ _add_edge(obj_nid, tbl_nid, "reads_from", line)
571
+
572
+ for child in node.children:
573
+ walk(child)
574
+
575
+ def _walk_from_refs(node, caller_nid: str, line: int,
576
+ cte_names: frozenset[str] = frozenset()) -> None:
577
+ """Recursively find FROM/JOIN table references inside a node, skipping CTEs.
578
+
579
+ A name bound by `WITH <name> AS (...)` is not a table: emitting it as a
580
+ `reads_from` target minted a bare `_ref_stub`, and because that stub is
581
+ intentionally sourceless (see `_ref_stub`) it carried no schema, file, or
582
+ language namespace, so a CTE named `levels` or `slug` collided with any
583
+ same-named node from another language during the build (#2577).
584
+
585
+ Scoping matters: a CTE is visible only inside the query that declares it,
586
+ and a `WITH` inside a subquery is scoped to that subquery alone. So the
587
+ active set is extended PER SUBTREE — each node's directly-owned `cte`
588
+ children (`create_query` for a statement-level WITH, `subquery` for a
589
+ nested one) join the set passed down into that node's recursion only. A
590
+ single statement-wide pre-collect would also suppress an OUTER reference
591
+ to a real table that merely shares a subquery-CTE's name
592
+ (`... FROM t2 JOIN (WITH t2 AS (...) SELECT ...) sub`), dropping the
593
+ real `-> t2` edge.
594
+ """
595
+ own: set[str] = set()
596
+ for c in node.children:
597
+ if c.type != "cte":
598
+ continue
599
+ # First identifier is the CTE's name; later ones are its column
600
+ # list (`WITH levels(a, b) AS (...)`), which must not be skipped.
601
+ for cc in c.children:
602
+ if cc.type in ("identifier", "object_reference"):
603
+ own.add(_norm_ident(_read(cc)))
604
+ break
605
+ if own:
606
+ cte_names = frozenset(cte_names | own)
607
+ if node.type in ("from", "join"):
608
+ for c in node.children:
609
+ if c.type == "relation":
610
+ for cc in c.children:
611
+ if cc.type == "object_reference":
612
+ tbl = _read(cc)
613
+ if _norm_ident(tbl) in cte_names:
614
+ continue
615
+ tbl_nid = table_nids.get(_norm_ident(tbl)) or _ref_stub(tbl)
616
+ _add_edge(caller_nid, tbl_nid, "reads_from",
617
+ c.start_point[0] + 1)
618
+ for child in node.children:
619
+ _walk_from_refs(child, caller_nid, line, cte_names)
620
+
621
+ # Pre-pass: register every table/view DEFINED in this file before walking,
622
+ # so forward references (a FK to a table created later in the same file)
623
+ # still resolve to the real sourced node instead of falling back to a stub.
624
+ def _collect_defined_names(node) -> None:
625
+ if node.type in ("create_table", "create_view"):
626
+ name = _obj_name(node)
627
+ if name:
628
+ table_nids[_norm_ident(name)] = _make_id(stem, name)
629
+ for child in node.children:
630
+ _collect_defined_names(child)
631
+
632
+ _collect_defined_names(root)
633
+
634
+ # Secondary bare-name aliases: a reference written without a schema
635
+ # (`REFERENCES users`) should resolve to a schema-qualified definition
636
+ # (`public.users`) when that is unambiguous. Never shadow an explicit
637
+ # definition, and skip bare names defined under more than one schema.
638
+ bare_candidates: dict[str, str | None] = {}
639
+ for key, alias_nid in table_nids.items():
640
+ if "." in key:
641
+ bare = key.rsplit(".", 1)[1]
642
+ bare_candidates[bare] = (
643
+ alias_nid if bare_candidates.get(bare, alias_nid) == alias_nid else None
644
+ )
645
+ for bare, alias_nid in bare_candidates.items():
646
+ if alias_nid is not None and bare not in table_nids:
647
+ table_nids[bare] = alias_nid
648
+
649
+ for stmt in root.children:
650
+ if stmt.type == "statement":
651
+ for child in stmt.children:
652
+ walk(child)
653
+ elif stmt.type == "transaction":
654
+ # BEGIN; ... COMMIT; wraps DDL in a transaction node whose children
655
+ # are statement nodes, not direct create_table nodes (#2953).
656
+ walk(stmt)
657
+ elif stmt.type in ("fb_proc_or_trigger", "set_term", "declare_external_function", "ERROR"):
658
+ walk(stmt)
659
+
660
+ # Global regex fallback: catch any REFERENCES missed due to ERROR nodes in the parse tree
661
+ # (e.g. Firebird COMPUTED BY columns push constraints out of the tree entirely).
662
+ # Snapshot after tree walk so we don't re-emit edges already captured above.
663
+ emitted = {(e["source"], e["target"]) for e in edges if e["relation"] == "references"}
664
+ src_text = source.decode("utf-8", errors="replace")
665
+ for m in re.finditer(r"CREATE\s+TABLE\s+([\w$]+)\s*\(", src_text, re.IGNORECASE):
666
+ tbl_name = m.group(1)
667
+ tbl_nid = table_nids.get(_norm_ident(tbl_name))
668
+ if tbl_nid is None:
669
+ continue
670
+ tbl_line = src_text[: m.start()].count("\n") + 1
671
+ tail = src_text[m.start():]
672
+ end = re.search(r"(?:^|\n)(?:CREATE|SET\s+TERM|ALTER)\s", tail[1:], re.IGNORECASE)
673
+ block = tail[: end.start() + 1] if end else tail
674
+ for rm in re.finditer(r"\bREFERENCES\s+([\w$]+)", block, re.IGNORECASE):
675
+ ref_name = rm.group(1)
676
+ ref_nid = table_nids.get(_norm_ident(ref_name)) or _ref_stub(ref_name)
677
+ if (tbl_nid, ref_nid) not in emitted:
678
+ _add_edge(tbl_nid, ref_nid, "references", tbl_line)
679
+ emitted.add((tbl_nid, ref_nid))
680
+
681
+ # Global regex fallback for routines (#2180). PL/pgSQL bodies break the parse
682
+ # in more than one shape, and only the first was recovered before:
683
+ # 1. the whole CREATE lands in one ERROR node -> handled in walk()
684
+ # 2. the statement is shredded into loose top-level tokens
685
+ # (keyword_create/keyword_function/object_reference/... ) and the ERROR
686
+ # node holds only the offending body line, e.g. `PERFORM x();` or
687
+ # `x := 1;` -- so no CREATE text is inside any ERROR node at all
688
+ # 3. the name is a delimited identifier — quoted ("public"."fn") or
689
+ # T-SQL-bracketed ([dbo].[usp_Load]) — which a bare [\w$.]+ pattern
690
+ # cannot match
691
+ # Shapes 2 and 3 silently dropped the routine: no node, no warning, exit 0.
692
+ # Scanning the raw source catches all three, and _add_node dedupes by id so
693
+ # routines already recovered from the tree are not emitted twice.
694
+ #
695
+ # Gate on a failed parse: a cleanly-parsing file must NOT have routines
696
+ # fabricated from MySQL `CREATE FUNCTION IF NOT EXISTS` (which would
697
+ # capture `IF`) or other shapes the mask does not model (double-quoted
698
+ # strings in MySQL's default mode, PostgreSQL dollar-quoted bodies). Every
699
+ # observed drop shape leaves an ERROR node in the tree, so has_error loses
700
+ # nothing while protecting clean corpora (#2180 follow-up).
701
+ if root.has_error:
702
+ # The mask blanks comments (nesting-aware) and single-quoted strings
703
+ # (offset-preserving), so commented-out DDL and single-quoted dynamic
704
+ # SQL cannot fabricate a routine when an unrelated error arms this
705
+ # scan; string shapes the mask does not model rely on the has_error
706
+ # gate alone. Preserved delimited identifiers keep their text
707
+ # verbatim (they carry the recoverable names), so a match whose
708
+ # CREATE keyword STARTS inside one is identifier data, not DDL
709
+ # ('SELECT 1 AS [CREATE PROCEDURE x pending]') and is skipped — the
710
+ # name a genuine statement captures is allowed to be a delimited
711
+ # identifier; its CREATE never is.
712
+ masked_src, ident_spans = _scan_sql(src_text)
713
+ for m in _ROUTINE_RECOVERY_RX.finditer(masked_src):
714
+ if any(s <= m.start() < e for s, e in ident_spans):
715
+ continue
716
+ fn_name = m.group(1)
717
+ fn_line = src_text[: m.start()].count("\n") + 1
718
+ _add_node(_make_id(stem, fn_name), f"{fn_name}()", fn_line)
719
+
720
+ return {"nodes": nodes, "edges": edges}