graphitect 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graphify/__init__.py +30 -0
- graphify/__main__.py +757 -0
- graphify/_minhash.py +107 -0
- graphify/affected.py +318 -0
- graphify/always_on/agents-md.md +12 -0
- graphify/always_on/antigravity-rules.md +14 -0
- graphify/always_on/claude-md.md +9 -0
- graphify/always_on/gemini-md.md +9 -0
- graphify/always_on/kiro-steering.md +5 -0
- graphify/always_on/vscode-instructions.md +17 -0
- graphify/analyze.py +769 -0
- graphify/benchmark.py +152 -0
- graphify/build.py +2300 -0
- graphify/cache.py +1746 -0
- graphify/callflow_html.py +2051 -0
- graphify/cargo_introspect.py +109 -0
- graphify/cli.py +4745 -0
- graphify/cluster.py +409 -0
- graphify/command-kilo.md +15 -0
- graphify/cross_repo_calls.py +216 -0
- graphify/cross_repo_types.py +75 -0
- graphify/csharp_dispatch.py +154 -0
- graphify/dedup.py +1213 -0
- graphify/detect.py +2566 -0
- graphify/diagnostics.py +406 -0
- graphify/export.py +1349 -0
- graphify/exporters/__init__.py +1 -0
- graphify/exporters/base.py +14 -0
- graphify/exporters/graphdb.py +173 -0
- graphify/exporters/html.py +637 -0
- graphify/extract.py +7856 -0
- graphify/extractors/MIGRATION.md +107 -0
- graphify/extractors/__init__.py +66 -0
- graphify/extractors/apex.py +215 -0
- graphify/extractors/base.py +85 -0
- graphify/extractors/bash.py +579 -0
- graphify/extractors/blade.py +53 -0
- graphify/extractors/commonlisp.py +540 -0
- graphify/extractors/csharp.py +448 -0
- graphify/extractors/dart.py +564 -0
- graphify/extractors/dm.py +494 -0
- graphify/extractors/elixir.py +241 -0
- graphify/extractors/engine.py +6509 -0
- graphify/extractors/fortran.py +311 -0
- graphify/extractors/go.py +527 -0
- graphify/extractors/json_config.py +240 -0
- graphify/extractors/julia.py +289 -0
- graphify/extractors/markdown.py +408 -0
- graphify/extractors/models.py +131 -0
- graphify/extractors/objc.py +566 -0
- graphify/extractors/ocaml.py +289 -0
- graphify/extractors/pascal.py +688 -0
- graphify/extractors/pascal_forms.py +196 -0
- graphify/extractors/powershell.py +522 -0
- graphify/extractors/razor.py +192 -0
- graphify/extractors/resolution.py +3584 -0
- graphify/extractors/robot.py +296 -0
- graphify/extractors/rust.py +470 -0
- graphify/extractors/sln.py +92 -0
- graphify/extractors/sql.py +720 -0
- graphify/extractors/terraform.py +181 -0
- graphify/extractors/verilog.py +329 -0
- graphify/extractors/zig.py +181 -0
- graphify/file_slice.py +246 -0
- graphify/global_graph.py +194 -0
- graphify/google_workspace.py +237 -0
- graphify/hooks.py +933 -0
- graphify/ids.py +93 -0
- graphify/ingest.py +358 -0
- graphify/install.py +2366 -0
- graphify/llm.py +3544 -0
- graphify/manifest.py +4 -0
- graphify/manifest_ingest.py +311 -0
- graphify/mcp_ingest.py +386 -0
- graphify/multigraph_compat.py +212 -0
- graphify/pascal_resolution.py +129 -0
- graphify/paths.py +436 -0
- graphify/pg_introspect.py +165 -0
- graphify/prs.py +770 -0
- graphify/querylog.py +80 -0
- graphify/reflect.py +882 -0
- graphify/report.py +346 -0
- graphify/resolver_registry.py +85 -0
- graphify/ruby_resolution.py +242 -0
- graphify/scip_ingest.py +363 -0
- graphify/security.py +460 -0
- graphify/semantic_cleanup.py +336 -0
- graphify/serve.py +2608 -0
- graphify/skill-agents.md +710 -0
- graphify/skill-aider.md +1283 -0
- graphify/skill-amp.md +710 -0
- graphify/skill-claw.md +713 -0
- graphify/skill-codex.md +710 -0
- graphify/skill-copilot.md +713 -0
- graphify/skill-devin.md +1410 -0
- graphify/skill-droid.md +710 -0
- graphify/skill-kilo.md +722 -0
- graphify/skill-kiro.md +713 -0
- graphify/skill-opencode.md +705 -0
- graphify/skill-pi.md +713 -0
- graphify/skill-trae.md +711 -0
- graphify/skill-vscode.md +709 -0
- graphify/skill-windows.md +755 -0
- graphify/skill.md +713 -0
- graphify/skills/agents/references/add-watch.md +56 -0
- graphify/skills/agents/references/exports.md +87 -0
- graphify/skills/agents/references/extraction-spec.md +70 -0
- graphify/skills/agents/references/github-and-merge.md +46 -0
- graphify/skills/agents/references/hooks.md +33 -0
- graphify/skills/agents/references/query.md +311 -0
- graphify/skills/agents/references/transcribe.md +52 -0
- graphify/skills/agents/references/update.md +210 -0
- graphify/skills/amp/references/add-watch.md +56 -0
- graphify/skills/amp/references/exports.md +87 -0
- graphify/skills/amp/references/extraction-spec.md +70 -0
- graphify/skills/amp/references/github-and-merge.md +46 -0
- graphify/skills/amp/references/hooks.md +33 -0
- graphify/skills/amp/references/query.md +311 -0
- graphify/skills/amp/references/transcribe.md +52 -0
- graphify/skills/amp/references/update.md +210 -0
- graphify/skills/claude/references/add-watch.md +56 -0
- graphify/skills/claude/references/exports.md +87 -0
- graphify/skills/claude/references/extraction-spec.md +70 -0
- graphify/skills/claude/references/github-and-merge.md +46 -0
- graphify/skills/claude/references/hooks.md +33 -0
- graphify/skills/claude/references/query.md +311 -0
- graphify/skills/claude/references/transcribe.md +52 -0
- graphify/skills/claude/references/update.md +210 -0
- graphify/skills/claw/references/add-watch.md +56 -0
- graphify/skills/claw/references/exports.md +87 -0
- graphify/skills/claw/references/extraction-spec.md +31 -0
- graphify/skills/claw/references/github-and-merge.md +46 -0
- graphify/skills/claw/references/hooks.md +33 -0
- graphify/skills/claw/references/query.md +311 -0
- graphify/skills/claw/references/transcribe.md +52 -0
- graphify/skills/claw/references/update.md +210 -0
- graphify/skills/codex/references/add-watch.md +56 -0
- graphify/skills/codex/references/exports.md +87 -0
- graphify/skills/codex/references/extraction-spec.md +31 -0
- graphify/skills/codex/references/github-and-merge.md +46 -0
- graphify/skills/codex/references/hooks.md +33 -0
- graphify/skills/codex/references/query.md +311 -0
- graphify/skills/codex/references/transcribe.md +52 -0
- graphify/skills/codex/references/update.md +210 -0
- graphify/skills/copilot/references/add-watch.md +56 -0
- graphify/skills/copilot/references/exports.md +87 -0
- graphify/skills/copilot/references/extraction-spec.md +70 -0
- graphify/skills/copilot/references/github-and-merge.md +46 -0
- graphify/skills/copilot/references/hooks.md +33 -0
- graphify/skills/copilot/references/query.md +311 -0
- graphify/skills/copilot/references/transcribe.md +52 -0
- graphify/skills/copilot/references/update.md +210 -0
- graphify/skills/droid/references/add-watch.md +56 -0
- graphify/skills/droid/references/exports.md +87 -0
- graphify/skills/droid/references/extraction-spec.md +70 -0
- graphify/skills/droid/references/github-and-merge.md +46 -0
- graphify/skills/droid/references/hooks.md +33 -0
- graphify/skills/droid/references/query.md +311 -0
- graphify/skills/droid/references/transcribe.md +52 -0
- graphify/skills/droid/references/update.md +210 -0
- graphify/skills/kilo/references/add-watch.md +56 -0
- graphify/skills/kilo/references/exports.md +87 -0
- graphify/skills/kilo/references/extraction-spec.md +70 -0
- graphify/skills/kilo/references/github-and-merge.md +46 -0
- graphify/skills/kilo/references/hooks.md +33 -0
- graphify/skills/kilo/references/query.md +311 -0
- graphify/skills/kilo/references/transcribe.md +52 -0
- graphify/skills/kilo/references/update.md +210 -0
- graphify/skills/kiro/references/add-watch.md +56 -0
- graphify/skills/kiro/references/exports.md +87 -0
- graphify/skills/kiro/references/extraction-spec.md +31 -0
- graphify/skills/kiro/references/github-and-merge.md +46 -0
- graphify/skills/kiro/references/hooks.md +33 -0
- graphify/skills/kiro/references/query.md +311 -0
- graphify/skills/kiro/references/transcribe.md +52 -0
- graphify/skills/kiro/references/update.md +210 -0
- graphify/skills/opencode/references/add-watch.md +56 -0
- graphify/skills/opencode/references/exports.md +87 -0
- graphify/skills/opencode/references/extraction-spec.md +70 -0
- graphify/skills/opencode/references/github-and-merge.md +46 -0
- graphify/skills/opencode/references/hooks.md +33 -0
- graphify/skills/opencode/references/query.md +311 -0
- graphify/skills/opencode/references/transcribe.md +52 -0
- graphify/skills/opencode/references/update.md +210 -0
- graphify/skills/pi/references/add-watch.md +56 -0
- graphify/skills/pi/references/exports.md +87 -0
- graphify/skills/pi/references/extraction-spec.md +31 -0
- graphify/skills/pi/references/github-and-merge.md +46 -0
- graphify/skills/pi/references/hooks.md +33 -0
- graphify/skills/pi/references/query.md +311 -0
- graphify/skills/pi/references/transcribe.md +52 -0
- graphify/skills/pi/references/update.md +210 -0
- graphify/skills/trae/references/add-watch.md +56 -0
- graphify/skills/trae/references/exports.md +87 -0
- graphify/skills/trae/references/extraction-spec.md +70 -0
- graphify/skills/trae/references/github-and-merge.md +46 -0
- graphify/skills/trae/references/hooks.md +35 -0
- graphify/skills/trae/references/query.md +311 -0
- graphify/skills/trae/references/transcribe.md +52 -0
- graphify/skills/trae/references/update.md +210 -0
- graphify/skills/vscode/references/add-watch.md +56 -0
- graphify/skills/vscode/references/exports.md +87 -0
- graphify/skills/vscode/references/extraction-spec.md +70 -0
- graphify/skills/vscode/references/github-and-merge.md +46 -0
- graphify/skills/vscode/references/hooks.md +33 -0
- graphify/skills/vscode/references/query.md +311 -0
- graphify/skills/vscode/references/transcribe.md +52 -0
- graphify/skills/vscode/references/update.md +210 -0
- graphify/skills/windows/references/add-watch.md +56 -0
- graphify/skills/windows/references/exports.md +87 -0
- graphify/skills/windows/references/extraction-spec.md +70 -0
- graphify/skills/windows/references/github-and-merge.md +46 -0
- graphify/skills/windows/references/hooks.md +33 -0
- graphify/skills/windows/references/query.md +311 -0
- graphify/skills/windows/references/transcribe.md +52 -0
- graphify/skills/windows/references/update.md +210 -0
- graphify/symbol_resolution.py +556 -0
- graphify/transcribe.py +186 -0
- graphify/tree_html.py +603 -0
- graphify/validate.py +95 -0
- graphify/watch.py +2280 -0
- graphify/wiki.py +405 -0
- graphitect/__init__.py +28 -0
- graphitect/__main__.py +4 -0
- graphitect/_vendor/__init__.py +2 -0
- graphitect/_vendor/archify/LICENSE +22 -0
- graphitect/_vendor/archify/SKILL.md +137 -0
- graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
- graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
- graphitect/_vendor/archify/assets/template.html +14935 -0
- graphitect/_vendor/archify/bin/archify.mjs +2091 -0
- graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
- graphitect/_vendor/archify/bin/preview.mjs +653 -0
- graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
- graphitect/_vendor/archify/brand-marks/README.md +31 -0
- graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
- graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
- graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
- graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
- graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
- graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
- graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
- graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
- graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
- graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
- graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
- graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
- graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
- graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
- graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
- graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
- graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
- graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
- graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
- graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
- graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
- graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
- graphitect/_vendor/archify/package-lock.json +149 -0
- graphitect/_vendor/archify/package.json +39 -0
- graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
- graphitect/_vendor/archify/references/authoring-contract.md +243 -0
- graphitect/_vendor/archify/references/brand-marks.md +65 -0
- graphitect/_vendor/archify/references/delivery-contract.md +120 -0
- graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
- graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
- graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
- graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
- graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
- graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
- graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
- graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
- graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
- graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
- graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
- graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
- graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
- graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
- graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
- graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
- graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
- graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
- graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
- graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
- graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
- graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
- graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
- graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
- graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
- graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
- graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
- graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
- graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
- graphitect/_vendor/archify/schemas/README.md +211 -0
- graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
- graphitect/_vendor/archify/schemas/common.schema.json +115 -0
- graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
- graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
- graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
- graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
- graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
- graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
- graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
- graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
- graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
- graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
- graphitect/_vendor/archify/skill-release.json +10 -0
- graphitect/cli.py +981 -0
- graphitect/deliver/__init__.py +5 -0
- graphitect/deliver/archify_adapter.py +1877 -0
- graphitect/deliver/archify_ir.py +160 -0
- graphitect/deliver/archify_repair.py +135 -0
- graphitect/deliver/doc_compiler.py +916 -0
- graphitect/ground/__init__.py +5 -0
- graphitect/ground/describe_source.py +27 -0
- graphitect/ground/fullread_source.py +56 -0
- graphitect/ground/graphify_source.py +107 -0
- graphitect/models.py +118 -0
- graphitect/skill/SKILL.md +80 -0
- graphitect/skill/agents/openai.yaml +4 -0
- graphitect/synthesize/__init__.py +5 -0
- graphitect/synthesize/engine.py +281 -0
- graphitect/synthesize/llm_backend.py +331 -0
- graphitect/synthesize/questions.py +139 -0
- graphitect/synthesize/rubric.py +104 -0
- graphitect-0.2.0.dist-info/METADATA +284 -0
- graphitect-0.2.0.dist-info/RECORD +336 -0
- graphitect-0.2.0.dist-info/WHEEL +5 -0
- graphitect-0.2.0.dist-info/entry_points.txt +2 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
- graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
- graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
- graphitect-0.2.0.dist-info/top_level.txt +2 -0
|
@@ -0,0 +1,720 @@
|
|
|
1
|
+
"""Sql extractor. Moved verbatim from graphify/extract.py."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import re
|
|
5
|
+
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from graphify.extractors.base import _file_stem, _make_id
|
|
8
|
+
|
|
9
|
+
# Recovers CREATE FUNCTION/PROCEDURE statements the grammar could not parse
|
|
10
|
+
# structurally. Used by BOTH recovery sites — the walk-time ERROR-node scan and
|
|
11
|
+
# the whole-file has_error fallback (#2180). They MUST share one pattern: when
|
|
12
|
+
# they disagreed, the same statement produced two nodes with different names
|
|
13
|
+
# (the ERROR scan captured `dbo.[usp_Mixed]`, the fallback stopped at `dbo`,
|
|
14
|
+
# and _add_node's id-dedupe never fired because the ids differed).
|
|
15
|
+
#
|
|
16
|
+
# Each name part is a bare identifier, a double-quoted (delimited) one, or a
|
|
17
|
+
# T-SQL bracket-delimited one, so CREATE OR REPLACE FUNCTION "public"."fn"(...)
|
|
18
|
+
# and CREATE PROCEDURE [dbo].[usp_Load] ... are both recovered. A bare [\w$.]+
|
|
19
|
+
# stops dead at the leading delimiter, which silently dropped every quoted
|
|
20
|
+
# PL/pgSQL routine (#2180) and every bracket-named T-SQL procedure. T-SQL's
|
|
21
|
+
# AS BEGIN...END body idiom always lands in recovery — the grammar has no
|
|
22
|
+
# create_procedure parse for it — and T-SQL spells re-creation CREATE OR ALTER
|
|
23
|
+
# (it has no OR REPLACE), so accept that form too, mirroring fb_proc_or_trigger.
|
|
24
|
+
# Inside a bracket-delimited part, a literal ] is escaped by doubling
|
|
25
|
+
# ([a]]b] names the identifier a]b), so consume ]] before treating a
|
|
26
|
+
# lone ] as the closing delimiter — stopping at the first ] truncated
|
|
27
|
+
# the name and minted a phantom that could collide with a real [a].
|
|
28
|
+
# PROC is T-SQL's official shorthand for PROCEDURE and equally common in the
|
|
29
|
+
# wild; the optional (?:EDURE)? still requires trailing whitespace, so a word
|
|
30
|
+
# that merely starts with PROC cannot match.
|
|
31
|
+
# \bCREATE: without the boundary, CREATE matched inside a bare word, so
|
|
32
|
+
# 'SELECT AUTOCREATE PROCEDURE x FROM t;' in an error-bearing file minted a
|
|
33
|
+
# phantom routine x() (delimited identifiers are span-skipped at the scan
|
|
34
|
+
# site, but a bare word has no span).
|
|
35
|
+
_ROUTINE_RECOVERY_RX = re.compile(
|
|
36
|
+
r"\bCREATE\s+(?:OR\s+(?:REPLACE|ALTER)\s+)?(?:FUNCTION|PROC(?:EDURE)?)\s+"
|
|
37
|
+
r"(?:IF\s+NOT\s+EXISTS\s+)?"
|
|
38
|
+
r"((?:\"(?:[^\"\n]|\"\")+\"|\[(?:[^\]\n]|\]\])+\]|[\w$]+)"
|
|
39
|
+
r"(?:\s*\.\s*(?:\"(?:[^\"\n]|\"\")+\"|\[(?:[^\]\n]|\]\])+\]|[\w$]+))*)",
|
|
40
|
+
re.IGNORECASE,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
# _mask_sql_comments is a linear character scanner, not a regex: the four
|
|
44
|
+
# span kinds interact in ways a single pattern cannot express safely —
|
|
45
|
+
# comment-opener parity inside strings, nested block comments, and
|
|
46
|
+
# end-of-line abandonment of unclosed literals. Span handling:
|
|
47
|
+
#
|
|
48
|
+
# - single-quoted strings are BLANKED like comments: routine names never
|
|
49
|
+
# live in single quotes, and dynamic SQL (EXEC(N'CREATE PROC [dbo].[Fake]
|
|
50
|
+
# ...')) would otherwise fabricate a routine node whenever an unrelated
|
|
51
|
+
# parse error arms the whole-file scan. (Dialects where other quoting
|
|
52
|
+
# carries strings — MySQL double quotes, PostgreSQL dollar-quoting — are
|
|
53
|
+
# NOT modelled; DDL inside those still reaches the scan.)
|
|
54
|
+
# - double-quoted and bracket-delimited identifiers are PRESERVED verbatim —
|
|
55
|
+
# they are exactly the delimited names the recovery regex must see ("" and
|
|
56
|
+
# ]] escapes consumed, mirroring _ROUTINE_RECOVERY_RX).
|
|
57
|
+
# - line comments blank to end-of-line; block comments blank to their
|
|
58
|
+
# matching */ with NESTING tracked (SQL Server and PostgreSQL both nest
|
|
59
|
+
# /* */, and a lazy first-*/ match let commented-out DDL inside a nested
|
|
60
|
+
# comment fabricate nodes). MySQL and Oracle do NOT nest — there the
|
|
61
|
+
# depth tracking over-blanks, losing (never fabricating) a routine after
|
|
62
|
+
# an inner */; the primary T-SQL/PostgreSQL targets win that trade. An
|
|
63
|
+
# UNCLOSED block comment blanks to end-of-file, matching SQL semantics.
|
|
64
|
+
#
|
|
65
|
+
# Literals are deliberately line-scoped: this mask only runs on files that
|
|
66
|
+
# already failed to parse, where an unclosed delimiter is likely, and a
|
|
67
|
+
# multi-line literal span would let one unclosed quote swallow real DDL
|
|
68
|
+
# below it. The cost is asymmetric by kind. A single-quoted string that
|
|
69
|
+
# continues past its line is blanked only up to the newline, so a comment
|
|
70
|
+
# opener inside it cannot fire on that line — but its CONTINUATION lines are
|
|
71
|
+
# scanned as code, and DDL there fabricates: multi-line dynamic SQL
|
|
72
|
+
# (SET @sql = N\'\n CREATE PROC ...\') is a KNOWN HOLE, alongside the
|
|
73
|
+
# unmodelled quoting dialects above; only same-line dynamic SQL is blanked.
|
|
74
|
+
# Closing it would need a real string heuristic (e.g. an end-of-line opening
|
|
75
|
+
# quote), judged not worth the swallow risk in a recovery-only path.
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _scan_sql(text: str) -> tuple[str, list[tuple[int, int]]]:
|
|
79
|
+
"""Blank comment and string-literal spans, preserving every offset.
|
|
80
|
+
|
|
81
|
+
One output character per input character: non-newline characters inside
|
|
82
|
+
a blanked span become spaces and newlines are kept, so positions and
|
|
83
|
+
line numbers computed against the masked text are valid against the
|
|
84
|
+
original. Double-quoted and bracket-delimited identifiers are preserved
|
|
85
|
+
verbatim (they carry recoverable routine names); single-quoted strings,
|
|
86
|
+
line comments, and (nesting-aware) block comments are blanked. Used by
|
|
87
|
+
the routine-recovery scan so CREATE PROCEDURE/FUNCTION DDL reachable
|
|
88
|
+
only through a comment or a single-quoted string cannot fabricate a
|
|
89
|
+
routine node when an unrelated parse error arms recovery.
|
|
90
|
+
|
|
91
|
+
Returns (masked_text, ident_spans) where ident_spans holds the [start,
|
|
92
|
+
end) of every PRESERVED delimited identifier: preserved spans keep their
|
|
93
|
+
text verbatim, so DDL keywords inside one are still visible in the
|
|
94
|
+
masked text, and the recovery scan must skip a match that starts there
|
|
95
|
+
(identifier data, not DDL — 'SELECT 1 AS [CREATE PROCEDURE x pending]'
|
|
96
|
+
must not mint a routine).
|
|
97
|
+
"""
|
|
98
|
+
ident_spans: list[tuple[int, int]] = []
|
|
99
|
+
out: list[str] = []
|
|
100
|
+
i, n = 0, len(text)
|
|
101
|
+
|
|
102
|
+
def _blank(upto: int) -> int:
|
|
103
|
+
"""Blank [i, upto), keeping newlines; return upto."""
|
|
104
|
+
for c in text[i:upto]:
|
|
105
|
+
out.append("\n" if c == "\n" else " ")
|
|
106
|
+
return upto
|
|
107
|
+
|
|
108
|
+
def _blank_tail_and_carry(start: int) -> int:
|
|
109
|
+
"""Blank from start to end-of-line, carrying comment state forward.
|
|
110
|
+
|
|
111
|
+
The union-of-readings blank for an ambiguous stretch: the rest of the
|
|
112
|
+
line is blanked outright; if the raw text of that stretch leaves a /*
|
|
113
|
+
unclosed on its own line (the maximum comment depth any reading could
|
|
114
|
+
be left holding), blanking continues, nesting-aware, to the closing
|
|
115
|
+
*/ or EOF. Where a carry closes MID-line the same rule applies to the
|
|
116
|
+
remainder of that line — under the reading where the carry never
|
|
117
|
+
opened, that whole line may be a comment or a string, so emitting the
|
|
118
|
+
post-*/ text verbatim exposed it (found by differential fuzzing).
|
|
119
|
+
Repeats until a line ends with no carry pending. Appends one output
|
|
120
|
+
character per input character; returns the resume index.
|
|
121
|
+
"""
|
|
122
|
+
k = start
|
|
123
|
+
while True:
|
|
124
|
+
eol = text.find("\n", k)
|
|
125
|
+
eol = n if eol == -1 else eol
|
|
126
|
+
depth = 0
|
|
127
|
+
m2 = k
|
|
128
|
+
while m2 < eol:
|
|
129
|
+
if text.startswith("/*", m2):
|
|
130
|
+
depth += 1
|
|
131
|
+
m2 += 2
|
|
132
|
+
elif text.startswith("*/", m2):
|
|
133
|
+
if depth:
|
|
134
|
+
depth -= 1
|
|
135
|
+
m2 += 2
|
|
136
|
+
else:
|
|
137
|
+
m2 += 1
|
|
138
|
+
for ch in text[k:eol]:
|
|
139
|
+
out.append("\n" if ch == "\n" else " ")
|
|
140
|
+
k = eol
|
|
141
|
+
if not depth:
|
|
142
|
+
return k
|
|
143
|
+
j2 = k
|
|
144
|
+
while j2 < n and depth:
|
|
145
|
+
if text.startswith("/*", j2):
|
|
146
|
+
depth += 1
|
|
147
|
+
j2 += 2
|
|
148
|
+
elif text.startswith("*/", j2):
|
|
149
|
+
depth -= 1
|
|
150
|
+
j2 += 2
|
|
151
|
+
else:
|
|
152
|
+
j2 += 1
|
|
153
|
+
for ch in text[k:j2]:
|
|
154
|
+
out.append("\n" if ch == "\n" else " ")
|
|
155
|
+
k = j2
|
|
156
|
+
if k >= n:
|
|
157
|
+
return k
|
|
158
|
+
# the carry closed mid-line: the remainder of THIS line is the
|
|
159
|
+
# same ambiguous stretch — loop and blank it too
|
|
160
|
+
|
|
161
|
+
while i < n:
|
|
162
|
+
c = text[i]
|
|
163
|
+
if c == "'":
|
|
164
|
+
# Single-quoted string: blank it. '' is an escaped quote; a
|
|
165
|
+
# newline abandons the literal (see the comment above).
|
|
166
|
+
j = i + 1
|
|
167
|
+
while j < n and text[j] != "\n":
|
|
168
|
+
if text[j] == "'":
|
|
169
|
+
if j + 1 < n and text[j + 1] == "'":
|
|
170
|
+
j += 2
|
|
171
|
+
continue
|
|
172
|
+
j += 1
|
|
173
|
+
break
|
|
174
|
+
j += 1
|
|
175
|
+
i = _blank(j)
|
|
176
|
+
elif c == '"' or c == "[":
|
|
177
|
+
# Delimited identifier: preserve verbatim. Doubled closers are
|
|
178
|
+
# escapes. A span is DISTRUSTED when it is unterminated (no
|
|
179
|
+
# closer before the newline) or would swallow a comment opener
|
|
180
|
+
# on its way to the closer ('SELECT [Col FROM t -- CREATE PROC
|
|
181
|
+
# [dbo]' closes on [dbo]'s bracket) — a stray delimiter is
|
|
182
|
+
# ordinary in exactly the broken files this mask runs on.
|
|
183
|
+
#
|
|
184
|
+
# A distrusted span is irreducibly ambiguous (identifier data vs
|
|
185
|
+
# stray delimiter before real comments/strings), and any attempt
|
|
186
|
+
# to pick one reading exposed text the other reading blanks —
|
|
187
|
+
# re-emitting the delimiter and rescanning even re-paired later
|
|
188
|
+
# single quotes and uncovered dynamic SQL. So blank the UNION of
|
|
189
|
+
# every reading: the rest of the line is blanked outright, and
|
|
190
|
+
# any raw /* on it with no later */ on the same line carries
|
|
191
|
+
# forward as (nesting-aware) comment state, since some reading
|
|
192
|
+
# may have left it open — and where that carry closes mid-line,
|
|
193
|
+
# the remainder of THAT line gets the same treatment, repeated
|
|
194
|
+
# until a line ends carry-free (under the no-carry reading the
|
|
195
|
+
# close line may itself be all comment or string, so emitting its
|
|
196
|
+
# post-*/ tail verbatim was an exposure). Over-blanking loses at
|
|
197
|
+
# most routines on lines already entangled with the broken one (a
|
|
198
|
+
# conservative false negative; a routine named like [a--b] is
|
|
199
|
+
# inside that loss); under-blanking is what fabricates, and every
|
|
200
|
+
# reading's blank set stays a subset of this one — except where a
|
|
201
|
+
# */ + * versus * + /* token split makes two readings consume the
|
|
202
|
+
# same /, an irreducible divergence whose only closure would be
|
|
203
|
+
# blanking to EOF on every */* sequence (accepted, documented
|
|
204
|
+
# limitation; the token sequence appears in no dialect's idiom).
|
|
205
|
+
closer = '"' if c == '"' else "]"
|
|
206
|
+
j = i + 1
|
|
207
|
+
closed = False
|
|
208
|
+
while j < n and text[j] != "\n":
|
|
209
|
+
if text[j] == closer:
|
|
210
|
+
if j + 1 < n and text[j + 1] == closer:
|
|
211
|
+
j += 2
|
|
212
|
+
continue
|
|
213
|
+
j += 1
|
|
214
|
+
closed = True
|
|
215
|
+
break
|
|
216
|
+
j += 1
|
|
217
|
+
span = text[i:j]
|
|
218
|
+
if closed and "--" not in span and "/*" not in span:
|
|
219
|
+
ident_spans.append((i, j))
|
|
220
|
+
out.append(span)
|
|
221
|
+
i = j
|
|
222
|
+
else:
|
|
223
|
+
i = _blank_tail_and_carry(i)
|
|
224
|
+
elif c == "-" and i + 1 < n and text[i + 1] == "-":
|
|
225
|
+
j = i
|
|
226
|
+
while j < n and text[j] != "\n":
|
|
227
|
+
j += 1
|
|
228
|
+
i = _blank(j)
|
|
229
|
+
elif c == "/" and i + 1 < n and text[i + 1] == "*":
|
|
230
|
+
depth, j = 1, i + 2
|
|
231
|
+
while j < n and depth:
|
|
232
|
+
if text[j] == "/" and j + 1 < n and text[j + 1] == "*":
|
|
233
|
+
depth += 1
|
|
234
|
+
j += 2
|
|
235
|
+
elif text[j] == "*" and j + 1 < n and text[j + 1] == "/":
|
|
236
|
+
depth -= 1
|
|
237
|
+
j += 2
|
|
238
|
+
else:
|
|
239
|
+
j += 1
|
|
240
|
+
i = _blank(j) # unclosed comment: j == n, blanks to end-of-file
|
|
241
|
+
else:
|
|
242
|
+
out.append(c)
|
|
243
|
+
i += 1
|
|
244
|
+
return "".join(out), ident_spans
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _mask_sql_comments(text: str) -> str:
|
|
248
|
+
"""Masked text only — see _scan_sql for the span-reporting form."""
|
|
249
|
+
return _scan_sql(text)[0]
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _norm_ident(name: str) -> str:
|
|
253
|
+
"""Normalize a SQL identifier for name-based reference resolution.
|
|
254
|
+
|
|
255
|
+
Splits on `.`, strips one pair of surrounding delimiters from each part
|
|
256
|
+
(double quotes for Postgres/ANSI, backticks for MySQL, brackets for
|
|
257
|
+
T-SQL), lowercases, and rejoins. So `"public"."users"`, `public.users`,
|
|
258
|
+
and `PUBLIC.USERS` all normalize to `public.users`. Used ONLY for
|
|
259
|
+
`table_nids` keys and lookups — node ids and display labels keep the
|
|
260
|
+
original text.
|
|
261
|
+
"""
|
|
262
|
+
parts = []
|
|
263
|
+
for part in name.split("."):
|
|
264
|
+
p = part.strip()
|
|
265
|
+
if len(p) >= 2 and ((p[0] == p[-1] and p[0] in ('"', "`"))
|
|
266
|
+
or (p[0] == "[" and p[-1] == "]")):
|
|
267
|
+
p = p[1:-1]
|
|
268
|
+
parts.append(p.lower())
|
|
269
|
+
return ".".join(parts)
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def extract_sql(path: Path, content: str | bytes | None = None) -> dict:
|
|
273
|
+
"""Extract tables, views, functions, and relationships from .sql files via tree-sitter."""
|
|
274
|
+
try:
|
|
275
|
+
import tree_sitter_sql as tssql
|
|
276
|
+
from tree_sitter import Language, Parser
|
|
277
|
+
except ImportError as e:
|
|
278
|
+
import importlib.util
|
|
279
|
+
# An installed-but-broken grammar (e.g. a C extension built for a
|
|
280
|
+
# different Python ABI, #2602) raises ImportError here too. Reporting
|
|
281
|
+
# that as "not installed" sends the user to a no-op `pip install`, so
|
|
282
|
+
# distinguish a genuinely-absent module from one that failed to load
|
|
283
|
+
# and surface the real exception in the latter case.
|
|
284
|
+
if importlib.util.find_spec("tree_sitter_sql") is None:
|
|
285
|
+
return {"nodes": [], "edges": [],
|
|
286
|
+
"error": "tree_sitter_sql not installed. Run: pip install tree-sitter-sql"}
|
|
287
|
+
return {"nodes": [], "edges": [],
|
|
288
|
+
"error": f"tree_sitter_sql is installed but failed to load: {e}"}
|
|
289
|
+
|
|
290
|
+
try:
|
|
291
|
+
language = Language(tssql.language())
|
|
292
|
+
parser = Parser(language)
|
|
293
|
+
source = (
|
|
294
|
+
content.encode("utf-8") if isinstance(content, str)
|
|
295
|
+
else content if content is not None
|
|
296
|
+
else path.read_bytes()
|
|
297
|
+
)
|
|
298
|
+
tree = parser.parse(source)
|
|
299
|
+
root = tree.root_node
|
|
300
|
+
except Exception as e:
|
|
301
|
+
return {"nodes": [], "edges": [], "error": str(e)}
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
stem = _file_stem(path)
|
|
305
|
+
str_path = str(path)
|
|
306
|
+
file_nid = _make_id(str_path)
|
|
307
|
+
nodes: list[dict] = [{"id": file_nid, "label": path.name, "file_type": "code",
|
|
308
|
+
"source_file": str_path, "source_location": None}]
|
|
309
|
+
edges: list[dict] = []
|
|
310
|
+
seen_ids: set[str] = {file_nid}
|
|
311
|
+
table_nids: dict[str, str] = {} # name → nid for reference resolution
|
|
312
|
+
|
|
313
|
+
def _read(n) -> str:
|
|
314
|
+
return source[n.start_byte:n.end_byte].decode("utf-8", errors="replace")
|
|
315
|
+
|
|
316
|
+
def _obj_name(n) -> str | None:
|
|
317
|
+
for c in n.children:
|
|
318
|
+
if c.type == "object_reference":
|
|
319
|
+
return _read(c)
|
|
320
|
+
return None
|
|
321
|
+
|
|
322
|
+
def _add_node(nid: str, label: str, line: int) -> None:
|
|
323
|
+
if nid not in seen_ids:
|
|
324
|
+
seen_ids.add(nid)
|
|
325
|
+
nodes.append({"id": nid, "label": label, "file_type": "code",
|
|
326
|
+
"source_file": str_path, "source_location": f"L{line}"})
|
|
327
|
+
edges.append({"source": file_nid, "target": nid, "relation": "contains",
|
|
328
|
+
"confidence": "EXTRACTED", "source_file": str_path,
|
|
329
|
+
"source_location": f"L{line}", "weight": 1.0})
|
|
330
|
+
|
|
331
|
+
def _add_edge(src: str, tgt: str, relation: str, line: int) -> None:
|
|
332
|
+
edges.append({"source": src, "target": tgt, "relation": relation,
|
|
333
|
+
"confidence": "EXTRACTED", "source_file": str_path,
|
|
334
|
+
"source_location": f"L{line}", "weight": 1.0})
|
|
335
|
+
|
|
336
|
+
def _ref_stub(name: str) -> str:
|
|
337
|
+
"""Sourceless bare-name stub for a table referenced but not defined here.
|
|
338
|
+
|
|
339
|
+
SQL references are NAME-based, so a table defined in another file (e.g.
|
|
340
|
+
prisma migration m2 referencing a table created in m1) can only resolve
|
|
341
|
+
at the corpus level. Minting `_make_id(stem, name)` under THIS file's
|
|
342
|
+
stem fabricated a node-less compound id — an absolute-path slug when the
|
|
343
|
+
input path was absolute — that could never match the real definition
|
|
344
|
+
(#2324). Instead emit a SOURCELESS stub, mirroring the Go extractor's
|
|
345
|
+
cross-file pattern (#1402): `_rewire_unique_stub_nodes` collapses it
|
|
346
|
+
onto the unique real table definition, and an unresolvable name survives
|
|
347
|
+
as a portable name-only node instead of dangling. No contains edge: a
|
|
348
|
+
sourced/contained stub would get the referencing file's path baked into
|
|
349
|
+
its id by disambiguation, blocking the rewire.
|
|
350
|
+
"""
|
|
351
|
+
nid = _make_id(name)
|
|
352
|
+
if nid not in seen_ids:
|
|
353
|
+
seen_ids.add(nid)
|
|
354
|
+
nodes.append({"id": nid, "label": name, "file_type": "code",
|
|
355
|
+
"source_file": "", "source_location": "",
|
|
356
|
+
"origin_file": str_path})
|
|
357
|
+
return nid
|
|
358
|
+
|
|
359
|
+
def walk(node) -> None:
|
|
360
|
+
t = node.type
|
|
361
|
+
line = node.start_point[0] + 1
|
|
362
|
+
|
|
363
|
+
if t == "create_table":
|
|
364
|
+
name = _obj_name(node)
|
|
365
|
+
if name:
|
|
366
|
+
nid = _make_id(stem, name)
|
|
367
|
+
_add_node(nid, name, line)
|
|
368
|
+
table_nids[_norm_ident(name)] = nid
|
|
369
|
+
# Foreign key REFERENCES
|
|
370
|
+
for col in node.children:
|
|
371
|
+
if col.type == "column_definitions":
|
|
372
|
+
has_error = any(cd.type == "ERROR" for cd in col.children)
|
|
373
|
+
seen_refs: set[str] = set()
|
|
374
|
+
for cd in col.children:
|
|
375
|
+
if cd.type == "column_definition":
|
|
376
|
+
# Inline column-level REFERENCES
|
|
377
|
+
ref_name: str | None = None
|
|
378
|
+
found_ref = False
|
|
379
|
+
for cc in cd.children:
|
|
380
|
+
if cc.type == "keyword_references":
|
|
381
|
+
found_ref = True
|
|
382
|
+
elif found_ref and cc.type == "object_reference":
|
|
383
|
+
ref_name = _read(cc)
|
|
384
|
+
break
|
|
385
|
+
if ref_name:
|
|
386
|
+
ref_nid = table_nids.get(_norm_ident(ref_name)) or _ref_stub(ref_name)
|
|
387
|
+
_add_edge(nid, ref_nid, "references", line)
|
|
388
|
+
seen_refs.add(_norm_ident(ref_name))
|
|
389
|
+
elif cd.type == "constraints":
|
|
390
|
+
# Table-level FOREIGN KEY ... REFERENCES ... constraints
|
|
391
|
+
for constraint in cd.children:
|
|
392
|
+
if constraint.type != "constraint":
|
|
393
|
+
continue
|
|
394
|
+
ref_name = None
|
|
395
|
+
found_ref = False
|
|
396
|
+
for cc in constraint.children:
|
|
397
|
+
if cc.type == "keyword_references":
|
|
398
|
+
found_ref = True
|
|
399
|
+
elif found_ref and cc.type == "object_reference":
|
|
400
|
+
ref_name = _read(cc)
|
|
401
|
+
break
|
|
402
|
+
if ref_name:
|
|
403
|
+
ref_nid = table_nids.get(_norm_ident(ref_name)) or _ref_stub(ref_name)
|
|
404
|
+
_add_edge(nid, ref_nid, "references", line)
|
|
405
|
+
seen_refs.add(_norm_ident(ref_name))
|
|
406
|
+
if has_error:
|
|
407
|
+
# Dialect-specific syntax (e.g. Firebird COMPUTED BY) causes ERROR
|
|
408
|
+
# nodes that make the parser drop the trailing constraints block.
|
|
409
|
+
# Regex-scan the raw column_definitions text as fallback.
|
|
410
|
+
col_text = _read(col)
|
|
411
|
+
for rm in re.finditer(r"\bREFERENCES\s+([\w$]+)", col_text, re.IGNORECASE):
|
|
412
|
+
ref_name = rm.group(1)
|
|
413
|
+
if _norm_ident(ref_name) not in seen_refs:
|
|
414
|
+
ref_nid = table_nids.get(_norm_ident(ref_name)) or _ref_stub(ref_name)
|
|
415
|
+
_add_edge(nid, ref_nid, "references", line)
|
|
416
|
+
seen_refs.add(_norm_ident(ref_name))
|
|
417
|
+
|
|
418
|
+
elif t == "create_view":
|
|
419
|
+
name = _obj_name(node)
|
|
420
|
+
if name:
|
|
421
|
+
nid = _make_id(stem, name)
|
|
422
|
+
_add_node(nid, name, line)
|
|
423
|
+
table_nids[_norm_ident(name)] = nid
|
|
424
|
+
# FROM/JOIN table references inside view body
|
|
425
|
+
_walk_from_refs(node, nid, line)
|
|
426
|
+
|
|
427
|
+
elif t == "create_function":
|
|
428
|
+
name = _obj_name(node)
|
|
429
|
+
if name:
|
|
430
|
+
nid = _make_id(stem, name)
|
|
431
|
+
_add_node(nid, f"{name}()", line)
|
|
432
|
+
_walk_from_refs(node, nid, line)
|
|
433
|
+
|
|
434
|
+
elif t == "create_procedure":
|
|
435
|
+
name = _obj_name(node)
|
|
436
|
+
if name:
|
|
437
|
+
nid = _make_id(stem, name)
|
|
438
|
+
_add_node(nid, f"{name}()", line)
|
|
439
|
+
_walk_from_refs(node, nid, line)
|
|
440
|
+
|
|
441
|
+
elif t == "alter_table":
|
|
442
|
+
name = _obj_name(node)
|
|
443
|
+
if name:
|
|
444
|
+
src_nid = table_nids.get(_norm_ident(name))
|
|
445
|
+
if not src_nid:
|
|
446
|
+
# Subject table not defined in this file: sourceless stub,
|
|
447
|
+
# not a sourced wrong-stem node (#2324).
|
|
448
|
+
src_nid = _ref_stub(name)
|
|
449
|
+
table_nids[_norm_ident(name)] = src_nid
|
|
450
|
+
for child in node.children:
|
|
451
|
+
if child.type == "add_constraint":
|
|
452
|
+
for cc in child.children:
|
|
453
|
+
if cc.type != "constraint":
|
|
454
|
+
continue
|
|
455
|
+
found_ref = False
|
|
456
|
+
ref_name: str | None = None
|
|
457
|
+
for ccc in cc.children:
|
|
458
|
+
if ccc.type == "keyword_references":
|
|
459
|
+
found_ref = True
|
|
460
|
+
elif found_ref and ccc.type == "object_reference":
|
|
461
|
+
ref_name = _read(ccc)
|
|
462
|
+
break
|
|
463
|
+
if ref_name:
|
|
464
|
+
ref_nid = (table_nids.get(_norm_ident(ref_name))
|
|
465
|
+
or _ref_stub(ref_name))
|
|
466
|
+
_add_edge(src_nid, ref_nid, "references", line)
|
|
467
|
+
|
|
468
|
+
elif t == "create_trigger":
|
|
469
|
+
trig_name: str | None = None
|
|
470
|
+
tbl_name: str | None = None
|
|
471
|
+
after_trigger = False
|
|
472
|
+
after_for = False
|
|
473
|
+
for c in node.children:
|
|
474
|
+
if c.type == "keyword_trigger":
|
|
475
|
+
after_trigger = True
|
|
476
|
+
elif after_trigger and not trig_name and c.type == "object_reference":
|
|
477
|
+
trig_name = _read(c)
|
|
478
|
+
elif c.type == "keyword_for":
|
|
479
|
+
after_for = True
|
|
480
|
+
elif after_for and not tbl_name and c.type == "object_reference":
|
|
481
|
+
tbl_name = _read(c)
|
|
482
|
+
if trig_name:
|
|
483
|
+
trig_nid = _make_id(stem, trig_name)
|
|
484
|
+
_add_node(trig_nid, trig_name, line)
|
|
485
|
+
if tbl_name:
|
|
486
|
+
tbl_nid = table_nids.get(_norm_ident(tbl_name)) or _ref_stub(tbl_name)
|
|
487
|
+
_add_edge(trig_nid, tbl_nid, "triggers", line)
|
|
488
|
+
|
|
489
|
+
elif t == "create_index":
|
|
490
|
+
# CREATE [UNIQUE] INDEX [CONCURRENTLY] [IF NOT EXISTS] <name>
|
|
491
|
+
# ON <table> (...). Unlike CREATE POLICY (#3401) the grammar
|
|
492
|
+
# parses this statement fine; the walk simply never dispatched on
|
|
493
|
+
# it, so every index was silently dropped (#3467). The name is the
|
|
494
|
+
# identifier (or a quoted literal) before ON; the table is the
|
|
495
|
+
# object_reference after it. An unnamed index (`CREATE INDEX ON
|
|
496
|
+
# t (c)`) has nothing to name a node after and is skipped.
|
|
497
|
+
index_name: str | None = None
|
|
498
|
+
index_table: str | None = None
|
|
499
|
+
after_on = False
|
|
500
|
+
for c in node.children:
|
|
501
|
+
if c.type == "keyword_on":
|
|
502
|
+
after_on = True
|
|
503
|
+
elif not after_on and index_name is None and c.type in ("identifier", "literal"):
|
|
504
|
+
index_name = _read(c).strip('"`')
|
|
505
|
+
elif after_on and index_table is None and c.type == "object_reference":
|
|
506
|
+
index_table = _read(c)
|
|
507
|
+
if index_name:
|
|
508
|
+
index_nid = _make_id(stem, index_name)
|
|
509
|
+
_add_node(index_nid, index_name, line)
|
|
510
|
+
if index_table:
|
|
511
|
+
index_tbl_nid = (table_nids.get(_norm_ident(index_table))
|
|
512
|
+
or _ref_stub(index_table))
|
|
513
|
+
_add_edge(index_nid, index_tbl_nid, "indexes", line)
|
|
514
|
+
|
|
515
|
+
# NOTE: there is deliberately NO recovery scan on individual ERROR
|
|
516
|
+
# nodes. Any ERROR node anywhere makes root.has_error true, so the
|
|
517
|
+
# whole-file masked scan below this walk already recovers everything
|
|
518
|
+
# a per-node scan could — from the SAME shared _ROUTINE_RECOVERY_RX,
|
|
519
|
+
# with _add_node deduping by id. A per-node scan is not just
|
|
520
|
+
# redundant, it is unsound: _mask_sql_comments needs file-level
|
|
521
|
+
# context, and an ERROR fragment can begin MID-comment (tree-sitter's
|
|
522
|
+
# lexer does not nest /* */, so the text after an inner */ parses as
|
|
523
|
+
# code and lands in an ERROR blob with no comment opener in sight),
|
|
524
|
+
# which let commented-out DDL fabricate routine nodes.
|
|
525
|
+
|
|
526
|
+
elif t == "fb_proc_or_trigger":
|
|
527
|
+
text = _read(node)
|
|
528
|
+
m = re.match(
|
|
529
|
+
r"CREATE\s+(?:OR\s+(?:REPLACE|ALTER)\s+)?"
|
|
530
|
+
r"(PROCEDURE|TRIGGER|FUNCTION)\s+([\w$]+)",
|
|
531
|
+
text, re.IGNORECASE,
|
|
532
|
+
)
|
|
533
|
+
if m:
|
|
534
|
+
obj_type = m.group(1).upper()
|
|
535
|
+
obj_name = m.group(2)
|
|
536
|
+
obj_nid = _make_id(stem, obj_name)
|
|
537
|
+
label = obj_name if obj_type == "TRIGGER" else f"{obj_name}()"
|
|
538
|
+
_add_node(obj_nid, label, line)
|
|
539
|
+
if obj_type == "TRIGGER":
|
|
540
|
+
fm = re.search(r"\bFOR\s+([\w$]+)", text, re.IGNORECASE)
|
|
541
|
+
if fm:
|
|
542
|
+
tbl = fm.group(1)
|
|
543
|
+
tbl_nid = table_nids.get(_norm_ident(tbl)) or _ref_stub(tbl)
|
|
544
|
+
_add_edge(obj_nid, tbl_nid, "triggers", line)
|
|
545
|
+
_NON_TABLES = {
|
|
546
|
+
"select", "where", "set", "dual", "null", "true", "false",
|
|
547
|
+
"first", "skip", "rows", "next", "only", "lateral",
|
|
548
|
+
}
|
|
549
|
+
# Same CTE-blindness as the AST path (#2577): a `WITH <name> AS (`
|
|
550
|
+
# binding is statement-local, not a table, so its name must not
|
|
551
|
+
# become a reads_from stub. The regex has no scope tree, so the
|
|
552
|
+
# skip is body-wide — the right trade for a recovery path.
|
|
553
|
+
for cm in re.finditer(
|
|
554
|
+
r"(?:\bWITH\s+(?:RECURSIVE\s+)?|,\s*)([\w$]+)\s*(?:\([^()]*\))?\s+AS\s*\(",
|
|
555
|
+
text, re.IGNORECASE,
|
|
556
|
+
):
|
|
557
|
+
_NON_TABLES.add(_norm_ident(cm.group(1)))
|
|
558
|
+
seen_tbls: set[str] = set()
|
|
559
|
+
for rm in re.finditer(r"\b(?:FROM|JOIN|INTO)\s+([\w$]+)", text, re.IGNORECASE):
|
|
560
|
+
tbl = rm.group(1)
|
|
561
|
+
if _norm_ident(tbl) not in _NON_TABLES and _norm_ident(tbl) not in seen_tbls:
|
|
562
|
+
seen_tbls.add(_norm_ident(tbl))
|
|
563
|
+
tbl_nid = table_nids.get(_norm_ident(tbl)) or _ref_stub(tbl)
|
|
564
|
+
_add_edge(obj_nid, tbl_nid, "reads_from", line)
|
|
565
|
+
for rm in re.finditer(r"\bUPDATE\s+([\w$]+)", text, re.IGNORECASE):
|
|
566
|
+
tbl = rm.group(1)
|
|
567
|
+
if _norm_ident(tbl) not in _NON_TABLES and _norm_ident(tbl) not in seen_tbls:
|
|
568
|
+
seen_tbls.add(_norm_ident(tbl))
|
|
569
|
+
tbl_nid = table_nids.get(_norm_ident(tbl)) or _ref_stub(tbl)
|
|
570
|
+
_add_edge(obj_nid, tbl_nid, "reads_from", line)
|
|
571
|
+
|
|
572
|
+
for child in node.children:
|
|
573
|
+
walk(child)
|
|
574
|
+
|
|
575
|
+
def _walk_from_refs(node, caller_nid: str, line: int,
|
|
576
|
+
cte_names: frozenset[str] = frozenset()) -> None:
|
|
577
|
+
"""Recursively find FROM/JOIN table references inside a node, skipping CTEs.
|
|
578
|
+
|
|
579
|
+
A name bound by `WITH <name> AS (...)` is not a table: emitting it as a
|
|
580
|
+
`reads_from` target minted a bare `_ref_stub`, and because that stub is
|
|
581
|
+
intentionally sourceless (see `_ref_stub`) it carried no schema, file, or
|
|
582
|
+
language namespace, so a CTE named `levels` or `slug` collided with any
|
|
583
|
+
same-named node from another language during the build (#2577).
|
|
584
|
+
|
|
585
|
+
Scoping matters: a CTE is visible only inside the query that declares it,
|
|
586
|
+
and a `WITH` inside a subquery is scoped to that subquery alone. So the
|
|
587
|
+
active set is extended PER SUBTREE — each node's directly-owned `cte`
|
|
588
|
+
children (`create_query` for a statement-level WITH, `subquery` for a
|
|
589
|
+
nested one) join the set passed down into that node's recursion only. A
|
|
590
|
+
single statement-wide pre-collect would also suppress an OUTER reference
|
|
591
|
+
to a real table that merely shares a subquery-CTE's name
|
|
592
|
+
(`... FROM t2 JOIN (WITH t2 AS (...) SELECT ...) sub`), dropping the
|
|
593
|
+
real `-> t2` edge.
|
|
594
|
+
"""
|
|
595
|
+
own: set[str] = set()
|
|
596
|
+
for c in node.children:
|
|
597
|
+
if c.type != "cte":
|
|
598
|
+
continue
|
|
599
|
+
# First identifier is the CTE's name; later ones are its column
|
|
600
|
+
# list (`WITH levels(a, b) AS (...)`), which must not be skipped.
|
|
601
|
+
for cc in c.children:
|
|
602
|
+
if cc.type in ("identifier", "object_reference"):
|
|
603
|
+
own.add(_norm_ident(_read(cc)))
|
|
604
|
+
break
|
|
605
|
+
if own:
|
|
606
|
+
cte_names = frozenset(cte_names | own)
|
|
607
|
+
if node.type in ("from", "join"):
|
|
608
|
+
for c in node.children:
|
|
609
|
+
if c.type == "relation":
|
|
610
|
+
for cc in c.children:
|
|
611
|
+
if cc.type == "object_reference":
|
|
612
|
+
tbl = _read(cc)
|
|
613
|
+
if _norm_ident(tbl) in cte_names:
|
|
614
|
+
continue
|
|
615
|
+
tbl_nid = table_nids.get(_norm_ident(tbl)) or _ref_stub(tbl)
|
|
616
|
+
_add_edge(caller_nid, tbl_nid, "reads_from",
|
|
617
|
+
c.start_point[0] + 1)
|
|
618
|
+
for child in node.children:
|
|
619
|
+
_walk_from_refs(child, caller_nid, line, cte_names)
|
|
620
|
+
|
|
621
|
+
# Pre-pass: register every table/view DEFINED in this file before walking,
|
|
622
|
+
# so forward references (a FK to a table created later in the same file)
|
|
623
|
+
# still resolve to the real sourced node instead of falling back to a stub.
|
|
624
|
+
def _collect_defined_names(node) -> None:
|
|
625
|
+
if node.type in ("create_table", "create_view"):
|
|
626
|
+
name = _obj_name(node)
|
|
627
|
+
if name:
|
|
628
|
+
table_nids[_norm_ident(name)] = _make_id(stem, name)
|
|
629
|
+
for child in node.children:
|
|
630
|
+
_collect_defined_names(child)
|
|
631
|
+
|
|
632
|
+
_collect_defined_names(root)
|
|
633
|
+
|
|
634
|
+
# Secondary bare-name aliases: a reference written without a schema
|
|
635
|
+
# (`REFERENCES users`) should resolve to a schema-qualified definition
|
|
636
|
+
# (`public.users`) when that is unambiguous. Never shadow an explicit
|
|
637
|
+
# definition, and skip bare names defined under more than one schema.
|
|
638
|
+
bare_candidates: dict[str, str | None] = {}
|
|
639
|
+
for key, alias_nid in table_nids.items():
|
|
640
|
+
if "." in key:
|
|
641
|
+
bare = key.rsplit(".", 1)[1]
|
|
642
|
+
bare_candidates[bare] = (
|
|
643
|
+
alias_nid if bare_candidates.get(bare, alias_nid) == alias_nid else None
|
|
644
|
+
)
|
|
645
|
+
for bare, alias_nid in bare_candidates.items():
|
|
646
|
+
if alias_nid is not None and bare not in table_nids:
|
|
647
|
+
table_nids[bare] = alias_nid
|
|
648
|
+
|
|
649
|
+
for stmt in root.children:
|
|
650
|
+
if stmt.type == "statement":
|
|
651
|
+
for child in stmt.children:
|
|
652
|
+
walk(child)
|
|
653
|
+
elif stmt.type == "transaction":
|
|
654
|
+
# BEGIN; ... COMMIT; wraps DDL in a transaction node whose children
|
|
655
|
+
# are statement nodes, not direct create_table nodes (#2953).
|
|
656
|
+
walk(stmt)
|
|
657
|
+
elif stmt.type in ("fb_proc_or_trigger", "set_term", "declare_external_function", "ERROR"):
|
|
658
|
+
walk(stmt)
|
|
659
|
+
|
|
660
|
+
# Global regex fallback: catch any REFERENCES missed due to ERROR nodes in the parse tree
|
|
661
|
+
# (e.g. Firebird COMPUTED BY columns push constraints out of the tree entirely).
|
|
662
|
+
# Snapshot after tree walk so we don't re-emit edges already captured above.
|
|
663
|
+
emitted = {(e["source"], e["target"]) for e in edges if e["relation"] == "references"}
|
|
664
|
+
src_text = source.decode("utf-8", errors="replace")
|
|
665
|
+
for m in re.finditer(r"CREATE\s+TABLE\s+([\w$]+)\s*\(", src_text, re.IGNORECASE):
|
|
666
|
+
tbl_name = m.group(1)
|
|
667
|
+
tbl_nid = table_nids.get(_norm_ident(tbl_name))
|
|
668
|
+
if tbl_nid is None:
|
|
669
|
+
continue
|
|
670
|
+
tbl_line = src_text[: m.start()].count("\n") + 1
|
|
671
|
+
tail = src_text[m.start():]
|
|
672
|
+
end = re.search(r"(?:^|\n)(?:CREATE|SET\s+TERM|ALTER)\s", tail[1:], re.IGNORECASE)
|
|
673
|
+
block = tail[: end.start() + 1] if end else tail
|
|
674
|
+
for rm in re.finditer(r"\bREFERENCES\s+([\w$]+)", block, re.IGNORECASE):
|
|
675
|
+
ref_name = rm.group(1)
|
|
676
|
+
ref_nid = table_nids.get(_norm_ident(ref_name)) or _ref_stub(ref_name)
|
|
677
|
+
if (tbl_nid, ref_nid) not in emitted:
|
|
678
|
+
_add_edge(tbl_nid, ref_nid, "references", tbl_line)
|
|
679
|
+
emitted.add((tbl_nid, ref_nid))
|
|
680
|
+
|
|
681
|
+
# Global regex fallback for routines (#2180). PL/pgSQL bodies break the parse
|
|
682
|
+
# in more than one shape, and only the first was recovered before:
|
|
683
|
+
# 1. the whole CREATE lands in one ERROR node -> handled in walk()
|
|
684
|
+
# 2. the statement is shredded into loose top-level tokens
|
|
685
|
+
# (keyword_create/keyword_function/object_reference/... ) and the ERROR
|
|
686
|
+
# node holds only the offending body line, e.g. `PERFORM x();` or
|
|
687
|
+
# `x := 1;` -- so no CREATE text is inside any ERROR node at all
|
|
688
|
+
# 3. the name is a delimited identifier — quoted ("public"."fn") or
|
|
689
|
+
# T-SQL-bracketed ([dbo].[usp_Load]) — which a bare [\w$.]+ pattern
|
|
690
|
+
# cannot match
|
|
691
|
+
# Shapes 2 and 3 silently dropped the routine: no node, no warning, exit 0.
|
|
692
|
+
# Scanning the raw source catches all three, and _add_node dedupes by id so
|
|
693
|
+
# routines already recovered from the tree are not emitted twice.
|
|
694
|
+
#
|
|
695
|
+
# Gate on a failed parse: a cleanly-parsing file must NOT have routines
|
|
696
|
+
# fabricated from MySQL `CREATE FUNCTION IF NOT EXISTS` (which would
|
|
697
|
+
# capture `IF`) or other shapes the mask does not model (double-quoted
|
|
698
|
+
# strings in MySQL's default mode, PostgreSQL dollar-quoted bodies). Every
|
|
699
|
+
# observed drop shape leaves an ERROR node in the tree, so has_error loses
|
|
700
|
+
# nothing while protecting clean corpora (#2180 follow-up).
|
|
701
|
+
if root.has_error:
|
|
702
|
+
# The mask blanks comments (nesting-aware) and single-quoted strings
|
|
703
|
+
# (offset-preserving), so commented-out DDL and single-quoted dynamic
|
|
704
|
+
# SQL cannot fabricate a routine when an unrelated error arms this
|
|
705
|
+
# scan; string shapes the mask does not model rely on the has_error
|
|
706
|
+
# gate alone. Preserved delimited identifiers keep their text
|
|
707
|
+
# verbatim (they carry the recoverable names), so a match whose
|
|
708
|
+
# CREATE keyword STARTS inside one is identifier data, not DDL
|
|
709
|
+
# ('SELECT 1 AS [CREATE PROCEDURE x pending]') and is skipped — the
|
|
710
|
+
# name a genuine statement captures is allowed to be a delimited
|
|
711
|
+
# identifier; its CREATE never is.
|
|
712
|
+
masked_src, ident_spans = _scan_sql(src_text)
|
|
713
|
+
for m in _ROUTINE_RECOVERY_RX.finditer(masked_src):
|
|
714
|
+
if any(s <= m.start() < e for s, e in ident_spans):
|
|
715
|
+
continue
|
|
716
|
+
fn_name = m.group(1)
|
|
717
|
+
fn_line = src_text[: m.start()].count("\n") + 1
|
|
718
|
+
_add_node(_make_id(stem, fn_name), f"{fn_name}()", fn_line)
|
|
719
|
+
|
|
720
|
+
return {"nodes": nodes, "edges": edges}
|