graphitect 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (336) hide show
  1. graphify/__init__.py +30 -0
  2. graphify/__main__.py +757 -0
  3. graphify/_minhash.py +107 -0
  4. graphify/affected.py +318 -0
  5. graphify/always_on/agents-md.md +12 -0
  6. graphify/always_on/antigravity-rules.md +14 -0
  7. graphify/always_on/claude-md.md +9 -0
  8. graphify/always_on/gemini-md.md +9 -0
  9. graphify/always_on/kiro-steering.md +5 -0
  10. graphify/always_on/vscode-instructions.md +17 -0
  11. graphify/analyze.py +769 -0
  12. graphify/benchmark.py +152 -0
  13. graphify/build.py +2300 -0
  14. graphify/cache.py +1746 -0
  15. graphify/callflow_html.py +2051 -0
  16. graphify/cargo_introspect.py +109 -0
  17. graphify/cli.py +4745 -0
  18. graphify/cluster.py +409 -0
  19. graphify/command-kilo.md +15 -0
  20. graphify/cross_repo_calls.py +216 -0
  21. graphify/cross_repo_types.py +75 -0
  22. graphify/csharp_dispatch.py +154 -0
  23. graphify/dedup.py +1213 -0
  24. graphify/detect.py +2566 -0
  25. graphify/diagnostics.py +406 -0
  26. graphify/export.py +1349 -0
  27. graphify/exporters/__init__.py +1 -0
  28. graphify/exporters/base.py +14 -0
  29. graphify/exporters/graphdb.py +173 -0
  30. graphify/exporters/html.py +637 -0
  31. graphify/extract.py +7856 -0
  32. graphify/extractors/MIGRATION.md +107 -0
  33. graphify/extractors/__init__.py +66 -0
  34. graphify/extractors/apex.py +215 -0
  35. graphify/extractors/base.py +85 -0
  36. graphify/extractors/bash.py +579 -0
  37. graphify/extractors/blade.py +53 -0
  38. graphify/extractors/commonlisp.py +540 -0
  39. graphify/extractors/csharp.py +448 -0
  40. graphify/extractors/dart.py +564 -0
  41. graphify/extractors/dm.py +494 -0
  42. graphify/extractors/elixir.py +241 -0
  43. graphify/extractors/engine.py +6509 -0
  44. graphify/extractors/fortran.py +311 -0
  45. graphify/extractors/go.py +527 -0
  46. graphify/extractors/json_config.py +240 -0
  47. graphify/extractors/julia.py +289 -0
  48. graphify/extractors/markdown.py +408 -0
  49. graphify/extractors/models.py +131 -0
  50. graphify/extractors/objc.py +566 -0
  51. graphify/extractors/ocaml.py +289 -0
  52. graphify/extractors/pascal.py +688 -0
  53. graphify/extractors/pascal_forms.py +196 -0
  54. graphify/extractors/powershell.py +522 -0
  55. graphify/extractors/razor.py +192 -0
  56. graphify/extractors/resolution.py +3584 -0
  57. graphify/extractors/robot.py +296 -0
  58. graphify/extractors/rust.py +470 -0
  59. graphify/extractors/sln.py +92 -0
  60. graphify/extractors/sql.py +720 -0
  61. graphify/extractors/terraform.py +181 -0
  62. graphify/extractors/verilog.py +329 -0
  63. graphify/extractors/zig.py +181 -0
  64. graphify/file_slice.py +246 -0
  65. graphify/global_graph.py +194 -0
  66. graphify/google_workspace.py +237 -0
  67. graphify/hooks.py +933 -0
  68. graphify/ids.py +93 -0
  69. graphify/ingest.py +358 -0
  70. graphify/install.py +2366 -0
  71. graphify/llm.py +3544 -0
  72. graphify/manifest.py +4 -0
  73. graphify/manifest_ingest.py +311 -0
  74. graphify/mcp_ingest.py +386 -0
  75. graphify/multigraph_compat.py +212 -0
  76. graphify/pascal_resolution.py +129 -0
  77. graphify/paths.py +436 -0
  78. graphify/pg_introspect.py +165 -0
  79. graphify/prs.py +770 -0
  80. graphify/querylog.py +80 -0
  81. graphify/reflect.py +882 -0
  82. graphify/report.py +346 -0
  83. graphify/resolver_registry.py +85 -0
  84. graphify/ruby_resolution.py +242 -0
  85. graphify/scip_ingest.py +363 -0
  86. graphify/security.py +460 -0
  87. graphify/semantic_cleanup.py +336 -0
  88. graphify/serve.py +2608 -0
  89. graphify/skill-agents.md +710 -0
  90. graphify/skill-aider.md +1283 -0
  91. graphify/skill-amp.md +710 -0
  92. graphify/skill-claw.md +713 -0
  93. graphify/skill-codex.md +710 -0
  94. graphify/skill-copilot.md +713 -0
  95. graphify/skill-devin.md +1410 -0
  96. graphify/skill-droid.md +710 -0
  97. graphify/skill-kilo.md +722 -0
  98. graphify/skill-kiro.md +713 -0
  99. graphify/skill-opencode.md +705 -0
  100. graphify/skill-pi.md +713 -0
  101. graphify/skill-trae.md +711 -0
  102. graphify/skill-vscode.md +709 -0
  103. graphify/skill-windows.md +755 -0
  104. graphify/skill.md +713 -0
  105. graphify/skills/agents/references/add-watch.md +56 -0
  106. graphify/skills/agents/references/exports.md +87 -0
  107. graphify/skills/agents/references/extraction-spec.md +70 -0
  108. graphify/skills/agents/references/github-and-merge.md +46 -0
  109. graphify/skills/agents/references/hooks.md +33 -0
  110. graphify/skills/agents/references/query.md +311 -0
  111. graphify/skills/agents/references/transcribe.md +52 -0
  112. graphify/skills/agents/references/update.md +210 -0
  113. graphify/skills/amp/references/add-watch.md +56 -0
  114. graphify/skills/amp/references/exports.md +87 -0
  115. graphify/skills/amp/references/extraction-spec.md +70 -0
  116. graphify/skills/amp/references/github-and-merge.md +46 -0
  117. graphify/skills/amp/references/hooks.md +33 -0
  118. graphify/skills/amp/references/query.md +311 -0
  119. graphify/skills/amp/references/transcribe.md +52 -0
  120. graphify/skills/amp/references/update.md +210 -0
  121. graphify/skills/claude/references/add-watch.md +56 -0
  122. graphify/skills/claude/references/exports.md +87 -0
  123. graphify/skills/claude/references/extraction-spec.md +70 -0
  124. graphify/skills/claude/references/github-and-merge.md +46 -0
  125. graphify/skills/claude/references/hooks.md +33 -0
  126. graphify/skills/claude/references/query.md +311 -0
  127. graphify/skills/claude/references/transcribe.md +52 -0
  128. graphify/skills/claude/references/update.md +210 -0
  129. graphify/skills/claw/references/add-watch.md +56 -0
  130. graphify/skills/claw/references/exports.md +87 -0
  131. graphify/skills/claw/references/extraction-spec.md +31 -0
  132. graphify/skills/claw/references/github-and-merge.md +46 -0
  133. graphify/skills/claw/references/hooks.md +33 -0
  134. graphify/skills/claw/references/query.md +311 -0
  135. graphify/skills/claw/references/transcribe.md +52 -0
  136. graphify/skills/claw/references/update.md +210 -0
  137. graphify/skills/codex/references/add-watch.md +56 -0
  138. graphify/skills/codex/references/exports.md +87 -0
  139. graphify/skills/codex/references/extraction-spec.md +31 -0
  140. graphify/skills/codex/references/github-and-merge.md +46 -0
  141. graphify/skills/codex/references/hooks.md +33 -0
  142. graphify/skills/codex/references/query.md +311 -0
  143. graphify/skills/codex/references/transcribe.md +52 -0
  144. graphify/skills/codex/references/update.md +210 -0
  145. graphify/skills/copilot/references/add-watch.md +56 -0
  146. graphify/skills/copilot/references/exports.md +87 -0
  147. graphify/skills/copilot/references/extraction-spec.md +70 -0
  148. graphify/skills/copilot/references/github-and-merge.md +46 -0
  149. graphify/skills/copilot/references/hooks.md +33 -0
  150. graphify/skills/copilot/references/query.md +311 -0
  151. graphify/skills/copilot/references/transcribe.md +52 -0
  152. graphify/skills/copilot/references/update.md +210 -0
  153. graphify/skills/droid/references/add-watch.md +56 -0
  154. graphify/skills/droid/references/exports.md +87 -0
  155. graphify/skills/droid/references/extraction-spec.md +70 -0
  156. graphify/skills/droid/references/github-and-merge.md +46 -0
  157. graphify/skills/droid/references/hooks.md +33 -0
  158. graphify/skills/droid/references/query.md +311 -0
  159. graphify/skills/droid/references/transcribe.md +52 -0
  160. graphify/skills/droid/references/update.md +210 -0
  161. graphify/skills/kilo/references/add-watch.md +56 -0
  162. graphify/skills/kilo/references/exports.md +87 -0
  163. graphify/skills/kilo/references/extraction-spec.md +70 -0
  164. graphify/skills/kilo/references/github-and-merge.md +46 -0
  165. graphify/skills/kilo/references/hooks.md +33 -0
  166. graphify/skills/kilo/references/query.md +311 -0
  167. graphify/skills/kilo/references/transcribe.md +52 -0
  168. graphify/skills/kilo/references/update.md +210 -0
  169. graphify/skills/kiro/references/add-watch.md +56 -0
  170. graphify/skills/kiro/references/exports.md +87 -0
  171. graphify/skills/kiro/references/extraction-spec.md +31 -0
  172. graphify/skills/kiro/references/github-and-merge.md +46 -0
  173. graphify/skills/kiro/references/hooks.md +33 -0
  174. graphify/skills/kiro/references/query.md +311 -0
  175. graphify/skills/kiro/references/transcribe.md +52 -0
  176. graphify/skills/kiro/references/update.md +210 -0
  177. graphify/skills/opencode/references/add-watch.md +56 -0
  178. graphify/skills/opencode/references/exports.md +87 -0
  179. graphify/skills/opencode/references/extraction-spec.md +70 -0
  180. graphify/skills/opencode/references/github-and-merge.md +46 -0
  181. graphify/skills/opencode/references/hooks.md +33 -0
  182. graphify/skills/opencode/references/query.md +311 -0
  183. graphify/skills/opencode/references/transcribe.md +52 -0
  184. graphify/skills/opencode/references/update.md +210 -0
  185. graphify/skills/pi/references/add-watch.md +56 -0
  186. graphify/skills/pi/references/exports.md +87 -0
  187. graphify/skills/pi/references/extraction-spec.md +31 -0
  188. graphify/skills/pi/references/github-and-merge.md +46 -0
  189. graphify/skills/pi/references/hooks.md +33 -0
  190. graphify/skills/pi/references/query.md +311 -0
  191. graphify/skills/pi/references/transcribe.md +52 -0
  192. graphify/skills/pi/references/update.md +210 -0
  193. graphify/skills/trae/references/add-watch.md +56 -0
  194. graphify/skills/trae/references/exports.md +87 -0
  195. graphify/skills/trae/references/extraction-spec.md +70 -0
  196. graphify/skills/trae/references/github-and-merge.md +46 -0
  197. graphify/skills/trae/references/hooks.md +35 -0
  198. graphify/skills/trae/references/query.md +311 -0
  199. graphify/skills/trae/references/transcribe.md +52 -0
  200. graphify/skills/trae/references/update.md +210 -0
  201. graphify/skills/vscode/references/add-watch.md +56 -0
  202. graphify/skills/vscode/references/exports.md +87 -0
  203. graphify/skills/vscode/references/extraction-spec.md +70 -0
  204. graphify/skills/vscode/references/github-and-merge.md +46 -0
  205. graphify/skills/vscode/references/hooks.md +33 -0
  206. graphify/skills/vscode/references/query.md +311 -0
  207. graphify/skills/vscode/references/transcribe.md +52 -0
  208. graphify/skills/vscode/references/update.md +210 -0
  209. graphify/skills/windows/references/add-watch.md +56 -0
  210. graphify/skills/windows/references/exports.md +87 -0
  211. graphify/skills/windows/references/extraction-spec.md +70 -0
  212. graphify/skills/windows/references/github-and-merge.md +46 -0
  213. graphify/skills/windows/references/hooks.md +33 -0
  214. graphify/skills/windows/references/query.md +311 -0
  215. graphify/skills/windows/references/transcribe.md +52 -0
  216. graphify/skills/windows/references/update.md +210 -0
  217. graphify/symbol_resolution.py +556 -0
  218. graphify/transcribe.py +186 -0
  219. graphify/tree_html.py +603 -0
  220. graphify/validate.py +95 -0
  221. graphify/watch.py +2280 -0
  222. graphify/wiki.py +405 -0
  223. graphitect/__init__.py +28 -0
  224. graphitect/__main__.py +4 -0
  225. graphitect/_vendor/__init__.py +2 -0
  226. graphitect/_vendor/archify/LICENSE +22 -0
  227. graphitect/_vendor/archify/SKILL.md +137 -0
  228. graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
  229. graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
  230. graphitect/_vendor/archify/assets/template.html +14935 -0
  231. graphitect/_vendor/archify/bin/archify.mjs +2091 -0
  232. graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
  233. graphitect/_vendor/archify/bin/preview.mjs +653 -0
  234. graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
  235. graphitect/_vendor/archify/brand-marks/README.md +31 -0
  236. graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
  237. graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
  238. graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
  239. graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
  240. graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
  241. graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
  242. graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
  243. graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
  244. graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
  245. graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
  246. graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
  247. graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
  248. graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
  249. graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
  250. graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
  251. graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
  252. graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
  253. graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
  254. graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
  255. graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
  256. graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
  257. graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
  258. graphitect/_vendor/archify/package-lock.json +149 -0
  259. graphitect/_vendor/archify/package.json +39 -0
  260. graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
  261. graphitect/_vendor/archify/references/authoring-contract.md +243 -0
  262. graphitect/_vendor/archify/references/brand-marks.md +65 -0
  263. graphitect/_vendor/archify/references/delivery-contract.md +120 -0
  264. graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
  265. graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
  266. graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
  267. graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
  268. graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
  269. graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
  270. graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
  271. graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
  272. graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
  273. graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
  274. graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
  275. graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
  276. graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
  277. graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
  278. graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
  279. graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
  280. graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
  281. graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
  282. graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
  283. graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
  284. graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
  285. graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
  286. graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
  287. graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
  288. graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
  289. graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
  290. graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
  291. graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
  292. graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
  293. graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
  294. graphitect/_vendor/archify/schemas/README.md +211 -0
  295. graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
  296. graphitect/_vendor/archify/schemas/common.schema.json +115 -0
  297. graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
  298. graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
  299. graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
  300. graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
  301. graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
  302. graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
  303. graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
  304. graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
  305. graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
  306. graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
  307. graphitect/_vendor/archify/skill-release.json +10 -0
  308. graphitect/cli.py +981 -0
  309. graphitect/deliver/__init__.py +5 -0
  310. graphitect/deliver/archify_adapter.py +1877 -0
  311. graphitect/deliver/archify_ir.py +160 -0
  312. graphitect/deliver/archify_repair.py +135 -0
  313. graphitect/deliver/doc_compiler.py +916 -0
  314. graphitect/ground/__init__.py +5 -0
  315. graphitect/ground/describe_source.py +27 -0
  316. graphitect/ground/fullread_source.py +56 -0
  317. graphitect/ground/graphify_source.py +107 -0
  318. graphitect/models.py +118 -0
  319. graphitect/skill/SKILL.md +80 -0
  320. graphitect/skill/agents/openai.yaml +4 -0
  321. graphitect/synthesize/__init__.py +5 -0
  322. graphitect/synthesize/engine.py +281 -0
  323. graphitect/synthesize/llm_backend.py +331 -0
  324. graphitect/synthesize/questions.py +139 -0
  325. graphitect/synthesize/rubric.py +104 -0
  326. graphitect-0.2.0.dist-info/METADATA +284 -0
  327. graphitect-0.2.0.dist-info/RECORD +336 -0
  328. graphitect-0.2.0.dist-info/WHEEL +5 -0
  329. graphitect-0.2.0.dist-info/entry_points.txt +2 -0
  330. graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
  331. graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
  332. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
  333. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
  334. graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
  335. graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
  336. graphitect-0.2.0.dist-info/top_level.txt +2 -0
graphify/cache.py ADDED
@@ -0,0 +1,1746 @@
1
+ # per-file extraction cache - skip unchanged files on re-run
2
+ from __future__ import annotations
3
+
4
+ import atexit
5
+ import hashlib
6
+ import json
7
+ import os
8
+ import re
9
+ import tempfile
10
+ import time
11
+ import warnings
12
+ from collections.abc import Callable, Iterable
13
+ from pathlib import Path
14
+
15
+ # Output directory name — override with GRAPHIFY_OUT env var for worktrees or
16
+ # shared-output setups. Accepts a relative name ("graphify-out-feature") or an
17
+ # absolute path ("/shared/graphify-out"). Single source of truth in graphify.paths
18
+ # (#1423); re-exported here as _GRAPHIFY_OUT for the existing call sites.
19
+ from graphify.paths import GRAPHIFY_OUT as _GRAPHIFY_OUT
20
+
21
+ # AST cache entries are the output of graphify's own extractor code, so they
22
+ # are only valid for the version that wrote them: keying purely on file
23
+ # content means extractor fixes shipped in a new release keep serving stale
24
+ # pre-fix results. The AST cache is therefore namespaced by package version
25
+ # and cache-key schema (cache/ast/v{version}-s{schema}/), with entries from
26
+ # other versions or schemas removed on first
27
+ # use. The semantic cache is deliberately NOT versioned — its entries are
28
+ # produced by the LLM from file contents, and invalidating them on every
29
+ # release would re-bill extraction for unchanged files.
30
+ try:
31
+ from importlib.metadata import version as _pkg_version
32
+
33
+ _EXTRACTOR_VERSION = _pkg_version("graphifyy")
34
+ except Exception:
35
+ _EXTRACTOR_VERSION = "0.9.58-bundled"
36
+
37
+ # Bump when AST cache-key semantics change independently of the package version.
38
+ _AST_CACHE_SCHEMA = 2
39
+
40
+ # Version dirs already swept this process — cleanup runs once per (base, version).
41
+ _cleaned_ast_dirs: set[str] = set()
42
+
43
+
44
+ def _cleanup_stale_ast_entries(ast_base: Path, current_dir: Path) -> None:
45
+ """Remove AST cache entries left behind by other graphify versions.
46
+
47
+ Sweeps sibling ``v*/`` directories and unversioned ``*.json`` entries
48
+ (the pre-versioning layout) under ``cache/ast/``. Best-effort: failures
49
+ are ignored, stragglers are retried on the next run.
50
+ """
51
+ key = str(current_dir)
52
+ if key in _cleaned_ast_dirs:
53
+ return
54
+ _cleaned_ast_dirs.add(key)
55
+ if not ast_base.is_dir():
56
+ return
57
+ import shutil
58
+
59
+ for child in ast_base.iterdir():
60
+ if child == current_dir:
61
+ continue
62
+ try:
63
+ if child.is_dir() and child.name.startswith("v"):
64
+ shutil.rmtree(child, ignore_errors=True)
65
+ elif child.suffix == ".json":
66
+ child.unlink()
67
+ except OSError:
68
+ pass
69
+
70
+
71
+ # Semantic cache entries are LLM output, so they depend on the extraction prompt
72
+ # that produced them, not just on file contents. Keying purely on content means a
73
+ # release that changes the prompt keeps replaying entries from the older prompt on
74
+ # every unchanged file, silently mixing extraction vintages in one graph (#1939).
75
+ # Versioning them by package version (as the AST cache does) would re-bill LLM
76
+ # extraction on every patch release — the reason #1252 deliberately left them
77
+ # unversioned. Fingerprinting the prompt itself keeps both properties: entries
78
+ # survive releases that don't touch the prompt, and invalidate only when it
79
+ # actually changed. Entries live under cache/semantic/p{fingerprint}/ when the
80
+ # caller supplies its prompt; callers that don't keep the historical flat layout.
81
+ _PROMPT_FP_LEN = 12
82
+
83
+ # Count of pre-fingerprint (flat-layout) entries served this process, so
84
+ # check_semantic_cache can report N to the user (#1939).
85
+ _legacy_semantic_hits = 0
86
+
87
+ # Count of cache entries that failed to parse as JSON this process. A corrupt
88
+ # entry is not a miss: left in place it fails on every future run, silently
89
+ # re-extracting (and, for semantic kinds, re-billing) the file forever. The
90
+ # counter lets check_semantic_cache surface one aggregate warning (#2405).
91
+ _corrupt_cache_entries = 0
92
+
93
+ # Prompt-file fingerprints already computed, keyed by (path, size, mtime_ns) —
94
+ # the same stat signature the hash index uses. check_semantic_cache resolves the
95
+ # prompt once per FILE in the corpus, so without this a 500-doc run re-reads and
96
+ # re-hashes the same spec 500 times (and warns 500 times when it is unreadable).
97
+ _prompt_fp_cache: dict[tuple, str] = {}
98
+
99
+
100
+ def prompt_fingerprint(prompt: "str | Path") -> str:
101
+ """Return a short stable fingerprint of an extraction prompt.
102
+
103
+ ``prompt`` is either the prompt text itself (the Python extraction path owns
104
+ its system prompt, :func:`graphify.llm._extraction_system`) or a Path to the
105
+ prompt file an agent loaded (the skill path's
106
+ ``references/extraction-spec.md``).
107
+
108
+ Line endings and trailing whitespace are normalized before hashing: the same
109
+ spec file checked out with CRLF on Windows must not fingerprint differently
110
+ from the LF checkout that wrote the cache, or every Windows run would look
111
+ like a prompt change and re-bill extraction.
112
+ """
113
+ if isinstance(prompt, Path):
114
+ text = prompt.read_text(encoding="utf-8", errors="replace")
115
+ else:
116
+ text = prompt
117
+ normalized = "\n".join(
118
+ line.rstrip() for line in text.replace("\r\n", "\n").replace("\r", "\n").split("\n")
119
+ ).strip()
120
+ return hashlib.sha256(normalized.encode()).hexdigest()[:_PROMPT_FP_LEN]
121
+
122
+
123
+ def _resolve_prompt_fp(prompt: "str | Path | None" = None,
124
+ prompt_file: "str | Path | None" = None) -> str | None:
125
+ """Fingerprint the caller's extraction prompt, or None when it supplied none.
126
+
127
+ ``prompt`` is prompt TEXT; ``prompt_file`` is a path to a file CONTAINING the
128
+ prompt. They are separate parameters rather than one overloaded argument
129
+ because the skill-driven callers are markdown snippets an agent copies with a
130
+ path substituted in — passing that path as ``prompt`` would hash the path
131
+ string itself, yielding a fingerprint that is stable, plausible, and tracks
132
+ nothing about the prompt. A silent wrong fingerprint is the exact failure
133
+ class #1939 is about, so the two are not inferred from each other.
134
+
135
+ Best-effort: an unreadable ``prompt_file`` falls back to the flat, unattributed
136
+ layout rather than failing the run — a cache is never worth aborting an
137
+ extraction over. It warns rather than falling back quietly, because that
138
+ fallback silently restores the very behavior this fixes, and the skill-side
139
+ caller substitutes this path by hand.
140
+ """
141
+ memo_key = None
142
+ if prompt_file is not None:
143
+ prompt = Path(prompt_file)
144
+ try:
145
+ st = prompt.stat()
146
+ memo_key = (str(prompt), st.st_size, st.st_mtime_ns)
147
+ if memo_key in _prompt_fp_cache:
148
+ return _prompt_fp_cache[memo_key]
149
+ except OSError:
150
+ pass # unreadable — fall through to the warning below
151
+ if prompt is None:
152
+ return None
153
+ try:
154
+ fp = prompt_fingerprint(prompt)
155
+ if memo_key is not None:
156
+ _prompt_fp_cache[memo_key] = fp
157
+ return fp
158
+ except (OSError, UnicodeError) as exc:
159
+ warnings.warn(
160
+ f"could not read extraction prompt {str(prompt)!r} ({exc}); semantic cache "
161
+ "entries cannot be attributed to a prompt version and fall back to the "
162
+ "unversioned layout, so this run may replay entries from an older "
163
+ "extraction prompt (#1939).",
164
+ RuntimeWarning,
165
+ stacklevel=3,
166
+ )
167
+ return None
168
+
169
+
170
+ # A frontmatter delimiter is a whole line of exactly three dashes (optional
171
+ # trailing whitespace). Substring checks like startswith("---") /
172
+ # find("\n---") also match `----` thematic breaks and `--- text` prose,
173
+ # silently dropping everything above them from the hash (#1259).
174
+ _FRONTMATTER_DELIM = re.compile(r"^---[ \t]*\r?$", re.MULTILINE)
175
+
176
+
177
+ def _body_content(content: bytes) -> bytes:
178
+ """Strip YAML frontmatter from Markdown content, returning only the body."""
179
+ text = content.decode(errors="replace")
180
+ opener = _FRONTMATTER_DELIM.match(text)
181
+ if opener is None:
182
+ return content
183
+ closer = _FRONTMATTER_DELIM.search(text, opener.end())
184
+ if closer is None:
185
+ return content
186
+ # Slice right after the closing `---` (not after its line) so the output
187
+ # stays byte-identical with the historical implementation for well-formed
188
+ # frontmatter -- existing semantic-cache hashes must not churn.
189
+ return text[closer.start() + 3:].encode()
190
+
191
+
192
+ # Stat-based index: maps absolute path → {size, mtime_ns, indexed_at_ns, ...}.
193
+ # Loaded once per process, flushed via atexit. Skips full file reads when
194
+ # size+mtime_ns are unchanged — same trade-off as make(1).
195
+ # Correctness risks: `touch` causes a harmless extra re-hash. Same-size edits
196
+ # inside one mtime tick used to return the PREVIOUS content's digest; the
197
+ # racily-clean guard below closes that hole (see _stat_sig_fresh).
198
+ # `graphify extract --force` / `graphify update --force` (or GRAPHIFY_FORCE=1)
199
+ # skip the cache reads and re-dispatch everything when needed (#1894).
200
+ _stat_index: dict[str, dict] = {}
201
+ _stat_index_root: Path | None = None
202
+ # Key anchor for the ON-DISK index (#2199): the first caller's key-root, i.e.
203
+ # the corpus. Distinct from _stat_index_root, which is the cache-FILE location
204
+ # (cache_root, #1774) — the two differ under --out and must not be conflated.
205
+ _stat_index_anchor: Path | None = None
206
+ _stat_index_dirty: bool = False
207
+
208
+
209
+ # Filesystem mtime granularity, in nanoseconds. A stat signature only proves a
210
+ # file is unchanged when the clock that stamped its mtime is finer-grained than
211
+ # the interval between two writes — which is false almost everywhere: NTFS
212
+ # advances mtime on the ~15.6 ms system tick, FAT/exFAT on 2 s, and Linux
213
+ # stamps from the coarse (jiffies) clock even though ext4 stores nanoseconds.
214
+ # 2 s is the conservative default that covers all of them. It costs nothing in
215
+ # practice: only files modified within the last 2 s lose the fastpath, and in a
216
+ # real corpus those are exactly the handful of files that changed and have to be
217
+ # read anyway. Override with GRAPHIFY_MTIME_GRANULARITY_MS (0 disables the
218
+ # guard and restores the pre-fix behaviour).
219
+ _MTIME_GRANULARITY_NS = 2_000_000_000
220
+
221
+
222
+ def _mtime_granularity_ns() -> int:
223
+ """Return the assumed filesystem mtime granularity in nanoseconds.
224
+
225
+ Read fresh on every call so the env var can be set after import (and so
226
+ tests can flip it without reloading the module).
227
+ """
228
+ raw = os.environ.get("GRAPHIFY_MTIME_GRANULARITY_MS", "").strip()
229
+ if raw:
230
+ try:
231
+ ms = float(raw)
232
+ except ValueError:
233
+ return _MTIME_GRANULARITY_NS
234
+ if ms >= 0:
235
+ return int(ms * 1_000_000)
236
+ return _MTIME_GRANULARITY_NS
237
+
238
+
239
+ def _stat_sig_fresh(entry: object, st: "os.stat_result") -> bool:
240
+ """True if ``entry`` provably describes the file's CURRENT content.
241
+
242
+ Beyond matching (size, mtime_ns), the entry must be *racily clean* in git's
243
+ sense: we must have read the content strictly after the file's mtime tick
244
+ had already closed. Otherwise a write that landed between our read and the
245
+ end of that tick would have left mtime (and, for a same-length edit, size)
246
+ untouched, and the stored digest would describe content that is no longer
247
+ on disk.
248
+
249
+ ``indexed_at_ns`` is the wall clock captured immediately BEFORE the content
250
+ was read. Requiring ``mtime + granularity <= indexed_at`` means any later
251
+ write necessarily lands in a new tick and so changes mtime, making it
252
+ visible to the next signature comparison.
253
+
254
+ Entries written by an older graphify carry no ``indexed_at_ns``; they are
255
+ treated as untrusted (one re-read each), and gain the field when rewritten.
256
+ """
257
+ if not isinstance(entry, dict):
258
+ return False
259
+ if entry.get("size") != st.st_size or entry.get("mtime_ns") != st.st_mtime_ns:
260
+ return False
261
+ indexed_at = entry.get("indexed_at_ns")
262
+ if not isinstance(indexed_at, int):
263
+ return False
264
+ return st.st_mtime_ns + _mtime_granularity_ns() <= indexed_at
265
+
266
+
267
+ def _stat_entry_for(abs_key: str, st: "os.stat_result", observed_at_ns: int) -> dict:
268
+ """Get-or-reset the index entry for ``abs_key`` and stamp when it was read.
269
+
270
+ Reuses the existing dict when the stat signature still matches, so
271
+ co-located values (other salts' digests, ``word_count``) survive; resets it
272
+ otherwise, so a stale ``word_count`` cannot outlive the content it counted.
273
+
274
+ ``observed_at_ns`` must be the clock reading taken *before* the content was
275
+ read — see :func:`_stat_sig_fresh` for why the ordering matters.
276
+ """
277
+ entry = _stat_index.get(abs_key)
278
+ if (not isinstance(entry, dict)
279
+ or entry.get("size") != st.st_size
280
+ or entry.get("mtime_ns") != st.st_mtime_ns):
281
+ entry = {"size": st.st_size, "mtime_ns": st.st_mtime_ns}
282
+ _stat_index[abs_key] = entry
283
+ entry["indexed_at_ns"] = observed_at_ns
284
+ return entry
285
+
286
+
287
+ def _stat_key_to_relative(key: str, anchor: Path) -> str:
288
+ """Return ``key`` as a forward-slash relative path from ``anchor``.
289
+
290
+ Local duplicate of :func:`graphify.detect._to_relative_for_storage` —
291
+ detect imports cache, so cache cannot import detect without a cycle
292
+ (and pulling detect in during the atexit flush would be fragile).
293
+ Out-of-anchor and already-relative keys pass through unchanged, and
294
+ ``..``-escaping relpaths are rejected (kept absolute), mirroring the
295
+ manifest's portability rules.
296
+ """
297
+ p = Path(key)
298
+ if not p.is_absolute():
299
+ return key
300
+ try:
301
+ rel = os.path.relpath(p, anchor)
302
+ except (ValueError, OSError):
303
+ return key # outside anchor (e.g. Windows cross-drive)
304
+ if rel == ".." or rel.startswith(".." + os.sep) or rel.startswith("../"):
305
+ return key # escaped anchor — keep absolute
306
+ return rel.replace(os.sep, "/")
307
+
308
+
309
+ def _stat_key_to_absolute(key: str, anchor: Path) -> str:
310
+ """Inverse of :func:`_stat_key_to_relative`.
311
+
312
+ Re-anchor a stored relative key against ``anchor``. Already-absolute keys
313
+ (legacy indexes, out-of-anchor entries) pass through unchanged so an index
314
+ written by an older graphify remains readable.
315
+ """
316
+ p = Path(key)
317
+ if p.is_absolute():
318
+ return str(p)
319
+ return str(anchor / p)
320
+
321
+
322
+ def _stat_index_file(root: Path) -> Path:
323
+ _out = Path(_GRAPHIFY_OUT)
324
+ base = _out if _out.is_absolute() else Path(root).resolve() / _out
325
+ return base / "cache" / "stat-index.json"
326
+
327
+
328
+ def _ensure_stat_index(root: Path, cache_root: "Path | None" = None) -> None:
329
+ global _stat_index, _stat_index_root, _stat_index_anchor, _stat_index_dirty
330
+ if _stat_index_root is not None:
331
+ return
332
+ # _stat_index_root determines the cache FILE location, so honoring an
333
+ # explicit cache_root keeps detect()'s word-count cache under the requested
334
+ # --out dir instead of polluting the scanned corpus with a stray
335
+ # graphify-out/ (#1747). _stat_index_anchor is the separate KEY anchor:
336
+ # in-memory keys stay absolute, but the on-disk index stores in-anchor keys
337
+ # relative so a moved/cloned corpus still hits (#2199) — same load/save
338
+ # re-anchoring the detect manifest uses.
339
+ _stat_index_root = Path(cache_root if cache_root is not None else root).resolve()
340
+ _stat_index_anchor = Path(root).resolve()
341
+ p = _stat_index_file(_stat_index_root)
342
+ _stat_index = {}
343
+ if p.exists():
344
+ try:
345
+ raw = json.loads(p.read_text(encoding="utf-8"))
346
+ if isinstance(raw, dict):
347
+ for k, v in raw.items():
348
+ if not isinstance(k, str):
349
+ continue
350
+ if Path(k).is_absolute():
351
+ # Legacy/out-of-anchor key: pass through, but never
352
+ # clobber a re-anchored relative (new-format) entry
353
+ # that resolved to the same absolute path.
354
+ _stat_index.setdefault(k, v)
355
+ else:
356
+ _stat_index[_stat_key_to_absolute(k, _stat_index_anchor)] = v
357
+ except (json.JSONDecodeError, OSError):
358
+ _stat_index = {}
359
+ atexit.register(_flush_stat_index)
360
+
361
+
362
+ def _flush_stat_index() -> None:
363
+ global _stat_index_dirty, _stat_index_root
364
+ if not _stat_index_dirty or _stat_index_root is None:
365
+ return
366
+ p = _stat_index_file(_stat_index_root)
367
+ # Build the on-disk form (#2199): prune entries whose file is gone (the
368
+ # index otherwise grows without bound), then store in-anchor keys as
369
+ # forward-slash relative paths so the index survives a corpus move/clone.
370
+ # Out-of-anchor keys stay absolute (same rule as the detect manifest); a
371
+ # reader tells the formats apart by absoluteness, so no version marker is
372
+ # needed. In-memory keys are untouched — only the serialization changes.
373
+ on_disk: dict[str, dict] = {}
374
+ for k, v in _stat_index.items():
375
+ try:
376
+ if not os.path.exists(k):
377
+ continue
378
+ except OSError:
379
+ continue
380
+ dk = _stat_key_to_relative(k, _stat_index_anchor) if _stat_index_anchor is not None else k
381
+ on_disk[dk] = v
382
+ # Never resurrect a corpus that was deleted while graphify was running
383
+ # (#2974): a hook-launched `graphify update . &` in a short-lived worktree
384
+ # outlives `git worktree remove`, and an unconditional `mkdir -p` here
385
+ # rebuilt the dead path as a husk holding nothing but this index. The
386
+ # index is a pure optimisation, so when its root is gone it is simply not
387
+ # written. Creating graphify-out/cache/ under a root that still exists is
388
+ # unchanged (a first run writes the index before anything else does).
389
+ try:
390
+ if not _stat_index_root.is_dir():
391
+ _stat_index_dirty = False
392
+ return
393
+ except OSError:
394
+ _stat_index_dirty = False
395
+ return
396
+ try:
397
+ p.parent.mkdir(parents=True, exist_ok=True)
398
+ fd, tmp = tempfile.mkstemp(dir=p.parent, prefix="stat-index.", suffix=".tmp")
399
+ try:
400
+ os.write(fd, json.dumps(on_disk, separators=(",", ":")).encode())
401
+ os.close(fd)
402
+ os.replace(tmp, p)
403
+ except Exception:
404
+ try:
405
+ os.close(fd)
406
+ except OSError:
407
+ pass
408
+ try:
409
+ os.unlink(tmp)
410
+ except OSError:
411
+ pass
412
+ except OSError:
413
+ pass
414
+ _stat_index_dirty = False
415
+
416
+
417
+ def _normalize_path(path: Path) -> Path:
418
+ """Normalize path for consistent cache keys across Windows path spellings."""
419
+ import sys
420
+ if sys.platform != "win32":
421
+ return path
422
+ s = str(path)
423
+ if s.startswith("\\\\?\\"):
424
+ s = s[4:] # strip extended-length prefix \\?\
425
+ return Path(os.path.normcase(s))
426
+
427
+
428
+ def file_hash(path: Path, root: Path = Path("."), cache_root: "Path | None" = None) -> str:
429
+ """SHA256 of file contents + path relative to root.
430
+
431
+ Uses a stat-based fastpath (size + mtime_ns) to skip full reads when the
432
+ file hasn't changed. Falls through to full SHA256 on first encounter, when
433
+ stat changes, and when the recorded signature is not yet provably stable
434
+ (see :func:`_stat_sig_fresh`) — so two different contents can never share a
435
+ digest. Index is flushed atomically at process exit.
436
+
437
+ Using the walked path relative to root keeps distinct symlink aliases from
438
+ sharing an extraction entry while preserving portability across machines
439
+ and checkout directories. Falls back to the resolved path when the walked
440
+ path cannot be expressed relative to root.
441
+
442
+ For Markdown files (.md), only the body below the YAML frontmatter is hashed,
443
+ so metadata-only changes (e.g. reviewed, status, tags) do not invalidate the cache.
444
+ """
445
+ global _stat_index_dirty
446
+ p = _normalize_path(Path(path))
447
+ root = _normalize_path(Path(root))
448
+ if not p.is_file():
449
+ raise IsADirectoryError(f"file_hash requires a file, got: {p}")
450
+
451
+ # The stat index is a cache artifact, so it must follow the cache location
452
+ # (cache_root), not the key-anchor root — otherwise it leaves a stray
453
+ # graphify-out/cache/stat-index.json inside the analyzed source tree even when
454
+ # the AST cache itself is redirected to CWD (#1774 completion).
455
+ _ensure_stat_index(root, cache_root=cache_root)
456
+ resolved = p.resolve()
457
+ abs_key = str(resolved)
458
+ # The salt is the path component that enters the digest (relative to root, or
459
+ # the absolute-path fallback). The stat-index memo MUST be keyed by it too:
460
+ # the same file hashed under two different roots yields two different digests
461
+ # (this happens within one `--out` run), and a memo keyed only by absolute
462
+ # path served whichever was computed first — making file_hash order-dependent
463
+ # and poisoning the persisted stat-index across runs (#1989). Store one digest
464
+ # per salt so alternating roots don't force re-reads.
465
+ resolved_root = root.resolve()
466
+ try:
467
+ resolved_rel = resolved.relative_to(resolved_root)
468
+ except ValueError:
469
+ # Preserve the existing fallback for a target outside the corpus. An
470
+ # in-root symlink to such a target is excluded by collect_files(), but
471
+ # direct cache callers still rely on the resolved external identity.
472
+ salt = resolved.as_posix().lower()
473
+ else:
474
+ walked = Path(os.path.abspath(p))
475
+ walked_root = Path(os.path.abspath(root))
476
+ try:
477
+ walked_rel = walked.relative_to(walked_root)
478
+ except ValueError:
479
+ # extract() resolves its operational root, while paths collected
480
+ # through a symlinked scan root retain that walked spelling. Find
481
+ # the lexical ancestor representing the resolved corpus root so a
482
+ # leaf symlink still contributes its own relative path to the key.
483
+ walked_rel = None
484
+ for parent in walked.parents:
485
+ try:
486
+ if parent.resolve() == resolved_root:
487
+ walked_rel = walked.relative_to(parent)
488
+ break
489
+ except OSError:
490
+ continue
491
+ if walked_rel is None:
492
+ walked_rel = resolved_rel
493
+ salt = walked_rel.as_posix().lower()
494
+
495
+ st: "os.stat_result | None" = None
496
+ try:
497
+ st = p.stat()
498
+ if _stat_sig_fresh(_stat_index.get(abs_key), st):
499
+ hashes = _stat_index[abs_key].get("hashes")
500
+ if isinstance(hashes, dict):
501
+ cached = hashes.get(salt)
502
+ if isinstance(cached, str):
503
+ return cached
504
+ # Legacy single-digest entries ("hash") don't record which salt
505
+ # produced them, so they are never trusted (#1989) — recompute once.
506
+ except OSError:
507
+ pass
508
+
509
+ # Captured BEFORE the read so the stamp can never post-date content that
510
+ # changed while we were reading it (see _stat_sig_fresh).
511
+ observed_at_ns = time.time_ns()
512
+ raw = p.read_bytes()
513
+ content = _body_content(raw) if p.suffix.lower() == ".md" else raw
514
+ h = hashlib.sha256()
515
+ h.update(content)
516
+ h.update(b"\x00")
517
+ h.update(salt.encode())
518
+ digest = h.hexdigest()
519
+
520
+ if st is not None:
521
+ entry = _stat_entry_for(abs_key, st, observed_at_ns)
522
+ hashes = entry.get("hashes")
523
+ if not isinstance(hashes, dict):
524
+ hashes = {}
525
+ entry["hashes"] = hashes
526
+ hashes[salt] = digest # preserve a co-located word_count / other salts
527
+ entry.pop("hash", None) # retire the un-salted legacy digest
528
+ _stat_index_dirty = True
529
+
530
+ return digest
531
+
532
+
533
+ def cached_word_count(path: Path, root: Path, compute, cache_root: "Path | None" = None) -> int:
534
+ """Word count with the same (size, mtime_ns) stat-fastpath cache as
535
+ :func:`file_hash`, persisted in the shared stat index.
536
+
537
+ ``detect()`` counts words in every PDF/docx/text file to size the corpus,
538
+ which re-opens and re-parses every binary on each run — minutes on a large
539
+ docs corpus even when only a handful of files changed (#1656). This caches
540
+ the count against the file's stat signature so an unchanged file is counted
541
+ once and read from the index thereafter. ``compute(path)`` produces the
542
+ count on a miss. A file that can't be stat'd (e.g. a Windows long path the
543
+ index normalization can't reach) simply recomputes and isn't cached —
544
+ correct, just not accelerated.
545
+ """
546
+ global _stat_index_dirty
547
+ p = _normalize_path(Path(path))
548
+ root = _normalize_path(Path(root))
549
+ _ensure_stat_index(root, cache_root=cache_root)
550
+ abs_key = str(p.resolve())
551
+ st: "os.stat_result | None" = None
552
+ try:
553
+ st = p.stat()
554
+ entry = _stat_index.get(abs_key)
555
+ if _stat_sig_fresh(entry, st) and "word_count" in entry:
556
+ return entry["word_count"]
557
+ except OSError:
558
+ pass
559
+
560
+ # Captured BEFORE compute() reads the file, for the same reason file_hash
561
+ # stamps before its read (see _stat_sig_fresh).
562
+ observed_at_ns = time.time_ns()
563
+ wc = compute(Path(path))
564
+
565
+ if st is not None:
566
+ _stat_entry_for(abs_key, st, observed_at_ns)["word_count"] = wc
567
+ _stat_index_dirty = True
568
+
569
+ return wc
570
+
571
+
572
+ def _relativize_source_files_in(payload: dict, root: Path) -> None:
573
+ """Mutate ``payload`` to rewrite absolute ``source_file`` fields as
574
+ forward-slash relative paths from ``root``.
575
+
576
+ Mirror of :func:`graphify.watch._relativize_source_files` so cached
577
+ extraction fragments persist in portable form (#777). Out-of-root paths
578
+ pass through unchanged.
579
+
580
+ A CWD-relative field is re-anchored too. Extractors stamp ``source_file``
581
+ with the path string ``extract()`` was handed, so relative inputs yield a
582
+ CWD-relative stamp — but the stored format is root-relative, and
583
+ :func:`_absolutize_source_files_in` reads it back as such. When CWD is not
584
+ the inferred root the two disagree and a warm hit resurrects a path that
585
+ names no file (``<root>/src/pages/index.astro`` for an input of
586
+ ``src/pages/index.astro`` under root ``<root>``). Every source_file-GATED
587
+ remap in ``extract()`` then misses — the file-stem prefix pass looks up
588
+ ``Path(source_file).resolve()`` in ``prefix_remap`` — so a warm hit keeps
589
+ the raw-path symbol ids a cold run canonicalizes: symbols stop sharing
590
+ their file node's stem, and for absolute inputs the on-disk path survives
591
+ into the persisted id (#2630). Only rewritten when the CWD-relative
592
+ reading is a real file and the root-relative reading is a different path,
593
+ so a fragment that already stores root-relative (a semantic subagent's,
594
+ see :func:`_normalize_source_file_value`) is left alone.
595
+
596
+ Only ``root`` is resolved — ``source_file`` itself is relativized
597
+ symbolically so in-root symlinks keep their original name rather than
598
+ pointing at the resolved target. Same reasoning as
599
+ :func:`graphify.detect._to_relative_for_storage`.
600
+ """
601
+ try:
602
+ root_resolved = Path(root).resolve()
603
+ except OSError:
604
+ return
605
+ # raw_calls (#: Pascal/Delphi cross-file inherited-call resolution) carries
606
+ # source_file the same way nodes/edges/hyperedges do, so it needs the same
607
+ # portable-path treatment for cache entries to round-trip correctly across
608
+ # machines/checkout directories.
609
+ # definition_file (#2990) is a path into the scanned tree exactly like
610
+ # source_file; a cache entry keeping it absolute replayed the build host's
611
+ # layout on every warm hit (#3223).
612
+ for bucket in ("nodes", "edges", "hyperedges", "raw_calls"):
613
+ for item in payload.get(bucket, []):
614
+ if not isinstance(item, dict):
615
+ continue
616
+ for key in ("source_file", "definition_file"):
617
+ source = item.get(key)
618
+ if not source:
619
+ continue
620
+ sp = Path(source)
621
+ if not sp.is_absolute():
622
+ # os.path.abspath is lexical (no symlink resolution),
623
+ # matching the symbolic relativization below.
624
+ cwd_form = Path(os.path.abspath(sp))
625
+ try:
626
+ if cwd_form == root_resolved / sp or not cwd_form.exists():
627
+ continue # already root-relative, or a ghost path
628
+ except OSError:
629
+ continue
630
+ sp = cwd_form
631
+ try:
632
+ rel = os.path.relpath(sp, root_resolved)
633
+ except (ValueError, OSError):
634
+ continue # out-of-root (e.g. Windows cross-drive)
635
+ if rel == ".." or rel.startswith(".." + os.sep) or rel.startswith("../"):
636
+ continue # escaped root — keep absolute
637
+ item[key] = rel.replace(os.sep, "/")
638
+
639
+
640
+ def _normalize_source_file_value(src: "str | Path", root_resolved: Path) -> str:
641
+ """Return ``src`` in portable form: backslashes flipped to forward slashes,
642
+ then relativized against ``root_resolved`` when the path is in-root.
643
+
644
+ Windows ``detect()`` emits absolute backslash paths, and a semantic
645
+ fragment carrying one verbatim used to be persisted as-is — poisoning later
646
+ ``graphify update`` runs with a machine-specific ``source_file`` (#2197).
647
+ Out-of-root absolute paths pass through (slash-normalized only), same
648
+ in/out rule and ``..``-rejection as :func:`_relativize_source_files_in`.
649
+ """
650
+ s = str(src).replace("\\", "/")
651
+ p = Path(s)
652
+ if not p.is_absolute():
653
+ return s
654
+ try:
655
+ rel = os.path.relpath(p, root_resolved)
656
+ except (ValueError, OSError):
657
+ return s # out-of-root (e.g. Windows cross-drive)
658
+ if rel == ".." or rel.startswith(".." + os.sep) or rel.startswith("../"):
659
+ return s # escaped root — keep absolute
660
+ return rel.replace(os.sep, "/")
661
+
662
+
663
+ def _semantic_entry_matches_path(result: dict, path: Path, root: Path) -> bool:
664
+ """Whether cached semantic groups belong to the requested walked path.
665
+
666
+ Before walked paths entered the cache salt, a symlink could overwrite its
667
+ target's unversioned semantic entry. Rejecting that mismatched legacy
668
+ payload makes the next extraction self-heal instead of replaying it forever.
669
+ """
670
+ expected = _normalize_path(Path(os.path.abspath(path)))
671
+ for bucket in ("nodes", "edges", "hyperedges"):
672
+ for item in result.get(bucket, []):
673
+ if not isinstance(item, dict):
674
+ continue
675
+ source = item.get("source_file")
676
+ if not source:
677
+ continue
678
+ source_path = Path(source)
679
+ if not source_path.is_absolute():
680
+ source_path = Path(root) / source_path
681
+ if _normalize_path(Path(os.path.abspath(source_path))) != expected:
682
+ return False
683
+ return True
684
+
685
+
686
+ # Storage marker standing in for the absolute root a cached id/path was minted
687
+ # under (#2257). Extractors mint node ids from the path STRING they are handed
688
+ # (``_make_id(str(path))``, ``_file_node_id(path)``), so a cache entry written
689
+ # under root A embeds A's slug in every id and edge endpoint. Those are only
690
+ # rewritten to the canonical root-relative form by extract()'s whole-graph
691
+ # id-remap, which keys its rewrites off the CURRENT run's paths — so on a warm
692
+ # hit under root B (a clone, a moved checkout, a second mount) the stored ids
693
+ # match no key and A's machine slug survives into graph.json. Entries are
694
+ # therefore stored root-anchored: the root's contribution is replaced by this
695
+ # marker on write and re-anchored to the current root on read, so a replay is
696
+ # portable by construction and reproduces exactly what a cold run under the
697
+ # current root would have minted (the pre-remap form every downstream pass in
698
+ # extract() expects). Same store-portable/re-anchor-on-load contract as
699
+ # ``source_file`` (#777) and the stat index (#2199). Neither ``$`` nor ``-`` can
700
+ # occur in a normalized id (``normalize_id`` drops every non-word character), and
701
+ # no plausible source literal — a shell ``$root``, a template ``${root}`` — opens
702
+ # with this exact token, so the marker cannot collide with extractor output.
703
+ _ROOT_MARKER = "$graphify-root$"
704
+
705
+
706
+ def _id_anchor(path_str: str, rel_str: str) -> str:
707
+ """Return the id-slug prefix ``path_str`` contributes above ``rel_str``.
708
+
709
+ ``_make_id`` normalizes a whole path string, and normalization distributes
710
+ over path joins (every separator run collapses to one ``_``), so an id
711
+ minted from an absolute path decomposes exactly as
712
+ ``normalize_id(root) + "_" + normalize_id(rel)``. Recovering the root half
713
+ from the file's own two spellings — rather than assuming it equals the
714
+ scan root — keeps the decomposition exact when the extractor was handed an
715
+ unresolved or relative path (a symlinked root such as macOS ``/tmp`` ->
716
+ ``/private/tmp``, or inputs given relative to CWD).
717
+
718
+ Returns "" when ``path_str`` contributes no prefix (it already IS the
719
+ relative form) or when the two spellings disagree about the tail — both
720
+ mean "nothing to re-anchor", which leaves the entry untouched.
721
+ """
722
+ from graphify.ids import normalize_id # ids imports only re/unicodedata: no cycle
723
+
724
+ full = normalize_id(path_str)
725
+ tail = normalize_id(rel_str)
726
+ if not tail or full == tail:
727
+ return ""
728
+ suffix = "_" + tail
729
+ return full[: -len(suffix)] if full.endswith(suffix) else ""
730
+
731
+
732
+ def _portability_anchors(path: "str | Path", root: "str | Path") -> tuple[list[str], str, list[str], str]:
733
+ """Root forms to strip from / restore into one cache entry (#2257).
734
+
735
+ Returns ``(id_anchors, id_restore, path_anchors, path_restore)``. The two
736
+ ``*_anchors`` lists are the forms a stored value may have been minted from,
737
+ longest first so a shorter form can't shadow a longer one; the two
738
+ ``*_restore`` values are what the CURRENT run's extractor would mint.
739
+
740
+ Several spellings are collected because one entry can hold ids minted from
741
+ different forms: this file's own path as passed and as resolved, plus
742
+ cross-file edge targets minted from ANOTHER in-root file's absolute path
743
+ (which share the scan root's spelling). They collapse to a single marker on
744
+ write; that is safe because extract()'s remap registers both the input-form
745
+ and absolute-resolved-form ids for every path (#1529) and maps them to the
746
+ same canonical id, so restoring either one canonicalizes identically.
747
+ """
748
+ from graphify.ids import normalize_id
749
+
750
+ try:
751
+ root_resolved = Path(root).resolve()
752
+ except OSError:
753
+ return [], "", [], ""
754
+ try:
755
+ path_resolved = Path(path).resolve()
756
+ except (OSError, RuntimeError):
757
+ path_resolved = Path(path)
758
+ try:
759
+ rel = os.path.relpath(path_resolved, root_resolved)
760
+ except (ValueError, OSError):
761
+ rel = ""
762
+
763
+ # Ordered by preference for the restore form: the spelling the extractor was
764
+ # actually handed first, then its resolved form, then the scan root.
765
+ from_given = _id_anchor(str(path), rel)
766
+ from_resolved = _id_anchor(str(path_resolved), rel)
767
+ id_restore = next(
768
+ (a for a in (from_given, from_resolved, normalize_id(str(root_resolved))) if a), ""
769
+ )
770
+ # Every strippable form must be one this same call would RESTORE, or an id
771
+ # is re-anchored under a prefix it was never minted with. That rules out a
772
+ # relative ``root`` spelling ("src"): with an absolute ``path`` the restore
773
+ # form is the resolved slug, so admitting ``src`` as an anchor would rewrite
774
+ # an already-canonical ``src_utils_foo`` into ``<abs-root-slug>_utils_foo``.
775
+ # The two path-derived forms are always safe — they ARE the restore
776
+ # candidates — and cover a relative root on their own whenever the extractor
777
+ # was handed a matching relative path.
778
+ root_id_forms = (normalize_id(str(root_resolved)),)
779
+ if Path(root).is_absolute():
780
+ root_id_forms += (normalize_id(str(root)),)
781
+ id_anchors = sorted(
782
+ {a for a in (from_given, from_resolved, *root_id_forms) if a},
783
+ key=len, reverse=True,
784
+ )
785
+ # Only absolute roots may anchor a PATH value: a relative one ("corpus")
786
+ # would also match a genuinely relative value that merely starts with the
787
+ # same segment, and there is no way to tell the two apart on read.
788
+ path_anchors = sorted(
789
+ {s for s in (str(root_resolved), str(root)) if Path(s).is_absolute()},
790
+ key=len, reverse=True,
791
+ )
792
+ return id_anchors, id_restore, path_anchors, str(root_resolved)
793
+
794
+
795
+ def _rewrite_id_keyed_table_keys(payload: object, fn) -> None:
796
+ """Apply ``fn`` to objc_field_types["tables"] KEYS (#3150).
797
+
798
+ That table is the one extractor bucket keyed BY node id, which
799
+ :func:`_rewrite_strings` deliberately never touches - so a cached ObjC
800
+ shard replayed under another root kept absolute-derived class ids as keys
801
+ while the node ids themselves were re-anchored, and the receiver-typing
802
+ pass missed every class.
803
+ """
804
+ ft = payload.get("objc_field_types") if isinstance(payload, dict) else None
805
+ tables = ft.get("tables") if isinstance(ft, dict) else None
806
+ if isinstance(tables, dict):
807
+ ft["tables"] = {
808
+ (fn(k) if isinstance(k, str) else k): v for k, v in tables.items()
809
+ }
810
+
811
+
812
+ def _rewrite_strings(obj: object, fn) -> None:
813
+ """Apply ``fn`` to every string VALUE reachable in ``obj``, in place.
814
+
815
+ Values only, never dict keys: rewriting keys blindly could silently
816
+ collide two entries into one. The single id-keyed bucket -
817
+ ``objc_field_types["tables"]`` - is handled by
818
+ :func:`_rewrite_id_keyed_table_keys` beside each call to this (#3150).
819
+ """
820
+ if isinstance(obj, dict):
821
+ items: "Iterable" = obj.items()
822
+ elif isinstance(obj, list):
823
+ items = enumerate(obj)
824
+ else:
825
+ return
826
+ for key, value in list(items):
827
+ if isinstance(value, str):
828
+ new = fn(value)
829
+ if new != value:
830
+ obj[key] = new # type: ignore[index]
831
+ else:
832
+ _rewrite_strings(value, fn)
833
+
834
+
835
+ def _relativize_ids_in(payload: dict, path: "str | Path", root: Path) -> None:
836
+ """Replace the absolute root inside every stored id / path with the marker.
837
+
838
+ Walks the WHOLE payload rather than a list of known buckets. The id form is
839
+ self-identifying — a long casefolded path slug that nothing but an
840
+ absolute-path-derived id can start with — so this reaches carriers a
841
+ hand-maintained bucket list misses or has yet to grow: ``nodes[].id``,
842
+ ``edges[].source``/``target``, hyperedge member lists under any of their
843
+ aliases, ``raw_calls[].caller_nid``, ``swift_extensions[].nid``, plus the
844
+ path-valued ``edges[].target_file``, ``bash_sources[].source_file`` and
845
+ ``*_type_table.path``, which resolution passes need pointing at the current
846
+ root to reproduce a cold run's edges.
847
+
848
+ Call AFTER :func:`_relativize_source_files_in`: that stores ``source_file``
849
+ as a bare relative path (the #777 format), and running this first would
850
+ leave it marker-prefixed instead, churning the on-disk format for nothing.
851
+ """
852
+ id_anchors, _, path_anchors, _ = _portability_anchors(path, root)
853
+ if not id_anchors and not path_anchors:
854
+ return
855
+
856
+ def anchor(value: str) -> str:
857
+ # Path form first: it requires a separator, which an id never contains.
858
+ for a in path_anchors:
859
+ if value == a:
860
+ return _ROOT_MARKER
861
+ for sep in ("/", "\\"):
862
+ if value.startswith(a + sep):
863
+ return _ROOT_MARKER + "/" + value[len(a) + 1:].replace("\\", "/")
864
+ for a in id_anchors:
865
+ if value.startswith(a + "_"):
866
+ return _ROOT_MARKER + "_" + value[len(a) + 1:]
867
+ return value
868
+
869
+ _rewrite_strings(payload, anchor)
870
+ _rewrite_id_keyed_table_keys(payload, anchor)
871
+
872
+
873
+ def _absolutize_ids_in(payload: dict, path: "str | Path", root: Path) -> None:
874
+ """Inverse of :func:`_relativize_ids_in` — re-anchor to the current root.
875
+
876
+ The restored string is assembled by slicing, never by re-normalizing: the
877
+ stored form (``$graphify-root$_pkg_mod_base``) is a storage encoding, not a valid id,
878
+ and running it back through ``normalize_id`` would drop the marker's ``$``
879
+ and fuse it into the slug. Entries written before #2257 carry no marker and
880
+ pass through untouched — they are swept anyway, since AST entries live under
881
+ a per-version directory (:func:`cache_dir`).
882
+ """
883
+ _, id_restore, _, path_restore = _portability_anchors(path, root)
884
+
885
+ def restore(value: str) -> str:
886
+ if not value.startswith(_ROOT_MARKER):
887
+ return value
888
+ rest = value[len(_ROOT_MARKER):]
889
+ if not rest:
890
+ return path_restore
891
+ if rest[0] == "/":
892
+ tail = rest[1:]
893
+ return str(Path(path_restore) / tail) if tail else path_restore
894
+ if rest[0] == "_":
895
+ return (id_restore + rest) if id_restore else rest[1:]
896
+ return value
897
+
898
+ _rewrite_strings(payload, restore)
899
+ _rewrite_id_keyed_table_keys(payload, restore)
900
+
901
+
902
+ def _absolutize_source_files_in(payload: dict, root: Path) -> None:
903
+ """Inverse of :func:`_relativize_source_files_in`.
904
+
905
+ Re-anchor relative ``source_file`` fields against ``root`` so callers
906
+ that load a cached fragment see the same absolute-path shape that a
907
+ fresh in-process extraction would produce. Legacy cache entries with
908
+ absolute ``source_file`` values pass through unchanged.
909
+ """
910
+ try:
911
+ root_resolved = Path(root).resolve()
912
+ except OSError:
913
+ return
914
+ for bucket in ("nodes", "edges", "hyperedges", "raw_calls"):
915
+ for item in payload.get(bucket, []):
916
+ if not isinstance(item, dict):
917
+ continue
918
+ # Mirror of the relativize side: definition_file re-anchors too (#3223).
919
+ for key in ("source_file", "definition_file"):
920
+ source = item.get(key)
921
+ if not source:
922
+ continue
923
+ sp = Path(source)
924
+ if sp.is_absolute():
925
+ continue
926
+ try:
927
+ item[key] = str(root_resolved / sp)
928
+ except (TypeError, OSError):
929
+ continue
930
+
931
+
932
+ def cache_dir(root: Path = Path("."), kind: str = "ast",
933
+ prompt_fp: str | None = None) -> Path:
934
+ """Returns the cache directory for ``kind`` - creates it if needed.
935
+
936
+ kind is "ast", "semantic", or a mode-namespaced semantic kind such as
937
+ "semantic-deep" (#1894). Separate subdirectories prevent semantic cache
938
+ entries from overwriting AST cache entries for the same source_file (#582).
939
+
940
+ AST entries live in graphify-out/cache/ast/v{version}-s{schema}/, namespaced
941
+ by graphify version and cache-key schema because they depend on extractor
942
+ code and key semantics, not just file contents. Semantic entries are still
943
+ NOT version-namespaced (re-extraction
944
+ costs LLM calls, #1252): they live in graphify-out/cache/semantic/, with
945
+ deep-mode entries beside them in graphify-out/cache/semantic-deep/.
946
+
947
+ ``prompt_fp`` (semantic kinds only) adds a p{fingerprint}/ subdirectory so
948
+ entries are attributed to the extraction prompt that produced them (#1939).
949
+ Omitting it yields the historical flat layout, where entries of unknown
950
+ vintage live.
951
+ """
952
+ _out = Path(_GRAPHIFY_OUT)
953
+ base = _out if _out.is_absolute() else Path(root).resolve() / _out
954
+ d = base / "cache" / kind
955
+ if kind == "ast":
956
+ d = d / f"v{_EXTRACTOR_VERSION}-s{_AST_CACHE_SCHEMA}"
957
+ _cleanup_stale_ast_entries(d.parent, d)
958
+ elif prompt_fp:
959
+ d = d / f"p{prompt_fp}"
960
+ d.mkdir(parents=True, exist_ok=True)
961
+ return d
962
+
963
+
964
+ def load_cached(path: Path, root: Path = Path("."), kind: str = "ast",
965
+ cache_root: Path | None = None, prompt: "str | Path | None" = None,
966
+ prompt_file: "str | Path | None" = None,
967
+ allow_legacy: bool = True,
968
+ allow_partial: bool = False) -> dict | None:
969
+ """Return cached extraction for this file if hash matches, else None.
970
+
971
+ Cache key: SHA256 of file contents.
972
+ Cache value: stored as graphify-out/cache/{kind}/{hash}.json (AST entries
973
+ under the per-version subdirectory, see :func:`cache_dir`).
974
+
975
+ ``root`` anchors the content-hash key and source_file relativization (it
976
+ must stay the inferred common parent so keys remain portable). ``cache_root``
977
+ decouples *where* the cache directory lives from that anchor — the cache is
978
+ an output and must not land inside a read-only/analyzed source tree (#1774).
979
+ When ``cache_root`` is None the location falls back to ``root`` (unchanged
980
+ behavior for existing callers).
981
+
982
+ AST entries written by other graphify versions — including the legacy
983
+ flat cache/ layout (pre-0.5.3) and the unversioned cache/ast/ layout —
984
+ are deliberately not consulted: they were produced by a different
985
+ extractor and may be stale.
986
+
987
+ ``prompt`` (semantic kinds) is the extraction prompt — text, or a Path to
988
+ the prompt file — that the caller is about to extract with. It selects the
989
+ p{fingerprint}/ namespace, so an entry produced by a different prompt is a
990
+ miss rather than a silent stale hit (#1939). When it is given and the
991
+ fingerprinted namespace misses, ``allow_legacy`` (default True) falls back
992
+ to a flat-layout entry: those predate fingerprinting, so their vintage is
993
+ unknowable — they are served rather than re-billed, and the hit is counted
994
+ so :func:`check_semantic_cache` can report N to the user. Callers that must
995
+ not mix vintages within one entry (see :func:`save_semantic_cache`'s
996
+ ``merge_existing``) pass allow_legacy=False.
997
+ Returns None if no cache entry or file has changed.
998
+ """
999
+ global _legacy_semantic_hits, _corrupt_cache_entries
1000
+ location = cache_root if cache_root is not None else root
1001
+ try:
1002
+ h = file_hash(path, root, cache_root=cache_root)
1003
+ except OSError:
1004
+ return None
1005
+ prompt_fp = _resolve_prompt_fp(prompt, prompt_file)
1006
+ entry = cache_dir(location, kind, prompt_fp) / f"{h}.json"
1007
+ legacy_hit = False
1008
+ if prompt_fp and not entry.exists() and allow_legacy:
1009
+ legacy = cache_dir(location, kind) / f"{h}.json"
1010
+ if legacy.exists():
1011
+ entry, legacy_hit = legacy, True
1012
+ if entry.exists():
1013
+ try:
1014
+ result = json.loads(entry.read_text(encoding="utf-8"))
1015
+ except json.JSONDecodeError:
1016
+ # Corrupt entry, not a miss: a truncated write or a bad producer
1017
+ # (e.g. unescaped Windows backslashes in source_file) leaves JSON
1018
+ # that fails to parse on every future run, so the file is silently
1019
+ # re-extracted forever. Count it so the run can report it (#2405).
1020
+ _corrupt_cache_entries += 1
1021
+ return None
1022
+ except OSError:
1023
+ return None
1024
+ # A ``partial`` entry was produced from a truncated LLM response and
1025
+ # covers only part of the file's symbols. Serving it as authoritative
1026
+ # would return the incomplete node set forever until the file is
1027
+ # re-extracted. Treat it as a cache MISS (the normal read path) so the
1028
+ # file is re-dispatched and retried. Self-heals: a later complete
1029
+ # extraction overwrites the same content-hash key with a non-partial
1030
+ # entry. ``allow_partial`` is the one exception — the merge_existing
1031
+ # checkpoint peeks at a partial prev so it can accumulate a file's slices
1032
+ # across chunks without losing the truncated one (it stays partial).
1033
+ if not allow_partial and isinstance(result, dict) and result.get("partial"):
1034
+ return None
1035
+ # A semantic entry with zero nodes and zero hyperedges is invalid (#2927):
1036
+ # an edge-only or empty result (e.g. LLM omitted entities for the file)
1037
+ # is not a valid standalone extraction. Treating it as a cache MISS
1038
+ # ensures the file is re-dispatched and retried (#933/#1666).
1039
+ if (
1040
+ not allow_partial
1041
+ and kind.startswith("semantic")
1042
+ and isinstance(result, dict)
1043
+ and not result.get("nodes")
1044
+ and not result.get("hyperedges")
1045
+ ):
1046
+ return None
1047
+ if (
1048
+ kind.startswith("semantic")
1049
+ and isinstance(result, dict)
1050
+ and not _semantic_entry_matches_path(result, Path(path), Path(root))
1051
+ ):
1052
+ return None
1053
+ if legacy_hit:
1054
+ _legacy_semantic_hits += 1
1055
+ # Re-anchor relative source_file fields so callers see the same
1056
+ # absolute-path shape that a fresh in-process extraction produces
1057
+ # (#777). Legacy entries with absolute source_file pass through.
1058
+ if isinstance(result, dict):
1059
+ _absolutize_source_files_in(result, root)
1060
+ # Same contract for the ids and remaining paths the entry embeds
1061
+ # (#2257): without this a warm hit under a different absolute root
1062
+ # replays ids minted from the ORIGINAL root, which extract()'s
1063
+ # id-remap cannot fix because they match none of its current-path
1064
+ # keys. Order is free — source_file never carries the marker.
1065
+ _absolutize_ids_in(result, path, root)
1066
+ return result
1067
+ return None
1068
+
1069
+
1070
+ def save_cached(path: Path, result: dict, root: Path = Path("."), kind: str = "ast",
1071
+ cache_root: Path | None = None, prompt: "str | Path | None" = None,
1072
+ prompt_file: "str | Path | None" = None) -> None:
1073
+ """Save extraction result for this file.
1074
+
1075
+ Stores as graphify-out/cache/{kind}/{hash}.json where hash = SHA256 of current file contents.
1076
+ result should be a dict with 'nodes' and 'edges' lists.
1077
+
1078
+ ``root`` anchors the content-hash key and source_file relativization;
1079
+ ``cache_root`` (when given) is where the cache directory is written, decoupled
1080
+ from ``root`` so the cache never lands inside the analyzed source tree (#1774).
1081
+
1082
+ ``prompt`` (semantic kinds) is the extraction prompt that produced ``result``
1083
+ — text, or a Path to the prompt file. It stamps the entry into the
1084
+ p{fingerprint}/ namespace so a later run under a different prompt does not
1085
+ replay it (#1939). Writes always land in the fingerprinted namespace when a
1086
+ prompt is given: an entry of known vintage is never written back into the
1087
+ flat unknown-vintage layout.
1088
+
1089
+ No-ops if `path` is not a regular file. Subagent-produced semantic fragments
1090
+ occasionally carry a directory path in `source_file`; skipping them prevents
1091
+ IsADirectoryError from aborting the whole batch.
1092
+ """
1093
+ p = Path(path)
1094
+ if not p.is_file():
1095
+ return
1096
+ # Relativize source_file fields against ``root`` before write so the
1097
+ # cache file on disk is portable across machines and checkout
1098
+ # directories (#777). The cache key is content-hashed so lookup is
1099
+ # already path-independent; this fixes the embedded path leak.
1100
+ #
1101
+ # Serialize a relativized copy rather than mutating the caller's dict —
1102
+ # downstream pipeline steps (notably extract.py's AST prefix remap, which
1103
+ # looks up Path(source_file).resolve() in a prefix table) depend on the
1104
+ # source_file field's original absolute form. Mutating the input here would
1105
+ # silently break those remaps on the first extraction pass.
1106
+ #
1107
+ # The copy is unconditional (it used to be gated on a non-empty
1108
+ # nodes/edges/hyperedges/raw_calls bucket): a truthiness gate skips the copy
1109
+ # for a result whose only payload lives in another bucket — an empty
1110
+ # ``nodes`` beside a populated ``bash_sources`` — and the id/path anchoring
1111
+ # below would then mutate the caller's dict for real.
1112
+ on_disk = result
1113
+ if isinstance(result, dict):
1114
+ import copy as _copy
1115
+ on_disk = _copy.deepcopy(result)
1116
+ _relativize_source_files_in(on_disk, root)
1117
+ # Then replace the absolute root inside the ids and remaining paths, so
1118
+ # the entry replays portably under any root (#2257). Strictly after the
1119
+ # source_file pass, which owns that field's bare-relative format.
1120
+ _relativize_ids_in(on_disk, p, root)
1121
+ h = file_hash(p, root, cache_root=cache_root)
1122
+ location = cache_root if cache_root is not None else root
1123
+ target_dir = cache_dir(location, kind, _resolve_prompt_fp(prompt, prompt_file))
1124
+ entry = target_dir / f"{h}.json"
1125
+ fd, tmp_path = tempfile.mkstemp(dir=target_dir, prefix=f"{h}.", suffix=".tmp")
1126
+ try:
1127
+ os.write(fd, json.dumps(on_disk).encode())
1128
+ os.close(fd)
1129
+ try:
1130
+ os.replace(tmp_path, entry)
1131
+ except PermissionError:
1132
+ # Windows: os.replace can fail with WinError 5 if the target is
1133
+ # briefly locked. Fall back to copy-then-delete.
1134
+ import shutil
1135
+ shutil.copy2(tmp_path, entry)
1136
+ os.unlink(tmp_path)
1137
+ except Exception:
1138
+ try:
1139
+ os.close(fd)
1140
+ except OSError:
1141
+ pass
1142
+ try:
1143
+ os.unlink(tmp_path)
1144
+ except OSError:
1145
+ pass
1146
+ raise
1147
+
1148
+
1149
+ def cached_files(root: Path = Path(".")) -> set[str]:
1150
+ """Return set of file hashes that have a valid cache entry (any kind)."""
1151
+ base = Path(root).resolve() / _GRAPHIFY_OUT / "cache"
1152
+ hashes: set[str] = set()
1153
+ # Legacy flat entries
1154
+ if base.is_dir():
1155
+ hashes.update(p.stem for p in base.glob("*.json"))
1156
+ # Namespaced entries, all globbed recursively: ast/ has per-version subdirs,
1157
+ # semantic-deep/ holds --mode deep entries (#1894), and both semantic kinds
1158
+ # have per-prompt-fingerprint subdirs alongside pre-fingerprint flat entries
1159
+ # (#1939).
1160
+ for kind in ("ast", "semantic", "semantic-deep"):
1161
+ d = base / kind
1162
+ if d.is_dir():
1163
+ hashes.update(p.stem for p in d.glob("**/*.json"))
1164
+ return hashes
1165
+
1166
+
1167
+ def clear_cache(root: Path = Path(".")) -> None:
1168
+ """Delete all cache entries (ast/, semantic/, semantic-deep/, and legacy
1169
+ flat entries)."""
1170
+ base = Path(root).resolve() / _GRAPHIFY_OUT / "cache"
1171
+ # Legacy flat entries
1172
+ if base.is_dir():
1173
+ for f in base.glob("*.json"):
1174
+ f.unlink()
1175
+ # Namespaced entries, all globbed recursively: ast/ has per-version subdirs,
1176
+ # semantic-deep/ holds --mode deep entries (#1894), and both semantic kinds
1177
+ # have per-prompt-fingerprint subdirs (#1939).
1178
+ for kind in ("ast", "semantic", "semantic-deep"):
1179
+ d = base / kind
1180
+ if d.is_dir():
1181
+ for f in d.glob("**/*.json"):
1182
+ f.unlink()
1183
+
1184
+
1185
+ def prune_semantic_cache(root: Path, live_hashes: set[str]) -> int:
1186
+ """Remove orphaned semantic cache entries, returning the count pruned.
1187
+
1188
+ The semantic cache is content-hash-keyed (``{file_hash}.json`` under
1189
+ ``cache/semantic/``) and deliberately UNVERSIONED — entries are produced by
1190
+ the LLM from file contents, so invalidating them on every release would
1191
+ re-bill extraction. Because it is unversioned it is also never swept by the
1192
+ AST version-cleanup, so every content change or file deletion leaves a
1193
+ permanent orphan entry that accumulates unbounded.
1194
+
1195
+ This sweeps ``cache/semantic/*.json`` AND ``cache/semantic-deep/*.json``
1196
+ (the ``--mode deep`` namespace, #1894) and deletes any entry whose stem
1197
+ (the content hash) is not in ``live_hashes`` — the hashes of the current
1198
+ live document set. Both namespaces are pruned against the SAME live set:
1199
+ liveness is content-based and mode-independent, so a hash that is live for
1200
+ one namespace is live for both. Skipping the deep namespace would re-grow
1201
+ the unbounded-orphan problem this function fixed (#1527). ``*.tmp``
1202
+ atomic-write temporaries are skipped, and only these directories are
1203
+ touched (never ``cache/ast/**`` or anything else). The unversioned design
1204
+ is preserved: we prune by liveness, not by version.
1205
+
1206
+ The sweep recurses into the per-prompt-fingerprint subdirs (#1939) for the
1207
+ same reason it covers the deep namespace: a glob that stopped at the top
1208
+ level would leave every fingerprinted entry permanently unprunable. Entries
1209
+ under a fingerprint other than the current one are pruned by liveness only,
1210
+ never swept wholesale the way :func:`_cleanup_stale_ast_entries` sweeps old
1211
+ AST versions — two hosts with different prompts (verbose vs compact
1212
+ extraction-spec) can share one graphify-out/, and a wholesale sweep would
1213
+ have each run delete the other's entries and re-bill extraction on every
1214
+ alternation. Liveness keeps the total bounded by live docs × prompts seen.
1215
+
1216
+ Best-effort, mirroring :func:`_cleanup_stale_ast_entries`: each unlink is
1217
+ wrapped in ``try/except OSError`` and a failure is ignored. The worst-case
1218
+ failure mode is benign — a surviving orphan costs only one re-extraction of
1219
+ one doc on a future run, never incorrect output.
1220
+ """
1221
+ _out = Path(_GRAPHIFY_OUT)
1222
+ base = _out if _out.is_absolute() else Path(root).resolve() / _out
1223
+ pruned = 0
1224
+ for kind in ("semantic", "semantic-deep"):
1225
+ semantic_dir = base / "cache" / kind
1226
+ if not semantic_dir.is_dir():
1227
+ continue
1228
+ for entry in semantic_dir.glob("**/*.json"):
1229
+ if entry.stem in live_hashes:
1230
+ continue
1231
+ try:
1232
+ entry.unlink()
1233
+ pruned += 1
1234
+ except OSError:
1235
+ pass
1236
+ return pruned
1237
+
1238
+
1239
+ def check_semantic_cache(
1240
+ files: list[str],
1241
+ root: Path = Path("."),
1242
+ mode: str | None = None,
1243
+ prompt: "str | Path | None" = None,
1244
+ prompt_file: "str | Path | None" = None,
1245
+ cache_root: "Path | None" = None,
1246
+ ) -> tuple[list[dict], list[dict], list[dict], list[str]]:
1247
+ """Check semantic extraction cache for a list of absolute file paths.
1248
+
1249
+ Returns (cached_nodes, cached_edges, cached_hyperedges, uncached_files).
1250
+ Uncached files need Claude extraction; cached files are merged directly.
1251
+
1252
+ ``mode`` selects the cache namespace: ``None`` (the default) reads
1253
+ ``cache/semantic/`` — byte-identical to the historical behavior, so
1254
+ existing callers that omit it (including older installed skill flows)
1255
+ are unaffected. A non-None mode (e.g. ``"deep"``) reads
1256
+ ``cache/semantic-{mode}/`` instead, so deep-mode results never shadow
1257
+ (or get shadowed by) standard-mode entries for the same content (#1894).
1258
+
1259
+ ``prompt`` is the extraction prompt this run will use for the uncached
1260
+ files — the prompt text (Python path) or a Path to the prompt file the
1261
+ agent loaded (skill path, ``references/extraction-spec.md``). Supplying it
1262
+ restricts hits to entries produced by that same prompt, so an upgrade that
1263
+ changed the prompt re-extracts instead of replaying the older vintage
1264
+ (#1939). Entries written before fingerprinting existed still hit — their
1265
+ vintage is unknowable and dropping them would re-bill a whole corpus — but
1266
+ a warning reports how many were served. Omitting ``prompt`` keeps the
1267
+ historical behavior for existing callers.
1268
+
1269
+ ``cache_root`` decouples *where* the cache is read from the key-anchor
1270
+ ``root``, mirroring :func:`load_cached` and :func:`save_semantic_cache`
1271
+ (#1774 / #1990). With ``--out``, pass the corpus as ``root`` (so content-hash
1272
+ keys and relative-path resolution stay anchored to the source tree) and the
1273
+ output directory as ``cache_root``. Omitting it keeps ``root`` for both.
1274
+ """
1275
+ global _legacy_semantic_hits
1276
+ kind = "semantic" if mode is None else f"semantic-{mode}"
1277
+ cached_nodes: list[dict] = []
1278
+ cached_edges: list[dict] = []
1279
+ cached_hyperedges: list[dict] = []
1280
+ uncached: list[str] = []
1281
+ legacy_before = _legacy_semantic_hits
1282
+ corrupt_before = _corrupt_cache_entries
1283
+
1284
+ for fpath in files:
1285
+ p = Path(fpath)
1286
+ if not p.is_absolute():
1287
+ p = Path(root) / p
1288
+ result = load_cached(p, root, kind=kind, cache_root=cache_root,
1289
+ prompt=prompt, prompt_file=prompt_file)
1290
+ if result is not None:
1291
+ cached_nodes.extend(result.get("nodes", []))
1292
+ cached_edges.extend(result.get("edges", []))
1293
+ cached_hyperedges.extend(result.get("hyperedges", []))
1294
+ else:
1295
+ uncached.append(fpath)
1296
+
1297
+ legacy = _legacy_semantic_hits - legacy_before
1298
+ if legacy:
1299
+ warnings.warn(
1300
+ f"{legacy} semantic cache entr{'y' if legacy == 1 else 'ies'} predate "
1301
+ "extraction-prompt fingerprinting and were written by an unknown prompt "
1302
+ "version; they were replayed as-is, so this graph may mix extraction "
1303
+ "vintages. Re-run with --force (or GRAPHIFY_FORCE=1) to re-extract them "
1304
+ "with the current prompt (#1939).",
1305
+ RuntimeWarning,
1306
+ stacklevel=2,
1307
+ )
1308
+
1309
+ corrupt = _corrupt_cache_entries - corrupt_before
1310
+ if corrupt:
1311
+ warnings.warn(
1312
+ f"{corrupt} semantic cache entr{'y' if corrupt == 1 else 'ies'} could "
1313
+ "not be parsed as JSON and were treated as misses, so those files were "
1314
+ "re-extracted. A corrupt entry stays on disk and fails again every run; "
1315
+ "run with --force (or GRAPHIFY_FORCE=1) to rewrite them, or clear the "
1316
+ "cache to stop paying for the re-extraction (#2405).",
1317
+ RuntimeWarning,
1318
+ stacklevel=2,
1319
+ )
1320
+
1321
+ return cached_nodes, cached_edges, cached_hyperedges, uncached
1322
+
1323
+
1324
+ def _group_has_partial_marker(group: dict) -> bool:
1325
+ """True if any node/edge/hyperedge in a per-file group carries the internal
1326
+ ``_partial`` truncation marker set by the adaptive-retry give-up sites.
1327
+
1328
+ The marker rides the item dicts up through every chunk merge, so it reaches
1329
+ ``save_semantic_cache`` on BOTH the incremental checkpoint path (llm.py) and
1330
+ the final authoritative save (cli.py) without either caller having to thread
1331
+ an extra argument — the final save would otherwise overwrite a checkpoint's
1332
+ ``partial`` flag with a clean-looking entry.
1333
+ """
1334
+ for bucket in ("nodes", "edges", "hyperedges"):
1335
+ for item in group.get(bucket, []):
1336
+ if isinstance(item, dict) and item.get("_partial"):
1337
+ return True
1338
+ return False
1339
+
1340
+
1341
+ def _semantic_source_matcher(
1342
+ root: Path,
1343
+ ) -> tuple[Callable[[str | Path], Path], Callable[[str | Path], str]]:
1344
+ """Shared path-identity machinery for the semantic-scope guards (#1757/#2926).
1345
+
1346
+ ``save_semantic_cache``'s write allowlist and ``scope_semantic_result``'s
1347
+ graph-feed filter must agree on exactly which ``source_file`` values are in
1348
+ scope, so both derive their matching from this one implementation rather
1349
+ than parallel copies that could drift apart.
1350
+
1351
+ Returns ``(source_identity, normalize_value)`` closed over ``root``:
1352
+
1353
+ - ``normalize_value(src)`` maps a raw ``source_file`` to its portable
1354
+ relative forward-slash form (#2197),
1355
+ - ``source_identity(value)`` maps any form (relative or absolute, against
1356
+ the walked or the resolved root) to the single canonical walked path
1357
+ that identities are compared against.
1358
+ """
1359
+ root_walked = _normalize_path(Path(os.path.abspath(root)))
1360
+ root_resolved = _normalize_path(Path(root).resolve())
1361
+
1362
+ def normalize_value(src: str | Path) -> str:
1363
+ norm = _normalize_source_file_value(src, root_walked)
1364
+ if Path(norm).is_absolute() and root_walked != root_resolved:
1365
+ norm = _normalize_source_file_value(src, root_resolved)
1366
+ return norm
1367
+
1368
+ def source_identity(value: str | Path) -> Path:
1369
+ path = Path(value)
1370
+ if not path.is_absolute():
1371
+ path = root_walked / path
1372
+ elif root_walked != root_resolved:
1373
+ normalized = _normalize_path(Path(os.path.abspath(path)))
1374
+ try:
1375
+ relative = normalized.relative_to(root_resolved)
1376
+ except ValueError:
1377
+ pass
1378
+ else:
1379
+ path = root_walked / relative
1380
+ return _normalize_path(Path(os.path.abspath(path)))
1381
+
1382
+ return source_identity, normalize_value
1383
+
1384
+
1385
+ def save_semantic_cache(
1386
+ nodes: list[dict],
1387
+ edges: list[dict],
1388
+ hyperedges: list[dict] | None = None,
1389
+ root: Path = Path("."),
1390
+ merge_existing: bool = False,
1391
+ allowed_source_files: Iterable[str | Path] | None = None,
1392
+ mode: str | None = None,
1393
+ prompt: "str | Path | None" = None,
1394
+ prompt_file: "str | Path | None" = None,
1395
+ partial_source_files: Iterable[str | Path] | None = None,
1396
+ cache_root: "Path | None" = None,
1397
+ ) -> int:
1398
+ """Save semantic extraction results to cache, keyed by source_file.
1399
+
1400
+ Groups nodes and edges by source_file, then saves one cache entry per file
1401
+ under cache/semantic/ (separate from AST entries in cache/ast/) to prevent
1402
+ hash-key collisions (#582).
1403
+
1404
+ ``mode`` selects the cache namespace, mirroring
1405
+ :func:`check_semantic_cache`: ``None`` (the default) writes
1406
+ ``cache/semantic/`` — byte-identical to the historical behavior for
1407
+ existing callers that omit it — while a non-None mode (e.g. ``"deep"``)
1408
+ writes ``cache/semantic-{mode}/`` so richer deep-mode results never
1409
+ overwrite standard-mode entries and vice versa (#1894).
1410
+
1411
+ When ``merge_existing`` is True, any already-cached entry for a file is
1412
+ unioned with the new results before saving instead of being overwritten.
1413
+ This lets callers checkpoint incrementally (e.g. once per chunk) without
1414
+ dropping a prior slice of a large file that was split across chunks.
1415
+
1416
+ When ``allowed_source_files`` is provided, only those files may be used as
1417
+ cache-write keys. Semantic nodes can legitimately mention another corpus
1418
+ file, but a model must not be able to replace that file's complete cache
1419
+ entry unless the file was part of the current extraction batch (#1757).
1420
+
1421
+ When ``partial_source_files`` is provided, entries for those files are
1422
+ stamped ``partial: True`` — the extraction was truncated, so the entry is
1423
+ incomplete and :func:`load_cached` must treat it as a miss. Partial-ness is
1424
+ ALSO detected intrinsically from a ``_partial`` marker on any grouped item,
1425
+ so the flag survives even when a caller (e.g. cli.py's final save) does not
1426
+ pass ``partial_source_files``.
1427
+
1428
+ ``prompt`` is the extraction prompt that produced these results — text, or
1429
+ a Path to the prompt file. It stamps entries into the p{fingerprint}/
1430
+ namespace so a later run under a different prompt re-extracts rather than
1431
+ replaying them (#1939). Pass the same prompt here as to
1432
+ :func:`check_semantic_cache`, or the write lands in a namespace the next
1433
+ read won't consult.
1434
+
1435
+ ``cache_root`` decouples *where* the cache directory is written from the
1436
+ source-key anchor ``root`` — mirroring the same split that :func:`load_cached`
1437
+ and :func:`save_cached` already expose (#1774). When given, cache files land
1438
+ under ``cache_root`` while ``source_file`` paths are still resolved and
1439
+ relativized against ``root``. When omitted, ``root`` is used for both
1440
+ purposes (unchanged behaviour for existing callers). This fixes checkpoints
1441
+ and the final save going to the corpus tree instead of ``--out`` (#1990,
1442
+ #1991).
1443
+
1444
+ Returns the number of files cached.
1445
+ """
1446
+ from collections import defaultdict
1447
+
1448
+ kind = "semantic" if mode is None else f"semantic-{mode}"
1449
+ source_path, _normalize_value = _semantic_source_matcher(root)
1450
+
1451
+ def _normalized(item: dict) -> dict:
1452
+ """Copy of ``item`` with a portable ``source_file`` (#2197).
1453
+
1454
+ Normalizing BEFORE grouping means both the group key and the persisted
1455
+ item carry the relative forward-slash form, so a fragment whose
1456
+ source_file arrived absolute (Windows detect() output) can never be
1457
+ cached verbatim. A shallow copy keeps the caller's dicts untouched —
1458
+ downstream steps may still rely on the original absolute shape (same
1459
+ reasoning as :func:`save_cached`'s on-disk deepcopy).
1460
+ """
1461
+ src = item.get("source_file")
1462
+ if not src:
1463
+ return item
1464
+ norm = _normalize_value(src)
1465
+ if norm != src:
1466
+ item = {**item, "source_file": norm}
1467
+ return item
1468
+
1469
+ by_file: dict[str, dict] = defaultdict(lambda: {"nodes": [], "edges": [], "hyperedges": []})
1470
+ for n in nodes:
1471
+ n = _normalized(n)
1472
+ src = n.get("source_file", "")
1473
+ if src:
1474
+ by_file[src]["nodes"].append(n)
1475
+ for e in edges:
1476
+ e = _normalized(e)
1477
+ src = e.get("source_file", "")
1478
+ if src:
1479
+ by_file[src]["edges"].append(e)
1480
+ for h in (hyperedges or []):
1481
+ h = _normalized(h)
1482
+ src = h.get("source_file", "")
1483
+ if src:
1484
+ by_file[src]["hyperedges"].append(h)
1485
+
1486
+ def resolved_source_path(value: str | Path) -> Path:
1487
+ path = source_path(value)
1488
+ try:
1489
+ return path.resolve()
1490
+ except (OSError, RuntimeError):
1491
+ # Keep the cache write best-effort for inaccessible paths or a
1492
+ # symlink loop emitted by an untrusted semantic result.
1493
+ return Path(os.path.abspath(path))
1494
+
1495
+ allowed_paths = None
1496
+ if allowed_source_files is not None:
1497
+ allowed_paths = {source_path(path) for path in allowed_source_files}
1498
+
1499
+ partial_paths = None
1500
+ if partial_source_files is not None:
1501
+ partial_paths = {source_path(path) for path in partial_source_files}
1502
+ # A chunk that truncated to an EMPTY parse contributes no grouped items,
1503
+ # so its file is absent from by_file and the write loop below would never
1504
+ # stamp it partial — leaving a prior clean slice looking complete (#1950
1505
+ # empty-parse gap). Seed an empty group for each named partial file that
1506
+ # isn't already present, so the loop merges its existing entry and stamps
1507
+ # it partial. Keyed by walked path (deduped against present groups).
1508
+ _present = {source_path(k) for k in by_file}
1509
+ for _pp in partial_paths:
1510
+ if _pp not in _present:
1511
+ by_file[str(_pp)] # defaultdict: create an empty {nodes,edges,hyperedges}
1512
+
1513
+ def group_skipped(fpath: str) -> bool:
1514
+ """Mirror the write-loop skip condition for one source_file group."""
1515
+ p = resolved_source_path(fpath)
1516
+ return not p.is_file() or (
1517
+ allowed_paths is not None and source_path(fpath) not in allowed_paths
1518
+ )
1519
+
1520
+ # Dangling-reference pruning (#1916). A node group is skipped by the write
1521
+ # loop below when its source_file is not a real file (ghost path) or is
1522
+ # out-of-scope per the #1757 guard — but an edge/hyperedge in an ALLOWED
1523
+ # group that references a node id from a skipped group used to be written
1524
+ # verbatim, so on replay (check_semantic_cache) it dangled forever (the
1525
+ # #1895 merged-result filter runs AFTER this checkpoint write and is
1526
+ # bypassed entirely on replay). Compute the node ids that will be skipped
1527
+ # and drop any to-be-written edge whose endpoint — or hyperedge whose
1528
+ # member (whole-hyperedge drop, mirroring #1895) — references one. Gated
1529
+ # on allowed_source_files so unscoped callers stay byte-identical.
1530
+ if allowed_paths is not None:
1531
+ skipped_ids: set = set()
1532
+ written_ids: set = set()
1533
+ for fpath, result in by_file.items():
1534
+ target = skipped_ids if group_skipped(fpath) else written_ids
1535
+ for n in result["nodes"]:
1536
+ nid = n.get("id")
1537
+ if nid is None:
1538
+ continue
1539
+ try:
1540
+ hash(nid)
1541
+ except TypeError:
1542
+ continue
1543
+ target.add(nid)
1544
+ # A duplicate-attribution node (defined in a skipped AND a written
1545
+ # group) still reaches the cache — don't over-prune references to it.
1546
+ skipped_ids -= written_ids
1547
+ if skipped_ids:
1548
+
1549
+ def edge_dangles(e: dict) -> bool:
1550
+ try:
1551
+ return e.get("source") in skipped_ids or e.get("target") in skipped_ids
1552
+ except TypeError:
1553
+ # Non-hashable endpoint from an untrusted result; leave it
1554
+ # to build-time validation rather than fail the save.
1555
+ return False
1556
+
1557
+ def hyperedge_dangles(h: dict) -> bool:
1558
+ try:
1559
+ return bool(skipped_ids & set(h.get("nodes") or []))
1560
+ except TypeError:
1561
+ return False
1562
+
1563
+ for fpath, result in by_file.items():
1564
+ if group_skipped(fpath):
1565
+ continue
1566
+ result["edges"] = [e for e in result["edges"] if not edge_dangles(e)]
1567
+ result["hyperedges"] = [
1568
+ h for h in result["hyperedges"] if not hyperedge_dangles(h)
1569
+ ]
1570
+
1571
+ saved = 0
1572
+ skipped_not_file = 0
1573
+ for fpath, result in by_file.items():
1574
+ cache_path = source_path(fpath)
1575
+ p = resolved_source_path(fpath)
1576
+ if p.is_file():
1577
+ if allowed_paths is not None and cache_path not in allowed_paths:
1578
+ warnings.warn(
1579
+ "semantic cache skipped out-of-scope source_file "
1580
+ f"{fpath!r}; the file was not dispatched for extraction",
1581
+ RuntimeWarning,
1582
+ stacklevel=2,
1583
+ )
1584
+ continue
1585
+ if merge_existing:
1586
+ # allow_legacy=False: merging a pre-fingerprint entry into this
1587
+ # write would fuse two prompt vintages inside a single entry and
1588
+ # then stamp the result as current-vintage — the exact mixing
1589
+ # #1939 is about, made unfixable because the entry now claims a
1590
+ # prompt that only produced half of it.
1591
+ # allow_partial=True: a file split into slices across chunks
1592
+ # accumulates here; if an earlier slice truncated, keep its nodes
1593
+ # in the union AND let the entry stay partial (the _partial
1594
+ # markers ride through, so is_partial below re-detects it) rather
1595
+ # than a later clean slice silently replacing it and promoting the
1596
+ # half-file to complete.
1597
+ prev = load_cached(cache_path, root, kind=kind, cache_root=cache_root,
1598
+ prompt=prompt, prompt_file=prompt_file,
1599
+ allow_legacy=False, allow_partial=True)
1600
+ _prev_partial = bool(prev.get("partial")) if prev else False
1601
+ if prev:
1602
+ result = {
1603
+ "nodes": (prev.get("nodes", []) or []) + result["nodes"],
1604
+ "edges": (prev.get("edges", []) or []) + result["edges"],
1605
+ "hyperedges": (prev.get("hyperedges", []) or []) + result["hyperedges"],
1606
+ }
1607
+ else:
1608
+ _prev_partial = False
1609
+ # A file is partial if the caller named it, any of its grouped items
1610
+ # carries the intrinsic ``_partial`` marker, OR the entry it merged
1611
+ # onto was already partial (an empty-parse truncation leaves a
1612
+ # ``partial: True`` entry with no item markers, so a later clean slice
1613
+ # merging over it must NOT silently promote the half-file to complete
1614
+ # — #1950). Copy so the caller's dict is never mutated. A genuine
1615
+ # complete re-extraction (merge_existing=False) overwrites the
1616
+ # content-hash key with a non-partial entry that then serves normally.
1617
+ is_partial = (
1618
+ (partial_paths is not None and cache_path in partial_paths)
1619
+ or _group_has_partial_marker(result)
1620
+ or _prev_partial
1621
+ )
1622
+ if is_partial:
1623
+ result = {**result, "partial": True}
1624
+ # A semantic extraction with zero nodes and zero hyperedges is not a valid
1625
+ # standalone extraction (#2927): edge-only or empty results must not be
1626
+ # cached, so that subsequent runs can re-dispatch and retry the file (#933/#1666).
1627
+ if not is_partial and not (result.get("nodes") or result.get("hyperedges")):
1628
+ continue
1629
+ save_cached(cache_path, result, root, kind=kind, cache_root=cache_root,
1630
+ prompt=prompt, prompt_file=prompt_file)
1631
+ saved += 1
1632
+ else:
1633
+ skipped_not_file += 1
1634
+ if skipped_not_file and skipped_not_file == len(by_file):
1635
+ warnings.warn(
1636
+ f"save_semantic_cache: all {skipped_not_file} source_file group(s) were "
1637
+ "skipped because their paths do not resolve to real files. This usually "
1638
+ "means ``root`` is anchored to the wrong directory (e.g. the --out "
1639
+ "directory instead of the corpus root). Pass the corpus directory as "
1640
+ "``root`` and the output directory as ``cache_root`` (#1991).",
1641
+ RuntimeWarning,
1642
+ stacklevel=2,
1643
+ )
1644
+ return saved
1645
+
1646
+
1647
+ def scope_semantic_result(
1648
+ result: dict,
1649
+ root: Path = Path("."),
1650
+ allowed_source_files: "Iterable[str | Path] | None" = None,
1651
+ ) -> tuple[set[str], int]:
1652
+ """Scope an extraction result in place to the files actually dispatched (#2926).
1653
+
1654
+ Graph-side mirror of the ``allowed_source_files`` write-guard in
1655
+ :func:`save_semantic_cache` (#1757). A model can attribute stray
1656
+ nodes/edges to a corpus file that was not part of the current extraction
1657
+ batch; :func:`build_merge` derives its replace-set from the source_files
1658
+ present in the new chunks, so such a stray fragment would REPLACE that
1659
+ file's entire prior contribution in graph.json — while its manifest entry
1660
+ still says unchanged, so no later incremental run re-dispatches it and the
1661
+ loss is permanent until a full rebuild.
1662
+
1663
+ Items whose ``source_file`` resolves outside ``allowed_source_files`` are
1664
+ dropped from ``result``'s ``nodes`` / ``edges`` / ``hyperedges`` lists
1665
+ (mutated in place); items without a ``source_file`` pass through. An edge
1666
+ or hyperedge that survives the scope filter but references a dropped node
1667
+ id is dropped too (#1916 mirror), unless that id is also defined by a kept
1668
+ node (duplicate attribution must not be over-pruned).
1669
+
1670
+ Path matching shares :func:`_semantic_source_matcher` with
1671
+ :func:`save_semantic_cache` (relative against ``root``, walked-path
1672
+ identity), so an item this function keeps can never still hit the save's
1673
+ out-of-scope skip, and vice versa.
1674
+
1675
+ Returns ``(dropped_source_files, dropped_item_count)`` for logging;
1676
+ ``dropped_source_files`` holds the normalized ``source_file`` strings of
1677
+ every group that had at least one item removed.
1678
+ """
1679
+ if allowed_source_files is None:
1680
+ return set(), 0
1681
+
1682
+ source_identity, normalize_value = _semantic_source_matcher(root)
1683
+
1684
+ def _item_identity(item: dict) -> tuple[str | None, Path | None]:
1685
+ """(display form, walked identity) of an item's source_file."""
1686
+ src = item.get("source_file")
1687
+ if not src:
1688
+ return None, None
1689
+ norm = normalize_value(src)
1690
+ return norm, source_identity(norm)
1691
+
1692
+ allowed_paths = {source_identity(str(path)) for path in allowed_source_files}
1693
+
1694
+ def _hashable(value) -> bool:
1695
+ try:
1696
+ hash(value)
1697
+ except TypeError:
1698
+ return False
1699
+ return True
1700
+
1701
+ dropped_files: set[str] = set()
1702
+ dropped_items = 0
1703
+ dropped_ids: set = set()
1704
+ kept_ids: set = set()
1705
+ for bucket in ("nodes", "edges", "hyperedges"):
1706
+ kept: list[dict] = []
1707
+ for item in result.get(bucket) or []:
1708
+ display, ident = _item_identity(item)
1709
+ if ident is not None and ident not in allowed_paths:
1710
+ dropped_files.add(display)
1711
+ dropped_items += 1
1712
+ if bucket == "nodes" and item.get("id") is not None:
1713
+ nid = item["id"]
1714
+ if _hashable(nid):
1715
+ dropped_ids.add(nid)
1716
+ continue
1717
+ if bucket == "nodes" and item.get("id") is not None and _hashable(item["id"]):
1718
+ kept_ids.add(item["id"])
1719
+ kept.append(item)
1720
+ result[bucket] = kept
1721
+
1722
+ # A duplicate-attribution node (defined in a dropped AND a kept group)
1723
+ # survives the filter — don't prune references to it.
1724
+ dropped_ids -= kept_ids
1725
+ if dropped_ids:
1726
+
1727
+ def edge_dangles(e: dict) -> bool:
1728
+ try:
1729
+ return e.get("source") in dropped_ids or e.get("target") in dropped_ids
1730
+ except TypeError:
1731
+ # Non-hashable endpoint from an untrusted result; leave it
1732
+ # to build-time validation rather than fail here.
1733
+ return False
1734
+
1735
+ def hyperedge_dangles(h: dict) -> bool:
1736
+ try:
1737
+ return bool(dropped_ids & set(h.get("nodes") or []))
1738
+ except TypeError:
1739
+ return False
1740
+
1741
+ result["edges"] = [e for e in result.get("edges") or [] if not edge_dangles(e)]
1742
+ result["hyperedges"] = [
1743
+ h for h in result.get("hyperedges") or [] if not hyperedge_dangles(h)
1744
+ ]
1745
+
1746
+ return dropped_files, dropped_items