graphitect 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (336) hide show
  1. graphify/__init__.py +30 -0
  2. graphify/__main__.py +757 -0
  3. graphify/_minhash.py +107 -0
  4. graphify/affected.py +318 -0
  5. graphify/always_on/agents-md.md +12 -0
  6. graphify/always_on/antigravity-rules.md +14 -0
  7. graphify/always_on/claude-md.md +9 -0
  8. graphify/always_on/gemini-md.md +9 -0
  9. graphify/always_on/kiro-steering.md +5 -0
  10. graphify/always_on/vscode-instructions.md +17 -0
  11. graphify/analyze.py +769 -0
  12. graphify/benchmark.py +152 -0
  13. graphify/build.py +2300 -0
  14. graphify/cache.py +1746 -0
  15. graphify/callflow_html.py +2051 -0
  16. graphify/cargo_introspect.py +109 -0
  17. graphify/cli.py +4745 -0
  18. graphify/cluster.py +409 -0
  19. graphify/command-kilo.md +15 -0
  20. graphify/cross_repo_calls.py +216 -0
  21. graphify/cross_repo_types.py +75 -0
  22. graphify/csharp_dispatch.py +154 -0
  23. graphify/dedup.py +1213 -0
  24. graphify/detect.py +2566 -0
  25. graphify/diagnostics.py +406 -0
  26. graphify/export.py +1349 -0
  27. graphify/exporters/__init__.py +1 -0
  28. graphify/exporters/base.py +14 -0
  29. graphify/exporters/graphdb.py +173 -0
  30. graphify/exporters/html.py +637 -0
  31. graphify/extract.py +7856 -0
  32. graphify/extractors/MIGRATION.md +107 -0
  33. graphify/extractors/__init__.py +66 -0
  34. graphify/extractors/apex.py +215 -0
  35. graphify/extractors/base.py +85 -0
  36. graphify/extractors/bash.py +579 -0
  37. graphify/extractors/blade.py +53 -0
  38. graphify/extractors/commonlisp.py +540 -0
  39. graphify/extractors/csharp.py +448 -0
  40. graphify/extractors/dart.py +564 -0
  41. graphify/extractors/dm.py +494 -0
  42. graphify/extractors/elixir.py +241 -0
  43. graphify/extractors/engine.py +6509 -0
  44. graphify/extractors/fortran.py +311 -0
  45. graphify/extractors/go.py +527 -0
  46. graphify/extractors/json_config.py +240 -0
  47. graphify/extractors/julia.py +289 -0
  48. graphify/extractors/markdown.py +408 -0
  49. graphify/extractors/models.py +131 -0
  50. graphify/extractors/objc.py +566 -0
  51. graphify/extractors/ocaml.py +289 -0
  52. graphify/extractors/pascal.py +688 -0
  53. graphify/extractors/pascal_forms.py +196 -0
  54. graphify/extractors/powershell.py +522 -0
  55. graphify/extractors/razor.py +192 -0
  56. graphify/extractors/resolution.py +3584 -0
  57. graphify/extractors/robot.py +296 -0
  58. graphify/extractors/rust.py +470 -0
  59. graphify/extractors/sln.py +92 -0
  60. graphify/extractors/sql.py +720 -0
  61. graphify/extractors/terraform.py +181 -0
  62. graphify/extractors/verilog.py +329 -0
  63. graphify/extractors/zig.py +181 -0
  64. graphify/file_slice.py +246 -0
  65. graphify/global_graph.py +194 -0
  66. graphify/google_workspace.py +237 -0
  67. graphify/hooks.py +933 -0
  68. graphify/ids.py +93 -0
  69. graphify/ingest.py +358 -0
  70. graphify/install.py +2366 -0
  71. graphify/llm.py +3544 -0
  72. graphify/manifest.py +4 -0
  73. graphify/manifest_ingest.py +311 -0
  74. graphify/mcp_ingest.py +386 -0
  75. graphify/multigraph_compat.py +212 -0
  76. graphify/pascal_resolution.py +129 -0
  77. graphify/paths.py +436 -0
  78. graphify/pg_introspect.py +165 -0
  79. graphify/prs.py +770 -0
  80. graphify/querylog.py +80 -0
  81. graphify/reflect.py +882 -0
  82. graphify/report.py +346 -0
  83. graphify/resolver_registry.py +85 -0
  84. graphify/ruby_resolution.py +242 -0
  85. graphify/scip_ingest.py +363 -0
  86. graphify/security.py +460 -0
  87. graphify/semantic_cleanup.py +336 -0
  88. graphify/serve.py +2608 -0
  89. graphify/skill-agents.md +710 -0
  90. graphify/skill-aider.md +1283 -0
  91. graphify/skill-amp.md +710 -0
  92. graphify/skill-claw.md +713 -0
  93. graphify/skill-codex.md +710 -0
  94. graphify/skill-copilot.md +713 -0
  95. graphify/skill-devin.md +1410 -0
  96. graphify/skill-droid.md +710 -0
  97. graphify/skill-kilo.md +722 -0
  98. graphify/skill-kiro.md +713 -0
  99. graphify/skill-opencode.md +705 -0
  100. graphify/skill-pi.md +713 -0
  101. graphify/skill-trae.md +711 -0
  102. graphify/skill-vscode.md +709 -0
  103. graphify/skill-windows.md +755 -0
  104. graphify/skill.md +713 -0
  105. graphify/skills/agents/references/add-watch.md +56 -0
  106. graphify/skills/agents/references/exports.md +87 -0
  107. graphify/skills/agents/references/extraction-spec.md +70 -0
  108. graphify/skills/agents/references/github-and-merge.md +46 -0
  109. graphify/skills/agents/references/hooks.md +33 -0
  110. graphify/skills/agents/references/query.md +311 -0
  111. graphify/skills/agents/references/transcribe.md +52 -0
  112. graphify/skills/agents/references/update.md +210 -0
  113. graphify/skills/amp/references/add-watch.md +56 -0
  114. graphify/skills/amp/references/exports.md +87 -0
  115. graphify/skills/amp/references/extraction-spec.md +70 -0
  116. graphify/skills/amp/references/github-and-merge.md +46 -0
  117. graphify/skills/amp/references/hooks.md +33 -0
  118. graphify/skills/amp/references/query.md +311 -0
  119. graphify/skills/amp/references/transcribe.md +52 -0
  120. graphify/skills/amp/references/update.md +210 -0
  121. graphify/skills/claude/references/add-watch.md +56 -0
  122. graphify/skills/claude/references/exports.md +87 -0
  123. graphify/skills/claude/references/extraction-spec.md +70 -0
  124. graphify/skills/claude/references/github-and-merge.md +46 -0
  125. graphify/skills/claude/references/hooks.md +33 -0
  126. graphify/skills/claude/references/query.md +311 -0
  127. graphify/skills/claude/references/transcribe.md +52 -0
  128. graphify/skills/claude/references/update.md +210 -0
  129. graphify/skills/claw/references/add-watch.md +56 -0
  130. graphify/skills/claw/references/exports.md +87 -0
  131. graphify/skills/claw/references/extraction-spec.md +31 -0
  132. graphify/skills/claw/references/github-and-merge.md +46 -0
  133. graphify/skills/claw/references/hooks.md +33 -0
  134. graphify/skills/claw/references/query.md +311 -0
  135. graphify/skills/claw/references/transcribe.md +52 -0
  136. graphify/skills/claw/references/update.md +210 -0
  137. graphify/skills/codex/references/add-watch.md +56 -0
  138. graphify/skills/codex/references/exports.md +87 -0
  139. graphify/skills/codex/references/extraction-spec.md +31 -0
  140. graphify/skills/codex/references/github-and-merge.md +46 -0
  141. graphify/skills/codex/references/hooks.md +33 -0
  142. graphify/skills/codex/references/query.md +311 -0
  143. graphify/skills/codex/references/transcribe.md +52 -0
  144. graphify/skills/codex/references/update.md +210 -0
  145. graphify/skills/copilot/references/add-watch.md +56 -0
  146. graphify/skills/copilot/references/exports.md +87 -0
  147. graphify/skills/copilot/references/extraction-spec.md +70 -0
  148. graphify/skills/copilot/references/github-and-merge.md +46 -0
  149. graphify/skills/copilot/references/hooks.md +33 -0
  150. graphify/skills/copilot/references/query.md +311 -0
  151. graphify/skills/copilot/references/transcribe.md +52 -0
  152. graphify/skills/copilot/references/update.md +210 -0
  153. graphify/skills/droid/references/add-watch.md +56 -0
  154. graphify/skills/droid/references/exports.md +87 -0
  155. graphify/skills/droid/references/extraction-spec.md +70 -0
  156. graphify/skills/droid/references/github-and-merge.md +46 -0
  157. graphify/skills/droid/references/hooks.md +33 -0
  158. graphify/skills/droid/references/query.md +311 -0
  159. graphify/skills/droid/references/transcribe.md +52 -0
  160. graphify/skills/droid/references/update.md +210 -0
  161. graphify/skills/kilo/references/add-watch.md +56 -0
  162. graphify/skills/kilo/references/exports.md +87 -0
  163. graphify/skills/kilo/references/extraction-spec.md +70 -0
  164. graphify/skills/kilo/references/github-and-merge.md +46 -0
  165. graphify/skills/kilo/references/hooks.md +33 -0
  166. graphify/skills/kilo/references/query.md +311 -0
  167. graphify/skills/kilo/references/transcribe.md +52 -0
  168. graphify/skills/kilo/references/update.md +210 -0
  169. graphify/skills/kiro/references/add-watch.md +56 -0
  170. graphify/skills/kiro/references/exports.md +87 -0
  171. graphify/skills/kiro/references/extraction-spec.md +31 -0
  172. graphify/skills/kiro/references/github-and-merge.md +46 -0
  173. graphify/skills/kiro/references/hooks.md +33 -0
  174. graphify/skills/kiro/references/query.md +311 -0
  175. graphify/skills/kiro/references/transcribe.md +52 -0
  176. graphify/skills/kiro/references/update.md +210 -0
  177. graphify/skills/opencode/references/add-watch.md +56 -0
  178. graphify/skills/opencode/references/exports.md +87 -0
  179. graphify/skills/opencode/references/extraction-spec.md +70 -0
  180. graphify/skills/opencode/references/github-and-merge.md +46 -0
  181. graphify/skills/opencode/references/hooks.md +33 -0
  182. graphify/skills/opencode/references/query.md +311 -0
  183. graphify/skills/opencode/references/transcribe.md +52 -0
  184. graphify/skills/opencode/references/update.md +210 -0
  185. graphify/skills/pi/references/add-watch.md +56 -0
  186. graphify/skills/pi/references/exports.md +87 -0
  187. graphify/skills/pi/references/extraction-spec.md +31 -0
  188. graphify/skills/pi/references/github-and-merge.md +46 -0
  189. graphify/skills/pi/references/hooks.md +33 -0
  190. graphify/skills/pi/references/query.md +311 -0
  191. graphify/skills/pi/references/transcribe.md +52 -0
  192. graphify/skills/pi/references/update.md +210 -0
  193. graphify/skills/trae/references/add-watch.md +56 -0
  194. graphify/skills/trae/references/exports.md +87 -0
  195. graphify/skills/trae/references/extraction-spec.md +70 -0
  196. graphify/skills/trae/references/github-and-merge.md +46 -0
  197. graphify/skills/trae/references/hooks.md +35 -0
  198. graphify/skills/trae/references/query.md +311 -0
  199. graphify/skills/trae/references/transcribe.md +52 -0
  200. graphify/skills/trae/references/update.md +210 -0
  201. graphify/skills/vscode/references/add-watch.md +56 -0
  202. graphify/skills/vscode/references/exports.md +87 -0
  203. graphify/skills/vscode/references/extraction-spec.md +70 -0
  204. graphify/skills/vscode/references/github-and-merge.md +46 -0
  205. graphify/skills/vscode/references/hooks.md +33 -0
  206. graphify/skills/vscode/references/query.md +311 -0
  207. graphify/skills/vscode/references/transcribe.md +52 -0
  208. graphify/skills/vscode/references/update.md +210 -0
  209. graphify/skills/windows/references/add-watch.md +56 -0
  210. graphify/skills/windows/references/exports.md +87 -0
  211. graphify/skills/windows/references/extraction-spec.md +70 -0
  212. graphify/skills/windows/references/github-and-merge.md +46 -0
  213. graphify/skills/windows/references/hooks.md +33 -0
  214. graphify/skills/windows/references/query.md +311 -0
  215. graphify/skills/windows/references/transcribe.md +52 -0
  216. graphify/skills/windows/references/update.md +210 -0
  217. graphify/symbol_resolution.py +556 -0
  218. graphify/transcribe.py +186 -0
  219. graphify/tree_html.py +603 -0
  220. graphify/validate.py +95 -0
  221. graphify/watch.py +2280 -0
  222. graphify/wiki.py +405 -0
  223. graphitect/__init__.py +28 -0
  224. graphitect/__main__.py +4 -0
  225. graphitect/_vendor/__init__.py +2 -0
  226. graphitect/_vendor/archify/LICENSE +22 -0
  227. graphitect/_vendor/archify/SKILL.md +137 -0
  228. graphitect/_vendor/archify/THIRD_PARTY_NOTICES.md +69 -0
  229. graphitect/_vendor/archify/assets/JetBrainsMono-OFL.txt +93 -0
  230. graphitect/_vendor/archify/assets/template.html +14935 -0
  231. graphitect/_vendor/archify/bin/archify.mjs +2091 -0
  232. graphitect/_vendor/archify/bin/open-artifact.mjs +86 -0
  233. graphitect/_vendor/archify/bin/preview.mjs +653 -0
  234. graphitect/_vendor/archify/bin/visual-check.mjs +829 -0
  235. graphitect/_vendor/archify/brand-marks/README.md +31 -0
  236. graphitect/_vendor/archify/brand-marks/catalog.json +131 -0
  237. graphitect/_vendor/archify/delta/architecture-delta.mjs +1221 -0
  238. graphitect/_vendor/archify/examples/agent-run.lifecycle.json +60 -0
  239. graphitect/_vendor/archify/examples/agent-tool-call.workflow.json +94 -0
  240. graphitect/_vendor/archify/examples/async-job-roundtrip.sequence.json +61 -0
  241. graphitect/_vendor/archify/examples/brand-aware-delivery.architecture.json +47 -0
  242. graphitect/_vendor/archify/examples/cache-miss-request.sequence.json +82 -0
  243. graphitect/_vendor/archify/examples/checkout-platform.base.architecture.json +31 -0
  244. graphitect/_vendor/archify/examples/checkout-platform.head.architecture.json +31 -0
  245. graphitect/_vendor/archify/examples/dataflow-product-analytics.html +15045 -0
  246. graphitect/_vendor/archify/examples/deployment-release.lifecycle.json +49 -0
  247. graphitect/_vendor/archify/examples/event-stream.dataflow.json +57 -0
  248. graphitect/_vendor/archify/examples/incident-response.workflow.json +64 -0
  249. graphitect/_vendor/archify/examples/lifecycle-agent-run.html +14980 -0
  250. graphitect/_vendor/archify/examples/product-analytics.dataflow.json +76 -0
  251. graphitect/_vendor/archify/examples/production-deployment.architecture.json +71 -0
  252. graphitect/_vendor/archify/examples/release-delivery.workflow.json +62 -0
  253. graphitect/_vendor/archify/examples/sequence-cache-miss-request.html +15060 -0
  254. graphitect/_vendor/archify/examples/web-app-rendered.html +15009 -0
  255. graphitect/_vendor/archify/examples/web-app.architecture.json +46 -0
  256. graphitect/_vendor/archify/examples/workflow-agent-tool-call-rendered.html +15051 -0
  257. graphitect/_vendor/archify/migrations/workflow-v2.mjs +279 -0
  258. graphitect/_vendor/archify/package-lock.json +149 -0
  259. graphitect/_vendor/archify/package.json +39 -0
  260. graphitect/_vendor/archify/recipes/scenarios.mjs +391 -0
  261. graphitect/_vendor/archify/references/authoring-contract.md +243 -0
  262. graphitect/_vendor/archify/references/brand-marks.md +65 -0
  263. graphitect/_vendor/archify/references/delivery-contract.md +120 -0
  264. graphitect/_vendor/archify/references/viewer-runtime.md +45 -0
  265. graphitect/_vendor/archify/renderers/architecture/grid.mjs +62 -0
  266. graphitect/_vendor/archify/renderers/architecture/render-architecture.mjs +1078 -0
  267. graphitect/_vendor/archify/renderers/dataflow/README.md +104 -0
  268. graphitect/_vendor/archify/renderers/dataflow/render-dataflow.mjs +483 -0
  269. graphitect/_vendor/archify/renderers/lifecycle/README.md +115 -0
  270. graphitect/_vendor/archify/renderers/lifecycle/render-lifecycle.mjs +561 -0
  271. graphitect/_vendor/archify/renderers/sequence/README.md +114 -0
  272. graphitect/_vendor/archify/renderers/sequence/render-sequence.mjs +464 -0
  273. graphitect/_vendor/archify/renderers/shared/brand-marks.mjs +563 -0
  274. graphitect/_vendor/archify/renderers/shared/cli.mjs +218 -0
  275. graphitect/_vendor/archify/renderers/shared/desktop-readability.mjs +26 -0
  276. graphitect/_vendor/archify/renderers/shared/diagnostics.mjs +127 -0
  277. graphitect/_vendor/archify/renderers/shared/engineering-profiles.mjs +157 -0
  278. graphitect/_vendor/archify/renderers/shared/generated-brand-marks.mjs +2003 -0
  279. graphitect/_vendor/archify/renderers/shared/generated-validators.mjs +13 -0
  280. graphitect/_vendor/archify/renderers/shared/geometry.mjs +1423 -0
  281. graphitect/_vendor/archify/renderers/shared/i18n.mjs +595 -0
  282. graphitect/_vendor/archify/renderers/shared/layout-report.mjs +40 -0
  283. graphitect/_vendor/archify/renderers/shared/legend.mjs +217 -0
  284. graphitect/_vendor/archify/renderers/shared/output-path.mjs +340 -0
  285. graphitect/_vendor/archify/renderers/shared/repository-evidence.mjs +238 -0
  286. graphitect/_vendor/archify/renderers/shared/repository-location.mjs +58 -0
  287. graphitect/_vendor/archify/renderers/shared/text-fit.mjs +49 -0
  288. graphitect/_vendor/archify/renderers/shared/utils.mjs +232 -0
  289. graphitect/_vendor/archify/renderers/shared/validator.mjs +86 -0
  290. graphitect/_vendor/archify/renderers/workflow/README.md +223 -0
  291. graphitect/_vendor/archify/renderers/workflow/render-workflow.mjs +35 -0
  292. graphitect/_vendor/archify/renderers/workflow/workflow-compiler.mjs +4400 -0
  293. graphitect/_vendor/archify/renderers/workflow/workflow-migration-geometry.mjs +144 -0
  294. graphitect/_vendor/archify/schemas/README.md +211 -0
  295. graphitect/_vendor/archify/schemas/architecture.schema.json +178 -0
  296. graphitect/_vendor/archify/schemas/common.schema.json +115 -0
  297. graphitect/_vendor/archify/schemas/dataflow.schema.json +243 -0
  298. graphitect/_vendor/archify/schemas/lifecycle.schema.json +266 -0
  299. graphitect/_vendor/archify/schemas/sequence.schema.json +223 -0
  300. graphitect/_vendor/archify/schemas/workflow.schema.json +428 -0
  301. graphitect/_vendor/archify/scripts/check-render-output.mjs +836 -0
  302. graphitect/_vendor/archify/scripts/check-update.mjs +1667 -0
  303. graphitect/_vendor/archify/scripts/generate-brand-marks.mjs +141 -0
  304. graphitect/_vendor/archify/scripts/generate-validators.mjs +66 -0
  305. graphitect/_vendor/archify/scripts/render-examples.mjs +26 -0
  306. graphitect/_vendor/archify/scripts/update-contract.mjs +182 -0
  307. graphitect/_vendor/archify/skill-release.json +10 -0
  308. graphitect/cli.py +981 -0
  309. graphitect/deliver/__init__.py +5 -0
  310. graphitect/deliver/archify_adapter.py +1877 -0
  311. graphitect/deliver/archify_ir.py +160 -0
  312. graphitect/deliver/archify_repair.py +135 -0
  313. graphitect/deliver/doc_compiler.py +916 -0
  314. graphitect/ground/__init__.py +5 -0
  315. graphitect/ground/describe_source.py +27 -0
  316. graphitect/ground/fullread_source.py +56 -0
  317. graphitect/ground/graphify_source.py +107 -0
  318. graphitect/models.py +118 -0
  319. graphitect/skill/SKILL.md +80 -0
  320. graphitect/skill/agents/openai.yaml +4 -0
  321. graphitect/synthesize/__init__.py +5 -0
  322. graphitect/synthesize/engine.py +281 -0
  323. graphitect/synthesize/llm_backend.py +331 -0
  324. graphitect/synthesize/questions.py +139 -0
  325. graphitect/synthesize/rubric.py +104 -0
  326. graphitect-0.2.0.dist-info/METADATA +284 -0
  327. graphitect-0.2.0.dist-info/RECORD +336 -0
  328. graphitect-0.2.0.dist-info/WHEEL +5 -0
  329. graphitect-0.2.0.dist-info/entry_points.txt +2 -0
  330. graphitect-0.2.0.dist-info/licenses/LICENSE +21 -0
  331. graphitect-0.2.0.dist-info/licenses/LICENSE-ARCHIFY-MIT +22 -0
  332. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-APACHE-2.0 +202 -0
  333. graphitect-0.2.0.dist-info/licenses/LICENSE-GRAPHIFY-MIT +21 -0
  334. graphitect-0.2.0.dist-info/licenses/NOTICE-ARCHIFY-THIRD-PARTY.md +69 -0
  335. graphitect-0.2.0.dist-info/licenses/NOTICE-GRAPHIFY +8 -0
  336. graphitect-0.2.0.dist-info/top_level.txt +2 -0
@@ -0,0 +1,336 @@
1
+ # Semantic fragment sanitizer — converts sentence-like rationale nodes into
2
+ # attributes on related nodes and removes invalid file_type values.
3
+ #
4
+ # Called from the skill merge path (see skill-devin.md) and from the in-process
5
+ # `graphify merge-chunks` command — both ingest untrusted agent-written chunk
6
+ # JSON, and validate_semantic_fragment() rejects malformed/oversized payloads and
7
+ # crafted node/edge IDs before they touch the graph. The primary build/load paths
8
+ # (build_from_json, load_graph_json) deliberately do NOT run this: they must keep
9
+ # loading valid pre-existing graphs whose AST node IDs predate the stricter
10
+ # semantic-ID charset.
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ import re
15
+ from pathlib import Path
16
+
17
+ from .build import _normalize_hyperedge_members
18
+
19
+ # Labels longer than this many characters, or containing >= this many words,
20
+ # are candidates for being sentence-like rationale text rather than entity names.
21
+ _RATIONALE_MIN_CHARS = 80
22
+ _RATIONALE_MIN_WORDS = 8
23
+
24
+ # Validation limits for untrusted semantic-fragment payloads. See
25
+ # validate_semantic_fragment(). Issue #825: returned-JSON normalization for
26
+ # OpenCode and Codex agents requires a Python enforcement boundary so a
27
+ # malicious or runaway agent response cannot exhaust memory or escape the
28
+ # graphify-out chunk directory via crafted node/edge IDs.
29
+ MAX_SEMANTIC_FRAGMENT_BYTES = 25 * 1024 * 1024
30
+ MAX_SEMANTIC_FRAGMENT_NODES = 10_000
31
+ MAX_SEMANTIC_FRAGMENT_EDGES = 100_000
32
+ MAX_SEMANTIC_FRAGMENT_HYPEREDGES = 10_000
33
+ MAX_SEMANTIC_HYPEREDGE_NODES = 256
34
+ MAX_SEMANTIC_ID_LENGTH = 256
35
+ VALID_SEMANTIC_FILE_TYPES = frozenset({"code", "document", "paper", "image", "rationale", "concept"})
36
+ # Unicode word characters are allowed: build's normalize_id preserves CJK /
37
+ # Cyrillic / accented-Latin identifiers, so an ASCII-only gate would reject valid
38
+ # ids that the loader accepts. The explicit path-separator / ".." check in
39
+ # _validate_semantic_id still blocks directory escape (#825); "/", "\\", spaces,
40
+ # "@", "#" etc. are not \w and remain rejected.
41
+ _SEMANTIC_ID_RE = re.compile(r"^[\w.:-]+$")
42
+
43
+
44
+ def validate_semantic_fragment(fragment: object) -> list[str]:
45
+ """Return validation errors for an untrusted semantic extraction fragment.
46
+
47
+ Empty list means valid. Called by skill merge code before
48
+ sanitize_semantic_fragment() so malformed or malicious agent JSON is
49
+ rejected before it touches the graph. Parameter is `object` (not `dict`)
50
+ because we may be handed arbitrary deserialized JSON — the first check
51
+ rejects anything that isn't a dict.
52
+ """
53
+ if not isinstance(fragment, dict):
54
+ return ["fragment must be a JSON object"]
55
+
56
+ errors: list[str] = []
57
+ try:
58
+ payload = json.dumps(fragment, ensure_ascii=False).encode("utf-8")
59
+ except (TypeError, ValueError) as exc:
60
+ return [f"fragment is not JSON-serializable: {exc}"]
61
+
62
+ if len(payload) > MAX_SEMANTIC_FRAGMENT_BYTES:
63
+ errors.append(f"payload is {len(payload)} bytes; max is {MAX_SEMANTIC_FRAGMENT_BYTES}")
64
+
65
+ nodes = fragment.get("nodes", [])
66
+ edges = fragment.get("edges", [])
67
+ if not isinstance(nodes, list):
68
+ errors.append("nodes must be a list")
69
+ nodes = []
70
+ elif len(nodes) > MAX_SEMANTIC_FRAGMENT_NODES:
71
+ errors.append(f"nodes has {len(nodes)} entries; max is {MAX_SEMANTIC_FRAGMENT_NODES}")
72
+
73
+ if not isinstance(edges, list):
74
+ errors.append("edges must be a list")
75
+ edges = []
76
+ elif len(edges) > MAX_SEMANTIC_FRAGMENT_EDGES:
77
+ errors.append(f"edges has {len(edges)} entries; max is {MAX_SEMANTIC_FRAGMENT_EDGES}")
78
+
79
+ for i, node in enumerate(nodes):
80
+ if not isinstance(node, dict):
81
+ errors.append(f"nodes[{i}] must be an object")
82
+ continue
83
+ _validate_semantic_id(errors, f"nodes[{i}].id", node.get("id"))
84
+ # file_type is intentionally NOT rejected here. It carries no security
85
+ # risk (it can't exhaust memory or escape a directory), and
86
+ # build_from_json already coerces every value via _FILE_TYPE_SYNONYMS
87
+ # (unknown -> "concept", #840). Rejecting a whole chunk over a synonym
88
+ # like "markdown"/"tool"/"framework" that the loader would happily map is
89
+ # pure data loss, so leave file_type normalization to build.
90
+
91
+ for i, edge in enumerate(edges):
92
+ if not isinstance(edge, dict):
93
+ errors.append(f"edges[{i}] must be an object")
94
+ continue
95
+ _validate_semantic_id(errors, f"edges[{i}].source", edge.get("source"))
96
+ _validate_semantic_id(errors, f"edges[{i}].target", edge.get("target"))
97
+
98
+ hyperedges = fragment.get("hyperedges", [])
99
+ if hyperedges is None:
100
+ hyperedges = []
101
+ if not isinstance(hyperedges, list):
102
+ errors.append("hyperedges must be a list")
103
+ else:
104
+ if len(hyperedges) > MAX_SEMANTIC_FRAGMENT_HYPEREDGES:
105
+ errors.append(
106
+ f"hyperedges has {len(hyperedges)} entries; "
107
+ f"max is {MAX_SEMANTIC_FRAGMENT_HYPEREDGES}"
108
+ )
109
+ for i, he in enumerate(hyperedges):
110
+ if not isinstance(he, dict):
111
+ errors.append(f"hyperedges[{i}] must be an object")
112
+ continue
113
+ # Fold alias member keys (members/node_ids) onto `nodes` (#1561) so
114
+ # an alias-keyed hyperedge isn't rejected here for "nodes must be a
115
+ # list" before it ever reaches build's normalization.
116
+ _normalize_hyperedge_members(he)
117
+ _validate_semantic_id(errors, f"hyperedges[{i}].id", he.get("id"))
118
+ he_nodes = he.get("nodes")
119
+ if not isinstance(he_nodes, list):
120
+ errors.append(f"hyperedges[{i}].nodes must be a list")
121
+ continue
122
+ if len(he_nodes) > MAX_SEMANTIC_HYPEREDGE_NODES:
123
+ errors.append(
124
+ f"hyperedges[{i}].nodes has {len(he_nodes)} entries; "
125
+ f"max is {MAX_SEMANTIC_HYPEREDGE_NODES}"
126
+ )
127
+ for j, ref in enumerate(he_nodes):
128
+ _validate_semantic_id(errors, f"hyperedges[{i}].nodes[{j}]", ref)
129
+
130
+ return errors
131
+
132
+
133
+ def load_validated_semantic_fragment(path: Path) -> tuple[dict | None, list[str]]:
134
+ """Load and validate a semantic chunk, rejecting oversize files before parsing.
135
+
136
+ The size guard runs against `path.stat().st_size` so an attacker-supplied
137
+ multi-gigabyte chunk file cannot blow up memory at `read_text()` time.
138
+ JSON decode errors are returned as validation errors rather than raised,
139
+ so callers can `continue` past bad chunks without a try/except.
140
+ """
141
+ try:
142
+ size = path.stat().st_size
143
+ except OSError as exc:
144
+ return None, [f"could not stat {path}: {exc}"]
145
+ if size > MAX_SEMANTIC_FRAGMENT_BYTES:
146
+ return None, [f"payload is {size} bytes; max is {MAX_SEMANTIC_FRAGMENT_BYTES}"]
147
+ try:
148
+ fragment = json.loads(path.read_text(encoding="utf-8"))
149
+ except json.JSONDecodeError as exc:
150
+ return None, [f"invalid JSON: {exc}"]
151
+ except OSError as exc:
152
+ return None, [f"could not read {path}: {exc}"]
153
+ errors = validate_semantic_fragment(fragment)
154
+ return (None, errors) if errors else (fragment, [])
155
+
156
+
157
+ def _validate_semantic_id(errors: list[str], field: str, value: object) -> None:
158
+ if not isinstance(value, str):
159
+ errors.append(f"{field} must be a string")
160
+ return
161
+ if not value:
162
+ errors.append(f"{field} must not be empty")
163
+ return
164
+ if len(value) > MAX_SEMANTIC_ID_LENGTH:
165
+ errors.append(f"{field} is {len(value)} chars; max is {MAX_SEMANTIC_ID_LENGTH}")
166
+ if "/" in value or "\\" in value or ".." in value:
167
+ errors.append(f"{field} must not contain path separators or '..'")
168
+ if not _SEMANTIC_ID_RE.fullmatch(value):
169
+ errors.append(f"{field} contains unsupported characters")
170
+
171
+
172
+ def sanitize_semantic_fragment(fragment: dict) -> dict:
173
+ """Clean up a semantic extraction fragment in-place.
174
+
175
+ Operations:
176
+ 1. Removes nodes with ``file_type: "rationale"`` or ``file_type: "concept"``
177
+ that were emitted by an LLM (these are not valid semantic entity types).
178
+ 2. Detects nodes whose label reads like a sentence / rationale paragraph
179
+ AND that participate in a ``rationale_for`` edge, then converts the
180
+ label into a ``rationale`` attribute on the target node and removes
181
+ the source-node + its edges. The ``rationale_for`` edge signal applies
182
+ regardless of the source node's ``file_type`` — sentence-like nodes
183
+ with allowed types (``document``, ``code``) are still cleaned up when
184
+ they're explicitly marked as rationale.
185
+ 3. Strips nodes whose only distinguishing field is the label itself
186
+ (empty id — likely LLM hallucination).
187
+ 4. Filters hyperedges so they cannot reference removed or unknown node
188
+ IDs after the cleanup passes above. A hyperedge with fewer than two
189
+ surviving members is dropped.
190
+
191
+ Returns the same dict for convenience.
192
+ """
193
+ _invalid_ft = frozenset({"rationale", "concept"})
194
+
195
+ nodes: list[dict] = fragment.get("nodes", [])
196
+ edges: list[dict] = fragment.get("edges", [])
197
+ hyperedges: list[dict] = fragment.get("hyperedges", []) or []
198
+
199
+ # ---- build lookup maps --------------------------------------------------
200
+ node_by_id: dict[str, dict] = {}
201
+ for n in nodes:
202
+ nid = n.get("id", "")
203
+ if nid:
204
+ node_by_id[nid] = n
205
+
206
+ # Pre-collect node IDs that source a `rationale_for` edge — these are
207
+ # candidates for sentence-like cleanup even when file_type is allowed.
208
+ rationale_for_sources: set[str] = set()
209
+ for e in edges:
210
+ if e.get("relation") == "rationale_for":
211
+ src = e.get("source", "")
212
+ if src:
213
+ rationale_for_sources.add(src)
214
+
215
+ # ---- pass 1: identify nodes to remove + rationale candidates -----------
216
+ rationale_candidates: list[dict] = []
217
+ remove_ids: set[str] = set()
218
+ keep_nodes: list[dict] = []
219
+ for n in nodes:
220
+ nid = n.get("id", "")
221
+ if not nid:
222
+ # Node without an id cannot be referenced — discard.
223
+ continue
224
+ ft = n.get("file_type", "")
225
+ label = n.get("label", "")
226
+ if ft in _invalid_ft:
227
+ # Explicitly-invalid file_type ("rationale" or "concept"): if
228
+ # the label looks like a sentence we may convert to attribute.
229
+ if _is_sentence_like_rationale_label(label):
230
+ rationale_candidates.append(n)
231
+ remove_ids.add(nid)
232
+ continue
233
+ if nid in rationale_for_sources and _is_sentence_like_rationale_label(label):
234
+ # Allowed file_type, but the node sources a `rationale_for` edge
235
+ # AND its label is sentence-like prose. Treat it as rationale
236
+ # cleanup material rather than a real graph entity.
237
+ rationale_candidates.append(n)
238
+ remove_ids.add(nid)
239
+ continue
240
+ keep_nodes.append(n)
241
+
242
+ # ---- pass 2: convert sentence-nodes → rationale attributes --------------
243
+ # Only `rationale_for` edges propagate the rationale text. Other outgoing
244
+ # edges (e.g. references, conceptually_related_to) are NOT used as
245
+ # attribute-propagation paths — that would corrupt unrelated nodes by
246
+ # attaching rationale meant for a different target.
247
+ rationale_attrs: dict[str, list[str]] = {}
248
+ for rn in rationale_candidates:
249
+ rn_id = rn.get("id", "")
250
+ text = rn.get("label", "").strip()
251
+ for e in edges:
252
+ if e.get("relation") != "rationale_for":
253
+ continue
254
+ if e.get("source") != rn_id:
255
+ continue
256
+ target_id = e.get("target")
257
+ if target_id not in node_by_id or target_id in remove_ids:
258
+ continue
259
+ rationale_attrs.setdefault(target_id, []).append(text)
260
+
261
+ for target_id, texts in rationale_attrs.items():
262
+ if target_id in node_by_id and target_id not in remove_ids:
263
+ _append_rationale_attr(node_by_id[target_id], texts)
264
+
265
+ # ---- pass 3: strip edges referencing removed nodes ----------------------
266
+ keep_edges: list[dict] = []
267
+ for e in edges:
268
+ src = e.get("source", "")
269
+ tgt = e.get("target", "")
270
+ if src in remove_ids or tgt in remove_ids:
271
+ continue
272
+ keep_edges.append(e)
273
+
274
+ # ---- pass 4: filter hyperedges to surviving node IDs --------------------
275
+ surviving_ids: set[str] = {n.get("id", "") for n in keep_nodes}
276
+ surviving_ids.discard("")
277
+ keep_hyperedges: list[dict] = []
278
+ for he in hyperedges:
279
+ if not isinstance(he, dict):
280
+ continue
281
+ # Fold alias member keys (members/node_ids) onto `nodes` (#1561) so an
282
+ # alias-keyed hyperedge isn't silently dropped below for a missing
283
+ # `nodes` list before build can canonicalize it.
284
+ _normalize_hyperedge_members(he)
285
+ he_nodes = he.get("nodes")
286
+ if not isinstance(he_nodes, list):
287
+ continue
288
+ filtered = [ref for ref in he_nodes if isinstance(ref, str) and ref in surviving_ids]
289
+ if len(filtered) < 2:
290
+ # A hyperedge needs at least two surviving members to be meaningful.
291
+ continue
292
+ if len(filtered) != len(he_nodes):
293
+ he = dict(he)
294
+ he["nodes"] = filtered
295
+ keep_hyperedges.append(he)
296
+
297
+ fragment["nodes"] = keep_nodes
298
+ fragment["edges"] = keep_edges
299
+ fragment["hyperedges"] = keep_hyperedges
300
+ return fragment
301
+
302
+
303
+ def _is_sentence_like_rationale_label(label: str) -> bool:
304
+ """Return True if *label* looks like prose / rationale text rather than an
305
+ entity or concept name.
306
+
307
+ Heuristics (no false positives on short-concept-edge-cases):
308
+ - Longer than *_RATIONALE_MIN_CHARS* chars, OR
309
+ - At least *_RATIONALE_MIN_WORDS* whitespace-delimited tokens, AND
310
+ - Contains at least one sentence-ending punctuation mark (``. ! ?``) or a
311
+ colon (common in "Decision: ..." rationales).
312
+ """
313
+ if not label:
314
+ return False
315
+ label = label.strip()
316
+ if len(label) < _RATIONALE_MIN_CHARS:
317
+ word_count = len(label.split())
318
+ if word_count < _RATIONALE_MIN_WORDS:
319
+ return False
320
+ # Must look like actual prose: has sentence-ending punctuation or a colon.
321
+ return bool(re.search(r"[.!?:]", label))
322
+
323
+
324
+ def _append_rationale_attr(node: dict, texts: list[str]) -> None:
325
+ """Append one or more rationale strings to *node*'s ``rationale`` attribute.
326
+
327
+ If the attribute already exists the new texts are appended with a
328
+ double-newline separator so downstream consumers can distinguish distinct
329
+ rationale fragments.
330
+ """
331
+ existing = node.get("rationale", "")
332
+ new_text = "\n\n".join(texts).strip()
333
+ if existing:
334
+ node["rationale"] = existing + "\n\n" + new_text
335
+ else:
336
+ node["rationale"] = new_text