blun-king-cli 9.1.587 → 9.1.588

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (436) hide show
  1. package/CHANGELOG.md +11 -185
  2. package/LIESMICH.txt +51 -13
  3. package/README.md +44 -47
  4. package/agent-spine-plugin/.codex-plugin/plugin.json +16 -4
  5. package/agent-spine-plugin/CHANGELOG.md +37 -5
  6. package/agent-spine-plugin/README.md +3 -3
  7. package/agent-spine-plugin/blun.plugin.json +45 -10
  8. package/agent-spine-plugin/docs/artifact-evaluation.md +93 -0
  9. package/agent-spine-plugin/docs/host-integration.md +42 -27
  10. package/agent-spine-plugin/docs/preflight-recall.md +4 -2
  11. package/agent-spine-plugin/docs/session-timeline.md +97 -236
  12. package/agent-spine-plugin/docs/world-model.md +25 -0
  13. package/agent-spine-plugin/hooks/codex.json +1 -1
  14. package/agent-spine-plugin/hooks/hooks.json +1 -1
  15. package/agent-spine-plugin/package.json +1 -3
  16. package/agent-spine-plugin/scripts/check-hosts.js +3 -3
  17. package/agent-spine-plugin/scripts/release-check.js +10 -5
  18. package/agent-spine-plugin/scripts/run-checks.js +4 -1
  19. package/agent-spine-plugin/scripts/run-tests-hermetic.js +32 -6
  20. package/agent-spine-plugin/src/cli-learning.js +15 -0
  21. package/agent-spine-plugin/src/cli.js +2 -0
  22. package/agent-spine-plugin/src/hook.js +32 -32
  23. package/agent-spine-plugin/src/lib/action-lesson-recall.js +73 -8
  24. package/agent-spine-plugin/src/lib/briefing.js +146 -36
  25. package/agent-spine-plugin/src/lib/channel-continuity.js +19 -0
  26. package/agent-spine-plugin/src/lib/delivery-agent-usage.js +14 -7
  27. package/agent-spine-plugin/src/lib/gateway-group-response.js +128 -0
  28. package/agent-spine-plugin/src/lib/gateway-runs.js +24 -15
  29. package/agent-spine-plugin/src/lib/hook-briefing-use.js +13 -3
  30. package/agent-spine-plugin/src/lib/hook-context.js +16 -3
  31. package/agent-spine-plugin/src/lib/hook-output.js +129 -5
  32. package/agent-spine-plugin/src/lib/hook-timeline.js +5 -3
  33. package/agent-spine-plugin/src/lib/indexed-memory.js +2 -2
  34. package/agent-spine-plugin/src/lib/learning-artifact-evaluator.js +114 -0
  35. package/agent-spine-plugin/src/lib/learning-context.js +11 -4
  36. package/agent-spine-plugin/src/lib/learning-measurements.js +2 -2
  37. package/agent-spine-plugin/src/lib/mcp-runtime.js +89 -3
  38. package/agent-spine-plugin/src/lib/mcp-source-context.js +12 -2
  39. package/agent-spine-plugin/src/lib/mcp-timeline-tools.js +91 -8
  40. package/agent-spine-plugin/src/lib/mcp-world-tools.js +2 -2
  41. package/agent-spine-plugin/src/lib/owned-file-lock.js +20 -1
  42. package/agent-spine-plugin/src/lib/persona-runtime.js +2 -2
  43. package/agent-spine-plugin/src/lib/preflight-delivery-id.js +27 -0
  44. package/agent-spine-plugin/src/lib/preflight.js +4 -4
  45. package/agent-spine-plugin/src/lib/session-timeline-codex.js +15 -0
  46. package/agent-spine-plugin/src/lib/session-timeline-contract.js +12 -4
  47. package/agent-spine-plugin/src/lib/session-timeline-event-extract.js +36 -7
  48. package/agent-spine-plugin/src/lib/session-timeline-host-origin.js +13 -10
  49. package/agent-spine-plugin/src/lib/session-timeline-invocation.js +1 -1
  50. package/agent-spine-plugin/src/lib/session-timeline-king.js +14 -0
  51. package/agent-spine-plugin/src/lib/session-timeline-prior.js +18 -12
  52. package/agent-spine-plugin/src/lib/session-timeline-provider.js +5 -0
  53. package/agent-spine-plugin/src/lib/session-timeline-query.js +2 -0
  54. package/agent-spine-plugin/src/lib/session-timeline-results.js +35 -10
  55. package/agent-spine-plugin/src/lib/session-timeline-source-open.js +30 -0
  56. package/agent-spine-plugin/src/lib/session-timeline.js +122 -75
  57. package/agent-spine-plugin/src/lib/source-roots.js +3 -2
  58. package/agent-spine-plugin/src/lib/task-knowledge-context.js +22 -1
  59. package/agent-spine-plugin/src/lib/timeline-continuation-update.js +100 -0
  60. package/agent-spine-plugin/src/lib/timeline-tool-guard.js +30 -7
  61. package/agent-spine-plugin/src/lib/timeline-user-feedback.js +217 -0
  62. package/agent-spine-plugin/src/lib/timeline-world-capture.js +233 -0
  63. package/agent-spine-plugin/src/lib/world-knowledge.js +59 -2
  64. package/agent-spine-plugin/src/lib/world-model.js +64 -9
  65. package/agent-spine-plugin/src/worker.js +13 -1
  66. package/bin/blun.js +43 -28
  67. package/bin/core-bootstrap.js +5 -4
  68. package/bin/king.js +43 -28
  69. package/bin/launcher-mode.js +1 -10
  70. package/bin/launcher-runtime.js +128 -295
  71. package/bin/managed-node.js +0 -0
  72. package/bin/managed-plugin-selection.cjs +0 -1
  73. package/bin/native-module-repair.js +0 -0
  74. package/bin/node-runtime.js +0 -0
  75. package/bin/node-version.js +0 -0
  76. package/bin/plugin-bootstrap.js +56 -120
  77. package/bin/private-paths.js +11 -34
  78. package/bin/standard-tools-bootstrap.js +34 -114
  79. package/bin/turn-thinking-policy.cjs +3 -11
  80. package/bin/update-copy.js +200 -0
  81. package/bin/update-lease.js +0 -0
  82. package/bin/update-notice.js +136 -289
  83. package/bin/verify-agent-behavior.cjs +122 -0
  84. package/bin/verify-agent-components.cjs +104 -0
  85. package/bin/verify-bundled-agent-sources.cjs +57 -0
  86. package/blun.mjs +143076 -135288
  87. package/bundled-agent-sources.json +701 -0
  88. package/package.json +12 -15
  89. package/standard-skills/translate-native/README.md +1293 -0
  90. package/standard-skills/translate-native/SKILL.md +172 -22
  91. package/standard-skills/translate-native/VERSION +1 -1
  92. package/standard-skills/translate-native/agents/openai.yaml +18 -0
  93. package/standard-skills/translate-native/assets/icon.svg +8 -0
  94. package/standard-skills/translate-native/docs/BLUN_CODE_INTEGRATION.md +76 -0
  95. package/standard-skills/translate-native/docs/PREMORTEM.md +489 -0
  96. package/standard-skills/translate-native/docs/WEBSITE_LOCALIZATION.md +2035 -0
  97. package/standard-skills/translate-native/docs/WEBSITE_LOCALIZATION_API.md +1302 -0
  98. package/standard-skills/translate-native/docs/WEBSITE_LOCALIZATION_EVIDENCE_HTTP.md +136 -0
  99. package/standard-skills/translate-native/docs/WEBSITE_LOCALIZATION_HEALTH_HTTP.md +130 -0
  100. package/standard-skills/translate-native/docs/WEBSITE_LOCALIZATION_HTTP_PROVIDER.md +175 -0
  101. package/standard-skills/translate-native/docs/WEBSITE_LOCALIZATION_RECEIPT_VERIFIER_HTTP.md +86 -0
  102. package/standard-skills/translate-native/integrations/AGENT_RULES.md +32 -0
  103. package/standard-skills/translate-native/integrations/adapters/blun-code-language-guard.js +514 -0
  104. package/standard-skills/translate-native/integrations/adapters/node-language-guard.js +230 -0
  105. package/standard-skills/translate-native/integrations/audit_log.py +327 -0
  106. package/standard-skills/translate-native/integrations/claude_language_hook.js +1536 -0
  107. package/standard-skills/translate-native/integrations/commercial_localization_profile.py +42 -0
  108. package/standard-skills/translate-native/integrations/delivery-policy.example.json +28 -0
  109. package/standard-skills/translate-native/integrations/enforced_delivery.py +543 -0
  110. package/standard-skills/translate-native/integrations/guard_service.py +435 -0
  111. package/standard-skills/translate-native/integrations/language_gateway.py +67 -0
  112. package/standard-skills/translate-native/integrations/mcp_auth_headers.py +198 -0
  113. package/standard-skills/translate-native/integrations/mcp_http_gateway.py +429 -0
  114. package/standard-skills/translate-native/integrations/non_language_html_entities.js +1485 -0
  115. package/standard-skills/translate-native/integrations/pre_output_guard.py +65 -0
  116. package/standard-skills/translate-native/integrations/task_router.py +101 -0
  117. package/standard-skills/translate-native/integrations/website_localization.py +401 -0
  118. package/standard-skills/translate-native/integrations/website_localization_api.py +581 -0
  119. package/standard-skills/translate-native/integrations/website_localization_benchmark.py +1885 -0
  120. package/standard-skills/translate-native/integrations/website_localization_benchmark_campaign.py +1772 -0
  121. package/standard-skills/translate-native/integrations/website_localization_benchmark_candidate.py +506 -0
  122. package/standard-skills/translate-native/integrations/website_localization_benchmark_http.py +400 -0
  123. package/standard-skills/translate-native/integrations/website_localization_benchmark_review_store.py +781 -0
  124. package/standard-skills/translate-native/integrations/website_localization_benchmark_reviewer_http.py +500 -0
  125. package/standard-skills/translate-native/integrations/website_localization_benchmark_runtime.py +1107 -0
  126. package/standard-skills/translate-native/integrations/website_localization_benchmark_suite.py +463 -0
  127. package/standard-skills/translate-native/integrations/website_localization_cms.py +2835 -0
  128. package/standard-skills/translate-native/integrations/website_localization_cms_client.py +875 -0
  129. package/standard-skills/translate-native/integrations/website_localization_cms_dispatch.py +805 -0
  130. package/standard-skills/translate-native/integrations/website_localization_cms_http.py +588 -0
  131. package/standard-skills/translate-native/integrations/website_localization_cms_lifecycle_monitor.py +991 -0
  132. package/standard-skills/translate-native/integrations/website_localization_cms_receiver.py +1441 -0
  133. package/standard-skills/translate-native/integrations/website_localization_cms_receiver_runtime.py +414 -0
  134. package/standard-skills/translate-native/integrations/website_localization_cms_receiver_store.py +1073 -0
  135. package/standard-skills/translate-native/integrations/website_localization_cms_removal_dispatch.py +865 -0
  136. package/standard-skills/translate-native/integrations/website_localization_cms_source_client.py +583 -0
  137. package/standard-skills/translate-native/integrations/website_localization_cms_source_delivery.py +964 -0
  138. package/standard-skills/translate-native/integrations/website_localization_cms_source_delivery_runtime.py +665 -0
  139. package/standard-skills/translate-native/integrations/website_localization_cms_source_http.py +1153 -0
  140. package/standard-skills/translate-native/integrations/website_localization_cms_source_runtime.py +675 -0
  141. package/standard-skills/translate-native/integrations/website_localization_cms_source_service.py +1125 -0
  142. package/standard-skills/translate-native/integrations/website_localization_cms_terminal_notification.py +674 -0
  143. package/standard-skills/translate-native/integrations/website_localization_cms_terminal_notification_http.py +444 -0
  144. package/standard-skills/translate-native/integrations/website_localization_cms_terminal_notification_receiver.py +1469 -0
  145. package/standard-skills/translate-native/integrations/website_localization_cms_terminal_notification_receiver_runtime.py +1142 -0
  146. package/standard-skills/translate-native/integrations/website_localization_cms_terminal_processing_monitor.py +634 -0
  147. package/standard-skills/translate-native/integrations/website_localization_cms_terminal_receiver_client.py +804 -0
  148. package/standard-skills/translate-native/integrations/website_localization_deepl_baseline.py +922 -0
  149. package/standard-skills/translate-native/integrations/website_localization_evidence_http.py +482 -0
  150. package/standard-skills/translate-native/integrations/website_localization_health.py +1541 -0
  151. package/standard-skills/translate-native/integrations/website_localization_health_http.py +372 -0
  152. package/standard-skills/translate-native/integrations/website_localization_http_provider.py +297 -0
  153. package/standard-skills/translate-native/integrations/website_localization_native_reference_http.py +479 -0
  154. package/standard-skills/translate-native/integrations/website_localization_native_reference_intake.py +363 -0
  155. package/standard-skills/translate-native/integrations/website_localization_native_reference_queue.py +1449 -0
  156. package/standard-skills/translate-native/integrations/website_localization_native_reference_store.py +420 -0
  157. package/standard-skills/translate-native/integrations/website_localization_quality_profiles.py +235 -0
  158. package/standard-skills/translate-native/integrations/website_localization_queue.py +671 -0
  159. package/standard-skills/translate-native/integrations/website_localization_receipt_verifier_http.py +516 -0
  160. package/standard-skills/translate-native/integrations/website_localization_release.py +928 -0
  161. package/standard-skills/translate-native/integrations/website_localization_release_coordinator.py +1008 -0
  162. package/standard-skills/translate-native/integrations/website_localization_runner.py +276 -0
  163. package/standard-skills/translate-native/integrations/website_localization_runtime.py +862 -0
  164. package/standard-skills/translate-native/integrations/website_localization_service.py +350 -0
  165. package/standard-skills/translate-native/integrations/website_localization_supervisor.py +511 -0
  166. package/standard-skills/translate-native/integrations/website_localization_worker.py +663 -0
  167. package/standard-skills/translate-native/provenance.json +3 -4
  168. package/standard-skills/translate-native/references/commercial-localization.md +177 -0
  169. package/standard-skills/translate-native/scripts/blun_language_guard.py +7 -1
  170. package/standard-skills/translate-native/scripts/check_commercial_review.py +80 -0
  171. package/standard-skills/translate-native/scripts/commercial_localization_profile.py +333 -0
  172. package/standard-tools/language-guard/LICENSE +21 -0
  173. package/standard-tools/language-guard/VERSION +1 -0
  174. package/standard-tools/language-guard/blun_language_guard.py +7 -1
  175. package/standard-tools/language-guard/check_commercial_review.py +80 -0
  176. package/standard-tools/language-guard/commercial_localization_profile.py +333 -0
  177. package/standard-tools/language-guard/language_gateway.py +62 -0
  178. package/standard-tools/language-guard/pre_output_guard.py +64 -0
  179. package/standard-tools/language-guard/provenance.json +4 -11
  180. package/standard-tools/manifest.json +34 -11
  181. package/telegram-plugin/commands/access.md +2 -10
  182. package/telegram-plugin/dist/bridge.mjs +64041 -687
  183. package/telegram-plugin/dist/mcp-server.mjs +72810 -9027
  184. package/telegram-plugin/dist/noise.mjs +28 -63511
  185. package/agent-spine-plugin/CONTRIBUTING.md +0 -52
  186. package/agent-spine-plugin/SECURITY.md +0 -47
  187. package/agent-spine-plugin/docs/assignment-continuation.md +0 -48
  188. package/agent-spine-plugin/docs/releasing.md +0 -85
  189. package/agent-spine-plugin/docs/structured-completion.md +0 -67
  190. package/bin/abort-listener-policy.cjs +0 -43
  191. package/bin/active-steer-priority-policy.cjs +0 -24
  192. package/bin/agent-api-http-adapter.mjs +0 -446
  193. package/bin/agent-api-private-http-server.mjs +0 -288
  194. package/bin/agent-api-runtime.mjs +0 -252
  195. package/bin/agent-api-service-environment.mjs +0 -236
  196. package/bin/agent-api-service-host.mjs +0 -209
  197. package/bin/agent-api-service-process.mjs +0 -171
  198. package/bin/agent-api-session-registry.mjs +0 -428
  199. package/bin/agent-api-tool-broker.cjs +0 -248
  200. package/bin/agent-api-turn-controller.mjs +0 -461
  201. package/bin/agent-api-usage-journal.cjs +0 -259
  202. package/bin/agent-resume-snapshot.cjs +0 -241
  203. package/bin/agentspine-king-goal-inbox.mjs +0 -111
  204. package/bin/agentspine-king-goal-intake.mjs +0 -106
  205. package/bin/approval-rejection-stop.cjs +0 -15
  206. package/bin/assistant-message-offload-policy.cjs +0 -284
  207. package/bin/baseline-skill-performance-policy.cjs +0 -39
  208. package/bin/bash-search-scope-policy.cjs +0 -49
  209. package/bin/codebase-search-runtime.cjs +0 -23
  210. package/bin/cognitive-action-checkpoint.cjs +0 -1104
  211. package/bin/cognitive-attention-delivery.cjs +0 -76
  212. package/bin/cognitive-attention-policy.cjs +0 -143
  213. package/bin/cognitive-attention-runtime.cjs +0 -91
  214. package/bin/cognitive-context-projection.cjs +0 -73
  215. package/bin/cognitive-cross-portal-acceptance.cjs +0 -443
  216. package/bin/cognitive-effective-view.cjs +0 -77
  217. package/bin/cognitive-focus-projection.cjs +0 -206
  218. package/bin/cognitive-focus-scope.cjs +0 -37
  219. package/bin/cognitive-goal-autostart-policy.cjs +0 -72
  220. package/bin/cognitive-goal-time-trigger-controller.cjs +0 -146
  221. package/bin/cognitive-memory-adapter.cjs +0 -282
  222. package/bin/cognitive-memory-command.cjs +0 -293
  223. package/bin/cognitive-memory-provider.cjs +0 -92
  224. package/bin/cognitive-salience-policy.cjs +0 -159
  225. package/bin/cognitive-state-store.cjs +0 -508
  226. package/bin/cognitive-turn-lifecycle.cjs +0 -624
  227. package/bin/cognitive-work-focus.cjs +0 -180
  228. package/bin/compaction-history-archive.cjs +0 -166
  229. package/bin/compaction-history-startup.cjs +0 -50
  230. package/bin/compaction-model-policy.cjs +0 -31
  231. package/bin/compaction-stage-policy.cjs +0 -21
  232. package/bin/compaction-transaction-policy.cjs +0 -122
  233. package/bin/config-write-dedup-policy.cjs +0 -27
  234. package/bin/context-budget-ledger.cjs +0 -31
  235. package/bin/context-doctor-policy.cjs +0 -70
  236. package/bin/context-insight-policy.cjs +0 -36
  237. package/bin/context-performance-policy.cjs +0 -19
  238. package/bin/context-pressure-policy.cjs +0 -20
  239. package/bin/cron-run-output.cjs +0 -45
  240. package/bin/cron-run-store.cjs +0 -145
  241. package/bin/curiosity-scout-policy.cjs +0 -49
  242. package/bin/default-model-output-budget-policy.cjs +0 -28
  243. package/bin/durable-task-resume-policy.cjs +0 -130
  244. package/bin/durable-task-resume-runtime.cjs +0 -117
  245. package/bin/durable-task-resume-store.cjs +0 -88
  246. package/bin/editable-tool-approval-policy.cjs +0 -540
  247. package/bin/editable-tool-approval-runtime.cjs +0 -99
  248. package/bin/effective-system-prompt-cache-policy.cjs +0 -33
  249. package/bin/error-memory-performance-policy.cjs +0 -113
  250. package/bin/file-observation-policy.cjs +0 -133
  251. package/bin/foreground-output-capture-policy.cjs +0 -41
  252. package/bin/generated-source-health.cjs +0 -142
  253. package/bin/glob-pattern-policy.cjs +0 -13
  254. package/bin/goal-completion-evidence-policy.cjs +0 -120
  255. package/bin/grep-output-limit-policy.cjs +0 -39
  256. package/bin/historical-media-projection-policy.cjs +0 -48
  257. package/bin/history-offload-pressure-policy.cjs +0 -33
  258. package/bin/html-to-research-markdown.cjs +0 -147
  259. package/bin/identity-context-policy.cjs +0 -764
  260. package/bin/identity-journal-policy.cjs +0 -107
  261. package/bin/input-draft-persistence.cjs +0 -77
  262. package/bin/king-tui-function-contract.json +0 -33
  263. package/bin/launcher-restart-policy.cjs +0 -150
  264. package/bin/live-response-repetition-guard.cjs +0 -196
  265. package/bin/llm-config-log-dedup-policy.cjs +0 -76
  266. package/bin/loop-event-record-policy.cjs +0 -174
  267. package/bin/managed-context-startup-policy.cjs +0 -27
  268. package/bin/media-activity-layout-policy.cjs +0 -34
  269. package/bin/media-auto-retrieval-policy.cjs +0 -90
  270. package/bin/media-result-policy.cjs +0 -59
  271. package/bin/micro-compaction-policy.cjs +0 -145
  272. package/bin/mistake-relevance-policy.cjs +0 -319
  273. package/bin/model-retry-progress-policy.cjs +0 -46
  274. package/bin/native-large-file-io.cjs +0 -42
  275. package/bin/native-runtime-cache.cjs +0 -76
  276. package/bin/natural-presence-policy.cjs +0 -28
  277. package/bin/noninteractive-shell-env-policy.cjs +0 -19
  278. package/bin/observer-hooks.cjs +0 -14
  279. package/bin/outbound-claim-provenance.cjs +0 -150
  280. package/bin/oversized-context-offload-policy.cjs +0 -86
  281. package/bin/pending-media-policy.cjs +0 -182
  282. package/bin/pending-token-estimate-policy.cjs +0 -41
  283. package/bin/personal-memory-consent-policy.cjs +0 -72
  284. package/bin/personal-memory-performance-policy.cjs +0 -12
  285. package/bin/personality-choice-policy.cjs +0 -101
  286. package/bin/personality-memory-adapter.cjs +0 -379
  287. package/bin/personality-mode.cjs +0 -46
  288. package/bin/personality-setup-policy.cjs +0 -197
  289. package/bin/proactive-compaction-policy.cjs +0 -25
  290. package/bin/profile-identity-resolution.cjs +0 -136
  291. package/bin/profile-runtime.cjs +0 -318
  292. package/bin/profile-tool-exclusion-policy.cjs +0 -37
  293. package/bin/programmatic-context-isolation.cjs +0 -25
  294. package/bin/programmatic-tool-runtime.mjs +0 -627
  295. package/bin/provider-idle-timeout-policy.cjs +0 -14
  296. package/bin/provider-model-refresh-deadline.cjs +0 -53
  297. package/bin/provider-model-refresh-policy.cjs +0 -107
  298. package/bin/rate-limit-recovery-policy.cjs +0 -47
  299. package/bin/read-batch-policy.cjs +0 -32
  300. package/bin/read-continuation-policy.cjs +0 -59
  301. package/bin/recurring-cron-history-policy.cjs +0 -124
  302. package/bin/relationship-continuity-policy.cjs +0 -143
  303. package/bin/relationship-curiosity-policy.cjs +0 -107
  304. package/bin/relationship-learning-policy.cjs +0 -168
  305. package/bin/release-artifact-freeze-policy.cjs +0 -30
  306. package/bin/reload-plugin-bootstrap.cjs +0 -18
  307. package/bin/reload-queue-policy.cjs +0 -38
  308. package/bin/repeated-assistant-response-policy.cjs +0 -232
  309. package/bin/repeated-injection-projection.cjs +0 -107
  310. package/bin/repeated-user-message-projection.cjs +0 -8
  311. package/bin/research-page-result.cjs +0 -74
  312. package/bin/retry-checkpoint-policy.cjs +0 -13
  313. package/bin/runtime-exit-ledger.cjs +0 -144
  314. package/bin/scoped-cron-run-policy.cjs +0 -358
  315. package/bin/session-checkpoint-policy.cjs +0 -25
  316. package/bin/session-compaction-policy.cjs +0 -84
  317. package/bin/session-replay-policy.cjs +0 -20
  318. package/bin/session-replay-window-policy.cjs +0 -40
  319. package/bin/session-resume-checkpoint.cjs +0 -254
  320. package/bin/session-scrollback-archive.cjs +0 -229
  321. package/bin/skill-activation-performance-policy.cjs +0 -69
  322. package/bin/skill-listing-performance-policy.cjs +0 -92
  323. package/bin/soul-organization-policy.cjs +0 -78
  324. package/bin/soul-preservation-policy.cjs +0 -20
  325. package/bin/startup-preferences.cjs +0 -131
  326. package/bin/streaming-flush-performance-policy.cjs +0 -28
  327. package/bin/structured-agent-swarm-output.cjs +0 -325
  328. package/bin/structured-subagent-output.cjs +0 -252
  329. package/bin/subagent-context-fork-policy.cjs +0 -155
  330. package/bin/subagent-max-tokens-handoff-policy.cjs +0 -69
  331. package/bin/subagent-parent-responsiveness.cjs +0 -19
  332. package/bin/subagent-skill-policy.cjs +0 -206
  333. package/bin/subagent-timeout-policy.cjs +0 -182
  334. package/bin/subagent-tool-policy.cjs +0 -60
  335. package/bin/subagent-usage-rollup-policy.cjs +0 -29
  336. package/bin/system-prompt-context-policy.cjs +0 -124
  337. package/bin/system-prompt-token-cache-policy.cjs +0 -60
  338. package/bin/telegram-addressed-focus.cjs +0 -55
  339. package/bin/telegram-addressed-priority.cjs +0 -12
  340. package/bin/telegram-approval-relay.cjs +0 -290
  341. package/bin/telegram-bot-priority.cjs +0 -17
  342. package/bin/telegram-console-status-policy.cjs +0 -174
  343. package/bin/telegram-context-projection-policy.cjs +0 -141
  344. package/bin/telegram-delivery-lifecycle.cjs +0 -125
  345. package/bin/telegram-direct-focus-policy.cjs +0 -273
  346. package/bin/telegram-mcp-compatibility.cjs +0 -49
  347. package/bin/telegram-media-delivery-policy.cjs +0 -42
  348. package/bin/telegram-private-conversation-policy.cjs +0 -185
  349. package/bin/telegram-queue-handoff-policy.cjs +0 -73
  350. package/bin/telegram-remote-status-policy.cjs +0 -120
  351. package/bin/telegram-session-queue-runtime.mjs +0 -306
  352. package/bin/telegram-text-chunk-policy.cjs +0 -63
  353. package/bin/telegram-truncated-reply-policy.cjs +0 -37
  354. package/bin/telegram-urgent-policy.cjs +0 -45
  355. package/bin/telemetry-spool-policy.cjs +0 -57
  356. package/bin/thinking-activity-status-policy.cjs +0 -132
  357. package/bin/thinking-only-guard.cjs +0 -80
  358. package/bin/todo-list-turn-policy.cjs +0 -131
  359. package/bin/tool-call-loop-policy.cjs +0 -51
  360. package/bin/tool-file-persistence.cjs +0 -141
  361. package/bin/tool-result-offload-policy.cjs +0 -359
  362. package/bin/tool-result-offload-telemetry.cjs +0 -12
  363. package/bin/tool-schema-token-cache-policy.cjs +0 -41
  364. package/bin/tool-stream-preview-policy.cjs +0 -9
  365. package/bin/tui-functional-contract.cjs +0 -55
  366. package/bin/turn-tool-performance-policy.cjs +0 -486
  367. package/bin/usage-cache-efficiency-policy.cjs +0 -26
  368. package/bin/user-home-path-policy.cjs +0 -13
  369. package/bin/user-message-offload-policy.cjs +0 -103
  370. package/bin/user-prompt-hook-origin-policy.cjs +0 -34
  371. package/bin/user-tool-record-policy.cjs +0 -7
  372. package/bin/validated-learning-insight-policy.cjs +0 -58
  373. package/bin/validated-learning-outcome-trace.cjs +0 -107
  374. package/bin/validated-learning-performance-policy.cjs +0 -53
  375. package/bin/validated-learning-signal.cjs +0 -463
  376. package/bin/windows-bash-dialect-policy.cjs +0 -25
  377. package/bin/windows-node-crash-dump.cjs +0 -110
  378. package/bin/write-continuation-policy.cjs +0 -69
  379. package/codebase-index/README.md +0 -82
  380. package/codebase-index/codebase_index.py +0 -470
  381. package/standard-skills/agent-browser/SKILL.md +0 -19
  382. package/standard-skills/agent-browser/references/runtime.md +0 -8
  383. package/standard-skills/blun-session-inspector/SKILL.md +0 -41
  384. package/standard-skills/blun-session-inspector/scripts/inspect-session.cjs +0 -437
  385. package/standard-skills/design-taste-frontend/SKILL.md +0 -1206
  386. package/standard-skills/full-output-enforcement/SKILL.md +0 -49
  387. package/standard-skills/high-end-visual-design/SKILL.md +0 -98
  388. package/standard-skills/image-to-code/SKILL.md +0 -1228
  389. package/standard-skills/industrial-brutalist-ui/SKILL.md +0 -92
  390. package/standard-skills/minimalist-ui/SKILL.md +0 -85
  391. package/standard-skills/motion-design-taste/SKILL.md +0 -74
  392. package/standard-skills/playwright-testing/SKILL.md +0 -19
  393. package/standard-skills/playwright-testing/references/runtime.md +0 -7
  394. package/standard-skills/premortem/SKILL.md +0 -148
  395. package/standard-skills/redesign-existing-projects/SKILL.md +0 -178
  396. package/standard-skills/research-evidence/SKILL.md +0 -39
  397. package/standard-skills/research-evidence/references/evidence-format.md +0 -104
  398. package/standard-skills/research-evidence/scripts/evidence-collection.cjs +0 -260
  399. package/standard-skills/research-evidence/scripts/score-report.cjs +0 -130
  400. package/standard-skills/screenshot-lesen/SKILL.md +0 -52
  401. package/standard-skills/stitch-design-taste/DESIGN.md +0 -121
  402. package/standard-skills/stitch-design-taste/SKILL.md +0 -184
  403. package/standard-skills/telegram-channel/SKILL.md +0 -18
  404. package/standard-skills/telegram-channel/references/runtime.md +0 -7
  405. package/standard-skills/venture-flywheel/SKILL.md +0 -32
  406. package/standard-skills/venture-flywheel/identity/project-identity.cjs +0 -146
  407. package/standard-skills/venture-flywheel/policy/capability-engine.cjs +0 -114
  408. package/standard-skills/venture-flywheel/policy/repository-trust.cjs +0 -229
  409. package/standard-skills/venture-flywheel/references/BEISPIELE-phase0.md +0 -146
  410. package/standard-skills/venture-flywheel/references/CAPABILITY-MAP.md +0 -34
  411. package/standard-skills/venture-flywheel/references/SPEC-phase0-identity-trust.md +0 -77
  412. package/standard-skills/venture-flywheel/references/SPEC-phase0-state-events.md +0 -93
  413. package/standard-skills/venture-flywheel/schemas/capability-decision.schema.json +0 -13
  414. package/standard-skills/venture-flywheel/schemas/execution-event.schema.json +0 -44
  415. package/standard-skills/venture-flywheel/schemas/project-identity.schema.json +0 -32
  416. package/standard-skills/venture-flywheel/schemas/repository-trust.schema.json +0 -57
  417. package/standard-skills/venture-flywheel/schemas/run-transition.schema.json +0 -59
  418. package/standard-skills/venture-flywheel/state/execution-event.cjs +0 -191
  419. package/standard-skills/venture-flywheel/state/task-state-machine.cjs +0 -190
  420. package/standard-skills/web-lesen/SKILL.md +0 -73
  421. package/standard-skills/web-lesen/scripts/crawl_public.py +0 -379
  422. package/standard-skills/windows-mcp/SKILL.md +0 -19
  423. package/standard-skills/windows-mcp/references/runtime.md +0 -9
  424. package/telegram-plugin/DELIVERY.md +0 -36
  425. package/telegram-plugin/bin/telegram-approval-relay.cjs +0 -290
  426. package/telegram-plugin/bin/telegram-console-status-policy.cjs +0 -175
  427. package/telegram-plugin/bin/telegram-delivery-lifecycle.cjs +0 -125
  428. package/telegram-plugin/bin/telegram-direct-reply-policy.cjs +0 -48
  429. package/telegram-plugin/bin/telegram-launcher-status-queue.cjs +0 -122
  430. package/telegram-plugin/bin/telegram-private-conversation-policy.cjs +0 -186
  431. package/telegram-plugin/bin/telegram-remote-status-policy.cjs +0 -121
  432. package/telegram-plugin/bin/telegram-reply-parts.cjs +0 -149
  433. package/telegram-plugin/bin/telegram-text-chunk-policy.cjs +0 -63
  434. package/telegram-plugin/bin/telegram-typing-keepalive.cjs +0 -89
  435. package/telegram-plugin/compat/mcp-server-fa511cd1.mjs +0 -73825
  436. /package/{bin → scripts}/fix-node-pty-perms.js +0 -0
@@ -0,0 +1,781 @@
1
+ #!/usr/bin/env python3
2
+ """Durable, attested storage for one anonymous benchmark review pass.
3
+
4
+ The campaign lease remains the concurrency boundary. This store makes each
5
+ successful reviewer response immutable before the campaign proceeds to the
6
+ next ordered phase, so a later retry reuses the exact response instead of
7
+ asking the reviewer to judge the same blind variants again.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import hashlib
13
+ import importlib.util
14
+ import json
15
+ import math
16
+ import re
17
+ import sqlite3
18
+ import sys
19
+ import time
20
+ from contextlib import contextmanager
21
+ from dataclasses import asdict, dataclass
22
+ from pathlib import Path
23
+ from typing import Any, Callable, Iterator, Mapping
24
+
25
+
26
+ STORE_SCHEMA = "blun.website-localization-benchmark-review-store.v1"
27
+ ARTIFACT_SCHEMA = "blun.website-localization-benchmark-review-evidence.v1"
28
+ HEALTH_SCHEMA = "blun.website-localization-benchmark-review-health.v1"
29
+ MAX_ARTIFACT_BYTES = 4 * 1024 * 1024
30
+ ERROR_CODE = re.compile(r"^[a-z][a-z0-9_.-]{0,127}$")
31
+ REVIEW_ID = re.compile(r"^benchmark-review-[0-9a-f]{64}$")
32
+ STORE_COLUMNS = (
33
+ "acquisition_id", "review_id", "route_id", "policy_sha256",
34
+ "request_sha256", "artifact_json", "artifact_sha256", "created_at",
35
+ )
36
+
37
+
38
+ def _load_module(name: str, path: Path):
39
+ spec = importlib.util.spec_from_file_location(name, path)
40
+ if spec is None or spec.loader is None:
41
+ raise RuntimeError(f"cannot load benchmark review dependency: {path.name}")
42
+ module = importlib.util.module_from_spec(spec)
43
+ sys.modules[spec.name] = module
44
+ spec.loader.exec_module(module)
45
+ return module
46
+
47
+
48
+ _ROOT = Path(__file__).resolve().parents[1]
49
+ _BENCHMARK = _load_module(
50
+ "blun_website_localization_review_store_benchmark",
51
+ _ROOT / "integrations" / "website_localization_benchmark.py",
52
+ )
53
+
54
+
55
+ class BenchmarkReviewEvidenceFailed(RuntimeError):
56
+ """Content-free failure understood by benchmark orchestration."""
57
+
58
+ benchmark_reviewer_failure = True
59
+
60
+ def __init__(self, code: str, *, retryable: bool):
61
+ if not isinstance(code, str) or ERROR_CODE.fullmatch(code) is None:
62
+ raise ValueError("benchmark review evidence code is invalid")
63
+ if not isinstance(retryable, bool):
64
+ raise ValueError("benchmark review evidence retryability is invalid")
65
+ super().__init__(code)
66
+ self.code = code
67
+ self.retryable = retryable
68
+
69
+
70
+ @dataclass(frozen=True)
71
+ class BenchmarkReviewEvidenceHealth:
72
+ """Content-free integrity summary for one active review route and policy."""
73
+
74
+ route_id: str
75
+ status: str
76
+ reasons: tuple[str, ...]
77
+ counts: tuple[tuple[str, int], ...]
78
+
79
+ def as_payload(self) -> dict[str, Any]:
80
+ return {
81
+ "schema": HEALTH_SCHEMA,
82
+ "route_id": self.route_id,
83
+ "status": self.status,
84
+ "reasons": list(self.reasons),
85
+ "counts": dict(self.counts),
86
+ }
87
+
88
+
89
+ def _pairs(items):
90
+ result = {}
91
+ for key, value in items:
92
+ if key in result:
93
+ raise ValueError("duplicate JSON key")
94
+ result[key] = value
95
+ return result
96
+
97
+
98
+ def _constant(value):
99
+ raise ValueError("non-finite JSON number")
100
+
101
+
102
+ def _json_bytes(value: Any, code: str) -> bytes:
103
+ try:
104
+ encoded = json.dumps(
105
+ value,
106
+ ensure_ascii=False,
107
+ allow_nan=False,
108
+ sort_keys=True,
109
+ separators=(",", ":"),
110
+ ).encode("utf-8")
111
+ except (TypeError, ValueError, UnicodeEncodeError, RecursionError):
112
+ raise BenchmarkReviewEvidenceFailed(code, retryable=False) from None
113
+ if not encoded or len(encoded) > MAX_ARTIFACT_BYTES:
114
+ raise BenchmarkReviewEvidenceFailed(code, retryable=False)
115
+ return encoded
116
+
117
+
118
+ def _parse_json(value: Any) -> Any:
119
+ if not isinstance(value, str) or not value:
120
+ raise BenchmarkReviewEvidenceFailed(
121
+ "review.store.state_invalid", retryable=False,
122
+ )
123
+ try:
124
+ encoded = value.encode("utf-8")
125
+ if len(encoded) > MAX_ARTIFACT_BYTES:
126
+ raise ValueError("stored review JSON is too large")
127
+ return json.loads(
128
+ value, object_pairs_hook=_pairs, parse_constant=_constant,
129
+ )
130
+ except (UnicodeEncodeError, ValueError, RecursionError):
131
+ raise BenchmarkReviewEvidenceFailed(
132
+ "review.store.state_invalid", retryable=False,
133
+ ) from None
134
+
135
+
136
+ def _hash_bytes(value: bytes) -> str:
137
+ return hashlib.sha256(value).hexdigest()
138
+
139
+
140
+ def _hash_text(value: str) -> str:
141
+ return _hash_bytes(value.encode("utf-8"))
142
+
143
+
144
+ def _timestamp(value: Any = None) -> float:
145
+ value = time.time() if value is None else value
146
+ if (
147
+ isinstance(value, bool)
148
+ or not isinstance(value, (int, float))
149
+ or not math.isfinite(float(value))
150
+ or float(value) < 0
151
+ ):
152
+ raise BenchmarkReviewEvidenceFailed(
153
+ "review.store.time_invalid", retryable=False,
154
+ )
155
+ return float(value)
156
+
157
+
158
+ def _identifier(value: Any, code: str) -> str:
159
+ if not isinstance(value, str) or _BENCHMARK.IDENTIFIER.fullmatch(value) is None:
160
+ raise BenchmarkReviewEvidenceFailed(code, retryable=False)
161
+ return value
162
+
163
+
164
+ def _request_payload(value: Any) -> dict[str, Any]:
165
+ try:
166
+ payload = value.as_payload()
167
+ except Exception:
168
+ raise BenchmarkReviewEvidenceFailed(
169
+ "review.store.request_invalid", retryable=False,
170
+ ) from None
171
+ expected = {
172
+ "schema", "review_id", "phase", "target_locale",
173
+ "system_instruction", "input",
174
+ }
175
+ if not isinstance(payload, dict) or set(payload) != expected:
176
+ raise BenchmarkReviewEvidenceFailed(
177
+ "review.store.request_invalid", retryable=False,
178
+ )
179
+ phase = payload.get("phase")
180
+ input_value = payload.get("input")
181
+ commercial_fidelity = (
182
+ phase == "source_fidelity"
183
+ and isinstance(input_value, dict)
184
+ and input_value.get("content_type") == "commercial"
185
+ )
186
+ expected_system = {
187
+ "target_native": _BENCHMARK._NATIVE_SYSTEM,
188
+ "source_fidelity": _BENCHMARK._FIDELITY_SYSTEM + (
189
+ "\n" + _BENCHMARK._COMMERCIAL_BENCHMARK_FIDELITY_SYSTEM
190
+ if commercial_fidelity else ""
191
+ ),
192
+ }.get(phase)
193
+ commercial_dimensions = (
194
+ input_value.get("benchmark_suite", {}).get("commercial_dimensions")
195
+ if commercial_fidelity
196
+ and isinstance(input_value.get("benchmark_suite"), dict)
197
+ else None
198
+ )
199
+ if (
200
+ payload.get("schema") != _BENCHMARK.BENCHMARK_SCHEMA
201
+ or REVIEW_ID.fullmatch(payload.get("review_id", "")) is None
202
+ or expected_system is None
203
+ or payload.get("system_instruction") != expected_system
204
+ or not isinstance(payload.get("target_locale"), str)
205
+ or not isinstance(input_value, dict)
206
+ or input_value.get("blind_id") is None
207
+ or (
208
+ commercial_fidelity
209
+ and commercial_dimensions
210
+ != list(_BENCHMARK._WORKER._COMMERCIAL.DIMENSIONS)
211
+ )
212
+ ):
213
+ raise BenchmarkReviewEvidenceFailed(
214
+ "review.store.request_invalid", retryable=False,
215
+ )
216
+ try:
217
+ return json.loads(_json_bytes(
218
+ payload, "review.store.request_invalid",
219
+ ).decode("utf-8"))
220
+ except (UnicodeDecodeError, ValueError):
221
+ raise BenchmarkReviewEvidenceFailed(
222
+ "review.store.request_invalid", retryable=False,
223
+ ) from None
224
+
225
+
226
+ def _binding(
227
+ request: Any,
228
+ policy: Any,
229
+ route_id: Any,
230
+ ) -> tuple[tuple[str, str, str, str, str], dict[str, Any], Any]:
231
+ route = _identifier(route_id, "review.store.route_invalid")
232
+ try:
233
+ validated_policy = _BENCHMARK._validate_policy(policy)
234
+ except Exception:
235
+ raise BenchmarkReviewEvidenceFailed(
236
+ "review.store.policy_invalid", retryable=False,
237
+ ) from None
238
+ payload = _request_payload(request)
239
+ if payload["target_locale"] not in validated_policy.required_locales:
240
+ raise BenchmarkReviewEvidenceFailed(
241
+ "review.store.binding_invalid", retryable=False,
242
+ )
243
+ policy_sha256 = _hash_bytes(_json_bytes(
244
+ asdict(validated_policy), "review.store.policy_invalid",
245
+ ))
246
+ request_sha256 = _hash_bytes(_json_bytes(
247
+ payload, "review.store.request_invalid",
248
+ ))
249
+ acquisition_id = "benchmark-review-evidence:" + _hash_bytes(_json_bytes(
250
+ {
251
+ "schema": STORE_SCHEMA,
252
+ "review_id": payload["review_id"],
253
+ "route_id": route,
254
+ "policy_sha256": policy_sha256,
255
+ "request_sha256": request_sha256,
256
+ },
257
+ "review.store.binding_invalid",
258
+ ))
259
+ return (
260
+ (
261
+ acquisition_id, payload["review_id"], route,
262
+ policy_sha256, request_sha256,
263
+ ),
264
+ payload,
265
+ validated_policy,
266
+ )
267
+
268
+
269
+ def _response(value: Any, request: dict[str, Any]) -> dict[str, Any]:
270
+ if not isinstance(value, Mapping):
271
+ raise BenchmarkReviewEvidenceFailed(
272
+ "review.store.response_invalid", retryable=False,
273
+ )
274
+ try:
275
+ response = json.loads(_json_bytes(
276
+ dict(value), "review.store.response_invalid",
277
+ ).decode("utf-8"))
278
+ _BENCHMARK._validate_review(
279
+ response,
280
+ phase=request["phase"],
281
+ locale=request["target_locale"],
282
+ blind_id=request["input"]["blind_id"],
283
+ commercial_dimensions=(
284
+ request["input"].get("benchmark_suite", {}).get(
285
+ "commercial_dimensions",
286
+ )
287
+ if request["phase"] == "source_fidelity"
288
+ and request["input"].get("content_type") == "commercial"
289
+ and isinstance(request["input"].get("benchmark_suite"), dict)
290
+ else None
291
+ ),
292
+ )
293
+ except BenchmarkReviewEvidenceFailed:
294
+ raise
295
+ except Exception:
296
+ raise BenchmarkReviewEvidenceFailed(
297
+ "review.store.response_invalid", retryable=False,
298
+ ) from None
299
+ return response
300
+
301
+
302
+ def _validate_artifact(
303
+ artifact: Any,
304
+ identity: tuple[str, str, str, str, str],
305
+ request: dict[str, Any],
306
+ policy: Any,
307
+ evidence_authority: Any,
308
+ ) -> dict[str, Any]:
309
+ expected = {
310
+ "schema", "acquisition_id", "review_id", "route_id",
311
+ "policy_sha256", "request_sha256", "response_sha256", "reviewer",
312
+ "response", "attestation",
313
+ }
314
+ if not isinstance(artifact, dict) or set(artifact) != expected:
315
+ raise BenchmarkReviewEvidenceFailed(
316
+ "review.store.artifact_invalid", retryable=False,
317
+ )
318
+ unsigned = {key: value for key, value in artifact.items() if key != "attestation"}
319
+ if (
320
+ unsigned["schema"] != ARTIFACT_SCHEMA
321
+ or tuple(unsigned[name] for name in STORE_COLUMNS[:5]) != identity
322
+ or unsigned["reviewer"] != {
323
+ "id": policy.reviewer_id,
324
+ "version": policy.reviewer_version,
325
+ }
326
+ ):
327
+ raise BenchmarkReviewEvidenceFailed(
328
+ "review.store.artifact_invalid", retryable=False,
329
+ )
330
+ unsigned["response"] = _response(unsigned["response"], request)
331
+ if unsigned["response_sha256"] != _hash_bytes(_json_bytes(
332
+ unsigned["response"], "review.store.artifact_invalid",
333
+ )):
334
+ raise BenchmarkReviewEvidenceFailed(
335
+ "review.store.artifact_invalid", retryable=False,
336
+ )
337
+ try:
338
+ _BENCHMARK._verify_attestation(
339
+ unsigned, artifact["attestation"], policy, evidence_authority,
340
+ )
341
+ except _BENCHMARK.BenchmarkBlocked as error:
342
+ retryable = error.code == "benchmark.attestation.verify_failed"
343
+ code = (
344
+ "review.store.attestation_unavailable"
345
+ if retryable else "review.store.artifact_invalid"
346
+ )
347
+ raise BenchmarkReviewEvidenceFailed(code, retryable=retryable) from None
348
+ except Exception:
349
+ raise BenchmarkReviewEvidenceFailed(
350
+ "review.store.artifact_invalid", retryable=False,
351
+ ) from None
352
+ verified = dict(unsigned)
353
+ verified["attestation"] = artifact["attestation"]
354
+ return json.loads(_json_bytes(
355
+ verified, "review.store.artifact_invalid",
356
+ ).decode("utf-8"))
357
+
358
+
359
+ def _create_artifact(
360
+ identity: tuple[str, str, str, str, str],
361
+ response: Any,
362
+ request: dict[str, Any],
363
+ policy: Any,
364
+ evidence_authority: Any,
365
+ ) -> dict[str, Any]:
366
+ validated_response = _response(response, request)
367
+ unsigned = {
368
+ "schema": ARTIFACT_SCHEMA,
369
+ "acquisition_id": identity[0],
370
+ "review_id": identity[1],
371
+ "route_id": identity[2],
372
+ "policy_sha256": identity[3],
373
+ "request_sha256": identity[4],
374
+ "response_sha256": _hash_bytes(_json_bytes(
375
+ validated_response, "review.store.response_invalid",
376
+ )),
377
+ "reviewer": {
378
+ "id": policy.reviewer_id,
379
+ "version": policy.reviewer_version,
380
+ },
381
+ "response": validated_response,
382
+ }
383
+ try:
384
+ artifact = _BENCHMARK._attest(unsigned, policy, evidence_authority)
385
+ except _BENCHMARK.BenchmarkBlocked as error:
386
+ retryable = error.code in {
387
+ "benchmark.attestation.sign_failed",
388
+ "benchmark.attestation.verify_failed",
389
+ }
390
+ code = (
391
+ "review.store.attestation_unavailable"
392
+ if retryable else "review.store.artifact_invalid"
393
+ )
394
+ raise BenchmarkReviewEvidenceFailed(code, retryable=retryable) from None
395
+ return _validate_artifact(
396
+ artifact, identity, request, policy, evidence_authority,
397
+ )
398
+
399
+
400
+ @contextmanager
401
+ def _transaction(connection: sqlite3.Connection) -> Iterator[None]:
402
+ if connection.in_transaction:
403
+ raise BenchmarkReviewEvidenceFailed(
404
+ "review.store.external_transaction", retryable=False,
405
+ )
406
+ try:
407
+ connection.execute("BEGIN IMMEDIATE")
408
+ yield
409
+ except Exception:
410
+ connection.rollback()
411
+ raise
412
+ else:
413
+ connection.commit()
414
+
415
+
416
+ class BenchmarkReviewEvidenceStore:
417
+ """Persist the first exact attested response for one bound review pass."""
418
+
419
+ def __init__(self, connection: sqlite3.Connection):
420
+ if not isinstance(connection, sqlite3.Connection):
421
+ raise BenchmarkReviewEvidenceFailed(
422
+ "review.store.connection_invalid", retryable=False,
423
+ )
424
+ self.connection = connection
425
+ self.connection.row_factory = sqlite3.Row
426
+ self.connection.execute("PRAGMA busy_timeout = 5000")
427
+ with _transaction(self.connection):
428
+ self.connection.execute("""
429
+ CREATE TABLE IF NOT EXISTS benchmark_review_evidence (
430
+ acquisition_id TEXT PRIMARY KEY,
431
+ review_id TEXT NOT NULL,
432
+ route_id TEXT NOT NULL,
433
+ policy_sha256 TEXT NOT NULL,
434
+ request_sha256 TEXT NOT NULL,
435
+ artifact_json TEXT NOT NULL,
436
+ artifact_sha256 TEXT NOT NULL,
437
+ created_at REAL NOT NULL,
438
+ UNIQUE(review_id, route_id, policy_sha256)
439
+ )
440
+ """)
441
+ self._verify_schema()
442
+
443
+ def _verify_schema(self) -> None:
444
+ columns = tuple(
445
+ row[1] for row in self.connection.execute(
446
+ "PRAGMA table_info(benchmark_review_evidence)"
447
+ ).fetchall()
448
+ )
449
+ if columns != STORE_COLUMNS:
450
+ raise BenchmarkReviewEvidenceFailed(
451
+ "review.store.schema_unsupported", retryable=False,
452
+ )
453
+
454
+ def health(
455
+ self,
456
+ policy: Any,
457
+ route_id: Any,
458
+ *,
459
+ evidence_authority: Any,
460
+ expected_passes: Any = (),
461
+ now: Any = None,
462
+ ) -> BenchmarkReviewEvidenceHealth:
463
+ """Verify stored evidence without returning review text or changing state."""
464
+ counts = {
465
+ "total": 0,
466
+ "scoped": 0,
467
+ "historical": 0,
468
+ "target_native": 0,
469
+ "source_fidelity": 0,
470
+ "required": 0,
471
+ "matched": 0,
472
+ }
473
+ reasons: set[str] = set()
474
+ safe_route = route_id if isinstance(route_id, str) else "invalid"
475
+ try:
476
+ safe_route = _identifier(route_id, "review.store.route_invalid")
477
+ validated_policy = _BENCHMARK._validate_policy(policy)
478
+ policy_sha256 = _hash_bytes(_json_bytes(
479
+ asdict(validated_policy), "review.store.policy_invalid",
480
+ ))
481
+ checked_at = _timestamp(now)
482
+ if self.connection.in_transaction:
483
+ raise BenchmarkReviewEvidenceFailed(
484
+ "review.store.external_transaction", retryable=False,
485
+ )
486
+ if not isinstance(expected_passes, (tuple, list)):
487
+ raise ValueError
488
+ required: dict[str, tuple[str, str]] = {}
489
+ for item in expected_passes:
490
+ if not isinstance(item, Mapping) or set(item) != {
491
+ "phase", "request_sha256", "response_sha256",
492
+ }:
493
+ raise ValueError
494
+ phase = item["phase"]
495
+ request_sha256 = item["request_sha256"]
496
+ response_sha256 = item["response_sha256"]
497
+ if (
498
+ phase not in _BENCHMARK.PHASES
499
+ or re.fullmatch(r"[0-9a-f]{64}", request_sha256 or "") is None
500
+ or re.fullmatch(r"[0-9a-f]{64}", response_sha256 or "") is None
501
+ or request_sha256 in required
502
+ ):
503
+ raise ValueError
504
+ required[request_sha256] = (phase, response_sha256)
505
+ counts["required"] = len(required)
506
+ observed: dict[str, tuple[str, str]] = {}
507
+ self._verify_schema()
508
+ rows = self.connection.execute(
509
+ "SELECT * FROM benchmark_review_evidence "
510
+ "ORDER BY route_id, policy_sha256, review_id"
511
+ ).fetchall()
512
+ counts["total"] = len(rows)
513
+ for row in rows:
514
+ if tuple(row.keys()) != STORE_COLUMNS:
515
+ raise ValueError
516
+ scoped = (
517
+ row["route_id"] == safe_route
518
+ and row["policy_sha256"] == policy_sha256
519
+ )
520
+ if not scoped:
521
+ counts["historical"] += 1
522
+ continue
523
+ counts["scoped"] += 1
524
+ created_at = _timestamp(row["created_at"])
525
+ if created_at > checked_at:
526
+ raise ValueError
527
+ identity = tuple(row[name] for name in STORE_COLUMNS[:5])
528
+ if (
529
+ not isinstance(row["artifact_json"], str)
530
+ or row["artifact_sha256"] != _hash_text(row["artifact_json"])
531
+ or REVIEW_ID.fullmatch(row["review_id"] or "") is None
532
+ or re.fullmatch(r"[0-9a-f]{64}", row["request_sha256"] or "") is None
533
+ ):
534
+ raise ValueError
535
+ artifact = _parse_json(row["artifact_json"])
536
+ if _json_bytes(
537
+ artifact, "review.store.artifact_invalid",
538
+ ).decode("utf-8") != row["artifact_json"]:
539
+ raise ValueError
540
+ response = artifact.get("response") if isinstance(artifact, dict) else None
541
+ if not isinstance(response, dict):
542
+ raise ValueError
543
+ phase = response.get("phase")
544
+ locale = response.get("target_locale")
545
+ blind_id = response.get("blind_id")
546
+ if (
547
+ phase not in _BENCHMARK.PHASES
548
+ or locale not in validated_policy.required_locales
549
+ or re.fullmatch(r"blind-[0-9a-f]{64}", blind_id or "") is None
550
+ ):
551
+ raise ValueError
552
+ request_stub = {
553
+ "phase": phase,
554
+ "target_locale": locale,
555
+ "input": {"blind_id": blind_id},
556
+ }
557
+ verified = _validate_artifact(
558
+ artifact, identity, request_stub, validated_policy,
559
+ evidence_authority,
560
+ )
561
+ response_sha256 = verified["response_sha256"]
562
+ if row["request_sha256"] in observed:
563
+ raise ValueError
564
+ observed[row["request_sha256"]] = (phase, response_sha256)
565
+ counts[phase] += 1
566
+ for request_sha256, expected in required.items():
567
+ actual = observed.get(request_sha256)
568
+ if actual is None:
569
+ reasons.add("review.store.required_missing")
570
+ elif actual != expected:
571
+ reasons.add("review.store.required_mismatch")
572
+ else:
573
+ counts["matched"] += 1
574
+ except BenchmarkReviewEvidenceFailed as error:
575
+ reasons = {error.code}
576
+ except Exception:
577
+ reasons = {"review.store.state_invalid"}
578
+ status = "blocked" if reasons else "healthy"
579
+ return BenchmarkReviewEvidenceHealth(
580
+ route_id=safe_route,
581
+ status=status,
582
+ reasons=tuple(sorted(reasons)),
583
+ counts=tuple(sorted(counts.items())),
584
+ )
585
+
586
+ def _row_for_identity(
587
+ self, identity: tuple[str, str, str, str, str],
588
+ ) -> sqlite3.Row | None:
589
+ row = self.connection.execute("""
590
+ SELECT * FROM benchmark_review_evidence
591
+ WHERE review_id = ? AND route_id = ? AND policy_sha256 = ?
592
+ """, (identity[1], identity[2], identity[3])).fetchone()
593
+ if row is not None and tuple(row[name] for name in STORE_COLUMNS[:5]) != identity:
594
+ raise BenchmarkReviewEvidenceFailed(
595
+ "review.store.conflict", retryable=False,
596
+ )
597
+ return row
598
+
599
+ def load(
600
+ self,
601
+ request: Any,
602
+ policy: Any,
603
+ route_id: Any,
604
+ *,
605
+ evidence_authority: Any,
606
+ ) -> dict[str, Any] | None:
607
+ identity, request_payload, validated_policy = _binding(
608
+ request, policy, route_id,
609
+ )
610
+ row = self._row_for_identity(identity)
611
+ if row is None:
612
+ return None
613
+ if (
614
+ tuple(row.keys()) != STORE_COLUMNS
615
+ or isinstance(row["created_at"], bool)
616
+ or not isinstance(row["created_at"], (int, float))
617
+ or not math.isfinite(float(row["created_at"]))
618
+ or float(row["created_at"]) < 0
619
+ or not isinstance(row["artifact_json"], str)
620
+ or row["artifact_sha256"] != _hash_text(row["artifact_json"])
621
+ ):
622
+ raise BenchmarkReviewEvidenceFailed(
623
+ "review.store.state_invalid", retryable=False,
624
+ )
625
+ artifact = _parse_json(row["artifact_json"])
626
+ verified = _validate_artifact(
627
+ artifact, identity, request_payload, validated_policy,
628
+ evidence_authority,
629
+ )
630
+ return verified["response"]
631
+
632
+ def save(
633
+ self,
634
+ request: Any,
635
+ policy: Any,
636
+ route_id: Any,
637
+ response: Any,
638
+ *,
639
+ evidence_authority: Any,
640
+ now: Any = None,
641
+ ) -> dict[str, Any]:
642
+ identity, request_payload, validated_policy = _binding(
643
+ request, policy, route_id,
644
+ )
645
+ artifact = _create_artifact(
646
+ identity, response, request_payload, validated_policy,
647
+ evidence_authority,
648
+ )
649
+ artifact_json = _json_bytes(
650
+ artifact, "review.store.artifact_invalid",
651
+ ).decode("utf-8")
652
+ values = identity + (
653
+ artifact_json,
654
+ _hash_text(artifact_json),
655
+ _timestamp(now),
656
+ )
657
+ with _transaction(self.connection):
658
+ self.connection.execute("""
659
+ INSERT OR IGNORE INTO benchmark_review_evidence
660
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?)
661
+ """, values)
662
+ stored = self.load(
663
+ request, policy, route_id, evidence_authority=evidence_authority,
664
+ )
665
+ if stored != artifact["response"]:
666
+ raise BenchmarkReviewEvidenceFailed(
667
+ "review.store.conflict", retryable=False,
668
+ )
669
+ return stored
670
+
671
+
672
+ class _GuardedAuthority:
673
+ def __init__(self, authority: Any, guard: Callable[[], None]):
674
+ self.authority = authority
675
+ self.guard = guard
676
+
677
+ def sign(self, payload: bytes):
678
+ self.guard()
679
+ return self.authority.sign(payload)
680
+
681
+ def verify(self, payload: bytes, signature: Any):
682
+ self.guard()
683
+ return self.authority.verify(payload, signature)
684
+
685
+
686
+ class DurableBenchmarkReviewer:
687
+ """Reuse a valid review response or obtain and attest it exactly once."""
688
+
689
+ def __init__(
690
+ self,
691
+ *,
692
+ store: BenchmarkReviewEvidenceStore,
693
+ policy: Any,
694
+ route_id: Any,
695
+ reviewer: Any,
696
+ evidence_authority: Any,
697
+ operation_guard: Callable[[], Any] | None = None,
698
+ clock: Callable[[], float] = time.time,
699
+ ):
700
+ if any(not callable(getattr(store, name, None)) for name in ("load", "save")):
701
+ raise BenchmarkReviewEvidenceFailed(
702
+ "review.store.invalid", retryable=False,
703
+ )
704
+ if not callable(getattr(reviewer, "review", None)):
705
+ raise BenchmarkReviewEvidenceFailed(
706
+ "review.adapter.invalid", retryable=False,
707
+ )
708
+ if any(
709
+ not callable(getattr(evidence_authority, name, None))
710
+ for name in ("sign", "verify")
711
+ ):
712
+ raise BenchmarkReviewEvidenceFailed(
713
+ "review.authority.invalid", retryable=False,
714
+ )
715
+ if operation_guard is not None and not callable(operation_guard):
716
+ raise BenchmarkReviewEvidenceFailed(
717
+ "review.operation_guard_invalid", retryable=False,
718
+ )
719
+ if not callable(clock):
720
+ raise BenchmarkReviewEvidenceFailed(
721
+ "review.clock_invalid", retryable=False,
722
+ )
723
+ self.store = store
724
+ self.policy = _BENCHMARK._validate_policy(policy)
725
+ self.route_id = _identifier(route_id, "review.store.route_invalid")
726
+ self.reviewer = reviewer
727
+ self.evidence_authority = evidence_authority
728
+ self.operation_guard = operation_guard
729
+ self.clock = clock
730
+
731
+ def _guard(self) -> None:
732
+ if self.operation_guard is None:
733
+ return
734
+ try:
735
+ self.operation_guard()
736
+ except BenchmarkReviewEvidenceFailed:
737
+ raise
738
+ except Exception:
739
+ raise BenchmarkReviewEvidenceFailed(
740
+ "review.operation_guard_failed", retryable=True,
741
+ ) from None
742
+
743
+ def review(self, request: Any) -> Mapping[str, Any]:
744
+ guarded_authority = _GuardedAuthority(
745
+ self.evidence_authority, self._guard,
746
+ )
747
+ cached = self.store.load(
748
+ request,
749
+ self.policy,
750
+ self.route_id,
751
+ evidence_authority=guarded_authority,
752
+ )
753
+ if cached is not None:
754
+ return cached
755
+ before = _request_payload(request)
756
+ self._guard()
757
+ try:
758
+ response = self.reviewer.review(request)
759
+ except Exception as error:
760
+ if getattr(error, "benchmark_reviewer_failure", None) is True:
761
+ raise
762
+ retryable = getattr(error, "retryable", True)
763
+ if not isinstance(retryable, bool):
764
+ retryable = True
765
+ raise BenchmarkReviewEvidenceFailed(
766
+ "review.adapter_unavailable", retryable=retryable,
767
+ ) from None
768
+ after = _request_payload(request)
769
+ if after != before:
770
+ raise BenchmarkReviewEvidenceFailed(
771
+ "review.request_mutated", retryable=False,
772
+ )
773
+ self._guard()
774
+ return self.store.save(
775
+ request,
776
+ self.policy,
777
+ self.route_id,
778
+ response,
779
+ evidence_authority=guarded_authority,
780
+ now=self.clock(),
781
+ )