blun-king-cli 9.1.587 → 9.1.588
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +11 -185
- package/LIESMICH.txt +51 -13
- package/README.md +44 -47
- package/agent-spine-plugin/.codex-plugin/plugin.json +16 -4
- package/agent-spine-plugin/CHANGELOG.md +37 -5
- package/agent-spine-plugin/README.md +3 -3
- package/agent-spine-plugin/blun.plugin.json +45 -10
- package/agent-spine-plugin/docs/artifact-evaluation.md +93 -0
- package/agent-spine-plugin/docs/host-integration.md +42 -27
- package/agent-spine-plugin/docs/preflight-recall.md +4 -2
- package/agent-spine-plugin/docs/session-timeline.md +97 -236
- package/agent-spine-plugin/docs/world-model.md +25 -0
- package/agent-spine-plugin/hooks/codex.json +1 -1
- package/agent-spine-plugin/hooks/hooks.json +1 -1
- package/agent-spine-plugin/package.json +1 -3
- package/agent-spine-plugin/scripts/check-hosts.js +3 -3
- package/agent-spine-plugin/scripts/release-check.js +10 -5
- package/agent-spine-plugin/scripts/run-checks.js +4 -1
- package/agent-spine-plugin/scripts/run-tests-hermetic.js +32 -6
- package/agent-spine-plugin/src/cli-learning.js +15 -0
- package/agent-spine-plugin/src/cli.js +2 -0
- package/agent-spine-plugin/src/hook.js +32 -32
- package/agent-spine-plugin/src/lib/action-lesson-recall.js +73 -8
- package/agent-spine-plugin/src/lib/briefing.js +146 -36
- package/agent-spine-plugin/src/lib/channel-continuity.js +19 -0
- package/agent-spine-plugin/src/lib/delivery-agent-usage.js +14 -7
- package/agent-spine-plugin/src/lib/gateway-group-response.js +128 -0
- package/agent-spine-plugin/src/lib/gateway-runs.js +24 -15
- package/agent-spine-plugin/src/lib/hook-briefing-use.js +13 -3
- package/agent-spine-plugin/src/lib/hook-context.js +16 -3
- package/agent-spine-plugin/src/lib/hook-output.js +129 -5
- package/agent-spine-plugin/src/lib/hook-timeline.js +5 -3
- package/agent-spine-plugin/src/lib/indexed-memory.js +2 -2
- package/agent-spine-plugin/src/lib/learning-artifact-evaluator.js +114 -0
- package/agent-spine-plugin/src/lib/learning-context.js +11 -4
- package/agent-spine-plugin/src/lib/learning-measurements.js +2 -2
- package/agent-spine-plugin/src/lib/mcp-runtime.js +89 -3
- package/agent-spine-plugin/src/lib/mcp-source-context.js +12 -2
- package/agent-spine-plugin/src/lib/mcp-timeline-tools.js +91 -8
- package/agent-spine-plugin/src/lib/mcp-world-tools.js +2 -2
- package/agent-spine-plugin/src/lib/owned-file-lock.js +20 -1
- package/agent-spine-plugin/src/lib/persona-runtime.js +2 -2
- package/agent-spine-plugin/src/lib/preflight-delivery-id.js +27 -0
- package/agent-spine-plugin/src/lib/preflight.js +4 -4
- package/agent-spine-plugin/src/lib/session-timeline-codex.js +15 -0
- package/agent-spine-plugin/src/lib/session-timeline-contract.js +12 -4
- package/agent-spine-plugin/src/lib/session-timeline-event-extract.js +36 -7
- package/agent-spine-plugin/src/lib/session-timeline-host-origin.js +13 -10
- package/agent-spine-plugin/src/lib/session-timeline-invocation.js +1 -1
- package/agent-spine-plugin/src/lib/session-timeline-king.js +14 -0
- package/agent-spine-plugin/src/lib/session-timeline-prior.js +18 -12
- package/agent-spine-plugin/src/lib/session-timeline-provider.js +5 -0
- package/agent-spine-plugin/src/lib/session-timeline-query.js +2 -0
- package/agent-spine-plugin/src/lib/session-timeline-results.js +35 -10
- package/agent-spine-plugin/src/lib/session-timeline-source-open.js +30 -0
- package/agent-spine-plugin/src/lib/session-timeline.js +122 -75
- package/agent-spine-plugin/src/lib/source-roots.js +3 -2
- package/agent-spine-plugin/src/lib/task-knowledge-context.js +22 -1
- package/agent-spine-plugin/src/lib/timeline-continuation-update.js +100 -0
- package/agent-spine-plugin/src/lib/timeline-tool-guard.js +30 -7
- package/agent-spine-plugin/src/lib/timeline-user-feedback.js +217 -0
- package/agent-spine-plugin/src/lib/timeline-world-capture.js +233 -0
- package/agent-spine-plugin/src/lib/world-knowledge.js +59 -2
- package/agent-spine-plugin/src/lib/world-model.js +64 -9
- package/agent-spine-plugin/src/worker.js +13 -1
- package/bin/blun.js +43 -28
- package/bin/core-bootstrap.js +5 -4
- package/bin/king.js +43 -28
- package/bin/launcher-mode.js +1 -10
- package/bin/launcher-runtime.js +128 -295
- package/bin/managed-node.js +0 -0
- package/bin/managed-plugin-selection.cjs +0 -1
- package/bin/native-module-repair.js +0 -0
- package/bin/node-runtime.js +0 -0
- package/bin/node-version.js +0 -0
- package/bin/plugin-bootstrap.js +56 -120
- package/bin/private-paths.js +11 -34
- package/bin/standard-tools-bootstrap.js +34 -114
- package/bin/turn-thinking-policy.cjs +3 -11
- package/bin/update-copy.js +200 -0
- package/bin/update-lease.js +0 -0
- package/bin/update-notice.js +136 -289
- package/bin/verify-agent-behavior.cjs +122 -0
- package/bin/verify-agent-components.cjs +104 -0
- package/bin/verify-bundled-agent-sources.cjs +57 -0
- package/blun.mjs +143076 -135288
- package/bundled-agent-sources.json +701 -0
- package/package.json +12 -15
- package/standard-skills/translate-native/README.md +1293 -0
- package/standard-skills/translate-native/SKILL.md +172 -22
- package/standard-skills/translate-native/VERSION +1 -1
- package/standard-skills/translate-native/agents/openai.yaml +18 -0
- package/standard-skills/translate-native/assets/icon.svg +8 -0
- package/standard-skills/translate-native/docs/BLUN_CODE_INTEGRATION.md +76 -0
- package/standard-skills/translate-native/docs/PREMORTEM.md +489 -0
- package/standard-skills/translate-native/docs/WEBSITE_LOCALIZATION.md +2035 -0
- package/standard-skills/translate-native/docs/WEBSITE_LOCALIZATION_API.md +1302 -0
- package/standard-skills/translate-native/docs/WEBSITE_LOCALIZATION_EVIDENCE_HTTP.md +136 -0
- package/standard-skills/translate-native/docs/WEBSITE_LOCALIZATION_HEALTH_HTTP.md +130 -0
- package/standard-skills/translate-native/docs/WEBSITE_LOCALIZATION_HTTP_PROVIDER.md +175 -0
- package/standard-skills/translate-native/docs/WEBSITE_LOCALIZATION_RECEIPT_VERIFIER_HTTP.md +86 -0
- package/standard-skills/translate-native/integrations/AGENT_RULES.md +32 -0
- package/standard-skills/translate-native/integrations/adapters/blun-code-language-guard.js +514 -0
- package/standard-skills/translate-native/integrations/adapters/node-language-guard.js +230 -0
- package/standard-skills/translate-native/integrations/audit_log.py +327 -0
- package/standard-skills/translate-native/integrations/claude_language_hook.js +1536 -0
- package/standard-skills/translate-native/integrations/commercial_localization_profile.py +42 -0
- package/standard-skills/translate-native/integrations/delivery-policy.example.json +28 -0
- package/standard-skills/translate-native/integrations/enforced_delivery.py +543 -0
- package/standard-skills/translate-native/integrations/guard_service.py +435 -0
- package/standard-skills/translate-native/integrations/language_gateway.py +67 -0
- package/standard-skills/translate-native/integrations/mcp_auth_headers.py +198 -0
- package/standard-skills/translate-native/integrations/mcp_http_gateway.py +429 -0
- package/standard-skills/translate-native/integrations/non_language_html_entities.js +1485 -0
- package/standard-skills/translate-native/integrations/pre_output_guard.py +65 -0
- package/standard-skills/translate-native/integrations/task_router.py +101 -0
- package/standard-skills/translate-native/integrations/website_localization.py +401 -0
- package/standard-skills/translate-native/integrations/website_localization_api.py +581 -0
- package/standard-skills/translate-native/integrations/website_localization_benchmark.py +1885 -0
- package/standard-skills/translate-native/integrations/website_localization_benchmark_campaign.py +1772 -0
- package/standard-skills/translate-native/integrations/website_localization_benchmark_candidate.py +506 -0
- package/standard-skills/translate-native/integrations/website_localization_benchmark_http.py +400 -0
- package/standard-skills/translate-native/integrations/website_localization_benchmark_review_store.py +781 -0
- package/standard-skills/translate-native/integrations/website_localization_benchmark_reviewer_http.py +500 -0
- package/standard-skills/translate-native/integrations/website_localization_benchmark_runtime.py +1107 -0
- package/standard-skills/translate-native/integrations/website_localization_benchmark_suite.py +463 -0
- package/standard-skills/translate-native/integrations/website_localization_cms.py +2835 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_client.py +875 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_dispatch.py +805 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_http.py +588 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_lifecycle_monitor.py +991 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_receiver.py +1441 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_receiver_runtime.py +414 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_receiver_store.py +1073 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_removal_dispatch.py +865 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_source_client.py +583 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_source_delivery.py +964 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_source_delivery_runtime.py +665 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_source_http.py +1153 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_source_runtime.py +675 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_source_service.py +1125 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_terminal_notification.py +674 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_terminal_notification_http.py +444 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_terminal_notification_receiver.py +1469 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_terminal_notification_receiver_runtime.py +1142 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_terminal_processing_monitor.py +634 -0
- package/standard-skills/translate-native/integrations/website_localization_cms_terminal_receiver_client.py +804 -0
- package/standard-skills/translate-native/integrations/website_localization_deepl_baseline.py +922 -0
- package/standard-skills/translate-native/integrations/website_localization_evidence_http.py +482 -0
- package/standard-skills/translate-native/integrations/website_localization_health.py +1541 -0
- package/standard-skills/translate-native/integrations/website_localization_health_http.py +372 -0
- package/standard-skills/translate-native/integrations/website_localization_http_provider.py +297 -0
- package/standard-skills/translate-native/integrations/website_localization_native_reference_http.py +479 -0
- package/standard-skills/translate-native/integrations/website_localization_native_reference_intake.py +363 -0
- package/standard-skills/translate-native/integrations/website_localization_native_reference_queue.py +1449 -0
- package/standard-skills/translate-native/integrations/website_localization_native_reference_store.py +420 -0
- package/standard-skills/translate-native/integrations/website_localization_quality_profiles.py +235 -0
- package/standard-skills/translate-native/integrations/website_localization_queue.py +671 -0
- package/standard-skills/translate-native/integrations/website_localization_receipt_verifier_http.py +516 -0
- package/standard-skills/translate-native/integrations/website_localization_release.py +928 -0
- package/standard-skills/translate-native/integrations/website_localization_release_coordinator.py +1008 -0
- package/standard-skills/translate-native/integrations/website_localization_runner.py +276 -0
- package/standard-skills/translate-native/integrations/website_localization_runtime.py +862 -0
- package/standard-skills/translate-native/integrations/website_localization_service.py +350 -0
- package/standard-skills/translate-native/integrations/website_localization_supervisor.py +511 -0
- package/standard-skills/translate-native/integrations/website_localization_worker.py +663 -0
- package/standard-skills/translate-native/provenance.json +3 -4
- package/standard-skills/translate-native/references/commercial-localization.md +177 -0
- package/standard-skills/translate-native/scripts/blun_language_guard.py +7 -1
- package/standard-skills/translate-native/scripts/check_commercial_review.py +80 -0
- package/standard-skills/translate-native/scripts/commercial_localization_profile.py +333 -0
- package/standard-tools/language-guard/LICENSE +21 -0
- package/standard-tools/language-guard/VERSION +1 -0
- package/standard-tools/language-guard/blun_language_guard.py +7 -1
- package/standard-tools/language-guard/check_commercial_review.py +80 -0
- package/standard-tools/language-guard/commercial_localization_profile.py +333 -0
- package/standard-tools/language-guard/language_gateway.py +62 -0
- package/standard-tools/language-guard/pre_output_guard.py +64 -0
- package/standard-tools/language-guard/provenance.json +4 -11
- package/standard-tools/manifest.json +34 -11
- package/telegram-plugin/commands/access.md +2 -10
- package/telegram-plugin/dist/bridge.mjs +64041 -687
- package/telegram-plugin/dist/mcp-server.mjs +72810 -9027
- package/telegram-plugin/dist/noise.mjs +28 -63511
- package/agent-spine-plugin/CONTRIBUTING.md +0 -52
- package/agent-spine-plugin/SECURITY.md +0 -47
- package/agent-spine-plugin/docs/assignment-continuation.md +0 -48
- package/agent-spine-plugin/docs/releasing.md +0 -85
- package/agent-spine-plugin/docs/structured-completion.md +0 -67
- package/bin/abort-listener-policy.cjs +0 -43
- package/bin/active-steer-priority-policy.cjs +0 -24
- package/bin/agent-api-http-adapter.mjs +0 -446
- package/bin/agent-api-private-http-server.mjs +0 -288
- package/bin/agent-api-runtime.mjs +0 -252
- package/bin/agent-api-service-environment.mjs +0 -236
- package/bin/agent-api-service-host.mjs +0 -209
- package/bin/agent-api-service-process.mjs +0 -171
- package/bin/agent-api-session-registry.mjs +0 -428
- package/bin/agent-api-tool-broker.cjs +0 -248
- package/bin/agent-api-turn-controller.mjs +0 -461
- package/bin/agent-api-usage-journal.cjs +0 -259
- package/bin/agent-resume-snapshot.cjs +0 -241
- package/bin/agentspine-king-goal-inbox.mjs +0 -111
- package/bin/agentspine-king-goal-intake.mjs +0 -106
- package/bin/approval-rejection-stop.cjs +0 -15
- package/bin/assistant-message-offload-policy.cjs +0 -284
- package/bin/baseline-skill-performance-policy.cjs +0 -39
- package/bin/bash-search-scope-policy.cjs +0 -49
- package/bin/codebase-search-runtime.cjs +0 -23
- package/bin/cognitive-action-checkpoint.cjs +0 -1104
- package/bin/cognitive-attention-delivery.cjs +0 -76
- package/bin/cognitive-attention-policy.cjs +0 -143
- package/bin/cognitive-attention-runtime.cjs +0 -91
- package/bin/cognitive-context-projection.cjs +0 -73
- package/bin/cognitive-cross-portal-acceptance.cjs +0 -443
- package/bin/cognitive-effective-view.cjs +0 -77
- package/bin/cognitive-focus-projection.cjs +0 -206
- package/bin/cognitive-focus-scope.cjs +0 -37
- package/bin/cognitive-goal-autostart-policy.cjs +0 -72
- package/bin/cognitive-goal-time-trigger-controller.cjs +0 -146
- package/bin/cognitive-memory-adapter.cjs +0 -282
- package/bin/cognitive-memory-command.cjs +0 -293
- package/bin/cognitive-memory-provider.cjs +0 -92
- package/bin/cognitive-salience-policy.cjs +0 -159
- package/bin/cognitive-state-store.cjs +0 -508
- package/bin/cognitive-turn-lifecycle.cjs +0 -624
- package/bin/cognitive-work-focus.cjs +0 -180
- package/bin/compaction-history-archive.cjs +0 -166
- package/bin/compaction-history-startup.cjs +0 -50
- package/bin/compaction-model-policy.cjs +0 -31
- package/bin/compaction-stage-policy.cjs +0 -21
- package/bin/compaction-transaction-policy.cjs +0 -122
- package/bin/config-write-dedup-policy.cjs +0 -27
- package/bin/context-budget-ledger.cjs +0 -31
- package/bin/context-doctor-policy.cjs +0 -70
- package/bin/context-insight-policy.cjs +0 -36
- package/bin/context-performance-policy.cjs +0 -19
- package/bin/context-pressure-policy.cjs +0 -20
- package/bin/cron-run-output.cjs +0 -45
- package/bin/cron-run-store.cjs +0 -145
- package/bin/curiosity-scout-policy.cjs +0 -49
- package/bin/default-model-output-budget-policy.cjs +0 -28
- package/bin/durable-task-resume-policy.cjs +0 -130
- package/bin/durable-task-resume-runtime.cjs +0 -117
- package/bin/durable-task-resume-store.cjs +0 -88
- package/bin/editable-tool-approval-policy.cjs +0 -540
- package/bin/editable-tool-approval-runtime.cjs +0 -99
- package/bin/effective-system-prompt-cache-policy.cjs +0 -33
- package/bin/error-memory-performance-policy.cjs +0 -113
- package/bin/file-observation-policy.cjs +0 -133
- package/bin/foreground-output-capture-policy.cjs +0 -41
- package/bin/generated-source-health.cjs +0 -142
- package/bin/glob-pattern-policy.cjs +0 -13
- package/bin/goal-completion-evidence-policy.cjs +0 -120
- package/bin/grep-output-limit-policy.cjs +0 -39
- package/bin/historical-media-projection-policy.cjs +0 -48
- package/bin/history-offload-pressure-policy.cjs +0 -33
- package/bin/html-to-research-markdown.cjs +0 -147
- package/bin/identity-context-policy.cjs +0 -764
- package/bin/identity-journal-policy.cjs +0 -107
- package/bin/input-draft-persistence.cjs +0 -77
- package/bin/king-tui-function-contract.json +0 -33
- package/bin/launcher-restart-policy.cjs +0 -150
- package/bin/live-response-repetition-guard.cjs +0 -196
- package/bin/llm-config-log-dedup-policy.cjs +0 -76
- package/bin/loop-event-record-policy.cjs +0 -174
- package/bin/managed-context-startup-policy.cjs +0 -27
- package/bin/media-activity-layout-policy.cjs +0 -34
- package/bin/media-auto-retrieval-policy.cjs +0 -90
- package/bin/media-result-policy.cjs +0 -59
- package/bin/micro-compaction-policy.cjs +0 -145
- package/bin/mistake-relevance-policy.cjs +0 -319
- package/bin/model-retry-progress-policy.cjs +0 -46
- package/bin/native-large-file-io.cjs +0 -42
- package/bin/native-runtime-cache.cjs +0 -76
- package/bin/natural-presence-policy.cjs +0 -28
- package/bin/noninteractive-shell-env-policy.cjs +0 -19
- package/bin/observer-hooks.cjs +0 -14
- package/bin/outbound-claim-provenance.cjs +0 -150
- package/bin/oversized-context-offload-policy.cjs +0 -86
- package/bin/pending-media-policy.cjs +0 -182
- package/bin/pending-token-estimate-policy.cjs +0 -41
- package/bin/personal-memory-consent-policy.cjs +0 -72
- package/bin/personal-memory-performance-policy.cjs +0 -12
- package/bin/personality-choice-policy.cjs +0 -101
- package/bin/personality-memory-adapter.cjs +0 -379
- package/bin/personality-mode.cjs +0 -46
- package/bin/personality-setup-policy.cjs +0 -197
- package/bin/proactive-compaction-policy.cjs +0 -25
- package/bin/profile-identity-resolution.cjs +0 -136
- package/bin/profile-runtime.cjs +0 -318
- package/bin/profile-tool-exclusion-policy.cjs +0 -37
- package/bin/programmatic-context-isolation.cjs +0 -25
- package/bin/programmatic-tool-runtime.mjs +0 -627
- package/bin/provider-idle-timeout-policy.cjs +0 -14
- package/bin/provider-model-refresh-deadline.cjs +0 -53
- package/bin/provider-model-refresh-policy.cjs +0 -107
- package/bin/rate-limit-recovery-policy.cjs +0 -47
- package/bin/read-batch-policy.cjs +0 -32
- package/bin/read-continuation-policy.cjs +0 -59
- package/bin/recurring-cron-history-policy.cjs +0 -124
- package/bin/relationship-continuity-policy.cjs +0 -143
- package/bin/relationship-curiosity-policy.cjs +0 -107
- package/bin/relationship-learning-policy.cjs +0 -168
- package/bin/release-artifact-freeze-policy.cjs +0 -30
- package/bin/reload-plugin-bootstrap.cjs +0 -18
- package/bin/reload-queue-policy.cjs +0 -38
- package/bin/repeated-assistant-response-policy.cjs +0 -232
- package/bin/repeated-injection-projection.cjs +0 -107
- package/bin/repeated-user-message-projection.cjs +0 -8
- package/bin/research-page-result.cjs +0 -74
- package/bin/retry-checkpoint-policy.cjs +0 -13
- package/bin/runtime-exit-ledger.cjs +0 -144
- package/bin/scoped-cron-run-policy.cjs +0 -358
- package/bin/session-checkpoint-policy.cjs +0 -25
- package/bin/session-compaction-policy.cjs +0 -84
- package/bin/session-replay-policy.cjs +0 -20
- package/bin/session-replay-window-policy.cjs +0 -40
- package/bin/session-resume-checkpoint.cjs +0 -254
- package/bin/session-scrollback-archive.cjs +0 -229
- package/bin/skill-activation-performance-policy.cjs +0 -69
- package/bin/skill-listing-performance-policy.cjs +0 -92
- package/bin/soul-organization-policy.cjs +0 -78
- package/bin/soul-preservation-policy.cjs +0 -20
- package/bin/startup-preferences.cjs +0 -131
- package/bin/streaming-flush-performance-policy.cjs +0 -28
- package/bin/structured-agent-swarm-output.cjs +0 -325
- package/bin/structured-subagent-output.cjs +0 -252
- package/bin/subagent-context-fork-policy.cjs +0 -155
- package/bin/subagent-max-tokens-handoff-policy.cjs +0 -69
- package/bin/subagent-parent-responsiveness.cjs +0 -19
- package/bin/subagent-skill-policy.cjs +0 -206
- package/bin/subagent-timeout-policy.cjs +0 -182
- package/bin/subagent-tool-policy.cjs +0 -60
- package/bin/subagent-usage-rollup-policy.cjs +0 -29
- package/bin/system-prompt-context-policy.cjs +0 -124
- package/bin/system-prompt-token-cache-policy.cjs +0 -60
- package/bin/telegram-addressed-focus.cjs +0 -55
- package/bin/telegram-addressed-priority.cjs +0 -12
- package/bin/telegram-approval-relay.cjs +0 -290
- package/bin/telegram-bot-priority.cjs +0 -17
- package/bin/telegram-console-status-policy.cjs +0 -174
- package/bin/telegram-context-projection-policy.cjs +0 -141
- package/bin/telegram-delivery-lifecycle.cjs +0 -125
- package/bin/telegram-direct-focus-policy.cjs +0 -273
- package/bin/telegram-mcp-compatibility.cjs +0 -49
- package/bin/telegram-media-delivery-policy.cjs +0 -42
- package/bin/telegram-private-conversation-policy.cjs +0 -185
- package/bin/telegram-queue-handoff-policy.cjs +0 -73
- package/bin/telegram-remote-status-policy.cjs +0 -120
- package/bin/telegram-session-queue-runtime.mjs +0 -306
- package/bin/telegram-text-chunk-policy.cjs +0 -63
- package/bin/telegram-truncated-reply-policy.cjs +0 -37
- package/bin/telegram-urgent-policy.cjs +0 -45
- package/bin/telemetry-spool-policy.cjs +0 -57
- package/bin/thinking-activity-status-policy.cjs +0 -132
- package/bin/thinking-only-guard.cjs +0 -80
- package/bin/todo-list-turn-policy.cjs +0 -131
- package/bin/tool-call-loop-policy.cjs +0 -51
- package/bin/tool-file-persistence.cjs +0 -141
- package/bin/tool-result-offload-policy.cjs +0 -359
- package/bin/tool-result-offload-telemetry.cjs +0 -12
- package/bin/tool-schema-token-cache-policy.cjs +0 -41
- package/bin/tool-stream-preview-policy.cjs +0 -9
- package/bin/tui-functional-contract.cjs +0 -55
- package/bin/turn-tool-performance-policy.cjs +0 -486
- package/bin/usage-cache-efficiency-policy.cjs +0 -26
- package/bin/user-home-path-policy.cjs +0 -13
- package/bin/user-message-offload-policy.cjs +0 -103
- package/bin/user-prompt-hook-origin-policy.cjs +0 -34
- package/bin/user-tool-record-policy.cjs +0 -7
- package/bin/validated-learning-insight-policy.cjs +0 -58
- package/bin/validated-learning-outcome-trace.cjs +0 -107
- package/bin/validated-learning-performance-policy.cjs +0 -53
- package/bin/validated-learning-signal.cjs +0 -463
- package/bin/windows-bash-dialect-policy.cjs +0 -25
- package/bin/windows-node-crash-dump.cjs +0 -110
- package/bin/write-continuation-policy.cjs +0 -69
- package/codebase-index/README.md +0 -82
- package/codebase-index/codebase_index.py +0 -470
- package/standard-skills/agent-browser/SKILL.md +0 -19
- package/standard-skills/agent-browser/references/runtime.md +0 -8
- package/standard-skills/blun-session-inspector/SKILL.md +0 -41
- package/standard-skills/blun-session-inspector/scripts/inspect-session.cjs +0 -437
- package/standard-skills/design-taste-frontend/SKILL.md +0 -1206
- package/standard-skills/full-output-enforcement/SKILL.md +0 -49
- package/standard-skills/high-end-visual-design/SKILL.md +0 -98
- package/standard-skills/image-to-code/SKILL.md +0 -1228
- package/standard-skills/industrial-brutalist-ui/SKILL.md +0 -92
- package/standard-skills/minimalist-ui/SKILL.md +0 -85
- package/standard-skills/motion-design-taste/SKILL.md +0 -74
- package/standard-skills/playwright-testing/SKILL.md +0 -19
- package/standard-skills/playwright-testing/references/runtime.md +0 -7
- package/standard-skills/premortem/SKILL.md +0 -148
- package/standard-skills/redesign-existing-projects/SKILL.md +0 -178
- package/standard-skills/research-evidence/SKILL.md +0 -39
- package/standard-skills/research-evidence/references/evidence-format.md +0 -104
- package/standard-skills/research-evidence/scripts/evidence-collection.cjs +0 -260
- package/standard-skills/research-evidence/scripts/score-report.cjs +0 -130
- package/standard-skills/screenshot-lesen/SKILL.md +0 -52
- package/standard-skills/stitch-design-taste/DESIGN.md +0 -121
- package/standard-skills/stitch-design-taste/SKILL.md +0 -184
- package/standard-skills/telegram-channel/SKILL.md +0 -18
- package/standard-skills/telegram-channel/references/runtime.md +0 -7
- package/standard-skills/venture-flywheel/SKILL.md +0 -32
- package/standard-skills/venture-flywheel/identity/project-identity.cjs +0 -146
- package/standard-skills/venture-flywheel/policy/capability-engine.cjs +0 -114
- package/standard-skills/venture-flywheel/policy/repository-trust.cjs +0 -229
- package/standard-skills/venture-flywheel/references/BEISPIELE-phase0.md +0 -146
- package/standard-skills/venture-flywheel/references/CAPABILITY-MAP.md +0 -34
- package/standard-skills/venture-flywheel/references/SPEC-phase0-identity-trust.md +0 -77
- package/standard-skills/venture-flywheel/references/SPEC-phase0-state-events.md +0 -93
- package/standard-skills/venture-flywheel/schemas/capability-decision.schema.json +0 -13
- package/standard-skills/venture-flywheel/schemas/execution-event.schema.json +0 -44
- package/standard-skills/venture-flywheel/schemas/project-identity.schema.json +0 -32
- package/standard-skills/venture-flywheel/schemas/repository-trust.schema.json +0 -57
- package/standard-skills/venture-flywheel/schemas/run-transition.schema.json +0 -59
- package/standard-skills/venture-flywheel/state/execution-event.cjs +0 -191
- package/standard-skills/venture-flywheel/state/task-state-machine.cjs +0 -190
- package/standard-skills/web-lesen/SKILL.md +0 -73
- package/standard-skills/web-lesen/scripts/crawl_public.py +0 -379
- package/standard-skills/windows-mcp/SKILL.md +0 -19
- package/standard-skills/windows-mcp/references/runtime.md +0 -9
- package/telegram-plugin/DELIVERY.md +0 -36
- package/telegram-plugin/bin/telegram-approval-relay.cjs +0 -290
- package/telegram-plugin/bin/telegram-console-status-policy.cjs +0 -175
- package/telegram-plugin/bin/telegram-delivery-lifecycle.cjs +0 -125
- package/telegram-plugin/bin/telegram-direct-reply-policy.cjs +0 -48
- package/telegram-plugin/bin/telegram-launcher-status-queue.cjs +0 -122
- package/telegram-plugin/bin/telegram-private-conversation-policy.cjs +0 -186
- package/telegram-plugin/bin/telegram-remote-status-policy.cjs +0 -121
- package/telegram-plugin/bin/telegram-reply-parts.cjs +0 -149
- package/telegram-plugin/bin/telegram-text-chunk-policy.cjs +0 -63
- package/telegram-plugin/bin/telegram-typing-keepalive.cjs +0 -89
- package/telegram-plugin/compat/mcp-server-fa511cd1.mjs +0 -73825
- /package/{bin → scripts}/fix-node-pty-perms.js +0 -0
package/standard-skills/translate-native/integrations/website_localization_benchmark_review_store.py
ADDED
|
@@ -0,0 +1,781 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Durable, attested storage for one anonymous benchmark review pass.
|
|
3
|
+
|
|
4
|
+
The campaign lease remains the concurrency boundary. This store makes each
|
|
5
|
+
successful reviewer response immutable before the campaign proceeds to the
|
|
6
|
+
next ordered phase, so a later retry reuses the exact response instead of
|
|
7
|
+
asking the reviewer to judge the same blind variants again.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import hashlib
|
|
13
|
+
import importlib.util
|
|
14
|
+
import json
|
|
15
|
+
import math
|
|
16
|
+
import re
|
|
17
|
+
import sqlite3
|
|
18
|
+
import sys
|
|
19
|
+
import time
|
|
20
|
+
from contextlib import contextmanager
|
|
21
|
+
from dataclasses import asdict, dataclass
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Any, Callable, Iterator, Mapping
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
STORE_SCHEMA = "blun.website-localization-benchmark-review-store.v1"
|
|
27
|
+
ARTIFACT_SCHEMA = "blun.website-localization-benchmark-review-evidence.v1"
|
|
28
|
+
HEALTH_SCHEMA = "blun.website-localization-benchmark-review-health.v1"
|
|
29
|
+
MAX_ARTIFACT_BYTES = 4 * 1024 * 1024
|
|
30
|
+
ERROR_CODE = re.compile(r"^[a-z][a-z0-9_.-]{0,127}$")
|
|
31
|
+
REVIEW_ID = re.compile(r"^benchmark-review-[0-9a-f]{64}$")
|
|
32
|
+
STORE_COLUMNS = (
|
|
33
|
+
"acquisition_id", "review_id", "route_id", "policy_sha256",
|
|
34
|
+
"request_sha256", "artifact_json", "artifact_sha256", "created_at",
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _load_module(name: str, path: Path):
|
|
39
|
+
spec = importlib.util.spec_from_file_location(name, path)
|
|
40
|
+
if spec is None or spec.loader is None:
|
|
41
|
+
raise RuntimeError(f"cannot load benchmark review dependency: {path.name}")
|
|
42
|
+
module = importlib.util.module_from_spec(spec)
|
|
43
|
+
sys.modules[spec.name] = module
|
|
44
|
+
spec.loader.exec_module(module)
|
|
45
|
+
return module
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
_ROOT = Path(__file__).resolve().parents[1]
|
|
49
|
+
_BENCHMARK = _load_module(
|
|
50
|
+
"blun_website_localization_review_store_benchmark",
|
|
51
|
+
_ROOT / "integrations" / "website_localization_benchmark.py",
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class BenchmarkReviewEvidenceFailed(RuntimeError):
|
|
56
|
+
"""Content-free failure understood by benchmark orchestration."""
|
|
57
|
+
|
|
58
|
+
benchmark_reviewer_failure = True
|
|
59
|
+
|
|
60
|
+
def __init__(self, code: str, *, retryable: bool):
|
|
61
|
+
if not isinstance(code, str) or ERROR_CODE.fullmatch(code) is None:
|
|
62
|
+
raise ValueError("benchmark review evidence code is invalid")
|
|
63
|
+
if not isinstance(retryable, bool):
|
|
64
|
+
raise ValueError("benchmark review evidence retryability is invalid")
|
|
65
|
+
super().__init__(code)
|
|
66
|
+
self.code = code
|
|
67
|
+
self.retryable = retryable
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass(frozen=True)
|
|
71
|
+
class BenchmarkReviewEvidenceHealth:
|
|
72
|
+
"""Content-free integrity summary for one active review route and policy."""
|
|
73
|
+
|
|
74
|
+
route_id: str
|
|
75
|
+
status: str
|
|
76
|
+
reasons: tuple[str, ...]
|
|
77
|
+
counts: tuple[tuple[str, int], ...]
|
|
78
|
+
|
|
79
|
+
def as_payload(self) -> dict[str, Any]:
|
|
80
|
+
return {
|
|
81
|
+
"schema": HEALTH_SCHEMA,
|
|
82
|
+
"route_id": self.route_id,
|
|
83
|
+
"status": self.status,
|
|
84
|
+
"reasons": list(self.reasons),
|
|
85
|
+
"counts": dict(self.counts),
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _pairs(items):
|
|
90
|
+
result = {}
|
|
91
|
+
for key, value in items:
|
|
92
|
+
if key in result:
|
|
93
|
+
raise ValueError("duplicate JSON key")
|
|
94
|
+
result[key] = value
|
|
95
|
+
return result
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _constant(value):
|
|
99
|
+
raise ValueError("non-finite JSON number")
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _json_bytes(value: Any, code: str) -> bytes:
|
|
103
|
+
try:
|
|
104
|
+
encoded = json.dumps(
|
|
105
|
+
value,
|
|
106
|
+
ensure_ascii=False,
|
|
107
|
+
allow_nan=False,
|
|
108
|
+
sort_keys=True,
|
|
109
|
+
separators=(",", ":"),
|
|
110
|
+
).encode("utf-8")
|
|
111
|
+
except (TypeError, ValueError, UnicodeEncodeError, RecursionError):
|
|
112
|
+
raise BenchmarkReviewEvidenceFailed(code, retryable=False) from None
|
|
113
|
+
if not encoded or len(encoded) > MAX_ARTIFACT_BYTES:
|
|
114
|
+
raise BenchmarkReviewEvidenceFailed(code, retryable=False)
|
|
115
|
+
return encoded
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _parse_json(value: Any) -> Any:
|
|
119
|
+
if not isinstance(value, str) or not value:
|
|
120
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
121
|
+
"review.store.state_invalid", retryable=False,
|
|
122
|
+
)
|
|
123
|
+
try:
|
|
124
|
+
encoded = value.encode("utf-8")
|
|
125
|
+
if len(encoded) > MAX_ARTIFACT_BYTES:
|
|
126
|
+
raise ValueError("stored review JSON is too large")
|
|
127
|
+
return json.loads(
|
|
128
|
+
value, object_pairs_hook=_pairs, parse_constant=_constant,
|
|
129
|
+
)
|
|
130
|
+
except (UnicodeEncodeError, ValueError, RecursionError):
|
|
131
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
132
|
+
"review.store.state_invalid", retryable=False,
|
|
133
|
+
) from None
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _hash_bytes(value: bytes) -> str:
|
|
137
|
+
return hashlib.sha256(value).hexdigest()
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _hash_text(value: str) -> str:
|
|
141
|
+
return _hash_bytes(value.encode("utf-8"))
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _timestamp(value: Any = None) -> float:
|
|
145
|
+
value = time.time() if value is None else value
|
|
146
|
+
if (
|
|
147
|
+
isinstance(value, bool)
|
|
148
|
+
or not isinstance(value, (int, float))
|
|
149
|
+
or not math.isfinite(float(value))
|
|
150
|
+
or float(value) < 0
|
|
151
|
+
):
|
|
152
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
153
|
+
"review.store.time_invalid", retryable=False,
|
|
154
|
+
)
|
|
155
|
+
return float(value)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _identifier(value: Any, code: str) -> str:
|
|
159
|
+
if not isinstance(value, str) or _BENCHMARK.IDENTIFIER.fullmatch(value) is None:
|
|
160
|
+
raise BenchmarkReviewEvidenceFailed(code, retryable=False)
|
|
161
|
+
return value
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _request_payload(value: Any) -> dict[str, Any]:
|
|
165
|
+
try:
|
|
166
|
+
payload = value.as_payload()
|
|
167
|
+
except Exception:
|
|
168
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
169
|
+
"review.store.request_invalid", retryable=False,
|
|
170
|
+
) from None
|
|
171
|
+
expected = {
|
|
172
|
+
"schema", "review_id", "phase", "target_locale",
|
|
173
|
+
"system_instruction", "input",
|
|
174
|
+
}
|
|
175
|
+
if not isinstance(payload, dict) or set(payload) != expected:
|
|
176
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
177
|
+
"review.store.request_invalid", retryable=False,
|
|
178
|
+
)
|
|
179
|
+
phase = payload.get("phase")
|
|
180
|
+
input_value = payload.get("input")
|
|
181
|
+
commercial_fidelity = (
|
|
182
|
+
phase == "source_fidelity"
|
|
183
|
+
and isinstance(input_value, dict)
|
|
184
|
+
and input_value.get("content_type") == "commercial"
|
|
185
|
+
)
|
|
186
|
+
expected_system = {
|
|
187
|
+
"target_native": _BENCHMARK._NATIVE_SYSTEM,
|
|
188
|
+
"source_fidelity": _BENCHMARK._FIDELITY_SYSTEM + (
|
|
189
|
+
"\n" + _BENCHMARK._COMMERCIAL_BENCHMARK_FIDELITY_SYSTEM
|
|
190
|
+
if commercial_fidelity else ""
|
|
191
|
+
),
|
|
192
|
+
}.get(phase)
|
|
193
|
+
commercial_dimensions = (
|
|
194
|
+
input_value.get("benchmark_suite", {}).get("commercial_dimensions")
|
|
195
|
+
if commercial_fidelity
|
|
196
|
+
and isinstance(input_value.get("benchmark_suite"), dict)
|
|
197
|
+
else None
|
|
198
|
+
)
|
|
199
|
+
if (
|
|
200
|
+
payload.get("schema") != _BENCHMARK.BENCHMARK_SCHEMA
|
|
201
|
+
or REVIEW_ID.fullmatch(payload.get("review_id", "")) is None
|
|
202
|
+
or expected_system is None
|
|
203
|
+
or payload.get("system_instruction") != expected_system
|
|
204
|
+
or not isinstance(payload.get("target_locale"), str)
|
|
205
|
+
or not isinstance(input_value, dict)
|
|
206
|
+
or input_value.get("blind_id") is None
|
|
207
|
+
or (
|
|
208
|
+
commercial_fidelity
|
|
209
|
+
and commercial_dimensions
|
|
210
|
+
!= list(_BENCHMARK._WORKER._COMMERCIAL.DIMENSIONS)
|
|
211
|
+
)
|
|
212
|
+
):
|
|
213
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
214
|
+
"review.store.request_invalid", retryable=False,
|
|
215
|
+
)
|
|
216
|
+
try:
|
|
217
|
+
return json.loads(_json_bytes(
|
|
218
|
+
payload, "review.store.request_invalid",
|
|
219
|
+
).decode("utf-8"))
|
|
220
|
+
except (UnicodeDecodeError, ValueError):
|
|
221
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
222
|
+
"review.store.request_invalid", retryable=False,
|
|
223
|
+
) from None
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _binding(
|
|
227
|
+
request: Any,
|
|
228
|
+
policy: Any,
|
|
229
|
+
route_id: Any,
|
|
230
|
+
) -> tuple[tuple[str, str, str, str, str], dict[str, Any], Any]:
|
|
231
|
+
route = _identifier(route_id, "review.store.route_invalid")
|
|
232
|
+
try:
|
|
233
|
+
validated_policy = _BENCHMARK._validate_policy(policy)
|
|
234
|
+
except Exception:
|
|
235
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
236
|
+
"review.store.policy_invalid", retryable=False,
|
|
237
|
+
) from None
|
|
238
|
+
payload = _request_payload(request)
|
|
239
|
+
if payload["target_locale"] not in validated_policy.required_locales:
|
|
240
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
241
|
+
"review.store.binding_invalid", retryable=False,
|
|
242
|
+
)
|
|
243
|
+
policy_sha256 = _hash_bytes(_json_bytes(
|
|
244
|
+
asdict(validated_policy), "review.store.policy_invalid",
|
|
245
|
+
))
|
|
246
|
+
request_sha256 = _hash_bytes(_json_bytes(
|
|
247
|
+
payload, "review.store.request_invalid",
|
|
248
|
+
))
|
|
249
|
+
acquisition_id = "benchmark-review-evidence:" + _hash_bytes(_json_bytes(
|
|
250
|
+
{
|
|
251
|
+
"schema": STORE_SCHEMA,
|
|
252
|
+
"review_id": payload["review_id"],
|
|
253
|
+
"route_id": route,
|
|
254
|
+
"policy_sha256": policy_sha256,
|
|
255
|
+
"request_sha256": request_sha256,
|
|
256
|
+
},
|
|
257
|
+
"review.store.binding_invalid",
|
|
258
|
+
))
|
|
259
|
+
return (
|
|
260
|
+
(
|
|
261
|
+
acquisition_id, payload["review_id"], route,
|
|
262
|
+
policy_sha256, request_sha256,
|
|
263
|
+
),
|
|
264
|
+
payload,
|
|
265
|
+
validated_policy,
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _response(value: Any, request: dict[str, Any]) -> dict[str, Any]:
|
|
270
|
+
if not isinstance(value, Mapping):
|
|
271
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
272
|
+
"review.store.response_invalid", retryable=False,
|
|
273
|
+
)
|
|
274
|
+
try:
|
|
275
|
+
response = json.loads(_json_bytes(
|
|
276
|
+
dict(value), "review.store.response_invalid",
|
|
277
|
+
).decode("utf-8"))
|
|
278
|
+
_BENCHMARK._validate_review(
|
|
279
|
+
response,
|
|
280
|
+
phase=request["phase"],
|
|
281
|
+
locale=request["target_locale"],
|
|
282
|
+
blind_id=request["input"]["blind_id"],
|
|
283
|
+
commercial_dimensions=(
|
|
284
|
+
request["input"].get("benchmark_suite", {}).get(
|
|
285
|
+
"commercial_dimensions",
|
|
286
|
+
)
|
|
287
|
+
if request["phase"] == "source_fidelity"
|
|
288
|
+
and request["input"].get("content_type") == "commercial"
|
|
289
|
+
and isinstance(request["input"].get("benchmark_suite"), dict)
|
|
290
|
+
else None
|
|
291
|
+
),
|
|
292
|
+
)
|
|
293
|
+
except BenchmarkReviewEvidenceFailed:
|
|
294
|
+
raise
|
|
295
|
+
except Exception:
|
|
296
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
297
|
+
"review.store.response_invalid", retryable=False,
|
|
298
|
+
) from None
|
|
299
|
+
return response
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _validate_artifact(
|
|
303
|
+
artifact: Any,
|
|
304
|
+
identity: tuple[str, str, str, str, str],
|
|
305
|
+
request: dict[str, Any],
|
|
306
|
+
policy: Any,
|
|
307
|
+
evidence_authority: Any,
|
|
308
|
+
) -> dict[str, Any]:
|
|
309
|
+
expected = {
|
|
310
|
+
"schema", "acquisition_id", "review_id", "route_id",
|
|
311
|
+
"policy_sha256", "request_sha256", "response_sha256", "reviewer",
|
|
312
|
+
"response", "attestation",
|
|
313
|
+
}
|
|
314
|
+
if not isinstance(artifact, dict) or set(artifact) != expected:
|
|
315
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
316
|
+
"review.store.artifact_invalid", retryable=False,
|
|
317
|
+
)
|
|
318
|
+
unsigned = {key: value for key, value in artifact.items() if key != "attestation"}
|
|
319
|
+
if (
|
|
320
|
+
unsigned["schema"] != ARTIFACT_SCHEMA
|
|
321
|
+
or tuple(unsigned[name] for name in STORE_COLUMNS[:5]) != identity
|
|
322
|
+
or unsigned["reviewer"] != {
|
|
323
|
+
"id": policy.reviewer_id,
|
|
324
|
+
"version": policy.reviewer_version,
|
|
325
|
+
}
|
|
326
|
+
):
|
|
327
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
328
|
+
"review.store.artifact_invalid", retryable=False,
|
|
329
|
+
)
|
|
330
|
+
unsigned["response"] = _response(unsigned["response"], request)
|
|
331
|
+
if unsigned["response_sha256"] != _hash_bytes(_json_bytes(
|
|
332
|
+
unsigned["response"], "review.store.artifact_invalid",
|
|
333
|
+
)):
|
|
334
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
335
|
+
"review.store.artifact_invalid", retryable=False,
|
|
336
|
+
)
|
|
337
|
+
try:
|
|
338
|
+
_BENCHMARK._verify_attestation(
|
|
339
|
+
unsigned, artifact["attestation"], policy, evidence_authority,
|
|
340
|
+
)
|
|
341
|
+
except _BENCHMARK.BenchmarkBlocked as error:
|
|
342
|
+
retryable = error.code == "benchmark.attestation.verify_failed"
|
|
343
|
+
code = (
|
|
344
|
+
"review.store.attestation_unavailable"
|
|
345
|
+
if retryable else "review.store.artifact_invalid"
|
|
346
|
+
)
|
|
347
|
+
raise BenchmarkReviewEvidenceFailed(code, retryable=retryable) from None
|
|
348
|
+
except Exception:
|
|
349
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
350
|
+
"review.store.artifact_invalid", retryable=False,
|
|
351
|
+
) from None
|
|
352
|
+
verified = dict(unsigned)
|
|
353
|
+
verified["attestation"] = artifact["attestation"]
|
|
354
|
+
return json.loads(_json_bytes(
|
|
355
|
+
verified, "review.store.artifact_invalid",
|
|
356
|
+
).decode("utf-8"))
|
|
357
|
+
|
|
358
|
+
|
|
359
|
+
def _create_artifact(
|
|
360
|
+
identity: tuple[str, str, str, str, str],
|
|
361
|
+
response: Any,
|
|
362
|
+
request: dict[str, Any],
|
|
363
|
+
policy: Any,
|
|
364
|
+
evidence_authority: Any,
|
|
365
|
+
) -> dict[str, Any]:
|
|
366
|
+
validated_response = _response(response, request)
|
|
367
|
+
unsigned = {
|
|
368
|
+
"schema": ARTIFACT_SCHEMA,
|
|
369
|
+
"acquisition_id": identity[0],
|
|
370
|
+
"review_id": identity[1],
|
|
371
|
+
"route_id": identity[2],
|
|
372
|
+
"policy_sha256": identity[3],
|
|
373
|
+
"request_sha256": identity[4],
|
|
374
|
+
"response_sha256": _hash_bytes(_json_bytes(
|
|
375
|
+
validated_response, "review.store.response_invalid",
|
|
376
|
+
)),
|
|
377
|
+
"reviewer": {
|
|
378
|
+
"id": policy.reviewer_id,
|
|
379
|
+
"version": policy.reviewer_version,
|
|
380
|
+
},
|
|
381
|
+
"response": validated_response,
|
|
382
|
+
}
|
|
383
|
+
try:
|
|
384
|
+
artifact = _BENCHMARK._attest(unsigned, policy, evidence_authority)
|
|
385
|
+
except _BENCHMARK.BenchmarkBlocked as error:
|
|
386
|
+
retryable = error.code in {
|
|
387
|
+
"benchmark.attestation.sign_failed",
|
|
388
|
+
"benchmark.attestation.verify_failed",
|
|
389
|
+
}
|
|
390
|
+
code = (
|
|
391
|
+
"review.store.attestation_unavailable"
|
|
392
|
+
if retryable else "review.store.artifact_invalid"
|
|
393
|
+
)
|
|
394
|
+
raise BenchmarkReviewEvidenceFailed(code, retryable=retryable) from None
|
|
395
|
+
return _validate_artifact(
|
|
396
|
+
artifact, identity, request, policy, evidence_authority,
|
|
397
|
+
)
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
@contextmanager
|
|
401
|
+
def _transaction(connection: sqlite3.Connection) -> Iterator[None]:
|
|
402
|
+
if connection.in_transaction:
|
|
403
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
404
|
+
"review.store.external_transaction", retryable=False,
|
|
405
|
+
)
|
|
406
|
+
try:
|
|
407
|
+
connection.execute("BEGIN IMMEDIATE")
|
|
408
|
+
yield
|
|
409
|
+
except Exception:
|
|
410
|
+
connection.rollback()
|
|
411
|
+
raise
|
|
412
|
+
else:
|
|
413
|
+
connection.commit()
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
class BenchmarkReviewEvidenceStore:
|
|
417
|
+
"""Persist the first exact attested response for one bound review pass."""
|
|
418
|
+
|
|
419
|
+
def __init__(self, connection: sqlite3.Connection):
|
|
420
|
+
if not isinstance(connection, sqlite3.Connection):
|
|
421
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
422
|
+
"review.store.connection_invalid", retryable=False,
|
|
423
|
+
)
|
|
424
|
+
self.connection = connection
|
|
425
|
+
self.connection.row_factory = sqlite3.Row
|
|
426
|
+
self.connection.execute("PRAGMA busy_timeout = 5000")
|
|
427
|
+
with _transaction(self.connection):
|
|
428
|
+
self.connection.execute("""
|
|
429
|
+
CREATE TABLE IF NOT EXISTS benchmark_review_evidence (
|
|
430
|
+
acquisition_id TEXT PRIMARY KEY,
|
|
431
|
+
review_id TEXT NOT NULL,
|
|
432
|
+
route_id TEXT NOT NULL,
|
|
433
|
+
policy_sha256 TEXT NOT NULL,
|
|
434
|
+
request_sha256 TEXT NOT NULL,
|
|
435
|
+
artifact_json TEXT NOT NULL,
|
|
436
|
+
artifact_sha256 TEXT NOT NULL,
|
|
437
|
+
created_at REAL NOT NULL,
|
|
438
|
+
UNIQUE(review_id, route_id, policy_sha256)
|
|
439
|
+
)
|
|
440
|
+
""")
|
|
441
|
+
self._verify_schema()
|
|
442
|
+
|
|
443
|
+
def _verify_schema(self) -> None:
|
|
444
|
+
columns = tuple(
|
|
445
|
+
row[1] for row in self.connection.execute(
|
|
446
|
+
"PRAGMA table_info(benchmark_review_evidence)"
|
|
447
|
+
).fetchall()
|
|
448
|
+
)
|
|
449
|
+
if columns != STORE_COLUMNS:
|
|
450
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
451
|
+
"review.store.schema_unsupported", retryable=False,
|
|
452
|
+
)
|
|
453
|
+
|
|
454
|
+
def health(
|
|
455
|
+
self,
|
|
456
|
+
policy: Any,
|
|
457
|
+
route_id: Any,
|
|
458
|
+
*,
|
|
459
|
+
evidence_authority: Any,
|
|
460
|
+
expected_passes: Any = (),
|
|
461
|
+
now: Any = None,
|
|
462
|
+
) -> BenchmarkReviewEvidenceHealth:
|
|
463
|
+
"""Verify stored evidence without returning review text or changing state."""
|
|
464
|
+
counts = {
|
|
465
|
+
"total": 0,
|
|
466
|
+
"scoped": 0,
|
|
467
|
+
"historical": 0,
|
|
468
|
+
"target_native": 0,
|
|
469
|
+
"source_fidelity": 0,
|
|
470
|
+
"required": 0,
|
|
471
|
+
"matched": 0,
|
|
472
|
+
}
|
|
473
|
+
reasons: set[str] = set()
|
|
474
|
+
safe_route = route_id if isinstance(route_id, str) else "invalid"
|
|
475
|
+
try:
|
|
476
|
+
safe_route = _identifier(route_id, "review.store.route_invalid")
|
|
477
|
+
validated_policy = _BENCHMARK._validate_policy(policy)
|
|
478
|
+
policy_sha256 = _hash_bytes(_json_bytes(
|
|
479
|
+
asdict(validated_policy), "review.store.policy_invalid",
|
|
480
|
+
))
|
|
481
|
+
checked_at = _timestamp(now)
|
|
482
|
+
if self.connection.in_transaction:
|
|
483
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
484
|
+
"review.store.external_transaction", retryable=False,
|
|
485
|
+
)
|
|
486
|
+
if not isinstance(expected_passes, (tuple, list)):
|
|
487
|
+
raise ValueError
|
|
488
|
+
required: dict[str, tuple[str, str]] = {}
|
|
489
|
+
for item in expected_passes:
|
|
490
|
+
if not isinstance(item, Mapping) or set(item) != {
|
|
491
|
+
"phase", "request_sha256", "response_sha256",
|
|
492
|
+
}:
|
|
493
|
+
raise ValueError
|
|
494
|
+
phase = item["phase"]
|
|
495
|
+
request_sha256 = item["request_sha256"]
|
|
496
|
+
response_sha256 = item["response_sha256"]
|
|
497
|
+
if (
|
|
498
|
+
phase not in _BENCHMARK.PHASES
|
|
499
|
+
or re.fullmatch(r"[0-9a-f]{64}", request_sha256 or "") is None
|
|
500
|
+
or re.fullmatch(r"[0-9a-f]{64}", response_sha256 or "") is None
|
|
501
|
+
or request_sha256 in required
|
|
502
|
+
):
|
|
503
|
+
raise ValueError
|
|
504
|
+
required[request_sha256] = (phase, response_sha256)
|
|
505
|
+
counts["required"] = len(required)
|
|
506
|
+
observed: dict[str, tuple[str, str]] = {}
|
|
507
|
+
self._verify_schema()
|
|
508
|
+
rows = self.connection.execute(
|
|
509
|
+
"SELECT * FROM benchmark_review_evidence "
|
|
510
|
+
"ORDER BY route_id, policy_sha256, review_id"
|
|
511
|
+
).fetchall()
|
|
512
|
+
counts["total"] = len(rows)
|
|
513
|
+
for row in rows:
|
|
514
|
+
if tuple(row.keys()) != STORE_COLUMNS:
|
|
515
|
+
raise ValueError
|
|
516
|
+
scoped = (
|
|
517
|
+
row["route_id"] == safe_route
|
|
518
|
+
and row["policy_sha256"] == policy_sha256
|
|
519
|
+
)
|
|
520
|
+
if not scoped:
|
|
521
|
+
counts["historical"] += 1
|
|
522
|
+
continue
|
|
523
|
+
counts["scoped"] += 1
|
|
524
|
+
created_at = _timestamp(row["created_at"])
|
|
525
|
+
if created_at > checked_at:
|
|
526
|
+
raise ValueError
|
|
527
|
+
identity = tuple(row[name] for name in STORE_COLUMNS[:5])
|
|
528
|
+
if (
|
|
529
|
+
not isinstance(row["artifact_json"], str)
|
|
530
|
+
or row["artifact_sha256"] != _hash_text(row["artifact_json"])
|
|
531
|
+
or REVIEW_ID.fullmatch(row["review_id"] or "") is None
|
|
532
|
+
or re.fullmatch(r"[0-9a-f]{64}", row["request_sha256"] or "") is None
|
|
533
|
+
):
|
|
534
|
+
raise ValueError
|
|
535
|
+
artifact = _parse_json(row["artifact_json"])
|
|
536
|
+
if _json_bytes(
|
|
537
|
+
artifact, "review.store.artifact_invalid",
|
|
538
|
+
).decode("utf-8") != row["artifact_json"]:
|
|
539
|
+
raise ValueError
|
|
540
|
+
response = artifact.get("response") if isinstance(artifact, dict) else None
|
|
541
|
+
if not isinstance(response, dict):
|
|
542
|
+
raise ValueError
|
|
543
|
+
phase = response.get("phase")
|
|
544
|
+
locale = response.get("target_locale")
|
|
545
|
+
blind_id = response.get("blind_id")
|
|
546
|
+
if (
|
|
547
|
+
phase not in _BENCHMARK.PHASES
|
|
548
|
+
or locale not in validated_policy.required_locales
|
|
549
|
+
or re.fullmatch(r"blind-[0-9a-f]{64}", blind_id or "") is None
|
|
550
|
+
):
|
|
551
|
+
raise ValueError
|
|
552
|
+
request_stub = {
|
|
553
|
+
"phase": phase,
|
|
554
|
+
"target_locale": locale,
|
|
555
|
+
"input": {"blind_id": blind_id},
|
|
556
|
+
}
|
|
557
|
+
verified = _validate_artifact(
|
|
558
|
+
artifact, identity, request_stub, validated_policy,
|
|
559
|
+
evidence_authority,
|
|
560
|
+
)
|
|
561
|
+
response_sha256 = verified["response_sha256"]
|
|
562
|
+
if row["request_sha256"] in observed:
|
|
563
|
+
raise ValueError
|
|
564
|
+
observed[row["request_sha256"]] = (phase, response_sha256)
|
|
565
|
+
counts[phase] += 1
|
|
566
|
+
for request_sha256, expected in required.items():
|
|
567
|
+
actual = observed.get(request_sha256)
|
|
568
|
+
if actual is None:
|
|
569
|
+
reasons.add("review.store.required_missing")
|
|
570
|
+
elif actual != expected:
|
|
571
|
+
reasons.add("review.store.required_mismatch")
|
|
572
|
+
else:
|
|
573
|
+
counts["matched"] += 1
|
|
574
|
+
except BenchmarkReviewEvidenceFailed as error:
|
|
575
|
+
reasons = {error.code}
|
|
576
|
+
except Exception:
|
|
577
|
+
reasons = {"review.store.state_invalid"}
|
|
578
|
+
status = "blocked" if reasons else "healthy"
|
|
579
|
+
return BenchmarkReviewEvidenceHealth(
|
|
580
|
+
route_id=safe_route,
|
|
581
|
+
status=status,
|
|
582
|
+
reasons=tuple(sorted(reasons)),
|
|
583
|
+
counts=tuple(sorted(counts.items())),
|
|
584
|
+
)
|
|
585
|
+
|
|
586
|
+
def _row_for_identity(
|
|
587
|
+
self, identity: tuple[str, str, str, str, str],
|
|
588
|
+
) -> sqlite3.Row | None:
|
|
589
|
+
row = self.connection.execute("""
|
|
590
|
+
SELECT * FROM benchmark_review_evidence
|
|
591
|
+
WHERE review_id = ? AND route_id = ? AND policy_sha256 = ?
|
|
592
|
+
""", (identity[1], identity[2], identity[3])).fetchone()
|
|
593
|
+
if row is not None and tuple(row[name] for name in STORE_COLUMNS[:5]) != identity:
|
|
594
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
595
|
+
"review.store.conflict", retryable=False,
|
|
596
|
+
)
|
|
597
|
+
return row
|
|
598
|
+
|
|
599
|
+
def load(
|
|
600
|
+
self,
|
|
601
|
+
request: Any,
|
|
602
|
+
policy: Any,
|
|
603
|
+
route_id: Any,
|
|
604
|
+
*,
|
|
605
|
+
evidence_authority: Any,
|
|
606
|
+
) -> dict[str, Any] | None:
|
|
607
|
+
identity, request_payload, validated_policy = _binding(
|
|
608
|
+
request, policy, route_id,
|
|
609
|
+
)
|
|
610
|
+
row = self._row_for_identity(identity)
|
|
611
|
+
if row is None:
|
|
612
|
+
return None
|
|
613
|
+
if (
|
|
614
|
+
tuple(row.keys()) != STORE_COLUMNS
|
|
615
|
+
or isinstance(row["created_at"], bool)
|
|
616
|
+
or not isinstance(row["created_at"], (int, float))
|
|
617
|
+
or not math.isfinite(float(row["created_at"]))
|
|
618
|
+
or float(row["created_at"]) < 0
|
|
619
|
+
or not isinstance(row["artifact_json"], str)
|
|
620
|
+
or row["artifact_sha256"] != _hash_text(row["artifact_json"])
|
|
621
|
+
):
|
|
622
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
623
|
+
"review.store.state_invalid", retryable=False,
|
|
624
|
+
)
|
|
625
|
+
artifact = _parse_json(row["artifact_json"])
|
|
626
|
+
verified = _validate_artifact(
|
|
627
|
+
artifact, identity, request_payload, validated_policy,
|
|
628
|
+
evidence_authority,
|
|
629
|
+
)
|
|
630
|
+
return verified["response"]
|
|
631
|
+
|
|
632
|
+
def save(
|
|
633
|
+
self,
|
|
634
|
+
request: Any,
|
|
635
|
+
policy: Any,
|
|
636
|
+
route_id: Any,
|
|
637
|
+
response: Any,
|
|
638
|
+
*,
|
|
639
|
+
evidence_authority: Any,
|
|
640
|
+
now: Any = None,
|
|
641
|
+
) -> dict[str, Any]:
|
|
642
|
+
identity, request_payload, validated_policy = _binding(
|
|
643
|
+
request, policy, route_id,
|
|
644
|
+
)
|
|
645
|
+
artifact = _create_artifact(
|
|
646
|
+
identity, response, request_payload, validated_policy,
|
|
647
|
+
evidence_authority,
|
|
648
|
+
)
|
|
649
|
+
artifact_json = _json_bytes(
|
|
650
|
+
artifact, "review.store.artifact_invalid",
|
|
651
|
+
).decode("utf-8")
|
|
652
|
+
values = identity + (
|
|
653
|
+
artifact_json,
|
|
654
|
+
_hash_text(artifact_json),
|
|
655
|
+
_timestamp(now),
|
|
656
|
+
)
|
|
657
|
+
with _transaction(self.connection):
|
|
658
|
+
self.connection.execute("""
|
|
659
|
+
INSERT OR IGNORE INTO benchmark_review_evidence
|
|
660
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
|
661
|
+
""", values)
|
|
662
|
+
stored = self.load(
|
|
663
|
+
request, policy, route_id, evidence_authority=evidence_authority,
|
|
664
|
+
)
|
|
665
|
+
if stored != artifact["response"]:
|
|
666
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
667
|
+
"review.store.conflict", retryable=False,
|
|
668
|
+
)
|
|
669
|
+
return stored
|
|
670
|
+
|
|
671
|
+
|
|
672
|
+
class _GuardedAuthority:
|
|
673
|
+
def __init__(self, authority: Any, guard: Callable[[], None]):
|
|
674
|
+
self.authority = authority
|
|
675
|
+
self.guard = guard
|
|
676
|
+
|
|
677
|
+
def sign(self, payload: bytes):
|
|
678
|
+
self.guard()
|
|
679
|
+
return self.authority.sign(payload)
|
|
680
|
+
|
|
681
|
+
def verify(self, payload: bytes, signature: Any):
|
|
682
|
+
self.guard()
|
|
683
|
+
return self.authority.verify(payload, signature)
|
|
684
|
+
|
|
685
|
+
|
|
686
|
+
class DurableBenchmarkReviewer:
|
|
687
|
+
"""Reuse a valid review response or obtain and attest it exactly once."""
|
|
688
|
+
|
|
689
|
+
def __init__(
|
|
690
|
+
self,
|
|
691
|
+
*,
|
|
692
|
+
store: BenchmarkReviewEvidenceStore,
|
|
693
|
+
policy: Any,
|
|
694
|
+
route_id: Any,
|
|
695
|
+
reviewer: Any,
|
|
696
|
+
evidence_authority: Any,
|
|
697
|
+
operation_guard: Callable[[], Any] | None = None,
|
|
698
|
+
clock: Callable[[], float] = time.time,
|
|
699
|
+
):
|
|
700
|
+
if any(not callable(getattr(store, name, None)) for name in ("load", "save")):
|
|
701
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
702
|
+
"review.store.invalid", retryable=False,
|
|
703
|
+
)
|
|
704
|
+
if not callable(getattr(reviewer, "review", None)):
|
|
705
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
706
|
+
"review.adapter.invalid", retryable=False,
|
|
707
|
+
)
|
|
708
|
+
if any(
|
|
709
|
+
not callable(getattr(evidence_authority, name, None))
|
|
710
|
+
for name in ("sign", "verify")
|
|
711
|
+
):
|
|
712
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
713
|
+
"review.authority.invalid", retryable=False,
|
|
714
|
+
)
|
|
715
|
+
if operation_guard is not None and not callable(operation_guard):
|
|
716
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
717
|
+
"review.operation_guard_invalid", retryable=False,
|
|
718
|
+
)
|
|
719
|
+
if not callable(clock):
|
|
720
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
721
|
+
"review.clock_invalid", retryable=False,
|
|
722
|
+
)
|
|
723
|
+
self.store = store
|
|
724
|
+
self.policy = _BENCHMARK._validate_policy(policy)
|
|
725
|
+
self.route_id = _identifier(route_id, "review.store.route_invalid")
|
|
726
|
+
self.reviewer = reviewer
|
|
727
|
+
self.evidence_authority = evidence_authority
|
|
728
|
+
self.operation_guard = operation_guard
|
|
729
|
+
self.clock = clock
|
|
730
|
+
|
|
731
|
+
def _guard(self) -> None:
|
|
732
|
+
if self.operation_guard is None:
|
|
733
|
+
return
|
|
734
|
+
try:
|
|
735
|
+
self.operation_guard()
|
|
736
|
+
except BenchmarkReviewEvidenceFailed:
|
|
737
|
+
raise
|
|
738
|
+
except Exception:
|
|
739
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
740
|
+
"review.operation_guard_failed", retryable=True,
|
|
741
|
+
) from None
|
|
742
|
+
|
|
743
|
+
def review(self, request: Any) -> Mapping[str, Any]:
|
|
744
|
+
guarded_authority = _GuardedAuthority(
|
|
745
|
+
self.evidence_authority, self._guard,
|
|
746
|
+
)
|
|
747
|
+
cached = self.store.load(
|
|
748
|
+
request,
|
|
749
|
+
self.policy,
|
|
750
|
+
self.route_id,
|
|
751
|
+
evidence_authority=guarded_authority,
|
|
752
|
+
)
|
|
753
|
+
if cached is not None:
|
|
754
|
+
return cached
|
|
755
|
+
before = _request_payload(request)
|
|
756
|
+
self._guard()
|
|
757
|
+
try:
|
|
758
|
+
response = self.reviewer.review(request)
|
|
759
|
+
except Exception as error:
|
|
760
|
+
if getattr(error, "benchmark_reviewer_failure", None) is True:
|
|
761
|
+
raise
|
|
762
|
+
retryable = getattr(error, "retryable", True)
|
|
763
|
+
if not isinstance(retryable, bool):
|
|
764
|
+
retryable = True
|
|
765
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
766
|
+
"review.adapter_unavailable", retryable=retryable,
|
|
767
|
+
) from None
|
|
768
|
+
after = _request_payload(request)
|
|
769
|
+
if after != before:
|
|
770
|
+
raise BenchmarkReviewEvidenceFailed(
|
|
771
|
+
"review.request_mutated", retryable=False,
|
|
772
|
+
)
|
|
773
|
+
self._guard()
|
|
774
|
+
return self.store.save(
|
|
775
|
+
request,
|
|
776
|
+
self.policy,
|
|
777
|
+
self.route_id,
|
|
778
|
+
response,
|
|
779
|
+
evidence_authority=guarded_authority,
|
|
780
|
+
now=self.clock(),
|
|
781
|
+
)
|