@wardby/cli 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.env.example +131 -0
- package/LICENSE +202 -0
- package/NOTICE +2 -0
- package/README.md +282 -0
- package/bin/wardby.js +3 -0
- package/deploy/local/docker-compose.yml +19 -0
- package/dist/claude-coding-worker/driver.d.ts +70 -0
- package/dist/claude-coding-worker/driver.js +128 -0
- package/dist/claude-coding-worker/main.d.ts +2 -0
- package/dist/claude-coding-worker/main.js +41 -0
- package/dist/claude-coding-worker/sdk.d.ts +4 -0
- package/dist/claude-coding-worker/sdk.js +43 -0
- package/dist/claude-coding-worker/tool-relay.d.ts +1 -0
- package/dist/claude-coding-worker/tool-relay.js +9 -0
- package/dist/cli-help.d.ts +1 -0
- package/dist/cli-help.js +20 -0
- package/dist/cli.d.ts +18 -0
- package/dist/cli.js +629 -0
- package/dist/coding/observability.d.ts +59 -0
- package/dist/coding/observability.js +77 -0
- package/dist/coding/profile.d.ts +88 -0
- package/dist/coding/profile.js +137 -0
- package/dist/coding/protocol.d.ts +345 -0
- package/dist/coding/protocol.js +445 -0
- package/dist/coding/provider.d.ts +4 -0
- package/dist/coding/provider.js +18 -0
- package/dist/coding-proxy/main.d.ts +1 -0
- package/dist/coding-proxy/main.js +51 -0
- package/dist/coding-worker/artifact.d.ts +16 -0
- package/dist/coding-worker/artifact.js +37 -0
- package/dist/coding-worker/driver.d.ts +46 -0
- package/dist/coding-worker/driver.js +158 -0
- package/dist/coding-worker/errors.d.ts +1 -0
- package/dist/coding-worker/errors.js +23 -0
- package/dist/coding-worker/isolation-probe.d.ts +1 -0
- package/dist/coding-worker/isolation-probe.js +146 -0
- package/dist/coding-worker/keeper.d.ts +4 -0
- package/dist/coding-worker/keeper.js +29 -0
- package/dist/coding-worker/main.d.ts +2 -0
- package/dist/coding-worker/main.js +43 -0
- package/dist/coding-worker/sdk.d.ts +26 -0
- package/dist/coding-worker/sdk.js +28 -0
- package/dist/coding-worker/types.d.ts +64 -0
- package/dist/coding-worker/types.js +1 -0
- package/dist/config/providers.d.ts +158 -0
- package/dist/config/providers.js +162 -0
- package/dist/core/budget-groups.d.ts +72 -0
- package/dist/core/budget-groups.js +149 -0
- package/dist/core/budget.d.ts +63 -0
- package/dist/core/budget.js +69 -0
- package/dist/core/coding-queue.d.ts +32 -0
- package/dist/core/coding-queue.js +54 -0
- package/dist/core/cron.d.ts +31 -0
- package/dist/core/cron.js +43 -0
- package/dist/core/datastores.d.ts +41 -0
- package/dist/core/datastores.js +105 -0
- package/dist/core/db.d.ts +2 -0
- package/dist/core/db.js +2 -0
- package/dist/core/dispatch.d.ts +81 -0
- package/dist/core/dispatch.js +184 -0
- package/dist/core/engine-native.d.ts +29 -0
- package/dist/core/engine-native.js +259 -0
- package/dist/core/http-runtime.d.ts +7 -0
- package/dist/core/http-runtime.js +101 -0
- package/dist/core/lease.d.ts +13 -0
- package/dist/core/lease.js +22 -0
- package/dist/core/logger.d.ts +13 -0
- package/dist/core/logger.js +45 -0
- package/dist/core/memory-tools.d.ts +11 -0
- package/dist/core/memory-tools.js +122 -0
- package/dist/core/reconciler.d.ts +45 -0
- package/dist/core/reconciler.js +152 -0
- package/dist/core/runner.d.ts +87 -0
- package/dist/core/runner.js +529 -0
- package/dist/core/scheduler.d.ts +47 -0
- package/dist/core/scheduler.js +121 -0
- package/dist/core/secrets.d.ts +46 -0
- package/dist/core/secrets.js +100 -0
- package/dist/core/subagent-memory-tools.d.ts +18 -0
- package/dist/core/subagent-memory-tools.js +121 -0
- package/dist/core/timing.d.ts +7 -0
- package/dist/core/timing.js +7 -0
- package/dist/core/webhooks.d.ts +23 -0
- package/dist/core/webhooks.js +113 -0
- package/dist/env.d.ts +11 -0
- package/dist/env.js +13 -0
- package/dist/import/budgets.d.ts +16 -0
- package/dist/import/budgets.js +21 -0
- package/dist/import/bundle.d.ts +16 -0
- package/dist/import/bundle.js +61 -0
- package/dist/import/cli-args.d.ts +2 -0
- package/dist/import/cli-args.js +44 -0
- package/dist/import/create.d.ts +34 -0
- package/dist/import/create.js +363 -0
- package/dist/import/index.d.ts +25 -0
- package/dist/import/index.js +130 -0
- package/dist/import/model-gate.d.ts +1 -0
- package/dist/import/model-gate.js +3 -0
- package/dist/import/neutral-schema.d.ts +369 -0
- package/dist/import/neutral-schema.js +109 -0
- package/dist/import/preflight.d.ts +41 -0
- package/dist/import/preflight.js +71 -0
- package/dist/import/report.d.ts +5 -0
- package/dist/import/report.js +98 -0
- package/dist/import/tool-scan.d.ts +8 -0
- package/dist/import/tool-scan.js +20 -0
- package/dist/mcp/auth/ownership.d.ts +113 -0
- package/dist/mcp/auth/ownership.js +87 -0
- package/dist/mcp/auth/principal.d.ts +9 -0
- package/dist/mcp/auth/principal.js +9 -0
- package/dist/mcp/auth/resource-server.d.ts +47 -0
- package/dist/mcp/auth/resource-server.js +82 -0
- package/dist/mcp/auth/self-hosted/browser.d.ts +3 -0
- package/dist/mcp/auth/self-hosted/browser.js +273 -0
- package/dist/mcp/auth/self-hosted/cli.d.ts +2 -0
- package/dist/mcp/auth/self-hosted/cli.js +39 -0
- package/dist/mcp/auth/self-hosted/credentials.d.ts +58 -0
- package/dist/mcp/auth/self-hosted/credentials.js +106 -0
- package/dist/mcp/auth/self-hosted/rate-limit.d.ts +13 -0
- package/dist/mcp/auth/self-hosted/rate-limit.js +20 -0
- package/dist/mcp/auth/self-hosted/session.d.ts +36 -0
- package/dist/mcp/auth/self-hosted/session.js +110 -0
- package/dist/mcp/capabilities.d.ts +11 -0
- package/dist/mcp/capabilities.js +10 -0
- package/dist/mcp/context.d.ts +46 -0
- package/dist/mcp/context.js +1 -0
- package/dist/mcp/errors.d.ts +13 -0
- package/dist/mcp/errors.js +22 -0
- package/dist/mcp/index.d.ts +32 -0
- package/dist/mcp/index.js +187 -0
- package/dist/mcp/jsonrpc.d.ts +8 -0
- package/dist/mcp/jsonrpc.js +1 -0
- package/dist/mcp/server.d.ts +107 -0
- package/dist/mcp/server.js +203 -0
- package/dist/mcp/tasks/manager.d.ts +63 -0
- package/dist/mcp/tasks/manager.js +103 -0
- package/dist/mcp/tools/agents.d.ts +2 -0
- package/dist/mcp/tools/agents.js +375 -0
- package/dist/mcp/tools/budget-groups.d.ts +2 -0
- package/dist/mcp/tools/budget-groups.js +136 -0
- package/dist/mcp/tools/datastore.d.ts +2 -0
- package/dist/mcp/tools/datastore.js +217 -0
- package/dist/mcp/tools/memory.d.ts +3 -0
- package/dist/mcp/tools/memory.js +60 -0
- package/dist/mcp/tools/models.d.ts +8 -0
- package/dist/mcp/tools/models.js +15 -0
- package/dist/mcp/tools/runs.d.ts +2 -0
- package/dist/mcp/tools/runs.js +61 -0
- package/dist/mcp/tools/scheduling.d.ts +2 -0
- package/dist/mcp/tools/scheduling.js +48 -0
- package/dist/mcp/tools/secret-elicitation-form.d.ts +22 -0
- package/dist/mcp/tools/secret-elicitation-form.js +93 -0
- package/dist/mcp/tools/secret-elicitation-server.d.ts +5 -0
- package/dist/mcp/tools/secret-elicitation-server.js +57 -0
- package/dist/mcp/tools/secret-elicitation.d.ts +38 -0
- package/dist/mcp/tools/secret-elicitation.js +52 -0
- package/dist/mcp/tools/secrets.d.ts +13 -0
- package/dist/mcp/tools/secrets.js +124 -0
- package/dist/mcp/tools/subagents.d.ts +2 -0
- package/dist/mcp/tools/subagents.js +133 -0
- package/dist/mcp/tools/text-result.d.ts +6 -0
- package/dist/mcp/tools/text-result.js +3 -0
- package/dist/mcp/tools/tools.d.ts +2 -0
- package/dist/mcp/tools/tools.js +211 -0
- package/dist/mcp/tools/trigger.d.ts +2 -0
- package/dist/mcp/tools/trigger.js +120 -0
- package/dist/mcp/tools/webhooks.d.ts +2 -0
- package/dist/mcp/tools/webhooks.js +34 -0
- package/dist/mcp/transport/http-limits.d.ts +17 -0
- package/dist/mcp/transport/http-limits.js +92 -0
- package/dist/mcp/transport/stdio.d.ts +23 -0
- package/dist/mcp/transport/stdio.js +6 -0
- package/dist/mcp/transport/streamable-http.d.ts +34 -0
- package/dist/mcp/transport/streamable-http.js +158 -0
- package/dist/mcp/unattended-schedules.d.ts +15 -0
- package/dist/mcp/unattended-schedules.js +21 -0
- package/dist/mcp/webhooks/ingress.d.ts +17 -0
- package/dist/mcp/webhooks/ingress.js +33 -0
- package/dist/observability/config.d.ts +6 -0
- package/dist/observability/config.js +20 -0
- package/dist/observability/metrics-server.d.ts +11 -0
- package/dist/observability/metrics-server.js +47 -0
- package/dist/observability/metrics.d.ts +36 -0
- package/dist/observability/metrics.js +158 -0
- package/dist/providers/auth/authorization-server.d.ts +41 -0
- package/dist/providers/auth/authorization-server.js +1 -0
- package/dist/providers/auth/delegating.d.ts +44 -0
- package/dist/providers/auth/delegating.js +105 -0
- package/dist/providers/auth/index.d.ts +11 -0
- package/dist/providers/auth/index.js +20 -0
- package/dist/providers/auth/release-gate.d.ts +1 -0
- package/dist/providers/auth/release-gate.js +4 -0
- package/dist/providers/auth/self-hosted.d.ts +78 -0
- package/dist/providers/auth/self-hosted.js +377 -0
- package/dist/providers/auth/subject.d.ts +1 -0
- package/dist/providers/auth/subject.js +6 -0
- package/dist/providers/auth/types.d.ts +63 -0
- package/dist/providers/auth/types.js +22 -0
- package/dist/providers/coding-proxy/deny-port.d.ts +5 -0
- package/dist/providers/coding-proxy/deny-port.js +45 -0
- package/dist/providers/coding-proxy/environment-credentials.d.ts +7 -0
- package/dist/providers/coding-proxy/environment-credentials.js +16 -0
- package/dist/providers/coding-proxy/index.d.ts +7 -0
- package/dist/providers/coding-proxy/index.js +7 -0
- package/dist/providers/coding-proxy/memory-ledger.d.ts +17 -0
- package/dist/providers/coding-proxy/memory-ledger.js +103 -0
- package/dist/providers/coding-proxy/metering.d.ts +20 -0
- package/dist/providers/coding-proxy/metering.js +161 -0
- package/dist/providers/coding-proxy/prisma-ledger.d.ts +15 -0
- package/dist/providers/coding-proxy/prisma-ledger.js +213 -0
- package/dist/providers/coding-proxy/proxy.d.ts +75 -0
- package/dist/providers/coding-proxy/proxy.js +721 -0
- package/dist/providers/coding-proxy/runtime.d.ts +18 -0
- package/dist/providers/coding-proxy/runtime.js +39 -0
- package/dist/providers/coding-proxy/secure-fetch.d.ts +9 -0
- package/dist/providers/coding-proxy/secure-fetch.js +72 -0
- package/dist/providers/coding-proxy/server.d.ts +17 -0
- package/dist/providers/coding-proxy/server.js +139 -0
- package/dist/providers/coding-proxy/types.d.ts +99 -0
- package/dist/providers/coding-proxy/types.js +1 -0
- package/dist/providers/datastore/index.d.ts +2 -0
- package/dist/providers/datastore/index.js +2 -0
- package/dist/providers/datastore/postgres.d.ts +20 -0
- package/dist/providers/datastore/postgres.js +130 -0
- package/dist/providers/datastore/scoped.d.ts +10 -0
- package/dist/providers/datastore/scoped.js +46 -0
- package/dist/providers/datastore/types.d.ts +31 -0
- package/dist/providers/datastore/types.js +7 -0
- package/dist/providers/email/types.d.ts +46 -0
- package/dist/providers/email/types.js +8 -0
- package/dist/providers/engine/index.d.ts +1 -0
- package/dist/providers/engine/index.js +1 -0
- package/dist/providers/engine/types.d.ts +67 -0
- package/dist/providers/engine/types.js +16 -0
- package/dist/providers/executor/build.d.ts +6 -0
- package/dist/providers/executor/build.js +15 -0
- package/dist/providers/executor/composition.d.ts +14 -0
- package/dist/providers/executor/composition.js +123 -0
- package/dist/providers/executor/container.d.ts +184 -0
- package/dist/providers/executor/container.js +936 -0
- package/dist/providers/executor/dbos-status.d.ts +50 -0
- package/dist/providers/executor/dbos-status.js +64 -0
- package/dist/providers/executor/dbos.d.ts +53 -0
- package/dist/providers/executor/dbos.js +269 -0
- package/dist/providers/executor/in-process.d.ts +18 -0
- package/dist/providers/executor/in-process.js +38 -0
- package/dist/providers/executor/index.d.ts +8 -0
- package/dist/providers/executor/index.js +8 -0
- package/dist/providers/executor/routing.d.ts +48 -0
- package/dist/providers/executor/routing.js +70 -0
- package/dist/providers/executor/types.d.ts +55 -0
- package/dist/providers/executor/types.js +1 -0
- package/dist/providers/index.d.ts +43 -0
- package/dist/providers/index.js +18 -0
- package/dist/providers/jobs/docker-isolation.d.ts +166 -0
- package/dist/providers/jobs/docker-isolation.js +724 -0
- package/dist/providers/jobs/docker.d.ts +127 -0
- package/dist/providers/jobs/docker.js +1027 -0
- package/dist/providers/jobs/fake-kubernetes-api.d.ts +98 -0
- package/dist/providers/jobs/fake-kubernetes-api.js +142 -0
- package/dist/providers/jobs/fake.d.ts +23 -0
- package/dist/providers/jobs/fake.js +132 -0
- package/dist/providers/jobs/kubernetes-api.d.ts +60 -0
- package/dist/providers/jobs/kubernetes-api.js +20 -0
- package/dist/providers/jobs/kubernetes-client.d.ts +36 -0
- package/dist/providers/jobs/kubernetes-client.js +171 -0
- package/dist/providers/jobs/kubernetes-dry-run-fixture.d.ts +49 -0
- package/dist/providers/jobs/kubernetes-dry-run-fixture.js +141 -0
- package/dist/providers/jobs/kubernetes-isolation.d.ts +147 -0
- package/dist/providers/jobs/kubernetes-isolation.js +607 -0
- package/dist/providers/jobs/kubernetes-platform.d.ts +177 -0
- package/dist/providers/jobs/kubernetes-platform.js +280 -0
- package/dist/providers/jobs/kubernetes-preflight.d.ts +59 -0
- package/dist/providers/jobs/kubernetes-preflight.js +339 -0
- package/dist/providers/jobs/kubernetes-witness.d.ts +10 -0
- package/dist/providers/jobs/kubernetes-witness.js +193 -0
- package/dist/providers/jobs/kubernetes.d.ts +127 -0
- package/dist/providers/jobs/kubernetes.js +874 -0
- package/dist/providers/jobs/safe-extract.d.ts +9 -0
- package/dist/providers/jobs/safe-extract.js +325 -0
- package/dist/providers/jobs/types.d.ts +78 -0
- package/dist/providers/jobs/types.js +6 -0
- package/dist/providers/jobs/workspace-swap.d.ts +2 -0
- package/dist/providers/jobs/workspace-swap.js +47 -0
- package/dist/providers/llm/anthropic.d.ts +13 -0
- package/dist/providers/llm/anthropic.js +21 -0
- package/dist/providers/llm/bedrock.d.ts +19 -0
- package/dist/providers/llm/bedrock.js +33 -0
- package/dist/providers/llm/claude-messages.d.ts +106 -0
- package/dist/providers/llm/claude-messages.js +157 -0
- package/dist/providers/llm/claude-provider.d.ts +40 -0
- package/dist/providers/llm/claude-provider.js +31 -0
- package/dist/providers/llm/index.d.ts +10 -0
- package/dist/providers/llm/index.js +10 -0
- package/dist/providers/llm/openai.d.ts +23 -0
- package/dist/providers/llm/openai.js +156 -0
- package/dist/providers/llm/pricing-anthropic.d.ts +4 -0
- package/dist/providers/llm/pricing-anthropic.js +30 -0
- package/dist/providers/llm/pricing-bedrock-claude.d.ts +12 -0
- package/dist/providers/llm/pricing-bedrock-claude.js +37 -0
- package/dist/providers/llm/pricing-core.d.ts +26 -0
- package/dist/providers/llm/pricing-core.js +17 -0
- package/dist/providers/llm/pricing.d.ts +30 -0
- package/dist/providers/llm/pricing.js +74 -0
- package/dist/providers/llm/registration.d.ts +8 -0
- package/dist/providers/llm/registration.js +39 -0
- package/dist/providers/llm/routing.d.ts +25 -0
- package/dist/providers/llm/routing.js +32 -0
- package/dist/providers/llm/types.d.ts +85 -0
- package/dist/providers/llm/types.js +12 -0
- package/dist/providers/memory/index.d.ts +2 -0
- package/dist/providers/memory/index.js +2 -0
- package/dist/providers/memory/postgres.d.ts +20 -0
- package/dist/providers/memory/postgres.js +72 -0
- package/dist/providers/memory/types.d.ts +32 -0
- package/dist/providers/memory/types.js +17 -0
- package/dist/providers/secrets/app-key.d.ts +8 -0
- package/dist/providers/secrets/app-key.js +50 -0
- package/dist/providers/secrets/index.d.ts +10 -0
- package/dist/providers/secrets/index.js +13 -0
- package/dist/providers/secrets/transfer-envelope.d.ts +12 -0
- package/dist/providers/secrets/transfer-envelope.js +35 -0
- package/dist/providers/secrets/types.d.ts +18 -0
- package/dist/providers/secrets/types.js +12 -0
- package/dist/providers/storage/types.d.ts +18 -0
- package/dist/providers/storage/types.js +7 -0
- package/dist/providers/vcs/git.d.ts +105 -0
- package/dist/providers/vcs/git.js +710 -0
- package/dist/providers/vcs/github.d.ts +105 -0
- package/dist/providers/vcs/github.js +394 -0
- package/dist/providers/vcs/index.d.ts +6 -0
- package/dist/providers/vcs/index.js +26 -0
- package/dist/providers/vcs/types.d.ts +118 -0
- package/dist/providers/vcs/types.js +1 -0
- package/dist/sandbox/bounded-json.d.ts +2 -0
- package/dist/sandbox/bounded-json.js +36 -0
- package/dist/sandbox/bridge.d.ts +15 -0
- package/dist/sandbox/bridge.js +78 -0
- package/dist/sandbox/eval-core.d.ts +37 -0
- package/dist/sandbox/eval-core.js +159 -0
- package/dist/sandbox/fetch-policy.d.ts +24 -0
- package/dist/sandbox/fetch-policy.js +96 -0
- package/dist/sandbox/generated/zod-to-json-schema.bundle.js +5 -0
- package/dist/sandbox/generated/zod.bundle.js +1 -0
- package/dist/sandbox/host-functions.d.ts +29 -0
- package/dist/sandbox/host-functions.js +168 -0
- package/dist/sandbox/limits.d.ts +39 -0
- package/dist/sandbox/limits.js +39 -0
- package/dist/sandbox/parser-worker/__fixtures__/crash.d.ts +1 -0
- package/dist/sandbox/parser-worker/__fixtures__/crash.js +6 -0
- package/dist/sandbox/parser-worker/__fixtures__/echo.d.ts +1 -0
- package/dist/sandbox/parser-worker/__fixtures__/echo.js +5 -0
- package/dist/sandbox/parser-worker/__fixtures__/reject.d.ts +1 -0
- package/dist/sandbox/parser-worker/__fixtures__/reject.js +5 -0
- package/dist/sandbox/parser-worker/__fixtures__/spin-forever.d.ts +1 -0
- package/dist/sandbox/parser-worker/__fixtures__/spin-forever.js +11 -0
- package/dist/sandbox/parser-worker/pool.d.ts +13 -0
- package/dist/sandbox/parser-worker/pool.js +111 -0
- package/dist/sandbox/parser-worker/worker.d.ts +14 -0
- package/dist/sandbox/parser-worker/worker.js +63 -0
- package/dist/sandbox/pii-redaction.d.ts +1 -0
- package/dist/sandbox/pii-redaction.js +29 -0
- package/dist/sandbox/prelude.d.ts +26 -0
- package/dist/sandbox/prelude.js +408 -0
- package/dist/sandbox/quickjs-module.d.ts +13 -0
- package/dist/sandbox/quickjs-module.js +19 -0
- package/dist/sandbox/run-in-sandbox.d.ts +34 -0
- package/dist/sandbox/run-in-sandbox.js +53 -0
- package/dist/sandbox/safe-fetch.d.ts +26 -0
- package/dist/sandbox/safe-fetch.js +140 -0
- package/dist/sandbox/tool-capabilities.d.ts +66 -0
- package/dist/sandbox/tool-capabilities.js +103 -0
- package/dist/sandbox/zod-params.d.ts +22 -0
- package/dist/sandbox/zod-params.js +60 -0
- package/dist/serve.d.ts +14 -0
- package/dist/serve.js +54 -0
- package/dist/wardby-bin.d.ts +2 -0
- package/dist/wardby-bin.js +20 -0
- package/package.json +123 -0
- package/prisma/migrations/20260905000000_init/migration.sql +34 -0
- package/prisma/migrations/20260905010000_scheduler_durability/migration.sql +31 -0
- package/prisma/migrations/20260905020000_tools_multiturn/migration.sql +47 -0
- package/prisma/migrations/20260905030000_cache_tool_json_schema/migration.sql +13 -0
- package/prisma/migrations/20260906000000_mcp_phase4/migration.sql +115 -0
- package/prisma/migrations/20260906010000_run_final_text/migration.sql +7 -0
- package/prisma/migrations/20260906020000_run_turns/migration.sql +6 -0
- package/prisma/migrations/20260906030000_task_principal_id/migration.sql +5 -0
- package/prisma/migrations/20260906040000_secure_self_hosted_oauth/migration.sql +67 -0
- package/prisma/migrations/20260906050000_phase5_coding_agents/migration.sql +52 -0
- package/prisma/migrations/20260906060000_tool_capability_scoping/migration.sql +22 -0
- package/prisma/migrations/20260906070000_datastore_pii_flag/migration.sql +6 -0
- package/prisma/migrations/20260906080000_secret_elicitation_outcomes/migration.sql +13 -0
- package/prisma/migrations/20260907010000_coding_proxy_ledger/migration.sql +57 -0
- package/prisma/migrations/20260907020000_budget_groups/migration.sql +24 -0
- package/prisma/migrations/20260907030000_coding_webhook_task_override/migration.sql +4 -0
- package/prisma/migrations/20260908010000_run_execution_backend/migration.sql +2 -0
- package/prisma/migrations/20260908020000_coding_run_audit_metadata/migration.sql +7 -0
- package/prisma/migrations/20260910010000_agent_secret_bound_name/migration.sql +16 -0
- package/prisma/migrations/20260911010000_coding_worker_toolchains/migration.sql +5 -0
- package/prisma/migrations/20260912010000_coding_provider_contract/migration.sql +24 -0
- package/prisma/migrations/20260912020000_coding_proxy_protocol/migration.sql +10 -0
- package/prisma/migrations/20260912030000_agent_memory/migration.sql +18 -0
- package/prisma/migrations/20260912040000_named_shared_datastores/migration.sql +68 -0
- package/prisma/migrations/20260913000000_run_trigger_webhook/migration.sql +2 -0
- package/prisma/migrations/20260913010000_agent_sub_agent/migration.sql +32 -0
- package/prisma/migrations/20260913020000_run_task_override/migration.sql +5 -0
- package/prisma/migrations/20260913030000_coding_run_root/migration.sql +11 -0
- package/prisma/migrations/20260922010000_coding_run_queued_at/migration.sql +8 -0
- package/prisma/migrations/20260922020000_coding_workspace_disk/migration.sql +8 -0
- package/prisma/migrations/migration_lock.toml +3 -0
- package/prisma/schema.prisma +723 -0
- package/scripts/git-askpass.sh +11 -0
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Client-agnostic Claude Messages protocol. Pure functions, no SDK import,
|
|
3
|
+
* so a future Bedrock-Claude adapter reuses them unchanged — Bedrock and the
|
|
4
|
+
* direct API differ only in client/auth, model IDs, and pricing.
|
|
5
|
+
*/
|
|
6
|
+
import { encode as encodeO200kBase } from "gpt-tokenizer/encoding/o200k_base";
|
|
7
|
+
function parseArgs(argsJson) {
|
|
8
|
+
try {
|
|
9
|
+
return JSON.parse(argsJson || "{}");
|
|
10
|
+
}
|
|
11
|
+
catch {
|
|
12
|
+
return {};
|
|
13
|
+
}
|
|
14
|
+
}
|
|
15
|
+
/**
|
|
16
|
+
* Same shape sent to the API in `toClaudeRequest` — kept as one function so
|
|
17
|
+
* a pre-flight token estimate (AnthropicLlmProvider.countTokens) can never
|
|
18
|
+
* drift from what's actually serialized into the request. Mirrors the
|
|
19
|
+
* OpenAI adapter's `toOpenAiTools`, which guards against the same failure
|
|
20
|
+
* class: a duplicated tool-schema mapping once desynced the pre-flight
|
|
21
|
+
* estimate from the real request and broke the budget guardrail's turn-1
|
|
22
|
+
* refuse guarantee (measured 44-52% under-count).
|
|
23
|
+
*/
|
|
24
|
+
export function toClaudeTools(tools) {
|
|
25
|
+
return tools.map((t) => ({
|
|
26
|
+
name: t.name,
|
|
27
|
+
description: t.description,
|
|
28
|
+
input_schema: t.parameters,
|
|
29
|
+
}));
|
|
30
|
+
}
|
|
31
|
+
const TOKENS_PER_MESSAGE = 3;
|
|
32
|
+
/**
|
|
33
|
+
* Raw (pre-inflation) offline token estimate shared by every Claude-protocol
|
|
34
|
+
* adapter. Each caller applies its own inflation constant on top.
|
|
35
|
+
*/
|
|
36
|
+
export function estimateClaudeTokens(messages, tools) {
|
|
37
|
+
let raw = 0;
|
|
38
|
+
for (const m of messages) {
|
|
39
|
+
raw += TOKENS_PER_MESSAGE + encodeO200kBase(m.content).length + encodeO200kBase(m.role).length;
|
|
40
|
+
}
|
|
41
|
+
if (tools && tools.length > 0) {
|
|
42
|
+
raw += encodeO200kBase(JSON.stringify(toClaudeTools(tools))).length;
|
|
43
|
+
}
|
|
44
|
+
return raw;
|
|
45
|
+
}
|
|
46
|
+
export function toClaudeRequest(req, defaultMaxTokens) {
|
|
47
|
+
const system = [];
|
|
48
|
+
const messages = [];
|
|
49
|
+
for (const m of req.messages) {
|
|
50
|
+
if (m.role === "system") {
|
|
51
|
+
system.push({ type: "text", text: m.content });
|
|
52
|
+
continue;
|
|
53
|
+
}
|
|
54
|
+
if (m.role === "tool") {
|
|
55
|
+
messages.push({
|
|
56
|
+
role: "user",
|
|
57
|
+
content: [{ type: "tool_result", tool_use_id: m.toolCallId ?? "", content: m.content }],
|
|
58
|
+
});
|
|
59
|
+
continue;
|
|
60
|
+
}
|
|
61
|
+
if (m.role === "assistant") {
|
|
62
|
+
const content = [];
|
|
63
|
+
if (m.content)
|
|
64
|
+
content.push({ type: "text", text: m.content });
|
|
65
|
+
for (const tc of m.toolCalls ?? []) {
|
|
66
|
+
content.push({ type: "tool_use", id: tc.id, name: tc.name, input: parseArgs(tc.argsJson) });
|
|
67
|
+
}
|
|
68
|
+
messages.push({ role: "assistant", content });
|
|
69
|
+
continue;
|
|
70
|
+
}
|
|
71
|
+
// user
|
|
72
|
+
messages.push({ role: "user", content: [{ type: "text", text: m.content }] });
|
|
73
|
+
}
|
|
74
|
+
const tools = req.tools ? toClaudeTools(req.tools) : undefined;
|
|
75
|
+
return {
|
|
76
|
+
...(system.length > 0 ? { system } : {}),
|
|
77
|
+
messages,
|
|
78
|
+
...(tools && tools.length > 0 ? { tools } : {}),
|
|
79
|
+
max_tokens: req.maxTokens ?? defaultMaxTokens,
|
|
80
|
+
};
|
|
81
|
+
}
|
|
82
|
+
const EPHEMERAL = { type: "ephemeral" };
|
|
83
|
+
/**
|
|
84
|
+
* Two-breakpoint canonical caching. Anthropic's cache order is
|
|
85
|
+
* tools -> system -> messages, so a breakpoint on the last system block
|
|
86
|
+
* caches tools+system (static across the run). A rolling breakpoint on the
|
|
87
|
+
* last message's final block grows the conversation cache each turn.
|
|
88
|
+
* Sub-minimum prefixes (1024/2048 tokens) are silently un-cached by Anthropic.
|
|
89
|
+
*/
|
|
90
|
+
export function withCacheBreakpoints(req) {
|
|
91
|
+
const system = req.system?.map((b, i, arr) => (i === arr.length - 1 ? { ...b, cache_control: EPHEMERAL } : b));
|
|
92
|
+
const messages = req.messages.map((m, mi, marr) => {
|
|
93
|
+
if (mi !== marr.length - 1)
|
|
94
|
+
return m;
|
|
95
|
+
const content = m.content.map((b, bi, barr) => (bi === barr.length - 1 ? { ...b, cache_control: EPHEMERAL } : b));
|
|
96
|
+
return { ...m, content };
|
|
97
|
+
});
|
|
98
|
+
return { ...req, ...(system ? { system } : {}), messages };
|
|
99
|
+
}
|
|
100
|
+
export async function* mapClaudeStream(events, priceUsd) {
|
|
101
|
+
const toolBuffers = new Map();
|
|
102
|
+
let start = {};
|
|
103
|
+
let outputTokens = 0;
|
|
104
|
+
let stopReason = "stop";
|
|
105
|
+
for await (const ev of events) {
|
|
106
|
+
switch (ev.type) {
|
|
107
|
+
case "message_start":
|
|
108
|
+
start = ev.message.usage ?? {};
|
|
109
|
+
outputTokens = ev.message.usage?.output_tokens ?? 0;
|
|
110
|
+
break;
|
|
111
|
+
case "content_block_start":
|
|
112
|
+
if (ev.content_block.type === "tool_use") {
|
|
113
|
+
toolBuffers.set(ev.index, { id: ev.content_block.id ?? "", name: ev.content_block.name ?? "", argsJson: "" });
|
|
114
|
+
}
|
|
115
|
+
break;
|
|
116
|
+
case "content_block_delta":
|
|
117
|
+
if (ev.delta.type === "text_delta") {
|
|
118
|
+
yield { type: "text", delta: ev.delta.text };
|
|
119
|
+
}
|
|
120
|
+
else {
|
|
121
|
+
const buf = toolBuffers.get(ev.index);
|
|
122
|
+
if (buf)
|
|
123
|
+
buf.argsJson += ev.delta.partial_json;
|
|
124
|
+
}
|
|
125
|
+
break;
|
|
126
|
+
case "content_block_stop": {
|
|
127
|
+
const buf = toolBuffers.get(ev.index);
|
|
128
|
+
if (buf) {
|
|
129
|
+
yield { type: "tool_call", id: buf.id, name: buf.name, argsJson: buf.argsJson };
|
|
130
|
+
toolBuffers.delete(ev.index);
|
|
131
|
+
}
|
|
132
|
+
break;
|
|
133
|
+
}
|
|
134
|
+
case "message_delta":
|
|
135
|
+
if (ev.delta.stop_reason)
|
|
136
|
+
stopReason = ev.delta.stop_reason;
|
|
137
|
+
if (ev.usage?.output_tokens != null)
|
|
138
|
+
outputTokens = ev.usage.output_tokens;
|
|
139
|
+
break;
|
|
140
|
+
case "message_stop": {
|
|
141
|
+
const cachedInputTokens = start.cache_read_input_tokens ?? 0;
|
|
142
|
+
const cacheWriteTokens = start.cache_creation_input_tokens ?? 0;
|
|
143
|
+
const inputTokens = (start.input_tokens ?? 0) + cachedInputTokens;
|
|
144
|
+
const usage = {
|
|
145
|
+
inputTokens,
|
|
146
|
+
outputTokens,
|
|
147
|
+
cachedInputTokens,
|
|
148
|
+
cacheWriteTokens,
|
|
149
|
+
costUsd: 0,
|
|
150
|
+
};
|
|
151
|
+
usage.costUsd = priceUsd(usage);
|
|
152
|
+
yield { type: "done", stopReason, usage };
|
|
153
|
+
break;
|
|
154
|
+
}
|
|
155
|
+
}
|
|
156
|
+
}
|
|
157
|
+
}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* LlmProvider implementation shared by every Claude-protocol adapter (direct
|
|
3
|
+
* Anthropic API, Bedrock-Claude, and any future Claude-protocol client).
|
|
4
|
+
* Imports no SDK — the client and pricing module are injected, so this file
|
|
5
|
+
* stays exactly what each concrete adapter (anthropic.ts, bedrock.ts) has in
|
|
6
|
+
* common, and nothing more.
|
|
7
|
+
*/
|
|
8
|
+
import type { LlmMessage, LlmProvider, LlmRequest, LlmStreamEvent, LlmToolDef } from "./types.js";
|
|
9
|
+
import type { ModelPricing, UsageTokens } from "./pricing-core.js";
|
|
10
|
+
/**
|
|
11
|
+
* Structural shape every Claude-protocol SDK client satisfies. Loose on
|
|
12
|
+
* purpose — params/return are untyped here, mirroring the cast anthropic.ts
|
|
13
|
+
* already needed before this extraction (its own SDK's stream() params/
|
|
14
|
+
* return never lined up 1:1 with this codebase's own request/event shapes).
|
|
15
|
+
*/
|
|
16
|
+
export interface ClaudeMessagesClient {
|
|
17
|
+
messages: {
|
|
18
|
+
stream(params: unknown, options?: {
|
|
19
|
+
signal?: AbortSignal;
|
|
20
|
+
}): unknown;
|
|
21
|
+
};
|
|
22
|
+
}
|
|
23
|
+
export interface ClaudePricingModule {
|
|
24
|
+
/** Must throw on an unknown model — fail-closed, never price at zero. */
|
|
25
|
+
getPricing(model: string): ModelPricing;
|
|
26
|
+
priceUsd(model: string, usage: UsageTokens): number;
|
|
27
|
+
}
|
|
28
|
+
export declare class ClaudeLlmProvider implements LlmProvider {
|
|
29
|
+
private readonly client;
|
|
30
|
+
private readonly pricing;
|
|
31
|
+
constructor(client: ClaudeMessagesClient, pricing: ClaudePricingModule);
|
|
32
|
+
stream(req: LlmRequest, signal?: AbortSignal): AsyncIterable<LlmStreamEvent>;
|
|
33
|
+
countTokens(model: string, messages: LlmMessage[], tools?: LlmToolDef[]): Promise<number>;
|
|
34
|
+
priceUsd(model: string, usage: {
|
|
35
|
+
inputTokens: number;
|
|
36
|
+
outputTokens: number;
|
|
37
|
+
cachedInputTokens?: number;
|
|
38
|
+
cacheWriteTokens?: number;
|
|
39
|
+
}): number;
|
|
40
|
+
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
import { toClaudeRequest, withCacheBreakpoints, mapClaudeStream, estimateClaudeTokens, } from "./claude-messages.js";
|
|
2
|
+
// Anthropic's own guidance: don't lowball max_tokens — hitting the cap
|
|
3
|
+
// truncates output mid-thought with no error, silently handing engine-native.ts
|
|
4
|
+
// (which never sets LlmRequest.maxTokens, and never checks stopReason) a
|
|
5
|
+
// clipped "final" answer it treats as complete. This adapter streams, so a
|
|
6
|
+
// generous ceiling costs nothing in latency; the budget guardrail is driven
|
|
7
|
+
// by actual token counts (core/budget.ts), not by this cap.
|
|
8
|
+
const DEFAULT_MAX_TOKENS = 16000;
|
|
9
|
+
// Claude has no offline tokenizer; o200k_base is a proxy. Bias high so the
|
|
10
|
+
// pre-flight refuse never admits an over-budget run on an under-count.
|
|
11
|
+
const CLAUDE_TOKEN_INFLATION = 1.2;
|
|
12
|
+
export class ClaudeLlmProvider {
|
|
13
|
+
client;
|
|
14
|
+
pricing;
|
|
15
|
+
constructor(client, pricing) {
|
|
16
|
+
this.client = client;
|
|
17
|
+
this.pricing = pricing;
|
|
18
|
+
}
|
|
19
|
+
async *stream(req, signal) {
|
|
20
|
+
const claudeReq = withCacheBreakpoints(toClaudeRequest(req, DEFAULT_MAX_TOKENS));
|
|
21
|
+
const raw = this.client.messages.stream({ model: req.model, ...claudeReq }, { signal });
|
|
22
|
+
yield* mapClaudeStream(raw, (usage) => this.priceUsd(req.model, usage));
|
|
23
|
+
}
|
|
24
|
+
async countTokens(model, messages, tools) {
|
|
25
|
+
this.pricing.getPricing(model); // fail-closed on unknown model
|
|
26
|
+
return Math.ceil(estimateClaudeTokens(messages, tools) * CLAUDE_TOKEN_INFLATION);
|
|
27
|
+
}
|
|
28
|
+
priceUsd(model, usage) {
|
|
29
|
+
return this.pricing.priceUsd(model, usage);
|
|
30
|
+
}
|
|
31
|
+
}
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
export * from "./types.js";
|
|
2
|
+
export * from "./pricing.js";
|
|
3
|
+
export * from "./openai.js";
|
|
4
|
+
export { AnthropicLlmProvider, anthropicCredentialsPresent, anthropicSupportedModels } from "./anthropic.js";
|
|
5
|
+
export { BedrockClaudeLlmProvider, bedrockCredentialsPresent, bedrockClaudeSupportedModels } from "./bedrock.js";
|
|
6
|
+
export { ClaudeLlmProvider, type ClaudeMessagesClient, type ClaudePricingModule } from "./claude-provider.js";
|
|
7
|
+
export { openaiCredentialsPresent } from "./openai.js";
|
|
8
|
+
export { supportedModels as openaiSupportedModels } from "./pricing.js";
|
|
9
|
+
export { RoutingLlmProvider, type LlmRegistration } from "./routing.js";
|
|
10
|
+
export { resolveLlmRegistrations, type LlmRegistrationResult } from "./registration.js";
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
export * from "./types.js";
|
|
2
|
+
export * from "./pricing.js";
|
|
3
|
+
export * from "./openai.js";
|
|
4
|
+
export { AnthropicLlmProvider, anthropicCredentialsPresent, anthropicSupportedModels } from "./anthropic.js";
|
|
5
|
+
export { BedrockClaudeLlmProvider, bedrockCredentialsPresent, bedrockClaudeSupportedModels } from "./bedrock.js";
|
|
6
|
+
export { ClaudeLlmProvider } from "./claude-provider.js";
|
|
7
|
+
export { openaiCredentialsPresent } from "./openai.js";
|
|
8
|
+
export { supportedModels as openaiSupportedModels } from "./pricing.js";
|
|
9
|
+
export { RoutingLlmProvider } from "./routing.js";
|
|
10
|
+
export { resolveLlmRegistrations } from "./registration.js";
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OpenAI adapter for the `LlmProvider` seam — Phase 1's only concrete LLM
|
|
3
|
+
* adapter. The `openai` npm package is used only inside this file; core
|
|
4
|
+
* never imports it.
|
|
5
|
+
*/
|
|
6
|
+
import OpenAI from "openai";
|
|
7
|
+
import type { LlmMessage, LlmProvider, LlmRequest, LlmStreamEvent, LlmToolDef } from "./types.js";
|
|
8
|
+
/** Whether the OpenAI adapter has credentials to run (used by the router's enable-by-credential wiring). */
|
|
9
|
+
export declare function openaiCredentialsPresent(env?: NodeJS.ProcessEnv): boolean;
|
|
10
|
+
export declare function estimateTokens(model: string, messages: LlmMessage[], tools?: LlmToolDef[]): number;
|
|
11
|
+
export declare class OpenAiLlmProvider implements LlmProvider {
|
|
12
|
+
private readonly client;
|
|
13
|
+
/** `client` is an injection point for tests — a real adapter never passes it. */
|
|
14
|
+
constructor(apiKey?: string, client?: OpenAI);
|
|
15
|
+
stream(req: LlmRequest, signal?: AbortSignal): AsyncIterable<LlmStreamEvent>;
|
|
16
|
+
countTokens(model: string, messages: LlmMessage[], tools?: LlmToolDef[]): Promise<number>;
|
|
17
|
+
priceUsd(model: string, usage: {
|
|
18
|
+
inputTokens: number;
|
|
19
|
+
outputTokens: number;
|
|
20
|
+
cachedInputTokens?: number;
|
|
21
|
+
cacheWriteTokens?: number;
|
|
22
|
+
}): number;
|
|
23
|
+
}
|
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* OpenAI adapter for the `LlmProvider` seam — Phase 1's only concrete LLM
|
|
3
|
+
* adapter. The `openai` npm package is used only inside this file; core
|
|
4
|
+
* never imports it.
|
|
5
|
+
*/
|
|
6
|
+
import OpenAI from "openai";
|
|
7
|
+
import { encode as encodeCl100kBase } from "gpt-tokenizer/encoding/cl100k_base";
|
|
8
|
+
import { encode as encodeO200kBase } from "gpt-tokenizer/encoding/o200k_base";
|
|
9
|
+
import { getModelPricing, priceUsd as priceUsdFromTable } from "./pricing.js";
|
|
10
|
+
/** Whether the OpenAI adapter has credentials to run (used by the router's enable-by-credential wiring). */
|
|
11
|
+
export function openaiCredentialsPresent(env = process.env) {
|
|
12
|
+
return Boolean(env.OPENAI_API_KEY);
|
|
13
|
+
}
|
|
14
|
+
// Per-message token overhead from OpenAI's public chat-format guidance
|
|
15
|
+
// (role framing + a name field costs a few tokens beyond the raw content).
|
|
16
|
+
// This is a pre-flight *estimate* for the budget guardrail, not an exact
|
|
17
|
+
// match to server-side billing — the real count comes back in `usage` on
|
|
18
|
+
// the `done` event.
|
|
19
|
+
const TOKENS_PER_MESSAGE = 3;
|
|
20
|
+
const TOKENS_PER_NAME = 1;
|
|
21
|
+
const TOKENS_PRIMING_REPLY = 3;
|
|
22
|
+
// gpt-tokenizer's package default is cl100k_base, but every gpt-4o-and-later
|
|
23
|
+
// model (all of Phase 1's roster) actually uses o200k_base — using the
|
|
24
|
+
// wrong table skews the pre-flight token count, which is the budget
|
|
25
|
+
// guardrail's input. The encoding lives on the pricing table (pricing.ts)
|
|
26
|
+
// so a model can't be run without also declaring which tokenizer counts it.
|
|
27
|
+
function encodeForModel(model, text) {
|
|
28
|
+
const { encoding } = getModelPricing(model);
|
|
29
|
+
return encoding === "o200k_base" ? encodeO200kBase(text) : encodeCl100kBase(text);
|
|
30
|
+
}
|
|
31
|
+
/** Same shape sent to the API in `stream()` — kept as one function so the estimate can never drift from what's actually serialized. */
|
|
32
|
+
function toOpenAiTools(tools) {
|
|
33
|
+
return tools.map((t) => ({
|
|
34
|
+
type: "function",
|
|
35
|
+
function: { name: t.name, description: t.description, parameters: t.parameters },
|
|
36
|
+
}));
|
|
37
|
+
}
|
|
38
|
+
// Fixed: this used to count message tokens only. When a request carries
|
|
39
|
+
// `tools`, OpenAI also tokenizes the serialized tool JSON schemas into the
|
|
40
|
+
// prompt — measured live at a 44-52% under-count for a small tool roster,
|
|
41
|
+
// which is what let a turn-1 pre-flight refuse (budget.ts, cumulative
|
|
42
|
+
// spend still zero) admit a run whose real input cost already exceeded
|
|
43
|
+
// budget. Counting `JSON.stringify` of the same payload `stream()` sends
|
|
44
|
+
// is still an *approximation* (not OpenAI's undocumented exact tool-schema
|
|
45
|
+
// tokenization), not a byte-for-byte match — calibration logging
|
|
46
|
+
// (budget.ts's checkTokenCalibration) is what surfaces any remaining drift.
|
|
47
|
+
export function estimateTokens(model, messages, tools) {
|
|
48
|
+
let total = TOKENS_PRIMING_REPLY;
|
|
49
|
+
for (const message of messages) {
|
|
50
|
+
total += TOKENS_PER_MESSAGE;
|
|
51
|
+
total += encodeForModel(model, message.content).length;
|
|
52
|
+
total += encodeForModel(model, message.role).length;
|
|
53
|
+
if (message.name) {
|
|
54
|
+
total += encodeForModel(model, message.name).length + TOKENS_PER_NAME;
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
if (tools && tools.length > 0) {
|
|
58
|
+
total += encodeForModel(model, JSON.stringify(toOpenAiTools(tools))).length;
|
|
59
|
+
}
|
|
60
|
+
return total;
|
|
61
|
+
}
|
|
62
|
+
export class OpenAiLlmProvider {
|
|
63
|
+
client;
|
|
64
|
+
/** `client` is an injection point for tests — a real adapter never passes it. */
|
|
65
|
+
constructor(apiKey = process.env.OPENAI_API_KEY ?? "", client) {
|
|
66
|
+
if (client) {
|
|
67
|
+
this.client = client;
|
|
68
|
+
return;
|
|
69
|
+
}
|
|
70
|
+
if (!apiKey) {
|
|
71
|
+
throw new Error("OPENAI_API_KEY is not set — required by the OpenAI LlmProvider adapter.");
|
|
72
|
+
}
|
|
73
|
+
this.client = new OpenAI({ apiKey });
|
|
74
|
+
}
|
|
75
|
+
async *stream(req, signal) {
|
|
76
|
+
const stream = await this.client.chat.completions.create({
|
|
77
|
+
model: req.model,
|
|
78
|
+
messages: req.messages.map((m) => ({
|
|
79
|
+
role: m.role,
|
|
80
|
+
content: m.content,
|
|
81
|
+
...(m.name ? { name: m.name } : {}),
|
|
82
|
+
...(m.toolCallId ? { tool_call_id: m.toolCallId } : {}),
|
|
83
|
+
...(m.toolCalls && m.toolCalls.length > 0
|
|
84
|
+
? {
|
|
85
|
+
tool_calls: m.toolCalls.map((tc) => ({
|
|
86
|
+
id: tc.id,
|
|
87
|
+
type: "function",
|
|
88
|
+
function: { name: tc.name, arguments: tc.argsJson },
|
|
89
|
+
})),
|
|
90
|
+
}
|
|
91
|
+
: {}),
|
|
92
|
+
})),
|
|
93
|
+
tools: req.tools ? toOpenAiTools(req.tools) : undefined,
|
|
94
|
+
max_tokens: req.maxTokens,
|
|
95
|
+
temperature: req.temperature,
|
|
96
|
+
stop: req.stopSequences,
|
|
97
|
+
stream: true,
|
|
98
|
+
stream_options: { include_usage: true },
|
|
99
|
+
}, { signal });
|
|
100
|
+
let stopReason = "stop";
|
|
101
|
+
// OpenAI streams tool-call arguments fragmented across chunks, keyed by
|
|
102
|
+
// index — buffer per index until the turn's finish_reason confirms the
|
|
103
|
+
// call is complete, then emit one `tool_call` event per call.
|
|
104
|
+
const toolCallBuffers = new Map();
|
|
105
|
+
for await (const chunk of stream) {
|
|
106
|
+
const choice = chunk.choices[0];
|
|
107
|
+
if (choice?.delta?.content) {
|
|
108
|
+
yield { type: "text", delta: choice.delta.content };
|
|
109
|
+
}
|
|
110
|
+
if (choice?.delta?.tool_calls) {
|
|
111
|
+
for (const fragment of choice.delta.tool_calls) {
|
|
112
|
+
const buffered = toolCallBuffers.get(fragment.index) ?? { id: "", name: "", argsJson: "" };
|
|
113
|
+
if (fragment.id)
|
|
114
|
+
buffered.id = fragment.id;
|
|
115
|
+
if (fragment.function?.name)
|
|
116
|
+
buffered.name += fragment.function.name;
|
|
117
|
+
if (fragment.function?.arguments)
|
|
118
|
+
buffered.argsJson += fragment.function.arguments;
|
|
119
|
+
toolCallBuffers.set(fragment.index, buffered);
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
if (choice?.finish_reason) {
|
|
123
|
+
stopReason = choice.finish_reason;
|
|
124
|
+
if (stopReason === "tool_calls") {
|
|
125
|
+
for (const toolCall of toolCallBuffers.values()) {
|
|
126
|
+
yield { type: "tool_call", id: toolCall.id, name: toolCall.name, argsJson: toolCall.argsJson };
|
|
127
|
+
}
|
|
128
|
+
toolCallBuffers.clear();
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
if (chunk.usage) {
|
|
132
|
+
// OpenAI's prompt caching is automatic and read-only — no billed
|
|
133
|
+
// "cache write" step, so cacheWriteTokens stays unset here. Other
|
|
134
|
+
// future adapters (e.g. Bedrock/Claude) may report and bill one.
|
|
135
|
+
const cachedInputTokens = chunk.usage.prompt_tokens_details?.cached_tokens ?? 0;
|
|
136
|
+
const usage = {
|
|
137
|
+
inputTokens: chunk.usage.prompt_tokens,
|
|
138
|
+
outputTokens: chunk.usage.completion_tokens,
|
|
139
|
+
cachedInputTokens,
|
|
140
|
+
costUsd: this.priceUsd(req.model, {
|
|
141
|
+
inputTokens: chunk.usage.prompt_tokens,
|
|
142
|
+
outputTokens: chunk.usage.completion_tokens,
|
|
143
|
+
cachedInputTokens,
|
|
144
|
+
}),
|
|
145
|
+
};
|
|
146
|
+
yield { type: "done", stopReason, usage };
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
}
|
|
150
|
+
async countTokens(model, messages, tools) {
|
|
151
|
+
return estimateTokens(model, messages, tools);
|
|
152
|
+
}
|
|
153
|
+
priceUsd(model, usage) {
|
|
154
|
+
return priceUsdFromTable(model, usage);
|
|
155
|
+
}
|
|
156
|
+
}
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
import { type ModelPricing, type UsageTokens } from "./pricing-core.js";
|
|
2
|
+
export declare function getAnthropicPricing(model: string): ModelPricing;
|
|
3
|
+
export declare function anthropicSupportedModels(): string[];
|
|
4
|
+
export declare function anthropicPriceUsd(model: string, usage: UsageTokens): number;
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
import { computeCost } from "./pricing-core.js";
|
|
2
|
+
const CACHE_READ_MULT = 0.1;
|
|
3
|
+
const CACHE_WRITE_MULT = 1.25;
|
|
4
|
+
function claude(inputPerMTok, outputPerMTok) {
|
|
5
|
+
return {
|
|
6
|
+
encoding: "o200k_base",
|
|
7
|
+
inputPerMTok,
|
|
8
|
+
outputPerMTok,
|
|
9
|
+
cachedInputPerMTok: inputPerMTok * CACHE_READ_MULT,
|
|
10
|
+
cacheWritePerMTok: inputPerMTok * CACHE_WRITE_MULT,
|
|
11
|
+
};
|
|
12
|
+
}
|
|
13
|
+
const PRICING = {
|
|
14
|
+
"claude-opus-5": claude(5, 25),
|
|
15
|
+
"claude-sonnet-5": claude(2, 10),
|
|
16
|
+
"claude-fable-5": claude(10, 50),
|
|
17
|
+
"claude-haiku-4-5": claude(1, 5),
|
|
18
|
+
};
|
|
19
|
+
export function getAnthropicPricing(model) {
|
|
20
|
+
const p = PRICING[model];
|
|
21
|
+
if (!p)
|
|
22
|
+
throw new Error(`No pricing entry for model "${model}" — refusing to price at zero. Add it to src/providers/llm/pricing-anthropic.ts.`);
|
|
23
|
+
return p;
|
|
24
|
+
}
|
|
25
|
+
export function anthropicSupportedModels() {
|
|
26
|
+
return Object.keys(PRICING);
|
|
27
|
+
}
|
|
28
|
+
export function anthropicPriceUsd(model, usage) {
|
|
29
|
+
return computeCost(getAnthropicPricing(model), usage);
|
|
30
|
+
}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Bedrock-Claude pricing table. Every entry is a fully literal, hardcoded
|
|
3
|
+
* ModelPricing — every field, including cache read/write, is the provider's
|
|
4
|
+
* own published rate for that exact model, never derived from another
|
|
5
|
+
* field via a ratio (see CLAUDE.md "LLM pricing tables — STRICT": a
|
|
6
|
+
* multiplier that happens to be correct today silently goes stale the
|
|
7
|
+
* moment one model's real cache pricing diverges from the pattern).
|
|
8
|
+
*/
|
|
9
|
+
import { type ModelPricing, type UsageTokens } from "./pricing-core.js";
|
|
10
|
+
export declare function getBedrockClaudePricing(model: string): ModelPricing;
|
|
11
|
+
export declare function bedrockClaudeSupportedModels(): string[];
|
|
12
|
+
export declare function bedrockClaudePriceUsd(model: string, usage: UsageTokens): number;
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Bedrock-Claude pricing table. Every entry is a fully literal, hardcoded
|
|
3
|
+
* ModelPricing — every field, including cache read/write, is the provider's
|
|
4
|
+
* own published rate for that exact model, never derived from another
|
|
5
|
+
* field via a ratio (see CLAUDE.md "LLM pricing tables — STRICT": a
|
|
6
|
+
* multiplier that happens to be correct today silently goes stale the
|
|
7
|
+
* moment one model's real cache pricing diverges from the pattern).
|
|
8
|
+
*/
|
|
9
|
+
import { computeCost } from "./pricing-core.js";
|
|
10
|
+
function claude(inputPerMTok, outputPerMTok, cachedInputPerMTok, cacheWritePerMTok) {
|
|
11
|
+
return { encoding: "o200k_base", inputPerMTok, outputPerMTok, cachedInputPerMTok, cacheWritePerMTok };
|
|
12
|
+
}
|
|
13
|
+
// These are `us.` cross-region inference profiles (region us-east-1) — routability
|
|
14
|
+
// is keyed by model ID only; region is a separate adapter concern
|
|
15
|
+
// (BEDROCK_REGION/AWS_REGION). Rates below are the exact published 5-minute
|
|
16
|
+
// cache-TTL numbers (platform.claude.com/docs/en/about-claude/pricing,
|
|
17
|
+
// confirmed 2026-09-08) — this adapter only ever emits the default 5-minute
|
|
18
|
+
// cache_control breakpoint (see claude-messages.ts's withCacheBreakpoints),
|
|
19
|
+
// so the 1-hour-TTL rates are out of scope and not stored here.
|
|
20
|
+
const PRICING = {
|
|
21
|
+
"us.anthropic.claude-sonnet-4-6": claude(3, 15, 0.3, 3.75),
|
|
22
|
+
"us.anthropic.claude-opus-4-6-v1": claude(5, 25, 0.5, 6.25),
|
|
23
|
+
"us.anthropic.claude-opus-4-8": claude(5, 25, 0.5, 6.25),
|
|
24
|
+
"us.anthropic.claude-haiku-4-5-20251001-v1:0": claude(1, 5, 0.1, 1.25),
|
|
25
|
+
};
|
|
26
|
+
export function getBedrockClaudePricing(model) {
|
|
27
|
+
const p = PRICING[model];
|
|
28
|
+
if (!p)
|
|
29
|
+
throw new Error(`No pricing entry for model "${model}" — refusing to price at zero. Add it to src/providers/llm/pricing-bedrock-claude.ts.`);
|
|
30
|
+
return p;
|
|
31
|
+
}
|
|
32
|
+
export function bedrockClaudeSupportedModels() {
|
|
33
|
+
return Object.keys(PRICING);
|
|
34
|
+
}
|
|
35
|
+
export function bedrockClaudePriceUsd(model, usage) {
|
|
36
|
+
return computeCost(getBedrockClaudePricing(model), usage);
|
|
37
|
+
}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared pricing types + cost math, provider-agnostic. Each LLM adapter
|
|
3
|
+
* owns its own model table but prices through this one `computeCost`, so the
|
|
4
|
+
* cache-read/cache-write accounting can never diverge between providers.
|
|
5
|
+
*/
|
|
6
|
+
export type TokenizerEncoding = "cl100k_base" | "o200k_base";
|
|
7
|
+
export interface ModelPricing {
|
|
8
|
+
encoding: TokenizerEncoding;
|
|
9
|
+
inputPerMTok: number;
|
|
10
|
+
outputPerMTok: number;
|
|
11
|
+
cachedInputPerMTok?: number;
|
|
12
|
+
cacheWritePerMTok?: number;
|
|
13
|
+
}
|
|
14
|
+
export interface UsageTokens {
|
|
15
|
+
inputTokens: number;
|
|
16
|
+
outputTokens: number;
|
|
17
|
+
cachedInputTokens?: number;
|
|
18
|
+
cacheWriteTokens?: number;
|
|
19
|
+
}
|
|
20
|
+
/**
|
|
21
|
+
* A model missing `cachedInputPerMTok` / `cacheWritePerMTok` falls back to the
|
|
22
|
+
* full input rate — fail-toward-overestimate, never underestimate.
|
|
23
|
+
* `inputTokens` is the total billed at the *fresh* rate PLUS cache-read;
|
|
24
|
+
* cache-write tokens are billed separately and are NOT part of `inputTokens`.
|
|
25
|
+
*/
|
|
26
|
+
export declare function computeCost(pricing: ModelPricing, usage: UsageTokens): number;
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* A model missing `cachedInputPerMTok` / `cacheWritePerMTok` falls back to the
|
|
3
|
+
* full input rate — fail-toward-overestimate, never underestimate.
|
|
4
|
+
* `inputTokens` is the total billed at the *fresh* rate PLUS cache-read;
|
|
5
|
+
* cache-write tokens are billed separately and are NOT part of `inputTokens`.
|
|
6
|
+
*/
|
|
7
|
+
export function computeCost(pricing, usage) {
|
|
8
|
+
const cachedInputTokens = usage.cachedInputTokens ?? 0;
|
|
9
|
+
const cacheWriteTokens = usage.cacheWriteTokens ?? 0;
|
|
10
|
+
const freshInputTokens = usage.inputTokens - cachedInputTokens;
|
|
11
|
+
const cachedRate = pricing.cachedInputPerMTok ?? pricing.inputPerMTok;
|
|
12
|
+
const cacheWriteRate = pricing.cacheWritePerMTok ?? pricing.inputPerMTok;
|
|
13
|
+
return ((freshInputTokens / 1_000_000) * pricing.inputPerMTok +
|
|
14
|
+
(cachedInputTokens / 1_000_000) * cachedRate +
|
|
15
|
+
(cacheWriteTokens / 1_000_000) * cacheWriteRate +
|
|
16
|
+
(usage.outputTokens / 1_000_000) * pricing.outputPerMTok);
|
|
17
|
+
}
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Static price + tokenizer table for the budget guardrail.
|
|
3
|
+
*
|
|
4
|
+
* Sourced from OpenAI's public pricing page at write time. Deliberately
|
|
5
|
+
* static (no pricing API call) so `priceUsd` stays synchronous and
|
|
6
|
+
* network-free — the budget math must work offline and in tests. An unknown
|
|
7
|
+
* model throws rather than silently pricing at zero, so a misconfigured
|
|
8
|
+
* agent can never bypass the budget guardrail by accident.
|
|
9
|
+
*
|
|
10
|
+
* `encoding` is here too (not just price) because it drives which
|
|
11
|
+
* `gpt-tokenizer` BPE table `countTokens` uses — every gpt-4o-and-later
|
|
12
|
+
* model uses `o200k_base`, not the older `cl100k_base` gpt-tokenizer
|
|
13
|
+
* defaults to. Keeping both in one table means a model can't be priced
|
|
14
|
+
* without also being told which tokenizer estimates its input cost.
|
|
15
|
+
*
|
|
16
|
+
* `cachedInputPerMTok` / `cacheWritePerMTok` are optional: not every model's
|
|
17
|
+
* entry carries verified cache rates. When a model has cached or
|
|
18
|
+
* cache-write tokens but no rate for them, `priceUsd` falls back to the
|
|
19
|
+
* full (non-cached) input rate — the same fail-toward-overestimate
|
|
20
|
+
* direction as the unknown-model error, never fail-toward-underestimate.
|
|
21
|
+
*/
|
|
22
|
+
import { type ModelPricing, type UsageTokens } from "./pricing-core.js";
|
|
23
|
+
export { computeCost } from "./pricing-core.js";
|
|
24
|
+
export type { ModelPricing, TokenizerEncoding, UsageTokens } from "./pricing-core.js";
|
|
25
|
+
/** Bump whenever a rate changes so durable reservations retain their original terms. */
|
|
26
|
+
export declare const PRICING_VERSION = "2026-09-06";
|
|
27
|
+
/** Looks up a model's pricing/tokenizer entry, or throws (fail closed). */
|
|
28
|
+
export declare function getModelPricing(model: string): ModelPricing;
|
|
29
|
+
export declare function supportedModels(): string[];
|
|
30
|
+
export declare function priceUsd(model: string, usage: UsageTokens): number;
|