@arnilo/prism 0.3.2 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +50 -1
- package/README.md +42 -62
- package/dist/agent-run-lifecycle.js +4 -0
- package/dist/agent-run-state.d.ts +5 -2
- package/dist/agent-run-state.js +18 -8
- package/dist/agent-session/session/assemble.d.ts +6 -0
- package/dist/agent-session/session/assemble.js +391 -0
- package/dist/agent-session/session/persist.d.ts +28 -0
- package/dist/agent-session/session/persist.js +166 -0
- package/dist/agent-session/session/provider-round.d.ts +6 -0
- package/dist/agent-session/session/provider-round.js +231 -0
- package/dist/agent-session/session/tool-round.d.ts +31 -0
- package/dist/agent-session/session/tool-round.js +473 -0
- package/dist/agent-session/session/types.d.ts +115 -0
- package/dist/agent-session/session/types.js +5 -0
- package/dist/agent-session/session.d.ts +54 -41
- package/dist/agent-session/session.js +23 -1132
- package/dist/capture.d.ts +63 -0
- package/dist/capture.js +67 -0
- package/dist/cli-dev.d.ts +29 -0
- package/dist/cli-dev.js +52 -0
- package/dist/cli-init.d.ts +34 -3
- package/dist/cli-init.js +192 -24
- package/dist/cli-runner.d.ts +6 -2
- package/dist/cli-runner.js +57 -10
- package/dist/content.d.ts +3 -3
- package/dist/content.js +3 -1
- package/dist/contracts-core/agent.d.ts +8 -0
- package/dist/contracts-core/batch.d.ts +97 -0
- package/dist/contracts-core/batch.js +65 -0
- package/dist/contracts-core/content.d.ts +72 -1
- package/dist/contracts-core/embeddings.d.ts +30 -0
- package/dist/contracts-core/embeddings.js +17 -0
- package/dist/contracts-core/images.d.ts +60 -0
- package/dist/contracts-core/images.js +17 -0
- package/dist/contracts-core/moderation.d.ts +46 -0
- package/dist/contracts-core/moderation.js +34 -0
- package/dist/contracts-core/speech.d.ts +39 -0
- package/dist/contracts-core/speech.js +17 -0
- package/dist/contracts-core/transcription.d.ts +48 -0
- package/dist/contracts-core/transcription.js +17 -0
- package/dist/contracts-core/video.d.ts +61 -0
- package/dist/contracts-core/video.js +17 -0
- package/dist/contracts-core.d.ts +7 -0
- package/dist/contracts-core.js +7 -0
- package/dist/contracts-protocol.d.ts +18 -0
- package/dist/contracts-run-state.d.ts +1 -2
- package/dist/index.d.ts +7 -3
- package/dist/index.js +5 -3
- package/dist/input.d.ts +8 -0
- package/dist/input.js +4 -0
- package/dist/node/agent-definitions.d.ts +1 -8
- package/dist/node/agent-definitions.js +0 -34
- package/dist/node/settings.d.ts +0 -1
- package/dist/node/settings.js +0 -5
- package/dist/pinned-fetch.js +29 -3
- package/dist/provider-events.js +3 -4
- package/dist/providers/media.d.ts +1 -2
- package/dist/providers/media.js +1 -4
- package/dist/rpc.d.ts +1 -1
- package/dist/rpc.js +4 -4
- package/dist/testing/persistence-schema.d.ts +1 -1
- package/dist/testing/persistence-schema.js +32 -28
- package/dist/testing/provider-conformance.d.ts +114 -5
- package/dist/testing/provider-conformance.js +342 -0
- package/dist/testing/tool-conformance.d.ts +25 -0
- package/dist/testing/tool-conformance.js +128 -1
- package/dist/testing/tool-effect-store-conformance.d.ts +0 -1
- package/dist/testing/tool-effect-store-conformance.js +0 -3
- package/dist/thinking.d.ts +48 -9
- package/dist/thinking.js +134 -8
- package/dist/tool-search.d.ts +76 -0
- package/dist/tool-search.js +199 -0
- package/docs/0.1.0-readiness.md +3 -3
- package/docs/a2a.md +2 -2
- package/docs/acp-agent.md +1 -1
- package/docs/acp.md +3 -3
- package/docs/ag-ui-adoption.md +1 -1
- package/docs/ag-ui.md +1 -2
- package/docs/agent-definitions.md +1 -1
- package/docs/agent-events.md +5 -5
- package/docs/agent-identity.md +13 -2
- package/docs/audit-export.md +3 -3
- package/docs/batch-jobs.md +120 -0
- package/docs/browser-automation.md +5 -5
- package/docs/caveman.md +2 -2
- package/docs/cli-rpc.md +43 -9
- package/docs/coding-agent-tools.md +19 -19
- package/docs/coding-review-and-diagnostics.md +2 -2
- package/docs/coding-security.md +5 -5
- package/docs/coding-tools.md +82 -0
- package/docs/coding-workspaces.md +2 -2
- package/docs/compaction-and-retry.md +2 -2
- package/docs/compaction-llm.md +4 -4
- package/docs/compaction-observational-memory.md +3 -3
- package/docs/computer-use-linux.md +13 -2
- package/docs/context-and-skills.md +3 -1
- package/docs/conversations.md +4 -4
- package/docs/core.md +85 -0
- package/docs/credential-storage.md +12 -8
- package/docs/credentials-and-redaction.md +1 -1
- package/docs/data-classification.md +1 -1
- package/docs/database-persistence.md +7 -3
- package/docs/dev-inspector.md +103 -0
- package/docs/device-adapters.md +2 -2
- package/docs/diagrams.md +247 -0
- package/docs/document-reader.md +6 -6
- package/docs/documents.md +214 -0
- package/docs/embeddings.md +112 -0
- package/docs/enterprise-postgres-state.md +7 -7
- package/docs/evaluations.md +41 -7
- package/docs/extensions.md +3 -3
- package/docs/forge-integration.md +3 -3
- package/docs/graft.md +5 -5
- package/docs/guardrails.md +2 -2
- package/docs/host-security.md +16 -15
- package/docs/image-generation.md +129 -0
- package/docs/impeccable.md +7 -5
- package/docs/index.md +84 -46
- package/docs/indexed-code-search.md +2 -2
- package/docs/language-intelligence.md +4 -4
- package/docs/live-testing.md +126 -0
- package/docs/mcp-tools.md +44 -13
- package/docs/middleware-hooks.md +1 -1
- package/docs/migrate-to-0.4.md +312 -0
- package/docs/migrate-to-0.5.md +122 -0
- package/docs/migration.md +51 -1
- package/docs/model-registry.md +38 -0
- package/docs/model-routing.md +6 -6
- package/docs/moderation.md +117 -0
- package/docs/multi-agent-patterns.md +177 -0
- package/docs/multimodal-content.md +27 -3
- package/docs/obscura.md +12 -12
- package/docs/observability.md +32 -7
- package/docs/openapi-tools.md +14 -4
- package/docs/operations.md +11 -0
- package/docs/performance.md +30 -10
- package/docs/persistence-credentials-multimodality-primitives.md +7 -7
- package/docs/policy-and-audit.md +18 -8
- package/docs/ponytail.md +3 -3
- package/docs/postgres-persistence.md +5 -5
- package/docs/process-sessions.md +2 -2
- package/docs/prompt-registry.md +106 -0
- package/docs/provider-caching.md +36 -32
- package/docs/provider-conformance.md +24 -2
- package/docs/provider-packages.md +58 -22
- package/docs/provider-primitives.md +5 -5
- package/docs/provider-request-policies.md +1 -1
- package/docs/providers/ai-sdk.md +18 -6
- package/docs/providers/alibaba.md +10 -6
- package/docs/providers/anthropic.md +10 -6
- package/docs/providers/azure.md +20 -4
- package/docs/providers/bedrock.md +18 -3
- package/docs/providers/clinepass.md +7 -3
- package/docs/providers/commandcode.md +253 -0
- package/docs/providers/deepseek.md +7 -3
- package/docs/providers/google.md +8 -4
- package/docs/providers/hyper.md +284 -0
- package/docs/providers/kimi.md +7 -3
- package/docs/providers/neuralwatt.md +12 -8
- package/docs/providers/ollama.md +18 -3
- package/docs/providers/openai-compatible.md +5 -1
- package/docs/providers/openai.md +9 -5
- package/docs/providers/opencode-go.md +8 -4
- package/docs/providers/openrouter.md +8 -4
- package/docs/providers/vertex.md +21 -5
- package/docs/providers/xai.md +7 -3
- package/docs/providers/zai.md +7 -3
- package/docs/rag.md +31 -9
- package/docs/release-and-install.md +181 -76
- package/docs/resource-loading.md +1 -1
- package/docs/runs-and-usage.md +28 -3
- package/docs/server.md +94 -5
- package/docs/settings-auth-trust-security.md +7 -5
- package/docs/sheets.md +229 -0
- package/docs/speech.md +126 -0
- package/docs/sqlite-persistence.md +4 -4
- package/docs/supervisors.md +4 -3
- package/docs/thinking-and-reasoning.md +93 -60
- package/docs/tool-conformance.md +28 -3
- package/docs/tool-execution-primitives.md +8 -8
- package/docs/tools.md +32 -5
- package/docs/web-tools.md +3 -3
- package/docs/wiki.md +7 -7
- package/docs/work-artifacts-and-review.md +17 -6
- package/docs/work-connectors.md +4 -4
- package/docs/work-tools.md +5 -5
- package/docs/workflow-orchestration-primitives.md +35 -11
- package/docs/workflows.md +74 -13
- package/docs/working-and-semantic-memory.md +53 -5
- package/package.json +14 -31
- package/templates/README.md +23 -0
- package/templates/deep-research/README.md.tmpl +47 -0
- package/templates/deep-research/env.example.tmpl +12 -0
- package/templates/deep-research/gitignore.tmpl +7 -0
- package/templates/deep-research/manifest.json +12 -0
- package/templates/deep-research/package.json.tmpl +23 -0
- package/templates/deep-research/src/agent.ts.tmpl +81 -0
- package/templates/deep-research/src/index.ts.tmpl +53 -0
- package/templates/deep-research/src/tests/research.test.ts.tmpl +114 -0
- package/templates/deep-research/src/tools.ts.tmpl +86 -0
- package/templates/deep-research/src/types.ts.tmpl +45 -0
- package/templates/deep-research/src/workflow.ts.tmpl +156 -0
- package/templates/deep-research/tsconfig.json.tmpl +15 -0
- package/templates/init/manifest.json +5 -0
- package/templates/init/package.json.tmpl +2 -1
- package/templates/init/providers.json +40 -24
- package/docs/antigravity-agent.md +0 -207
package/docs/providers/ai-sdk.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## What it does
|
|
4
4
|
|
|
5
|
-
`@arnilo/prism-
|
|
5
|
+
`@arnilo/prism-providers/ai-sdk` adapts a host-supplied AI SDK `LanguageModelV4` into a Prism `AIProvider`. It maps Prism messages, tools, and structured-output options into `doStream` call options, then translates stream parts into Prism provider events incrementally.
|
|
6
6
|
|
|
7
7
|
Core `@arnilo/prism` does not depend on the AI SDK.
|
|
8
8
|
|
|
@@ -12,6 +12,7 @@ Core `@arnilo/prism` does not depend on the AI SDK.
|
|
|
12
12
|
| --- | --- | --- |
|
|
13
13
|
| `4.0.3` | `LanguageModelV4`, `specificationVersion: "v4"` | Supported and offline-tested |
|
|
14
14
|
| `4.0.4` | `LanguageModelV4`, `specificationVersion: "v4"` | Supported and offline-tested |
|
|
15
|
+
| `4.0.10` | `LanguageModelV4`, `specificationVersion: "v4"` | Current peer; supported and offline-tested |
|
|
15
16
|
|
|
16
17
|
The peer dependency is intentionally exact. `createAiSdkProvider()` reads its resolved `@ai-sdk/provider/package.json` version during setup and throws typed `AiSdkProviderError { code: "unsupported_version" }` for an unlisted version; it does not infer compatibility from a matching `"v4"` string.
|
|
17
18
|
|
|
@@ -24,7 +25,7 @@ Do not use it as a credential store, model catalog, or high-level `streamText`/`
|
|
|
24
25
|
## Inputs / request
|
|
25
26
|
|
|
26
27
|
```ts
|
|
27
|
-
import { createAiSdkProvider } from "@arnilo/prism-
|
|
28
|
+
import { createAiSdkProvider } from "@arnilo/prism-providers/ai-sdk";
|
|
28
29
|
|
|
29
30
|
createAiSdkProvider(options: {
|
|
30
31
|
model: LanguageModelV4;
|
|
@@ -91,7 +92,7 @@ No AI SDK stream part is silently coerced into Prism content: the table above ma
|
|
|
91
92
|
|
|
92
93
|
```ts
|
|
93
94
|
import { createAgent } from "@arnilo/prism";
|
|
94
|
-
import { createAiSdkProvider } from "@arnilo/prism-
|
|
95
|
+
import { createAiSdkProvider } from "@arnilo/prism-providers/ai-sdk";
|
|
95
96
|
|
|
96
97
|
const provider = createAiSdkProvider({ model: hostCreatedLanguageModelV4 });
|
|
97
98
|
|
|
@@ -131,7 +132,7 @@ See [Provider caching](../provider-caching.md) for the cross-provider matrix.
|
|
|
131
132
|
|
|
132
133
|
## Thinking and reasoning
|
|
133
134
|
|
|
134
|
-
Reasoning effort, budgets, and provider-specific thinking controls are **host-model-owned**. Prism does not map `ThinkingLevel` into AI SDK call options
|
|
135
|
+
Reasoning effort, budgets, and provider-specific thinking controls are **host-model-owned**. Prism does not map `ThinkingLevel` into AI SDK call options: the deliberate `noop` compat family (stamped `compat.thinkingFamily: "noop"` on AI SDK models, and the `thinkingFamilyForModel` fallback) leaves the request options unchanged. Hosts configure reasoning on the AI SDK model (for example OpenAI `reasoning.effort` via AI SDK `providerOptions`) and may pass per-turn overrides through `ProviderRequestOptions.compat` / `extra`, which the adapter forwards as `providerOptions.prism`.
|
|
135
136
|
|
|
136
137
|
Stream mapping:
|
|
137
138
|
|
|
@@ -144,8 +145,8 @@ Official evidence: [Custom providers / LanguageModelV4](https://ai-sdk.dev/provi
|
|
|
144
145
|
|
|
145
146
|
## Extension and configuration notes
|
|
146
147
|
|
|
147
|
-
- Peer dependency: `@ai-sdk/provider@4.0.
|
|
148
|
-
- First-party HTTP providers remain independent; this adapter is available directly
|
|
148
|
+
- Peer dependency: `@ai-sdk/provider@4.0.10` (matrix also lists `4.0.3` and `4.0.4`). Upgrade policy adds a matrix row and offline conformance fixture before accepting any new version.
|
|
149
|
+
- First-party HTTP providers remain independent; this adapter is available directly or through `@arnilo/prism-providers`. Installation does not select a model or invoke AI SDK.
|
|
149
150
|
- `options.compat` / `options.extra` pass through as AI SDK `providerOptions.prism`.
|
|
150
151
|
- Export helpers `toAiSdkCallOptions`, `toAiSdkPrompt`, and `mapAiSdkStream` for tests and custom hosts.
|
|
151
152
|
|
|
@@ -157,6 +158,17 @@ Official evidence: [Custom providers / LanguageModelV4](https://ai-sdk.dev/provi
|
|
|
157
158
|
- Unsupported content and stream parts fail closed before/at mapping; `structuredOutput.strict` is rejected because V4 cannot carry it.
|
|
158
159
|
- Pass `redactor` for direct use; agent runs apply their active redactor. Provider metadata/warnings never become prompt, tool, event, or telemetry content.
|
|
159
160
|
|
|
161
|
+
## Live probe
|
|
162
|
+
|
|
163
|
+
Runs the adapter over the real `@ai-sdk/openai` provider package (dev dependency) to prove the `LanguageModelV4` mapping on a genuine AI SDK wire:
|
|
164
|
+
|
|
165
|
+
```bash
|
|
166
|
+
PRISM_LIVE_PROVIDER_TESTS=1 OPENAI_API_KEY=... \
|
|
167
|
+
node --test packages/prism-providers/dist/ai-sdk/__tests__/live.test.js
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
`PRISM_LIVE_AISDK_MODEL` overrides the probed model (default `gpt-5.1`). Without a key the suite skips.
|
|
171
|
+
|
|
160
172
|
## Related APIs
|
|
161
173
|
|
|
162
174
|
- [Provider packages](../provider-packages.md)
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## What it does
|
|
4
4
|
|
|
5
|
-
`@arnilo/prism-
|
|
5
|
+
`@arnilo/prism-providers/alibaba` is a side-effect-free adapter for Alibaba Cloud
|
|
6
6
|
Model Studio / DashScope (including the Coding Plan) over the **OpenAI-compatible**
|
|
7
7
|
`POST {base}/chat/completions` endpoint.
|
|
8
8
|
|
|
@@ -64,7 +64,7 @@ import {
|
|
|
64
64
|
listAlibabaModels,
|
|
65
65
|
defineAlibabaModel,
|
|
66
66
|
alibabaBaseUrl,
|
|
67
|
-
} from "@arnilo/prism-
|
|
67
|
+
} from "@arnilo/prism-providers/alibaba";
|
|
68
68
|
|
|
69
69
|
createAlibabaProviderPackage(options: AlibabaProviderPackageOptions): ProviderPackage
|
|
70
70
|
createAlibabaProvider(options?: AlibabaProviderOptions): AIProvider
|
|
@@ -104,7 +104,7 @@ verbatim via `baseUrl`.
|
|
|
104
104
|
dependency-free).
|
|
105
105
|
|
|
106
106
|
```ts
|
|
107
|
-
import { createAlibabaEmbedder } from "@arnilo/prism-
|
|
107
|
+
import { createAlibabaEmbedder } from "@arnilo/prism-providers/alibaba";
|
|
108
108
|
|
|
109
109
|
const embedder = createAlibabaEmbedder({
|
|
110
110
|
apiKey: process.env.DASHSCOPE_API_KEY,
|
|
@@ -157,7 +157,7 @@ reranker. The verified compatible route is workspace-dedicated only:
|
|
|
157
157
|
(`qwen3-rerank`, ≤500 documents, 4,000 tokens/item; base path `compatible-api/v1`,
|
|
158
158
|
not `compatible-mode/v1`). A future `createAlibabaReranker` over that route is
|
|
159
159
|
demand-gated: implement when a caller supplies a workspace-dedicated `baseUrl` and
|
|
160
|
-
needs rerank (structural `Reranker` shape from `@arnilo/prism-rag`, no new
|
|
160
|
+
needs rerank (structural `Reranker` shape from `@arnilo/prism-memory/rag`, no new
|
|
161
161
|
dependency). Multimodal rerank (`qwen3-vl-rerank`) is native-only and stays out.
|
|
162
162
|
|
|
163
163
|
## Outputs / response / events
|
|
@@ -208,7 +208,7 @@ import { createExtensionKernel } from "@arnilo/prism";
|
|
|
208
208
|
import {
|
|
209
209
|
createAlibabaProviderPackage,
|
|
210
210
|
listAlibabaModels,
|
|
211
|
-
} from "@arnilo/prism-
|
|
211
|
+
} from "@arnilo/prism-providers/alibaba";
|
|
212
212
|
|
|
213
213
|
const kernel = createExtensionKernel();
|
|
214
214
|
|
|
@@ -255,7 +255,7 @@ await kernel.load([
|
|
|
255
255
|
`Authorization: Bearer`; keys are redacted from all thrown errors (including
|
|
256
256
|
discovery failures). No local filesystem paths enter request payloads.
|
|
257
257
|
- Opt-in live probe (never part of `npm test`/CI):
|
|
258
|
-
`PRISM_LIVE_DASHSCOPE_KEY=… npm run test:live --workspace @arnilo/prism-
|
|
258
|
+
`PRISM_LIVE_DASHSCOPE_KEY=… npm run test:live --workspace @arnilo/prism-providers/alibaba`
|
|
259
259
|
exercises an embeddings round-trip against the real endpoint (model override via
|
|
260
260
|
`PRISM_LIVE_DASHSCOPE_MODEL`); absent env = documented skip, never a failure.
|
|
261
261
|
- Caller-supplied `ProviderRequest.options.headers` can add non-owned headers, but
|
|
@@ -264,6 +264,10 @@ await kernel.load([
|
|
|
264
264
|
- Model discovery is caller-gated and never invoked in the provider hot path.
|
|
265
265
|
- Live tests stay opt-in; default tests are network-free.
|
|
266
266
|
|
|
267
|
+
## Thinking and reasoning
|
|
268
|
+
|
|
269
|
+
Qwen hybrid models route through the `thinking_type` family: `none` maps to `enable_thinking: false`, any other declared level to `true` — there are no effort levels upstream, so declared `capabilities.thinkingLevels` for hybrid models are the on/off set (`none`–`max` used as a toggle vocabulary). Thinking-only models (`qwq*`, `*-thinking`) always think and never receive `enable_thinking: false`; `thinking_budget` stays a package-local passthrough. See [Thinking and reasoning](../thinking-and-reasoning.md).
|
|
270
|
+
|
|
267
271
|
## Related APIs
|
|
268
272
|
|
|
269
273
|
- [Provider packages](../provider-packages.md): `defineProviderPackage`,
|
|
@@ -2,13 +2,13 @@
|
|
|
2
2
|
|
|
3
3
|
## What it does
|
|
4
4
|
|
|
5
|
-
`@arnilo/prism-
|
|
5
|
+
`@arnilo/prism-providers/anthropic` is the first-party Anthropic Messages provider for Prism (`POST /v1/messages`). Setup is side-effect-free: no network, env scan, or keychain lookup during import/setup. Wire format is package-local (OpenCode Go / Kimi Anthropic routes are pattern-only, not a shared core serializer).
|
|
6
6
|
|
|
7
7
|
## When to use it
|
|
8
8
|
|
|
9
9
|
Use for native Claude Messages (tools, `cache_control`, thinking/reasoning, media, usage, abort). Prefer this over the AI SDK escape hatch when Anthropic is a primary coding host.
|
|
10
10
|
|
|
11
|
-
Do **not** use for OpenCode Go Anthropic *route* hosting (`@arnilo/prism-
|
|
11
|
+
Do **not** use for OpenCode Go Anthropic *route* hosting (`@arnilo/prism-providers/opencode-go`), automatic credential discovery, Claude Code credential-file/setup-token import, or Claude.ai subscription login/routing. This package is API-key-only.
|
|
12
12
|
|
|
13
13
|
## Inputs / request
|
|
14
14
|
|
|
@@ -18,7 +18,7 @@ import {
|
|
|
18
18
|
createAnthropicMessagesProvider,
|
|
19
19
|
listAnthropicModels,
|
|
20
20
|
defineAnthropicModel,
|
|
21
|
-
} from "@arnilo/prism-
|
|
21
|
+
} from "@arnilo/prism-providers/anthropic";
|
|
22
22
|
|
|
23
23
|
createAnthropicProviderPackage(options?: AnthropicProviderPackageOptions): ProviderPackage
|
|
24
24
|
createAnthropicMessagesProvider(options?): AIProvider
|
|
@@ -60,7 +60,7 @@ Featured offline aliases: `claude-opus-4-8`, `claude-sonnet-5`, `claude-haiku-4-
|
|
|
60
60
|
|
|
61
61
|
```ts
|
|
62
62
|
import { createProviderRegistry, createModelRegistry } from "@arnilo/prism";
|
|
63
|
-
import { createAnthropicProviderPackage, listAnthropicModels } from "@arnilo/prism-
|
|
63
|
+
import { createAnthropicProviderPackage, listAnthropicModels } from "@arnilo/prism-providers/anthropic";
|
|
64
64
|
|
|
65
65
|
const api = /* ExtensionAPI or host registries */;
|
|
66
66
|
api.registerProviderPackage(createAnthropicProviderPackage({ apiKey: hostKey }));
|
|
@@ -73,7 +73,7 @@ api.registerProviderPackage(createAnthropicProviderPackage({ apiKey: hostKey, mo
|
|
|
73
73
|
## Extension and configuration notes
|
|
74
74
|
|
|
75
75
|
- Register via `defineProviderPackage` / host registries; no package auto-discovery.
|
|
76
|
-
- AI SDK (`@arnilo/prism-
|
|
76
|
+
- AI SDK (`@arnilo/prism-providers/ai-sdk`) remains an escape hatch, not the primary Anthropic path.
|
|
77
77
|
- Live smoke: `PRISM_LIVE_PROVIDER_TESTS=1` + `ANTHROPIC_API_KEY`.
|
|
78
78
|
- Anthropic says OAuth is for purchasers' ordinary Claude Code/native-app use; developers building products must use Claude Console API keys or a supported cloud provider and may not offer Claude.ai login or route Free/Pro/Max credentials ([legal and compliance](https://docs.anthropic.com/en/docs/claude-code/legal-and-compliance)). Prism therefore has no Anthropic subscription OAuth API or token-import shortcut.
|
|
79
79
|
|
|
@@ -84,10 +84,14 @@ api.registerProviderPackage(createAnthropicProviderPackage({ apiKey: hostKey, mo
|
|
|
84
84
|
- Media/SSRF bounds reuse `@arnilo/prism/providers/media` / transport helpers.
|
|
85
85
|
- Offline conformance: `@arnilo/prism/testing/provider-conformance`.
|
|
86
86
|
|
|
87
|
+
## Thinking and reasoning
|
|
88
|
+
|
|
89
|
+
Anthropic models route through the `output_config_effort` family: the adapter merges `compat.output_config.effort` (from `applyThinkingLevelForModel`) and the provider emits `output_config: { effort }` on Messages bodies. Declared levels per generation (from `capabilities.thinkingLevels`): Opus 4.8/4.7, Sonnet 5, Fable/Mythos 5, Opus 5 accept `low`–`max` incl. `xhigh`; Mythos Preview, Opus 4.6, Sonnet 4.6 accept `low/medium/high/max`; Opus 4.5 accepts `low/medium/high`; Haiku 4.5 declares `none`–`high` (upstream effort support undocumented — live-probe pending). Undeclared levels snap to the nearest declared level (ladder distance, ties up); values below the minimum snap up. Thinking type: `adaptive` on 4.6+/Sonnet 5/Fable/Mythos (bare `enabled` maps to `adaptive`); legacy 4.5 models get `enabled` plus a `budget_tokens` default of 10000 when absent. See [Thinking and reasoning](../thinking-and-reasoning.md).
|
|
90
|
+
|
|
87
91
|
## Related APIs
|
|
88
92
|
|
|
89
93
|
- [Provider packages](../provider-packages.md): package setup + discovery contract.
|
|
90
94
|
- [Provider caching](../provider-caching.md): `cache_control` breakpoints.
|
|
91
95
|
- [Thinking and reasoning](../thinking-and-reasoning.md): portable thinking helpers.
|
|
92
96
|
- [Provider conformance](../provider-conformance.md): network-free assertions.
|
|
93
|
-
- Package README: [
|
|
97
|
+
- Package README: [`@arnilo/prism-providers` family README](../../packages/prism-providers/README.md)
|
package/docs/providers/azure.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## What it does
|
|
4
4
|
|
|
5
|
-
`@arnilo/prism-
|
|
5
|
+
`@arnilo/prism-providers/azure` registers an Azure OpenAI / Foundry Chat Completions provider that uses host-supplied Entra workload identity (Bearer) or Azure resource keys (`api-key`). Deployment URLs keep the configured endpoint host (custom subdomain, private endpoint, or VNet FQDN).
|
|
6
6
|
|
|
7
7
|
## When to use it
|
|
8
8
|
|
|
@@ -11,7 +11,7 @@ Use it for enterprise Azure OpenAI / Foundry deployments with Managed Identity o
|
|
|
11
11
|
## Inputs / request
|
|
12
12
|
|
|
13
13
|
```ts
|
|
14
|
-
import { createAzureOpenAIProviderPackage } from "@arnilo/prism-
|
|
14
|
+
import { createAzureOpenAIProviderPackage } from "@arnilo/prism-providers/azure";
|
|
15
15
|
|
|
16
16
|
createAzureOpenAIProviderPackage({
|
|
17
17
|
endpoint: "https://my-resource.openai.azure.com",
|
|
@@ -56,7 +56,7 @@ Opt-in live canaries: inject real `fetch` + host credential behind host CI secre
|
|
|
56
56
|
|
|
57
57
|
## Extension and configuration notes
|
|
58
58
|
|
|
59
|
-
Register via `createExtensionKernel().load([createAzureOpenAIProviderPackage(...)])`. Pair with `@arnilo/prism-model-router` for residency allow-lists on Azure regions/endpoints.
|
|
59
|
+
Register via `createExtensionKernel().load([createAzureOpenAIProviderPackage(...)])`. Pair with `@arnilo/prism-core/governance/model-router` for residency allow-lists on Azure regions/endpoints.
|
|
60
60
|
|
|
61
61
|
## Security and performance notes
|
|
62
62
|
|
|
@@ -66,10 +66,26 @@ Register via `createExtensionKernel().load([createAzureOpenAIProviderPackage(...
|
|
|
66
66
|
- No Azure SDK dependency.
|
|
67
67
|
- Conformance-proven (Task 6): package `setup()` performs zero fetch and zero credential resolution; an already-aborted signal fails fast; a truncated SSE stream (no `data: [DONE]`) ends in an `error` event; Azure cache policy stays host-owned, so no cache wire fields (`cache_control`, `prompt_cache_*`) are emitted even when the request carries Prism cache hints — only upstream-reported `prompt_tokens_details.cached_tokens` maps to `Usage.cacheReadTokens`.
|
|
68
68
|
|
|
69
|
+
## Live probe
|
|
70
|
+
|
|
71
|
+
Opt-in smoke over your real Azure OpenAI resource:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
PRISM_LIVE_PROVIDER_TESTS=1 AZURE_OPENAI_ENDPOINT=https://<resource>.openai.azure.com \
|
|
75
|
+
AZURE_OPENAI_API_KEY=... PRISM_LIVE_AZURE_MODEL=<deployment-name> \
|
|
76
|
+
node --test packages/prism-providers/dist/azure/__tests__/live.test.js
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
The deployment name is the model knob (`AZURE_OPENAI_DEPLOYMENT` also works). Without any of these variables the suite skips — it never fails.
|
|
80
|
+
|
|
81
|
+
## Thinking and reasoning
|
|
82
|
+
|
|
83
|
+
Azure OpenAI deployments use the OpenAI-compatible wire: the provider forwards sanitized thinking compat (`reasoning_effort` with `effort`/`reasoningEffort` aliases, or a `reasoning` object with `effort` + preserved `summary`) via its body builder, snapping effort to the model's declared levels (gpt-5.1 → `none/low/medium/high`, ties up). Unrecognized compat keys are dropped. See [Thinking and reasoning](../thinking-and-reasoning.md).
|
|
84
|
+
|
|
69
85
|
## Related APIs
|
|
70
86
|
|
|
71
87
|
- [OpenAI-compatible provider](openai-compatible.md)
|
|
72
88
|
- [Provider packages](../provider-packages.md)
|
|
73
89
|
- [Model routing](../model-routing.md)
|
|
74
90
|
- [Credential storage](../credential-storage.md)
|
|
75
|
-
- Package README: [`@arnilo/prism-
|
|
91
|
+
- Package README: [`@arnilo/prism-providers` family README](../../packages/prism-providers/README.md)
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## What it does
|
|
4
4
|
|
|
5
|
-
`@arnilo/prism-
|
|
5
|
+
`@arnilo/prism-providers/bedrock` registers an Amazon Bedrock Runtime OpenAI-compatible Chat Completions provider. Hosts supply IAM/IRSA/assumed-role credentials; the package signs requests with SigV4 (no AWS SDK). Region and optional PrivateLink endpoint URLs are preserved.
|
|
6
6
|
|
|
7
7
|
## When to use it
|
|
8
8
|
|
|
@@ -11,7 +11,7 @@ Use it for enterprise Bedrock access under workload identity. Do not embed long-
|
|
|
11
11
|
## Inputs / request
|
|
12
12
|
|
|
13
13
|
```ts
|
|
14
|
-
import { createBedrockProviderPackage } from "@arnilo/prism-
|
|
14
|
+
import { createBedrockProviderPackage } from "@arnilo/prism-providers/bedrock";
|
|
15
15
|
|
|
16
16
|
createBedrockProviderPackage({
|
|
17
17
|
region: "eu-west-1",
|
|
@@ -66,9 +66,24 @@ Uses Bedrock’s OpenAI-compatible runtime route (not Converse eventstream). Hos
|
|
|
66
66
|
- Credential secrets are redacted from provider errors.
|
|
67
67
|
- No credential prefetch at import.
|
|
68
68
|
|
|
69
|
+
## Live probe
|
|
70
|
+
|
|
71
|
+
Opt-in smoke over real AWS Bedrock (package-local SigV4, static keys or session token):
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
PRISM_LIVE_PROVIDER_TESTS=1 AWS_ACCESS_KEY_ID=... AWS_SECRET_ACCESS_KEY=... AWS_REGION=us-east-1 \
|
|
75
|
+
node --test packages/prism-providers/dist/bedrock/__tests__/live.test.js
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
`PRISM_LIVE_BEDROCK_MODEL` overrides the probed model (default `us.anthropic.claude-haiku-4-5-20251001-v1:0`). Without credentials the suite skips.
|
|
79
|
+
|
|
80
|
+
## Thinking and reasoning
|
|
81
|
+
|
|
82
|
+
Bedrock OpenAI-compat chat expects snake_case `reasoning_effort` (with `effort`/`reasoningEffort` aliases) or a sanitized `reasoning` object. OpenAI-family models on Bedrock snap effort to their declared levels (gpt-5.1 → `none/low/medium/high`); non-OpenAI models pass through untouched. See [Thinking and reasoning](../thinking-and-reasoning.md).
|
|
83
|
+
|
|
69
84
|
## Related APIs
|
|
70
85
|
|
|
71
86
|
- [OpenAI-compatible provider](openai-compatible.md)
|
|
72
87
|
- [Model routing](../model-routing.md)
|
|
73
88
|
- [Provider packages](../provider-packages.md)
|
|
74
|
-
- Package README: [`@arnilo/prism-
|
|
89
|
+
- Package README: [`@arnilo/prism-providers` family README](../../packages/prism-providers/README.md)
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## What it does
|
|
4
4
|
|
|
5
|
-
`@arnilo/prism-
|
|
5
|
+
`@arnilo/prism-providers/clinepass` provides explicit, side-effect-free setup for
|
|
6
6
|
the ClinePass OpenAI-compatible Chat Completions API at
|
|
7
7
|
`https://api.cline.bot/api/v1`. Requests always stream. Model ids are official
|
|
8
8
|
`cline-pass/…` slugs from a static featured catalog.
|
|
@@ -19,7 +19,7 @@ or caller-gated `GET /models` (no documented OpenAI models endpoint).
|
|
|
19
19
|
## Inputs / request
|
|
20
20
|
|
|
21
21
|
```ts
|
|
22
|
-
import { createClinePassProviderPackage } from "@arnilo/prism-
|
|
22
|
+
import { createClinePassProviderPackage } from "@arnilo/prism-providers/clinepass";
|
|
23
23
|
|
|
24
24
|
createClinePassProviderPackage(options: ClinePassProviderPackageOptions): ProviderPackage
|
|
25
25
|
```
|
|
@@ -74,7 +74,7 @@ Completion budget is `max_completion_tokens` (not `max_tokens`).
|
|
|
74
74
|
|
|
75
75
|
```ts
|
|
76
76
|
import { createExtensionKernel } from "@arnilo/prism";
|
|
77
|
-
import { createClinePassProviderPackage } from "@arnilo/prism-
|
|
77
|
+
import { createClinePassProviderPackage } from "@arnilo/prism-providers/clinepass";
|
|
78
78
|
|
|
79
79
|
const kernel = createExtensionKernel();
|
|
80
80
|
await kernel.load([createClinePassProviderPackage({ apiKey: "fake-cline-key" })]);
|
|
@@ -106,6 +106,10 @@ await session.prompt("Plan the refactor", {
|
|
|
106
106
|
- Provider-owned headers win. One POST per generate. Bounded error bodies.
|
|
107
107
|
- Live tests: `PRISM_LIVE_PROVIDER_TESTS=1` plus `CLINE_API_KEY`.
|
|
108
108
|
|
|
109
|
+
## Thinking and reasoning
|
|
110
|
+
|
|
111
|
+
ClinePass routes through `reasoning_effort` with per-model slot maps (`compat.thinkingLevelMap`) as wire authority; declared `capabilities.thinkingLevels` mirror each map's portable slots (e.g. GLM: `none/low/medium/high/xhigh`). Portable `max` never reaches the wire (upstream 500s) — the map sends `high`; unsupported slots omit the field. See [Thinking and reasoning](../thinking-and-reasoning.md).
|
|
112
|
+
|
|
109
113
|
## Related APIs
|
|
110
114
|
|
|
111
115
|
- [Provider packages](../provider-packages.md)
|
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
# Command Code provider package
|
|
2
|
+
|
|
3
|
+
## What it does
|
|
4
|
+
|
|
5
|
+
`@arnilo/prism-providers/commandcode` provides explicit, side-effect-free setup
|
|
6
|
+
for the [Command Code Provider API](https://commandcode.ai/docs/provider) — an
|
|
7
|
+
aggregator exposing every top commercial and open model through OpenAI- and
|
|
8
|
+
Anthropic-compatible endpoints, billed at cost with deals auto-applied. The
|
|
9
|
+
package dual-routes by `ModelConfig.compat.route`:
|
|
10
|
+
|
|
11
|
+
| Route | Endpoint | Official model families |
|
|
12
|
+
| --- | --- | --- |
|
|
13
|
+
| `"openai"` (default) | `POST {baseUrl}/chat/completions` | everything except `claude-*` (GPT-5.6, DeepSeek, Kimi, GLM, MiniMax, Qwen, MiMo, Gemini flash, Grok) |
|
|
14
|
+
| `"anthropic"` | `POST {baseUrl}/messages` | `claude-*` tiers (Opus/Sonnet/Fable/Haiku) |
|
|
15
|
+
|
|
16
|
+
Default base URL is the official Provider API root:
|
|
17
|
+
|
|
18
|
+
```txt
|
|
19
|
+
https://api.commandcode.ai/provider/v1
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
Authentication: `Authorization: Bearer <key>` on the chat route,
|
|
23
|
+
`x-api-key` + `anthropic-version: 2023-06-01` on the messages route (Claude
|
|
24
|
+
Code compatibility). The same key authenticates the CLI and the API.
|
|
25
|
+
|
|
26
|
+
## When to use it
|
|
27
|
+
|
|
28
|
+
Use it when a host app wants Command Code models through Prism's `AgentSession`
|
|
29
|
+
runtime: dual-route serialization, `cache_control` breakpoints on Claude
|
|
30
|
+
models, implicit caching elsewhere, reasoning replay, caller-gated model
|
|
31
|
+
discovery, and optional zero-data-retention (`zdr: true` → provider-owned
|
|
32
|
+
`x-cmd-zdr: 1`, which routes only through ZDR-capable upstreams).
|
|
33
|
+
|
|
34
|
+
Do not use it for automatic credential discovery, setup-time catalog fetches,
|
|
35
|
+
or real-network tests (live probes are operator-gated, see below).
|
|
36
|
+
|
|
37
|
+
## Inputs / request
|
|
38
|
+
|
|
39
|
+
```ts
|
|
40
|
+
import {
|
|
41
|
+
createCommandCodeProviderPackage,
|
|
42
|
+
listCommandCodeModels,
|
|
43
|
+
} from "@arnilo/prism-providers/commandcode";
|
|
44
|
+
|
|
45
|
+
createCommandCodeProviderPackage(options: CommandCodeProviderPackageOptions): ProviderPackage
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
| Field | Type | Purpose |
|
|
49
|
+
| --- | --- | --- |
|
|
50
|
+
| `apiKey` | `CredentialValueSource` | Direct/callback/resolver API-key source. |
|
|
51
|
+
| `fetch` | `typeof fetch` | Optional fetch implementation for tests/hosts. |
|
|
52
|
+
| `baseUrl` | `string` | Overrides official `https://api.commandcode.ai/provider/v1`. |
|
|
53
|
+
| `models` | `readonly ModelConfig[]` | Overrides featured `commandCodeModels` defaults. |
|
|
54
|
+
| `zdr` | `boolean` | Enforce zero data retention (`x-cmd-zdr: 1`). May route to costlier upstreams or fail `422 cmd_zdr_no_providers`. |
|
|
55
|
+
|
|
56
|
+
`ProviderRequest.options.cache.breakpoints` select messages-route
|
|
57
|
+
`cache_control` markers for Claude models (max 4, no `ttl`).
|
|
58
|
+
|
|
59
|
+
## Outputs / response / events
|
|
60
|
+
|
|
61
|
+
| Surface | Behavior |
|
|
62
|
+
| --- | --- |
|
|
63
|
+
| Provider stream | Prism text, thinking, tool-call delta/final, `usage`, `done`, redacted `error`. |
|
|
64
|
+
| Stream completion | `done` only on completion evidence — chat route: `[DONE]` marker plus terminal `finish_reason`; messages route: `message_stop`. Truncated streams end with terminal `error`. |
|
|
65
|
+
| OpenAI thinking | `delta.reasoning_content` → thinking deltas; replay via top-level `reasoning_content` when `preserveThinking` (default), never folded into text. |
|
|
66
|
+
| Anthropic thinking | `thinking_delta` → thinking deltas; replay via Anthropic thinking blocks when `preserveThinking`. |
|
|
67
|
+
| Usage | Standard tokens + cache read/write per route; mapped through the shared OpenAI/Anthropic usage mappings. |
|
|
68
|
+
| Auth method | `api_key` for `commandcode`, credential name `apiKey`. |
|
|
69
|
+
|
|
70
|
+
## Request/response example
|
|
71
|
+
|
|
72
|
+
```json
|
|
73
|
+
{
|
|
74
|
+
"authorization": "Bearer cmd_…",
|
|
75
|
+
"content-type": "application/json"
|
|
76
|
+
}
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Messages route instead sends provider-owned `x-api-key: <key>` and
|
|
80
|
+
`anthropic-version: 2023-06-01`. All provider-owned headers are applied after
|
|
81
|
+
caller headers and cannot be overridden.
|
|
82
|
+
|
|
83
|
+
Chat-route body (thinking passthrough + preserved reasoning):
|
|
84
|
+
|
|
85
|
+
```json
|
|
86
|
+
{
|
|
87
|
+
"model": "Qwen/Qwen3.8-Flash",
|
|
88
|
+
"stream": true,
|
|
89
|
+
"stream_options": { "include_usage": true },
|
|
90
|
+
"max_tokens": 512,
|
|
91
|
+
"messages": [
|
|
92
|
+
{
|
|
93
|
+
"role": "assistant",
|
|
94
|
+
"tool_calls": [{ "id": "call_1", "type": "function", "function": { "name": "lookup", "arguments": "{}" } }],
|
|
95
|
+
"reasoning_content": "plan the lookup"
|
|
96
|
+
}
|
|
97
|
+
]
|
|
98
|
+
}
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
## Implementation example
|
|
102
|
+
|
|
103
|
+
```ts
|
|
104
|
+
import { createExtensionKernel } from "@arnilo/prism";
|
|
105
|
+
import { createCommandCodeProviderPackage } from "@arnilo/prism-providers/commandcode";
|
|
106
|
+
|
|
107
|
+
const kernel = createExtensionKernel();
|
|
108
|
+
await kernel.load([createCommandCodeProviderPackage({ apiKey: process.env.COMMAND_CODE_API_KEY })]);
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
Caller-gated live catalog (never runs during package setup):
|
|
112
|
+
|
|
113
|
+
```ts
|
|
114
|
+
const models = await listCommandCodeModels({ fetch }); // public endpoint, no auth needed
|
|
115
|
+
await kernel.load([createCommandCodeProviderPackage({ apiKey: process.env.COMMAND_CODE_API_KEY, models })]);
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
## Featured models and routes
|
|
119
|
+
|
|
120
|
+
Featured `commandCodeModels` is a curated 38-model bootstrap catalog: ids and
|
|
121
|
+
context windows from the live `GET /provider/v1/models` snapshot (2026-09, 67
|
|
122
|
+
ids), USD-per-million-token pricing from the docs table
|
|
123
|
+
(<https://commandcode.ai/docs/resources/pricing-limits>). `compat.pricing_source`
|
|
124
|
+
records caveats: open-source models bill at the **mean per-provider price**;
|
|
125
|
+
DeepSeek rates are **off-peak** (17h/day; peak 2× during 01–04 & 06–10 UTC);
|
|
126
|
+
deals (MiniMax M3 −50%, MiMo −98/99%) are already applied upstream. Custom
|
|
127
|
+
pricing metadata is always stripped before the wire.
|
|
128
|
+
|
|
129
|
+
| Model family | Route | Cache kind |
|
|
130
|
+
| --- | --- | --- |
|
|
131
|
+
| `claude-opus-5/4-8/4-7`, `claude-sonnet-5/4-6`, `claude-fable-5-1/5`, `claude-haiku-4-5` | `anthropic` | `cache_control` (max 4 breakpoints, no `ttl` — undocumented) |
|
|
132
|
+
| `gpt-5.6-sol/terra/luna` | `openai` | `implicit` (docs cache-write price recorded in `cost.cacheWrite`; explicit-key upgrade gated on live probe — Task 9) |
|
|
133
|
+
| `deepseek/*`, Kimi, GLM, MiniMax, Qwen, MiMo, Gemini flash, Grok | `openai` | `implicit` |
|
|
134
|
+
|
|
135
|
+
## Model discovery
|
|
136
|
+
|
|
137
|
+
```txt
|
|
138
|
+
GET https://api.commandcode.ai/provider/v1/models
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
Public endpoint — works without authentication and emits no auth header when no
|
|
142
|
+
key resolves. `listCommandCodeModels({ fetch?, baseUrl?, apiKey?, signal?,
|
|
143
|
+
headers? })` maps each `{ id, name, context_length }` entry to `ModelConfig`:
|
|
144
|
+
route from id (`claude-*` → anthropic), context window from the endpoint, and
|
|
145
|
+
featured docs metadata (cost/cache kind) applied when the id matches a curated
|
|
146
|
+
entry. The endpoint carries no pricing or capabilities — unknown ids get
|
|
147
|
+
route-derived cache kind and no cost. Discovery is **caller-gated** — setup
|
|
148
|
+
performs zero fetches.
|
|
149
|
+
|
|
150
|
+
## Thinking / reasoning
|
|
151
|
+
|
|
152
|
+
| Surface | Behavior |
|
|
153
|
+
| --- | --- |
|
|
154
|
+
| OpenAI route stream | `reasoning_content` → thinking deltas |
|
|
155
|
+
| OpenAI route replay | thinking blocks → top-level `reasoning_content` when `preserveThinking`; never folded into text |
|
|
156
|
+
| Anthropic route stream | `thinking_delta` → thinking deltas |
|
|
157
|
+
| Anthropic route replay | thinking blocks when `preserveThinking` |
|
|
158
|
+
|
|
159
|
+
Owned compat keys (`route`, `preserveThinking`, `pricing_source`) are stripped
|
|
160
|
+
before opaque compat spread so resolved values win.
|
|
161
|
+
|
|
162
|
+
## Extension and configuration notes
|
|
163
|
+
|
|
164
|
+
- Hosts choose base URL, model list, credential source, `fetch` impl, and ZDR.
|
|
165
|
+
- Route selection is explicit via `compat.route` (`"anthropic"` for `claude-*`
|
|
166
|
+
ids, default `"openai"`). Sending a model to the wrong endpoint 400s.
|
|
167
|
+
- Package contributes models via the extension `api` and an `api_key` auth method.
|
|
168
|
+
|
|
169
|
+
### Cache and session behavior
|
|
170
|
+
|
|
171
|
+
- The chat route sends **no** `cache_control` fields; it relies on OpenAI-style
|
|
172
|
+
implicit caching (upstream provider behavior, passed through). Read tokens
|
|
173
|
+
map from `prompt_tokens_details.cached_tokens` / `cache_write_tokens` /
|
|
174
|
+
`prompt_cache_hit_tokens`.
|
|
175
|
+
- The messages route applies `cache_control: { type: "ephemeral" }` markers
|
|
176
|
+
only to caller-selected `cache.breakpoints` (shared `applyCacheControl()`
|
|
177
|
+
helper) on the last content block of each selected message. A
|
|
178
|
+
`system_prompt` breakpoint serializes `system` as marked text blocks (plain
|
|
179
|
+
string otherwise). Caching is enabled unless disabled
|
|
180
|
+
(`cacheRetention: "none"` / `cache.mode: "off"`) and the model opts in via
|
|
181
|
+
`ModelConfig.cache.kind: "cache_control"`.
|
|
182
|
+
- **No `ttl` is ever emitted**: the upstream TTL window is undocumented on the
|
|
183
|
+
Provider API; `cacheRetention: "long"` must not produce a marker the gateway
|
|
184
|
+
may reject.
|
|
185
|
+
- Usage accounting per route: chat route maps
|
|
186
|
+
`prompt_tokens_details.cached_tokens`/`cache_write_tokens` (and
|
|
187
|
+
`prompt_cache_hit_tokens`); messages route maps
|
|
188
|
+
`cache_read_input_tokens`/`cache_creation_input_tokens`.
|
|
189
|
+
- Session identity is simple: no session header is emitted (undocumented).
|
|
190
|
+
|
|
191
|
+
### Live-verified mapping (findings ledger)
|
|
192
|
+
|
|
193
|
+
The following claims are encoded as operator-gated probes in
|
|
194
|
+
`packages/prism-providers/src/commandcode/__tests__/live.test.ts`. Each probe's
|
|
195
|
+
assertion encodes the documented claim, so a probe failure **is** the finding;
|
|
196
|
+
record the outcome here and adjust the mapping. Status: **pending operator
|
|
197
|
+
run** (no key in CI):
|
|
198
|
+
|
|
199
|
+
| # | Claim (documented) | Probe | Status |
|
|
200
|
+
| --- | --- | --- | --- |
|
|
201
|
+
| 1 | Warm chat-route replay reports cached tokens (implicit caching passes through) | `live_chat_route_reports_cached_tokens_on_warm_prefix_replay` | pending |
|
|
202
|
+
| 2 | `cache_control` on messages reports `cache_creation_input_tokens` on the creating call | `live_messages_route_cache_control_reports_creation_and_read_tokens` | pending |
|
|
203
|
+
| 3 | Same-prefix warm replay reads the created cache entry (TTL ≥ one request) | same probe (warm leg) | pending |
|
|
204
|
+
| 4 | OpenAI `prompt_cache_key` is accepted and honored for GPT-5.6 (explicit caching) — decides Task 9 | `live_gpt56_prompt_cache_key_passthrough_probe` | pending — Task 9 closed gated: **pass** → upgrade `gpt-5.6-*` to the OpenAI explicit mapping (`promptCacheKey`/`promptCacheOptions`/`applyPromptCacheBreakpoints` via `@arnilo/prism-providers/openai`, verified exported); **fail** (400/ignored/warm replay shows no cached tokens) → verified-negative, keep `implicit`, record here |
|
|
205
|
+
| 5 | OpenAI `reasoning_effort` is accepted (200) on the chat route | `live_reasoning_effort_is_accepted_on_chat_route` | pending |
|
|
206
|
+
| 6 | ZDR requests route (done) or fail `422 cmd_zdr_no_providers` when no ZDR-capable upstream exists | `live_zdr_route_probe_is_opt_in_and_routable` | pending |
|
|
207
|
+
|
|
208
|
+
Run the gate:
|
|
209
|
+
|
|
210
|
+
```sh
|
|
211
|
+
PRISM_LIVE_PROVIDER_TESTS=1 COMMAND_CODE_API_KEY=cmd_... \
|
|
212
|
+
npm run test --workspace=@arnilo/prism-providers/commandcode
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
## Security and performance notes
|
|
216
|
+
|
|
217
|
+
- SSE streams and HTTP error bodies use bounded `@arnilo/prism/providers/transport` helpers.
|
|
218
|
+
- No network calls during import, setup, build, or default tests.
|
|
219
|
+
- No automatic environment, file, keychain, or shell credential lookup.
|
|
220
|
+
- API keys are resolved per request from caller-supplied values or resolvers
|
|
221
|
+
and redacted from errors (upstream error bodies may carry the upstream
|
|
222
|
+
provider's message — always redacted).
|
|
223
|
+
- `403 upgrade_required` (Go plan — no API access) and
|
|
224
|
+
`422 cmd_zdr_no_providers` are non-retryable; `429` and `5xx` are retryable
|
|
225
|
+
with `retry-after` surfaced as `retry_after_ms`; `400/401/422` are
|
|
226
|
+
non-retryable.
|
|
227
|
+
- Caller headers cannot override provider-owned headers (`content-type`,
|
|
228
|
+
`authorization` on chat, `x-api-key`/`anthropic-version` on messages,
|
|
229
|
+
`x-cmd-zdr` when ZDR is opted in).
|
|
230
|
+
- Live tests stay opt-in behind `PRISM_LIVE_PROVIDER_TESTS=1` plus
|
|
231
|
+
`COMMAND_CODE_API_KEY`; default tests are network-free.
|
|
232
|
+
|
|
233
|
+
## Official evidence
|
|
234
|
+
|
|
235
|
+
- Command Code Provider API docs: `https://commandcode.ai/docs/provider`
|
|
236
|
+
- Pricing & limits (per-model USD, deals, off-peak): `https://commandcode.ai/docs/resources/pricing-limits`
|
|
237
|
+
- Live `GET https://api.commandcode.ai/provider/v1/models` snapshot (2026-09) — 67 ids, context windows
|
|
238
|
+
- Probe ledger above pending operator-gated live run; row 4 decides plan 055 Task 9
|
|
239
|
+
(explicit GPT-5.6 caching upgrade) — see `plans/055-First-Class-Hyper-And-Command-Code-Providers.md`
|
|
240
|
+
|
|
241
|
+
## Thinking and reasoning
|
|
242
|
+
|
|
243
|
+
Command Code models carry provenance-commented level tables: `claude-*` → `output_config_effort` (Anthropic-route effort sets mirroring the native Anthropic package), `gpt-5.6*` → `openai_reasoning` (`none`–`xhigh`), `deepseek-v4`/`kimi-k3`/`glm-5.3` → `reasoning_effort` (`low/high/max`), `glm-5.2` → `reasoning_effort` (`low`–`max`), Kimi-K2.x/MiniMax/Qwen → `thinking_type`, gemini-3.x → `noop` (the gateway chat route has no `thinking_level` wire, so no levels are declared), mimo/unknown → passthrough. Effort snaps to declared sets on both routes. See [Thinking and reasoning](../thinking-and-reasoning.md).
|
|
244
|
+
|
|
245
|
+
## Related APIs
|
|
246
|
+
|
|
247
|
+
- [Provider packages](../provider-packages.md): `defineProviderPackage`,
|
|
248
|
+
`ModelConfig`, discovery contract, request/cache policies.
|
|
249
|
+
- [Thinking and reasoning](../thinking-and-reasoning.md): per-turn `ThinkingLevel` → compat families.
|
|
250
|
+
- [Credentials and redaction](../credentials-and-redaction.md):
|
|
251
|
+
`resolveCredentialValue`, `redactSecrets`.
|
|
252
|
+
- [Provider caching](../provider-caching.md): per-provider cache behavior matrix.
|
|
253
|
+
- [Provider conformance](../provider-conformance.md): network-free adapter tests.
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## What it does
|
|
4
4
|
|
|
5
|
-
`@arnilo/prism-
|
|
5
|
+
`@arnilo/prism-providers/deepseek` provides explicit, side-effect-free setup for the
|
|
6
6
|
DeepSeek Chat Completions API (`POST /chat/completions`) with official thinking
|
|
7
7
|
mode, reasoning-effort mapping, tool-turn `reasoning_content` replay, and
|
|
8
8
|
implicit prefix caching.
|
|
@@ -26,7 +26,7 @@ import {
|
|
|
26
26
|
createDeepSeekProviderPackage,
|
|
27
27
|
defineDeepSeekModel,
|
|
28
28
|
listDeepSeekModels,
|
|
29
|
-
} from "@arnilo/prism-
|
|
29
|
+
} from "@arnilo/prism-providers/deepseek";
|
|
30
30
|
|
|
31
31
|
createDeepSeekProviderPackage(options: DeepSeekProviderPackageOptions): ProviderPackage
|
|
32
32
|
defineDeepSeekModel(config: DeepSeekModelConfig): ModelConfig
|
|
@@ -84,7 +84,7 @@ Unsupported media blocks fail before fetch. Text-only input.
|
|
|
84
84
|
|
|
85
85
|
```ts
|
|
86
86
|
import { createExtensionKernel } from "@arnilo/prism";
|
|
87
|
-
import { createDeepSeekProviderPackage, listDeepSeekModels } from "@arnilo/prism-
|
|
87
|
+
import { createDeepSeekProviderPackage, listDeepSeekModels } from "@arnilo/prism-providers/deepseek";
|
|
88
88
|
|
|
89
89
|
const kernel = createExtensionKernel();
|
|
90
90
|
await kernel.load([createDeepSeekProviderPackage({ apiKey: "fake-deepseek-key" })]);
|
|
@@ -129,6 +129,10 @@ await session.prompt("Plan the refactor", {
|
|
|
129
129
|
- One POST per generate. No provider retry loop.
|
|
130
130
|
- Live tests stay opt-in behind `PRISM_LIVE_PROVIDER_TESTS=1` plus `DEEPSEEK_API_KEY`.
|
|
131
131
|
|
|
132
|
+
## Thinking and reasoning
|
|
133
|
+
|
|
134
|
+
DeepSeek models declare `low/high/max` and stamp `reasoning_effort`; the wire table maps `medium`/`xhigh`→`high`, `none`/`minimal` stop thinking. `thinking.type: "enabled"/"disabled"` stays available (thinking on by default, `high`); a request-level `reasoning_effort: none` stops thinking only when no explicit `thinking` switch was sent. Tool turns must replay `reasoning_content` or the API returns 400. See [Thinking and reasoning](../thinking-and-reasoning.md).
|
|
135
|
+
|
|
132
136
|
## Related APIs
|
|
133
137
|
|
|
134
138
|
- [Provider packages](../provider-packages.md): `defineProviderPackage`,
|