@arnilo/prism 0.0.5 → 0.0.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/CHANGELOG.md +28 -1
  2. package/dist/agent-loops.d.ts +1 -0
  3. package/dist/agent-loops.js +26 -16
  4. package/dist/agents.js +2 -3
  5. package/dist/contracts.d.ts +2 -0
  6. package/dist/ids.d.ts +2 -0
  7. package/dist/ids.js +6 -0
  8. package/dist/index.d.ts +5 -1
  9. package/dist/index.js +3 -1
  10. package/dist/session-stores.js +2 -3
  11. package/dist/testing/persistence-schema.d.ts +45 -7
  12. package/dist/testing/persistence-schema.js +138 -24
  13. package/dist/thinking.d.ts +42 -0
  14. package/dist/thinking.js +92 -0
  15. package/dist/tools.js +2 -3
  16. package/dist/use-case-model.d.ts +63 -0
  17. package/dist/use-case-model.js +52 -0
  18. package/docs/a2a.md +4 -2
  19. package/docs/agent-events.md +10 -15
  20. package/docs/agent-loops.md +11 -8
  21. package/docs/coding-agent-tools.md +33 -12
  22. package/docs/coding-security.md +2 -2
  23. package/docs/compaction-llm.md +17 -7
  24. package/docs/compaction-observational-memory.md +28 -4
  25. package/docs/credential-storage.md +58 -9
  26. package/docs/credentials-and-redaction.md +1 -1
  27. package/docs/database-persistence.md +8 -3
  28. package/docs/host-security.md +10 -6
  29. package/docs/index.md +23 -20
  30. package/docs/mcp-tools.md +26 -10
  31. package/docs/migration.md +146 -2
  32. package/docs/node-filesystem-config.md +1 -0
  33. package/docs/node-jsonl-session-store.md +5 -4
  34. package/docs/postgres-persistence.md +3 -3
  35. package/docs/provider-caching.md +16 -4
  36. package/docs/provider-conformance.md +39 -1
  37. package/docs/provider-packages.md +60 -3
  38. package/docs/providers/ai-sdk.md +36 -0
  39. package/docs/providers/kimi.md +124 -61
  40. package/docs/providers/neuralwatt.md +19 -13
  41. package/docs/providers/openai.md +56 -13
  42. package/docs/providers/opencode-go.md +118 -30
  43. package/docs/providers/openrouter.md +105 -35
  44. package/docs/providers/zai.md +94 -45
  45. package/docs/release-and-install.md +47 -49
  46. package/docs/review-coverage-2026-07-17-provider-validation.md +192 -0
  47. package/docs/runs-and-usage.md +1 -1
  48. package/docs/sqlite-persistence.md +2 -2
  49. package/docs/structured-output.md +1 -1
  50. package/docs/thinking-and-reasoning.md +98 -0
  51. package/docs/tool-execution-primitives.md +3 -3
  52. package/docs/tools.md +15 -0
  53. package/docs/use-case-model-selection.md +109 -0
  54. package/docs/workflow-orchestration-primitives.md +1 -0
  55. package/docs/workflows.md +17 -10
  56. package/docs/working-and-semantic-memory.md +1 -0
  57. package/package.json +2 -2
@@ -0,0 +1,92 @@
1
+ import { mergeProviderRequestOptions } from "./provider-request-policy.js";
2
+ /**
3
+ * Portable thinking / reasoning effort levels shared across first-party providers.
4
+ * Model-dependent legality (which values a given model accepts) stays provider-owned.
5
+ */
6
+ export const THINKING_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh", "max"];
7
+ export function isThinkingLevel(value) {
8
+ return typeof value === "string" && THINKING_LEVELS.includes(value);
9
+ }
10
+ /**
11
+ * Normalize a host thinkingLevel string. Known levels are lowercased; other non-empty
12
+ * strings pass through as opaque effort values for forward-compatible provider fields.
13
+ */
14
+ export function normalizeThinkingLevel(level) {
15
+ const normalized = level.trim().toLowerCase();
16
+ if (!normalized)
17
+ return undefined;
18
+ return isThinkingLevel(normalized) ? normalized : normalized;
19
+ }
20
+ /**
21
+ * Build the `ProviderRequestOptions.compat` patch for a shared thinking level.
22
+ * Does not invent a second options tree — providers keep reading official fields from `compat`.
23
+ */
24
+ export function thinkingCompatFor(family, level) {
25
+ const normalized = typeof level === "string" ? normalizeThinkingLevel(level) : level;
26
+ if (!normalized || family === "noop")
27
+ return {};
28
+ switch (family) {
29
+ case "openai_reasoning":
30
+ return { reasoning: { effort: normalized } };
31
+ case "reasoning_effort":
32
+ return { reasoning_effort: normalized };
33
+ case "thinking_type":
34
+ return { thinking: { type: normalized === "none" ? "disabled" : "enabled" } };
35
+ default: {
36
+ const _exhaustive = family;
37
+ return _exhaustive;
38
+ }
39
+ }
40
+ }
41
+ /**
42
+ * Merge a shared thinking level into `providerOptions.compat` for the given family.
43
+ * Per-turn patches win over prior compat via {@link mergeProviderRequestOptions}.
44
+ */
45
+ export function applyThinkingLevel(options, level, family = "reasoning_effort") {
46
+ const normalized = normalizeThinkingLevel(String(level));
47
+ if (!normalized || family === "noop")
48
+ return options ?? {};
49
+ const patch = thinkingCompatFor(family, normalized);
50
+ if (family === "openai_reasoning" && options?.compat?.reasoning && typeof options.compat.reasoning === "object" && !Array.isArray(options.compat.reasoning)) {
51
+ return mergeProviderRequestOptions(options, {
52
+ compat: {
53
+ reasoning: {
54
+ ...options.compat.reasoning,
55
+ ...patch.reasoning,
56
+ },
57
+ },
58
+ });
59
+ }
60
+ return mergeProviderRequestOptions(options, { compat: patch });
61
+ }
62
+ /**
63
+ * Best-effort family inference from model metadata without a second options tree.
64
+ * Prefer an explicit family in hosts/use-case workers when the provider is known.
65
+ *
66
+ * Heuristics (ordered):
67
+ * 1. Existing `compat.thinking` object → `thinking_type`
68
+ * 2. Existing `compat.reasoning` → `openai_reasoning`
69
+ * 3. Existing `compat.reasoning_effort` → `reasoning_effort`
70
+ * 4. Provider id starting with `openai` → `openai_reasoning`
71
+ * 5. Provider id `neuralwatt` → `reasoning_effort`
72
+ * 6. `capabilities.reasoning` → `reasoning_effort` (portable string field)
73
+ * 7. Else `noop`
74
+ */
75
+ export function thinkingFamilyForModel(model) {
76
+ const compat = model.compat ?? {};
77
+ if (compat.thinking != null && typeof compat.thinking === "object")
78
+ return "thinking_type";
79
+ if (compat.reasoning != null)
80
+ return "openai_reasoning";
81
+ if (compat.reasoning_effort != null)
82
+ return "reasoning_effort";
83
+ const provider = model.provider.trim().toLowerCase();
84
+ if (provider === "openai" || provider.startsWith("openai"))
85
+ return "openai_reasoning";
86
+ if (provider === "neuralwatt")
87
+ return "reasoning_effort";
88
+ if (model.capabilities?.reasoning)
89
+ return "reasoning_effort";
90
+ return "noop";
91
+ }
92
+ //# sourceMappingURL=thinking.js.map
package/dist/tools.js CHANGED
@@ -1,4 +1,5 @@
1
1
  import { isJsonObject } from "./config.js";
2
+ import { createId } from "./ids.js";
2
3
  import { errorToErrorInfo, redactRunLedgerRecord, redactSecrets } from "./redaction.js";
3
4
  import { assertCanRegister } from "./registry-options.js";
4
5
  import { assertPermission } from "./security.js";
@@ -149,9 +150,7 @@ async function blocked(call, context, reason, error, options, startedAt) {
149
150
  function toErrorInfo(value, secrets) {
150
151
  return typeof value === "string" ? errorToErrorInfo(value, secrets) : redactSecrets(value, secrets);
151
152
  }
152
- function randomId(prefix) {
153
- return `${prefix}_${globalThis.crypto?.randomUUID?.() ?? Math.random().toString(36).slice(2)}`;
154
- }
153
+ const randomId = createId;
155
154
  function appendToolCallRecord(options, status, call, startedAt, fields) {
156
155
  if (!options.ledger)
157
156
  return undefined;
@@ -0,0 +1,63 @@
1
+ import type { ModelConfig, ProviderRequestOptions } from "./contracts.js";
2
+ /**
3
+ * Host binding for a non-session LLM job (observational memory, LLM compaction,
4
+ * declarative agents, evals, etc.). Omitting `model` means "use the session model"
5
+ * when a session fallback is supplied to {@link resolveUseCaseModel}.
6
+ *
7
+ * Workers must not write `model_change` session entries; they resolve a model for
8
+ * their own provider calls only.
9
+ */
10
+ export interface UseCaseModelBinding {
11
+ /** Explicit use-case model. When omitted, {@link resolveUseCaseModel} falls back to `sessionModel`. */
12
+ readonly model?: ModelConfig;
13
+ /**
14
+ * Optional provider id hint for docs / credential routing.
15
+ * When `model` is set, `model.provider` is authoritative.
16
+ */
17
+ readonly provider?: string;
18
+ readonly providerOptions?: ProviderRequestOptions;
19
+ /** Portable thinking level; packages map via `applyThinkingLevel` into `compat`. */
20
+ readonly thinkingLevel?: string;
21
+ /**
22
+ * When true, do not fall back to `sessionModel` — leave resolution empty if
23
+ * `model` is omitted (preserves historical explicit-worker `missing_model` behavior).
24
+ */
25
+ readonly requireExplicitModel?: boolean;
26
+ }
27
+ export interface ResolveUseCaseModelInput {
28
+ /** Explicit use-case model (or `binding.model`). */
29
+ readonly configured?: ModelConfig;
30
+ /** Active session / agent model used when `configured` is omitted. */
31
+ readonly sessionModel?: ModelConfig;
32
+ /** When true, skip session fallback (OM `missing_model` escape hatch). */
33
+ readonly requireExplicitModel?: boolean;
34
+ readonly providerOptions?: ProviderRequestOptions;
35
+ readonly thinkingLevel?: string;
36
+ }
37
+ export interface ResolvedUseCaseModel {
38
+ readonly model: ModelConfig;
39
+ /** Whether the model came from the use-case binding or session fallback. */
40
+ readonly source: "configured" | "session";
41
+ readonly providerOptions?: ProviderRequestOptions;
42
+ readonly thinkingLevel?: string;
43
+ }
44
+ /**
45
+ * Resolve the model for a non-session LLM job.
46
+ *
47
+ * Precedence:
48
+ * 1. `configured` → `source: "configured"`
49
+ * 2. Else `sessionModel` when `requireExplicitModel` is not set → `source: "session"`
50
+ * 3. Else `undefined` (caller skips / throws per package policy)
51
+ *
52
+ * O(1); no network. Does not mutate session history.
53
+ */
54
+ export declare function resolveUseCaseModel(input: ResolveUseCaseModelInput): ResolvedUseCaseModel | undefined;
55
+ /**
56
+ * Resolve from a {@link UseCaseModelBinding} plus optional session fallback.
57
+ */
58
+ export declare function resolveUseCaseModelBinding(binding: UseCaseModelBinding | undefined, sessionModel?: ModelConfig): ResolvedUseCaseModel | undefined;
59
+ /**
60
+ * Provider id for credential requests: always the **resolved** model's provider.
61
+ * Optional `binding.provider` is only a hint when no model resolved yet.
62
+ */
63
+ export declare function useCaseCredentialProviderId(resolved: ResolvedUseCaseModel | undefined, binding?: Pick<UseCaseModelBinding, "provider">): string | undefined;
@@ -0,0 +1,52 @@
1
+ /**
2
+ * Resolve the model for a non-session LLM job.
3
+ *
4
+ * Precedence:
5
+ * 1. `configured` → `source: "configured"`
6
+ * 2. Else `sessionModel` when `requireExplicitModel` is not set → `source: "session"`
7
+ * 3. Else `undefined` (caller skips / throws per package policy)
8
+ *
9
+ * O(1); no network. Does not mutate session history.
10
+ */
11
+ export function resolveUseCaseModel(input) {
12
+ const { providerOptions, thinkingLevel } = input;
13
+ if (input.configured) {
14
+ return {
15
+ model: input.configured,
16
+ source: "configured",
17
+ providerOptions,
18
+ thinkingLevel,
19
+ };
20
+ }
21
+ if (input.requireExplicitModel)
22
+ return undefined;
23
+ if (input.sessionModel) {
24
+ return {
25
+ model: input.sessionModel,
26
+ source: "session",
27
+ providerOptions,
28
+ thinkingLevel,
29
+ };
30
+ }
31
+ return undefined;
32
+ }
33
+ /**
34
+ * Resolve from a {@link UseCaseModelBinding} plus optional session fallback.
35
+ */
36
+ export function resolveUseCaseModelBinding(binding, sessionModel) {
37
+ return resolveUseCaseModel({
38
+ configured: binding?.model,
39
+ sessionModel,
40
+ requireExplicitModel: binding?.requireExplicitModel,
41
+ providerOptions: binding?.providerOptions,
42
+ thinkingLevel: binding?.thinkingLevel,
43
+ });
44
+ }
45
+ /**
46
+ * Provider id for credential requests: always the **resolved** model's provider.
47
+ * Optional `binding.provider` is only a hint when no model resolved yet.
48
+ */
49
+ export function useCaseCredentialProviderId(resolved, binding) {
50
+ return resolved?.model.provider ?? binding?.provider;
51
+ }
52
+ //# sourceMappingURL=use-case-model.js.map
package/docs/a2a.md CHANGED
@@ -21,7 +21,7 @@ Use it to expose one explicitly selected Prism agent at an A2A endpoint or call
21
21
 
22
22
  ## Outputs / response / events
23
23
 
24
- The handler serves `GET /.well-known/agent-card.json` and its configured POST endpoint. JSON-RPC returns `{ result: { task } }` or a bounded error. Streaming returns backpressure-driven SSE task envelopes. Client `send()` maps a terminal remote task to `AgentRunResult`; `stream()` yields validated/redacted text artifacts.
24
+ The handler serves `GET /.well-known/agent-card.json` and its configured POST endpoint. JSON-RPC returns `{ result: { task } }` or a bounded error. Streaming returns backpressure-driven SSE task envelopes. Client `send()` maps a terminal remote task to `AgentRunResult`; `stream()` incrementally yields validated/redacted text artifacts. Client SSE accepts LF, CRLF, mixed blank-line separators, comments/unknown fields, and multiline `data:` joined with LF.
25
25
 
26
26
  ## Request/response example
27
27
 
@@ -59,7 +59,9 @@ Only `text` parts are accepted. File/data parts, push notifications, task persis
59
59
  ## Security and performance notes
60
60
 
61
61
  - Endpoints and card URLs must be HTTPS and exactly origin-allow-listed before fetch; `redirect: "error"` prevents redirect SSRF.
62
- - Treat every remote card, error, task, status, artifact, and SSE frame as untrusted. Shape/count/byte/time limits apply before mapping.
62
+ - Treat every remote card, error, task, status, artifact, and SSE frame as untrusted. Shape/count/byte/time limits apply before mapping. Streaming keeps raw stream bytes, current frame bytes, and event count as separate existing limits.
63
+ - One fatal streaming UTF-8 decoder is reused across every body chunk and flushed once at EOF. Split multibyte code points are preserved; malformed/truncated UTF-8 fails rather than inserting `U+FFFD` into JSON. A small coalesced line buffer keeps one-byte chunk handling incremental.
64
+ - SSE frames require a terminating blank line. A non-whitespace final partial frame, malformed JSON, missing terminal task, failed/canceled task, or any event after a completed task fails with bounded package-owned text. Existing request/response/event/stream/count/timeout hard caps are unchanged.
63
65
  - Card verification pins `alg=ES256`, optional key ID, issue/expiry, optional maximum age, and canonical unsigned-card payload. Hosts provision trusted public keys; remote `jku` is never fetched automatically.
64
66
  - Card discovery is public; extended-card and invoke methods call host authorization. Use TLS, rate limits, and replay controls at the host edge.
65
67
  - Credentials remain in the client auth callback or server authorizer and never enter cards, messages, events, or metrics.
@@ -100,28 +100,23 @@ Artifact validation/refinement events (emitted only by `generateValidateReviseLo
100
100
  | `artifact_validation_finished` | `sessionId`, `runId`, `turn`, `attempt`, `result: ArtifactValidation` |
101
101
  | `artifact_revision_started` | `sessionId`, `runId`, `turn`, `attempt`, `failure: ArtifactValidation` |
102
102
  | `artifact_finished` | `sessionId`, `runId`, `turn`, `attempt`, `result: ArtifactValidation` (loop ended successfully) |
103
- | `artifact_failed` | `sessionId`, `runId`, `turn`, `attempt`, `result: ArtifactValidation` (budget exhausted) |
103
+ | `artifact_failed` | `sessionId`, `runId`, `turn`, `attempt`, `result: ArtifactValidation` (candidate budget exhausted, or `result.metadata.reason === "tool_round_limit"`) |
104
104
 
105
105
  ### Artifact event ordering
106
106
 
107
- A `generateValidateReviseLoop` run emits normal turn/message events for every provider turn, then a strictly ordered artifact sequence, correlated by `runId` / `turn` / `attempt`:
107
+ A call-free candidate in `generateValidateReviseLoop` emits normal turn/message events then a strictly ordered artifact sequence, correlated by `runId` / `turn` / `attempt`:
108
108
 
109
109
  ```
110
- turn_started
111
- → message_started
112
- → message_delta*
113
- → message_finished
114
- → turn_finished
115
- → artifact_validation_started
116
- → artifact_validation_finished
117
- → artifact_revision_started # when a revision will run next
118
- | artifact_finished # loop ended successfully
119
- | artifact_failed # budget exhausted (maxRevisions+1 attempts)
110
+ turn_started → message_started → message_delta* → message_finished → turn_finished
111
+ → artifact_validation_started → artifact_validation_finished
112
+ → artifact_revision_started | artifact_finished | artifact_failed
120
113
  ```
121
114
 
122
- - `attempt` is 1-indexed per validation attempt and equals the provider `turn` within `generateValidateReviseLoop`; it mirrors `retry_scheduled.attempt` and the `tool_execution_*` block/finish pairing.
115
+ With opt-in `toolCalls: "bounded"`, a provider turn containing calls emits its normal assistant envelope followed by existing `tool_execution_*` events and matching persisted tool results; it emits no validation event and the next provider turn consumes that transcript. A post-`maxToolRounds` call emits terminal `artifact_failed` directly after `turn_finished` and has no tool execution event.
116
+
117
+ - `attempt` is 1-indexed per call-free validation candidate. It can differ from provider `turn` when bounded tool calls occur.
123
118
  - Single-shot runs emit zero artifact events.
124
- - **Validation failure triggering a revision is recoverable and never an `error`.** Only terminal budget exhaustion emits `artifact_failed`. The `error` channel is reserved for real failures (provider failures not caught by retry, aborts, etc.), matching the existing convention used by `tool_execution_blocked`.
119
+ - **Validation failure triggering a revision is recoverable and never an `error`.** Terminal candidate-budget or `tool_round_limit` exhaustion emits `artifact_failed`; real failures remain on the `error` channel.
125
120
 
126
121
  ## Request/response example
127
122
 
@@ -193,7 +188,7 @@ for await (const event of session.stream("draft", { loop: { strategy: "generate-
193
188
  - Slow consumers are bounded by `SubscribeOptions`. Use `RunLedger` or host storage for durable replay; do not rely on a live subscriber as a queue.
194
189
  - Redaction is exact-string-match only and opt-in via `createSecretRedactor`; values not passed as known secrets are not redacted.
195
190
  - `ArtifactValidation.errors[].message` and `metadata` may echo model text; `redactAgentEvent` walks arbitrary nesting and replaces cyclic references with `"[Circular]"` (WeakSet cycle guard), so secret values in `result`/`failure` are redacted without crashing.
196
- - `artifact_*` events are bounded by `maxRevisions + 1` validation attempts; an always-failing validator cannot loop forever and emits exactly one terminal `artifact_failed`.
191
+ - `artifact_*` validation events are bounded by `maxRevisions + 1` call-free candidates. With opt-in bounded artifact tools, provider turns are additionally bounded by run-global `maxToolRounds` (maximum `1 + maxRevisions + maxToolRounds`); a post-cap call emits exactly one terminal `artifact_failed` with `result.metadata.reason === "tool_round_limit"` and has no tool lifecycle event because it never dispatches.
197
192
  - Runtime events contain messages/content only; do not put secrets in prompts, metadata, provider events, session entries, tool results, or artifact validation payloads.
198
193
 
199
194
  ## Related APIs
@@ -16,7 +16,7 @@ The `Artifact*` contracts (`ArtifactValidation`, `ArtifactContext`, `ArtifactPar
16
16
 
17
17
  Use the default `singleShotLoop` implicitly whenever you call `session.run()` — no configuration needed. Opt into `generateValidateReviseLoop` when a run should produce an artifact that must satisfy a host-supplied schema before it is considered complete (e.g. structured output, a validated JSON document, a generated file passing lint) and the host wants Prism to drive the revision turns.
18
18
 
19
- Do not use a loop to re-implement provider calls, retry, abort, store, or event emission — those stay runtime-owned and are exposed to the loop only through `LoopContext`. A loop that needs tools in revision turns is out of scope for `generateValidateReviseLoop`; use `singleShotLoop` or supply a custom `AgentLoopStrategy`.
19
+ Do not use a loop to re-implement provider calls, retry, abort, store, or event emission — those stay runtime-owned and are exposed to the loop only through `LoopContext`. Artifact-loop tools stay disabled by default; opt into bounded calls only for host-registered, least-privilege lookup tools that must inform an artifact candidate.
20
20
 
21
21
  ## Inputs / request
22
22
 
@@ -57,6 +57,7 @@ await session.run(input, {
57
57
  parser: hostParser, // optional; default treats assistant text as the value
58
58
  repairer: hostRepairer, // optional; default stringifies validation.errors[].message
59
59
  maxRevisions: 3, // optional; default 3
60
+ toolCalls: "bounded", // optional; default "disabled"; uses RunOptions.maxToolRounds
60
61
  },
61
62
  });
62
63
 
@@ -79,6 +80,8 @@ type AgentLoopOptions =
79
80
  readonly parser?: ArtifactParser<unknown>;
80
81
  readonly repairer?: ArtifactRepairer<unknown>;
81
82
  readonly maxRevisions?: number;
83
+ /** Default "disabled". "bounded" dispatches sequentially up to RunOptions.maxToolRounds. */
84
+ readonly toolCalls?: "disabled" | "bounded";
82
85
  };
83
86
  ```
84
87
 
@@ -112,7 +115,7 @@ Host callback contracts (all generic over host `T`):
112
115
 
113
116
  Events during a loop run are the existing `AgentEvent`s (`turn_started`, `message_started`, `message_delta`, `message_finished`, `turn_finished`, tool-execution events when the loop dispatches tools, `error` on real failures). Both built-in loops emit `turn_started` before each provider turn, `message_finished` for every assistant draft, and `turn_finished` after the assistant draft is appended. First-turn input is appended to live history once, matching the already-persisted user message.
114
117
 
115
- Validation-failure-triggering-a-revision is **not** an `error` event — it is recoverable, like `tool_execution_blocked`. `generateValidateReviseLoop` emits normal turn/message events around each provider turn, then the artifact event sequence `artifact_validation_started` → `artifact_validation_finished` → (`artifact_revision_started`)* → `artifact_finished` (success) | `artifact_failed` (budget exhausted), correlated by `runId`/`turn`/`attempt`; see [Agent events § Artifact event ordering](agent-events.md#artifact-event-ordering). `singleShotLoop` emits zero artifact events. Real failures stay on the `error` channel.
118
+ Validation-failure-triggering-a-revision is **not** an `error` event — it is recoverable, like `tool_execution_blocked`. In bounded artifact mode, a tool-calling provider response emits normal assistant/tool lifecycle events, skips artifact parsing/validation, then the next turn sees its persisted result. `generateValidateReviseLoop` emits artifact events only for call-free candidates: `artifact_validation_started` → `artifact_validation_finished` → (`artifact_revision_started`)* → `artifact_finished` | `artifact_failed`. A request beyond `maxToolRounds` executes nothing and emits terminal `artifact_failed` with `result.metadata.reason === "tool_round_limit"`; see [Agent events § Artifact event ordering](agent-events.md#artifact-event-ordering). `singleShotLoop` emits zero artifact events. Real failures stay on the `error` channel.
116
119
 
117
120
  A loop has no path to credentials, provider objects, or unredacted secrets. `LoopContext.generate` receives the already-policy-applied, middleware-run, redacted request; `LoopContext.emit` runs through `redactAgentEvent` with the active `SecretRedactor`.
118
121
 
@@ -198,17 +201,17 @@ await session.run(input, { loop: twoShotLoop });
198
201
  - `RunOptions.loop` wins over `AgentConfig.loop`; when neither is set the runtime uses `singleShotLoop`. This mirrors the other `RunOptions` overrides (`redactor`, `validate`, `activeSkills`).
199
202
  - `{ strategy: "single-shot" }` resolves to the exported `singleShotLoop`; `{ strategy: "generate-validate-revise", ... }` is mapped by `resolveLoop()` to `generateValidateReviseLoop(opts)`. An unknown `strategy` throws before the first turn. Passing an `AgentLoopStrategy` instance bypasses the options form entirely (custom-loop escape hatch).
200
203
  - The loop is resolved once per run inside `RuntimeAgentSession.run()`, after the usual setup (provider/skills/tools resolution, history rebuild, model-change entry, input append, auto-compaction). The runtime's outer try/catch/finally, run-exclusivity, abort bridging, and subscriber close remain in place around `loop.run(ctx)`.
201
- - `LoopContext.assemble(nextInput, toolResults?)` accepts an optional tool-result accumulator so `singleShotLoop` can pass its loop-local `toolResults`; `generateValidateReviseLoop` omits it (no tools in revision turns).
202
- - `maxToolRounds` bounds `singleShotLoop` tool rounds; `toolConcurrency` (default `1`) bounds how many independent tool calls from one provider turn may execute concurrently. Results and transcript rows are still appended in original call order. If any resolved `ToolDefinition` in a turn has `exclusive: true`, that turn uses concurrency `1`; later non-exclusive turns restore configured concurrency.
203
- - `maxRevisions` (default 3) bounds `generateValidateReviseLoop` revision turns. Budget exhaustion ends the loop and returns the last usage; it does not throw.
204
- - A revision cycle appends one assistant draft and one repair user message per revision to the session store, so store entries reflect every attempted draft. The original user input is stored once by the runtime and pushed into loop history once on the first turn.
204
+ - `LoopContext.assemble(nextInput, toolResults?)` accepts an optional tool-result accumulator so `singleShotLoop` can pass its loop-local results. Bounded artifact tools append results directly to shared history, then assemble the next turn with empty new input; no second transcript path exists.
205
+ - `maxToolRounds` bounds both `singleShotLoop` and opt-in bounded artifact tool rounds across the whole run. Artifact mode always dispatches sequentially, regardless of `toolConcurrency`; all dispatches still use existing registry/filter/permission/validator/middleware/redactor/ledger guards.
206
+ - `maxRevisions` (default 3) counts only failed call-free artifact candidates. Bounded artifact runs make at most `1 + maxRevisions + maxToolRounds` provider turns. A tool-round limit is terminal and returns last usage after `artifact_failed`; it does not throw.
207
+ - A revision cycle appends one assistant draft and one repair user message per revision to the session store, so store entries reflect every attempted draft. The original user input is stored once by the runtime and pushed into loop history once on the first turn. Repair messages are assembled as the next provider `nextInput` and only pushed into live history after that revision request has been generated, so the model never receives a duplicated repair instruction.
205
208
 
206
209
  ## Security and performance notes
207
210
 
208
211
  - Loops have no path to credentials, provider objects, or unredacted secrets. `LoopContext.generate` consumes an already-redacted request; `LoopContext.emit` runs through `redactAgentEvent` with the active `SecretRedactor`; `LoopContext.appendMessage` appends a redacted entry.
209
212
  - `ArtifactValidation.errors[].message` may echo model text — `artifact_*` event payloads flow through the same `redactAgentEvent` path as other `AgentEvent`s (see [Agent events](agent-events.md)).
210
- - `generateValidateReviseLoop` makes at most `maxRevisions + 1` provider turns; it cannot loop forever on an always-failing validator. Each revision costs one provider turn plus one store append.
211
- - Parallel tool dispatch uses a bounded worker pool over the calls in one turn; queue depth is `calls.length`, not unbounded. Exclusive turns use the same sequential path. Each call still runs through `dispatchToolCall` (permission + validation + execute). Tool lifecycle events may complete out of order; history/store appends stay in call order.
213
+ - `generateValidateReviseLoop` makes at most `1 + maxRevisions + maxToolRounds` provider turns when bounded tools are enabled (otherwise `maxRevisions + 1`); it cannot loop forever. Each revision costs one provider turn plus one store append.
214
+ - Bounded artifact tool calls run sequentially through `dispatchToolCall` (permission + validation + execute); their assistant call and result are persisted before the next provider request. `singleShotLoop` retains its bounded parallel worker pool and original call-order transcript behavior.
212
215
  - The loop is a plain object/factory; no class hierarchy, no background work, no extra dependencies. `LoopContext` is a single object literal of bound arrows built once per run.
213
216
  - The Synapta-free boundary is guarded by tests: `src/` imports no `synapta*` package, and the `Artifact*`/`AgentLoop*`/`LoopContext` contracts contain no `workflow`/`node`/`step` field names. Hosts supply their own schema; no host domain type is imported by `src/`.
214
217
 
@@ -15,6 +15,8 @@
15
15
  | `createAllTools(cwd, options?)` | Every tool the package provides (currently identical to `createCodingTools`). |
16
16
  | `detectSupportedImageMimeType(buf)` / `detectSupportedImageMimeTypeFromFile(path)` | Magic-byte image MIME detection (PNG/JPEG/GIF/WebP/BMP) used by `read`. |
17
17
  | `DEFAULT_MAX_IMAGE_BYTES` | Default `read` image size ceiling (10 MB). |
18
+ | `DEFAULT_*` / `HARD_*` coding limit constants | Published text-scan, image, write/edit, shell timeout, display, and total-output ceilings. |
19
+ | `ReadTextOptions` / `ReadTextResult` | Bounded text-page contract required by custom `ReadOperations`. |
18
20
  | `TransformImage` / `TransformImageInput` | Types for the optional `read` `transformImage` callback. |
19
21
  | `withFileMutationQueue(path, fn)` | Per-path serialization primitive re-exported for hosts. |
20
22
 
@@ -65,7 +67,7 @@ Run a shell command and return combined stdout+stderr.
65
67
  | Field | Type | Purpose |
66
68
  | --- | --- | --- |
67
69
  | `command` | `string` | Shell command to execute (required). |
68
- | `timeout` | `number` | Timeout in **seconds** (optional; no default). |
70
+ | `timeout` | `number` | Timeout in **seconds** (optional; defaults to 600, hard maximum 3600). |
69
71
 
70
72
  **Outputs:** a `ToolResult` whose `content[0]` is a `TextContent` with the combined output. Non-zero exit is **not** a tool error: it is returned as a normal result with `[Command exited with code N]` appended to the content and `exitCode` in metadata. Timeout and abort are error results that still carry the partial output captured so far.
71
73
 
@@ -75,9 +77,11 @@ Run a shell command and return combined stdout+stderr.
75
77
  | --- | --- | --- |
76
78
  | `exitCode` | always | Process exit code, or `null` when the process was killed by timeout/abort. |
77
79
  | `truncation` | always | `TruncationResult` from the bounded output accumulator. |
78
- | `fullOutputPath?` | truncated only | Path to the spilled temp file holding the full output. |
80
+ | `fullOutputPath?` | successful and truncated only | Host-owned path to retained output. Failed/aborted/timed-out/output-limited calls remove unpublished spills. |
81
+ | `totalOutputBytes` | shell executed | Raw bytes retained, never above `maxTotalOutputBytes`. |
82
+ | `outputLimitExceeded` / `outputStorageFailed` | shell executed | Attributable resource failure flags. |
79
83
 
80
- Shell resolution honors `options.shellPath` → `SHELL` env → `/bin/bash` → `sh`, and the process group is killed on timeout/abort (`process.kill(-pid)` on Unix, `taskkill /F /T` on Windows).
84
+ Shell resolution honors `options.shellPath` → `SHELL` env → `/bin/bash` → `sh`. The process group is killed on timeout, caller abort, spill failure, or total-output overflow (`process.kill(-pid)` on Unix, `taskkill /F /T` on Windows). Combined output defaults to a 64 MiB total cap (1 GiB hard cap). Spill files use random exclusive creation and Unix mode `0600`; hosts own and must delete a successful result's `fullOutputPath` after consumption.
81
85
 
82
86
  ### `read`
83
87
 
@@ -91,7 +95,7 @@ Read a text or image file.
91
95
  | `offset` | `number` | Line to start reading from (1-indexed). |
92
96
  | `limit` | `number` | Maximum number of lines to read. |
93
97
 
94
- **Outputs:** text files become a single `TextContent`, truncated to `maxLines`/`maxBytes` (defaults 2000 lines / 50 KB) with a `Use offset=N to continue` footer when more remains. Image files (PNG/JPEG/GIF/WebP/BMP by **magic bytes**, not extension) become `[TextContent note, ImageContent]` with base64 `data` and `mimeType`. Oversize images are rejected by `stat` (when available) or `buffer.length` against `maxImageBytes` (default 10 MB) before base64 encoding. An optional `transformImage` callback lets hosts resize or re-encode images without adding image-processing dependencies to the base package. Read failures (missing file, offset beyond end, oversize image, abort) are error results.
98
+ **Outputs:** text files are scanned incrementally until one requested page, `maxLines`/`maxBytes`, EOF, or `maxScanBytes` (default 64 MiB scanned per call; 1 GiB hard cap). The default path never loads the complete file and returns a `Use offset=N to continue` footer when more remains. Exact total line count is reported only when EOF was already reached in the bounded scan. Image files (PNG/JPEG/GIF/WebP/BMP by **magic bytes**, not extension) become `[TextContent note, ImageContent]` with base64 `data` and `mimeType`. Oversize images are rejected by `stat` (when available) or `buffer.length` against `maxImageBytes` (default 10 MB) before base64 encoding. An optional `transformImage` callback lets hosts resize or re-encode images without adding image-processing dependencies to the base package. Read failures (missing file, offset beyond end, oversize image, abort) are error results.
95
99
 
96
100
  `read` tool options (via `createReadTool(cwd, options)` or `ToolsOptions.read`):
97
101
 
@@ -100,8 +104,9 @@ Read a text or image file.
100
104
  | `maxImageBytes` | `DEFAULT_MAX_IMAGE_BYTES` (10 MB) | Reject image reads larger than this many bytes. |
101
105
  | `transformImage` | — | Host callback `( { buffer, mimeType } ) => Promise<Buffer>` run after read, before base64. |
102
106
  | `autoResizeImages` | — | **Deprecated.** Ignored unless `transformImage` is also set (use `transformImage` instead). |
103
- | `maxLines` / `maxBytes` | 2000 / 50 KB | Text head truncation limits. |
104
- | `operations` | local fs | Pluggable `ReadOperations` backend. |
107
+ | `maxLines` / `maxBytes` | 2000 / 50 KiB | Text page display limits (hard: 100,000 / 1 MiB). |
108
+ | `maxScanBytes` | 64 MiB | Raw bytes scanned to reach one page (hard: 1 GiB). |
109
+ | `operations` | local fs | Pluggable bounded `ReadOperations` backend. |
105
110
  | `executionPolicy` | — | Structured pre-execution policy (see [Coding security](coding-security.md)). |
106
111
 
107
112
  ```ts
@@ -133,7 +138,7 @@ Create or overwrite a file, creating parent directories as needed.
133
138
  | `path` | `string` | Path to the file to write (relative or absolute). Required. |
134
139
  | `content` | `string` | Content to write (empty string creates an empty file). Required. |
135
140
 
136
- **Outputs:** a `TextContent` confirmation naming the **absolute path** with UTF-8 byte and line counts (e.g. `Successfully wrote 42 bytes (3 lines) to /abs/path.txt`). Write failures and abort are error results. Empty `content` is valid.
141
+ **Outputs:** a `TextContent` confirmation naming the **absolute path** with UTF-8 byte and line counts (e.g. `Successfully wrote 42 bytes (3 lines) to /abs/path.txt`). `maxInputBytes` defaults to 8 MiB (64 MiB hard cap); oversized UTF-8 input fails before policy evaluation, directory creation, or write. Write failures and abort are error results. Empty `content` is valid.
137
142
 
138
143
  `write` result `metadata`: `{ bytes, lines, path }` (absolute path). Concurrent writes to the same path serialize through `withFileMutationQueue`; writes to different paths run in parallel.
139
144
 
@@ -148,7 +153,7 @@ Precise text replacement in an existing file via exact-then-fuzzy matching.
148
153
  | `path` | `string` | Path to the file to edit. Required. |
149
154
  | `edits` | `Array<{ oldText: string, newText: string }>` | Targeted replacements, each matched against the **original** file (not incrementally). No overlapping/nested edits. Required, non-empty. |
150
155
 
151
- Each `edits[].oldText` must match a unique, non-overlapping region of the original file. Matching is exact first, then fuzzy (unicode normalization / whitespace collapse). A BOM is stripped before matching and re-prepended on write; original line endings are restored.
156
+ Each `edits[].oldText` must match a unique, non-overlapping region of the original file. Matching is exact first, then fuzzy (unicode normalization / whitespace collapse). A BOM is stripped before matching and re-prepended on write; original line endings are restored. Defaults reject targets over 8 MiB, aggregate old/new UTF-8 input over 2 MiB, or more than 100 edits (hard caps: 64 MiB, 16 MiB, and 1,000). Stat and bounded read checks run before matching or mutation.
152
157
 
153
158
  **Outputs:** a `TextContent` confirmation (`Successfully replaced N block(s) in {path}.`) plus `metadata`. Any failure — missing/unreadable file, no match, duplicate (non-unique) match, overlap, empty `oldText`, no-op edit, or abort — is an error result, and the file is left **unchanged** (the match runs before the write).
154
159
 
@@ -156,7 +161,7 @@ Each `edits[].oldText` must match a unique, non-overlapping region of the origin
156
161
 
157
162
  ## Outputs / response / events
158
163
 
159
- Every tool returns a `ToolResult` with `toolCallId`, `name`, `content` (`readonly ContentBlock[]`), optional `error`, and optional `metadata`. Mutating tools (`shell` with same cwd, `write`, `edit`) serialize per realpath through `withFileMutationQueue` so concurrent calls targeting one file do not interleave. The package emits no events of its own; hosts observe tool execution through the normal Prism `AgentEvent` stream via `dispatchToolCall`.
164
+ Every tool returns a `ToolResult` with `toolCallId`, `name`, `content` (`readonly ContentBlock[]`), optional `error`, and optional `metadata`. `write` and `edit` serialize per realpath through `withFileMutationQueue` so concurrent calls targeting one file do not interleave. `shell` is marked `exclusive`; tool dispatch serializes it at the turn level. The package emits no events of its own; hosts observe tool execution through the normal Prism `AgentEvent` stream via `dispatchToolCall`.
160
165
 
161
166
  ## Request/response example
162
167
 
@@ -208,6 +213,8 @@ const shell = createShellTool("/repo", {
208
213
  shellPath: "/bin/bash",
209
214
  commandPrefix: "set -euo pipefail",
210
215
  maxLines: 500,
216
+ timeout: 600,
217
+ maxTotalOutputBytes: 64 * 1024 * 1024,
211
218
  });
212
219
 
213
220
  const remoteWrite = createWriteTool("/repo", {
@@ -220,8 +227,8 @@ const remoteWrite = createWriteTool("/repo", {
220
227
 
221
228
  ## Extension and configuration notes
222
229
 
223
- - **Pluggable operation backends.** Every tool accepts an `operations` seam so a host can delegate to a remote system (e.g. SSH) while keeping the tool's matching/serialization behavior: `BashOperations` (`shell`), `ReadOperations` (`read`), `WriteOperations` (`write`), `EditOperations` (`edit`).
224
- - **Per-tool options.** `ShellToolOptions` (`shellPath`, `commandPrefix`, `maxLines`, `maxBytes`, `tempFilePrefix`, `operations`, `spawnHook`, `executionPolicy`); `ReadToolOptions` (`operations`, `maxImageBytes`, `transformImage`, `maxLines`, `maxBytes`, `executionPolicy`; `autoResizeImages` deprecated); `WriteToolOptions` (`operations`, `executionPolicy`); `EditToolOptions` (`operations`, `executionPolicy`).
230
+ - **Pluggable operation backends.** Every tool accepts an `operations` seam. Custom `ReadOperations` must implement bounded `readText` plus `statFile`; custom `EditOperations` must implement `statFile`; read/write methods receive caps/signals. `BashOperations` must stream through `onData` and honor `signal`/`timeout`. A hostile custom backend can still violate its host-owned contract, so isolate it separately.
231
+ - **Per-tool options.** `ShellToolOptions` adds `timeout` and `maxTotalOutputBytes`; `ReadToolOptions` adds `maxScanBytes`; `WriteToolOptions` adds `maxInputBytes`; `EditToolOptions` adds `maxFileBytes`, `maxInputBytes`, and `maxEdits`. Invalid/non-finite/unsafe/above-hard-cap values throw during tool construction; request `timeout` errors before spawn.
225
232
  - **Aggregator options.** `ToolsOptions` (`{ executionPolicy?, shell?, read?, write?, edit? }`) threads each sub-object to the matching tool. `createCodingTools()`, `createAllTools()`, and `createReadOnlyTools()` apply the shared policy unless that tool has an explicit per-tool override.
226
233
  - **`ToolsOptions`** and the per-tool option types are exported from the package barrel for host configuration.
227
234
  - No auto-discovery or manifest registration: import and register explicitly. This package registers no extensions and owns no globals (the mutation queue is a process-wide per-path map — see `ponytail:` note in the source).
@@ -230,10 +237,24 @@ const remoteWrite = createWriteTool("/repo", {
230
237
 
231
238
  - **Host shell/filesystem access.** These tools run real commands and read/write real files. They provide **no sandbox**. Gate them with Prism `PermissionPolicy` / `ToolValidator` / trust policies before registering them for any provider turn. Shared `executionPolicy` applies to both full and read-only aggregators before filesystem/process side effects. See [Host security guide](host-security.md) and [Security/auth/trust](settings-auth-trust-security.md).
232
239
  - **Non-zero exit is not an error.** A failing command is a normal `shell` result (exit code in metadata); only timeout/abort/spawn failures are error results. Do not assume `error == undefined` means the command succeeded.
233
- - **Bounded output.** `shell`/`read` accumulate output into a rolling tail bounded by `maxLines`/`maxBytes`; oversized output spills to a temp file (`fullOutputPath`), so memory use is bounded regardless of command output size.
240
+ - **Bounded I/O.** `read` streams one page and bounds scan bytes; image/edit reads use stat plus a shared cap-enforcing reader; write/edit inputs are measured before mutation. `shell` retains only a rolling display tail and synchronously spills accepted raw chunks so stream backpressure cannot grow heap; wall time and total raw output remain finite.
234
241
  - **Per-path serialization.** Concurrent mutations to the same file serialize; concurrent mutations to different files do not block each other. The queue is a process-wide map — across sessions in one process, same-path writes still serialize (upgrade path: scope per registry if throughput matters).
235
242
  - **Bounded image reads.** `read` rejects images over `maxImageBytes` (default 10 MB) by `stat` before read when possible; MIME is detected from magic bytes only. Optional `transformImage` is host-owned — the base package has no image-processing dependency.
236
243
 
244
+ ### Resource-limit defaults and hard caps
245
+
246
+ | Boundary | Default | Hard cap | Failure point |
247
+ | --- | ---: | ---: | --- |
248
+ | Display lines / bytes | 2,000 / 50 KiB | 100,000 / 1 MiB | tool construction |
249
+ | Text scan per read | 64 MiB | 1 GiB | bounded scan before more input is retained |
250
+ | Image | 10,000,000 bytes | 32 MiB | stat and bounded read before base64/transform result use |
251
+ | Write UTF-8 input | 8 MiB | 64 MiB | before policy/filesystem mutation |
252
+ | Edit target / input / count | 8 MiB / 2 MiB / 100 | 64 MiB / 16 MiB / 1,000 | before target read/matching/write |
253
+ | Shell wall time | 600 seconds | 3,600 seconds | process-tree kill |
254
+ | Shell total stdout+stderr | 64 MiB | 1 GiB | process-tree kill; spill removal |
255
+
256
+ Every configurable value is a positive safe integer; Prism rejects rather than clamps invalid values. Limits control resources, not authority: they do not replace root containment, approval, validation, or a sandbox.
257
+
237
258
  ## Related APIs
238
259
 
239
260
  - [Tools](tools.md): the host-owned tool harness — `createToolRegistry`, `dispatchToolCall`, filtering, and the `ToolDefinition` contract these factories satisfy.
@@ -74,11 +74,11 @@ const tools = createCodingTools(workspaceRoot, {
74
74
 
75
75
  Policies are ordinary host values: attach one globally through `createCodingTools()`/`createReadOnlyTools()` or per tool. A per-tool policy overrides the shared policy. `SandboxAdapter` is replaceable and host-owned; approval policy and sandboxing are separate layers.
76
76
 
77
- Callback approval remains process-local. For approval that must survive restart, wrap the action in an opted-in workflow `toolNode({ approval: { reason, data?, resumeSchema? } })`. The workflow persists `suspended` state before any tool side effect. After explicit approve, it recomputes the action and invokes this package's current `ExecutionPolicy`; durable approval never populates or bypasses the process-local approval cache. Adapters should emit chunks through `request.onData` as they arrive and honor `request.signal`/`request.timeout`; buffering is unnecessary. Default caching is `none`; use run-scoped caching only when repeated approval within one run is desired, and session scope only when that wider lifecycle is intentional.
77
+ Callback approval remains process-local. For approval that must survive restart, wrap the action in an opted-in workflow `toolNode({ approval: { reason, data?, resumeSchema? } })`. The workflow persists `suspended` state before any tool side effect. After explicit approve, it recomputes the action and invokes this package's current `ExecutionPolicy`; durable approval never populates or bypasses the process-local approval cache. Adapters should emit chunks through `request.onData` as they arrive and honor `request.signal`/`request.timeout`; buffering is unnecessary. Coding-agent composes caller abort with its total-output controller, so ignoring the supplied signal defeats process termination even though Prism stops retaining output at the cap. Default caching is `none`; use run-scoped caching only when repeated approval within one run is desired, and session scope only when that wider lifecycle is intentional.
78
78
 
79
79
  ## Security and performance notes
80
80
 
81
- Containment resolves symlinks and rejects paths outside roots. Command rules are not a shell parser; shell metacharacters require approval. Approval waits and subprocess execution honor abort/timeouts. Durable workflow denial/cancellation is terminal and attributable; approved resume still fails if roots, command rules, read-only mode, or other policy changed while suspended. Cache keys are fixed-size SHA-256 digests of selected identity plus action shape; caches remain process-local, retain at most 1,000 decisions with oldest-entry eviction, and have no default/global mode. Path checks and cache lookup are local; sandbox latency belongs to the supplied adapter.
81
+ Containment resolves symlinks and rejects paths outside roots. Command rules are not a shell parser; shell metacharacters require approval. Approval waits and subprocess execution honor abort/timeouts. Coding-agent resource ceilings independently bound text scans, image/edit target reads, write/edit payloads, edit counts, shell wall time, and retained/spilled output. Those ceilings reduce exhaustion risk but do not grant path/command authority or make an unsandboxed shell safe. Durable workflow denial/cancellation is terminal and attributable; approved resume still fails if roots, command rules, read-only mode, or other policy changed while suspended. Cache keys are fixed-size SHA-256 digests of selected identity plus action shape; caches remain process-local, retain at most 1,000 decisions with oldest-entry eviction, and have no default/global mode. Path checks and cache lookup are local; sandbox latency belongs to the supplied adapter.
82
82
 
83
83
  ## Related APIs
84
84
 
@@ -25,15 +25,16 @@ Key exports:
25
25
  | Field | Purpose |
26
26
  | --- | --- |
27
27
  | `provider` / `summaryProvider` | Explicit `AIProvider`, or factory receiving a resolved credential. |
28
- | `model` / `summaryModel` | Explicit summary `ModelConfig`. |
28
+ | `model` / `summaryModel` | Summary `ModelConfig`. `summaryModel` wins; `model` is the session/host fallback via `resolveUseCaseModel`. See [Use-case model selection](use-case-model-selection.md). |
29
29
  | `credential`, `credentialRequest` | Optional per-call credential resolution for provider factories. |
30
30
  | `providerOptions` | Generic `ProviderRequest.options`, including cache fields. |
31
31
  | `providerRequestPolicies` | Optional Prism provider request policies applied before the summary call. |
32
32
  | `customInstructions` | Additional summary focus appended to prompts. |
33
- | `thinkingLevel` | Passed as `ProviderRequest.options.extra.thinkingLevel`. |
34
- | `reserveTokens` | Output budget basis; defaults to `16384`. |
33
+ | `thinkingLevel` | Mapped into `ProviderRequest.options.compat` via `applyThinkingLevel` / `thinkingFamilyForModel` (not inert `extra.thinkingLevel`). See [Thinking and reasoning](thinking-and-reasoning.md). |
34
+ | `reserveTokens` | Output budget basis; defaults to `16384`, hard cap `131072`. |
35
35
  | `keepRecentTokens` | Approximate recent-token budget; defaults to `20000`. |
36
- | `maxSummaryTokens` / `maxOutputTokens` | Computes the summary output budget, writes it to `summaryModel/model.parameters.maxTokens`, and truncates oversized collected summaries. First-party providers serialize that generic field to their real request field (`max_output_tokens` for OpenAI Responses, `max_tokens` for OpenAI-compatible/Anthropic-style providers). |
36
+ | `maxSummaryTokens` / `maxOutputTokens` | Summary retention/request ceiling; default `16384`, hard cap `131072`. `maxSummaryTokens` wins over the compatibility alias. The finite value is written to `model.parameters.maxTokens`; first-party providers map it to their wire field. |
37
+ | `maxErrorBytes` | Retained provider/factory/policy error detail; default `1024`, hard cap `8192`, UTF-8-safe and known-secret redacted. |
37
38
  | `maxToolResultChars` | Tool-result JSON truncation limit; defaults to `2000`. |
38
39
  | `trackFileOperations`, `includeFileOperations` | Control file path extraction and final summary blocks. |
39
40
  | `secrets` | Exact strings to redact from serialized prompts and final summaries. |
@@ -64,7 +65,8 @@ const strategy = createLlmCompactionStrategy({
64
65
  model: { provider: "openai", model: "gpt-4.1-mini" },
65
66
  keepRecentTokens: 20_000,
66
67
  reserveTokens: 16_384,
67
- maxOutputTokens: 800,
68
+ maxSummaryTokens: 800,
69
+ maxErrorBytes: 1_024,
68
70
  providerOptions: { cacheRetention: "short" },
69
71
  customInstructions: "Focus on current files and failing tests.",
70
72
  });
@@ -102,10 +104,18 @@ const agent = createAgent({ model, provider, compaction: { strategy, thresholdEn
102
104
  Registration only contributes an inert strategy. The host must resolve and pass it to runtime config.
103
105
 
104
106
  ## Security and performance notes
105
- Preparation is O(n) over branch entries and uses only arrays, strings, and JSON serialization. Output-budget calculation is O(1) and does not add an extra summarization call. The strategy makes only the needed provider call(s): one history summary plus one split-turn prefix summary when needed. It does not discover credentials, read files, start background jobs, or add provider SDK dependencies. Redaction is exact-string only; pass every known secret that may appear in history or provider output.
107
+ Preparation is O(n) over branch entries and uses only arrays, strings, and JSON serialization. Limit options must be positive safe integers at or below their hard caps and reject during strategy creation. Missing output options use a 16,384-token summary ceiling; reserve ratio/model metadata may narrow the provider request, never remove its finite `maxTokens`. A request policy that replaces `maxTokens` with NaN, Infinity, zero, an unsafe integer, or above-hard-cap input fails before provider generation.
108
+
109
+ Provider deltas are redacted while retained and stop at `maxSummaryTokens * 4` UTF-16 code units without splitting a surrogate pair. Provider iteration is closed/aborted on overflow. A derived finite event ceiling also stops endless empty/non-text deltas. Final history/turn/file composition receives the same cap. Provider error events, generator throws, provider-factory failures, and policy failures expose only bounded redacted detail; host abort remains authoritative.
110
+
111
+ The strategy makes only the needed provider call(s): one history summary plus one split-turn prefix summary when needed. It does not discover credentials, read files, start background jobs, or add provider SDK dependencies. Redaction is exact-string only; pass every known secret that may appear in history or provider output.
106
112
 
107
113
  ## Related APIs
108
- - [Compaction and retry policies](compaction-and-retry.md): core compaction strategy surface.
114
+
115
+ - [Use-case model selection](use-case-model-selection.md): `summaryModel` vs session `model` fallback.
116
+ - [Thinking and reasoning](thinking-and-reasoning.md): `thinkingLevel` → `compat`.
117
+ - [Compaction and retry policies](compaction-and-retry.md): replaceable compaction strategy boundary and core compaction strategy surface.
118
+ - [Observational memory compaction package](compaction-observational-memory.md): source-backed memory workers with the same use-case binding pattern.
109
119
  - [Agent/session runtime](agent-session-runtime.md): `AgentSession.compact()` and opt-in auto-compaction.
110
120
  - [Provider layer](provider-layer.md): mock providers and provider request contracts.
111
121
  - [Credentials and redaction](credentials-and-redaction.md): exact known-secret redaction behavior.