@arnilo/prism 0.0.5 → 0.0.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -1
- package/dist/agent-loops.d.ts +1 -0
- package/dist/agent-loops.js +26 -16
- package/dist/agents.js +2 -3
- package/dist/contracts.d.ts +2 -0
- package/dist/ids.d.ts +2 -0
- package/dist/ids.js +6 -0
- package/dist/index.d.ts +5 -1
- package/dist/index.js +3 -1
- package/dist/session-stores.js +2 -3
- package/dist/testing/persistence-schema.d.ts +45 -7
- package/dist/testing/persistence-schema.js +138 -24
- package/dist/thinking.d.ts +42 -0
- package/dist/thinking.js +92 -0
- package/dist/tools.js +2 -3
- package/dist/use-case-model.d.ts +63 -0
- package/dist/use-case-model.js +52 -0
- package/docs/a2a.md +4 -2
- package/docs/agent-events.md +10 -15
- package/docs/agent-loops.md +11 -8
- package/docs/coding-agent-tools.md +33 -12
- package/docs/coding-security.md +2 -2
- package/docs/compaction-llm.md +17 -7
- package/docs/compaction-observational-memory.md +28 -4
- package/docs/credential-storage.md +58 -9
- package/docs/credentials-and-redaction.md +1 -1
- package/docs/database-persistence.md +8 -3
- package/docs/host-security.md +10 -6
- package/docs/index.md +23 -20
- package/docs/mcp-tools.md +26 -10
- package/docs/migration.md +146 -2
- package/docs/node-filesystem-config.md +1 -0
- package/docs/node-jsonl-session-store.md +5 -4
- package/docs/postgres-persistence.md +3 -3
- package/docs/provider-caching.md +16 -4
- package/docs/provider-conformance.md +39 -1
- package/docs/provider-packages.md +60 -3
- package/docs/providers/ai-sdk.md +36 -0
- package/docs/providers/kimi.md +124 -61
- package/docs/providers/neuralwatt.md +19 -13
- package/docs/providers/openai.md +56 -13
- package/docs/providers/opencode-go.md +118 -30
- package/docs/providers/openrouter.md +105 -35
- package/docs/providers/zai.md +94 -45
- package/docs/release-and-install.md +47 -49
- package/docs/review-coverage-2026-07-17-provider-validation.md +192 -0
- package/docs/runs-and-usage.md +1 -1
- package/docs/sqlite-persistence.md +2 -2
- package/docs/structured-output.md +1 -1
- package/docs/thinking-and-reasoning.md +98 -0
- package/docs/tool-execution-primitives.md +3 -3
- package/docs/tools.md +15 -0
- package/docs/use-case-model-selection.md +109 -0
- package/docs/workflow-orchestration-primitives.md +1 -0
- package/docs/workflows.md +17 -10
- package/docs/working-and-semantic-memory.md +1 -0
- package/package.json +2 -2
package/dist/thinking.js
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
import { mergeProviderRequestOptions } from "./provider-request-policy.js";
|
|
2
|
+
/**
|
|
3
|
+
* Portable thinking / reasoning effort levels shared across first-party providers.
|
|
4
|
+
* Model-dependent legality (which values a given model accepts) stays provider-owned.
|
|
5
|
+
*/
|
|
6
|
+
export const THINKING_LEVELS = ["none", "minimal", "low", "medium", "high", "xhigh", "max"];
|
|
7
|
+
export function isThinkingLevel(value) {
|
|
8
|
+
return typeof value === "string" && THINKING_LEVELS.includes(value);
|
|
9
|
+
}
|
|
10
|
+
/**
|
|
11
|
+
* Normalize a host thinkingLevel string. Known levels are lowercased; other non-empty
|
|
12
|
+
* strings pass through as opaque effort values for forward-compatible provider fields.
|
|
13
|
+
*/
|
|
14
|
+
export function normalizeThinkingLevel(level) {
|
|
15
|
+
const normalized = level.trim().toLowerCase();
|
|
16
|
+
if (!normalized)
|
|
17
|
+
return undefined;
|
|
18
|
+
return isThinkingLevel(normalized) ? normalized : normalized;
|
|
19
|
+
}
|
|
20
|
+
/**
|
|
21
|
+
* Build the `ProviderRequestOptions.compat` patch for a shared thinking level.
|
|
22
|
+
* Does not invent a second options tree — providers keep reading official fields from `compat`.
|
|
23
|
+
*/
|
|
24
|
+
export function thinkingCompatFor(family, level) {
|
|
25
|
+
const normalized = typeof level === "string" ? normalizeThinkingLevel(level) : level;
|
|
26
|
+
if (!normalized || family === "noop")
|
|
27
|
+
return {};
|
|
28
|
+
switch (family) {
|
|
29
|
+
case "openai_reasoning":
|
|
30
|
+
return { reasoning: { effort: normalized } };
|
|
31
|
+
case "reasoning_effort":
|
|
32
|
+
return { reasoning_effort: normalized };
|
|
33
|
+
case "thinking_type":
|
|
34
|
+
return { thinking: { type: normalized === "none" ? "disabled" : "enabled" } };
|
|
35
|
+
default: {
|
|
36
|
+
const _exhaustive = family;
|
|
37
|
+
return _exhaustive;
|
|
38
|
+
}
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
/**
|
|
42
|
+
* Merge a shared thinking level into `providerOptions.compat` for the given family.
|
|
43
|
+
* Per-turn patches win over prior compat via {@link mergeProviderRequestOptions}.
|
|
44
|
+
*/
|
|
45
|
+
export function applyThinkingLevel(options, level, family = "reasoning_effort") {
|
|
46
|
+
const normalized = normalizeThinkingLevel(String(level));
|
|
47
|
+
if (!normalized || family === "noop")
|
|
48
|
+
return options ?? {};
|
|
49
|
+
const patch = thinkingCompatFor(family, normalized);
|
|
50
|
+
if (family === "openai_reasoning" && options?.compat?.reasoning && typeof options.compat.reasoning === "object" && !Array.isArray(options.compat.reasoning)) {
|
|
51
|
+
return mergeProviderRequestOptions(options, {
|
|
52
|
+
compat: {
|
|
53
|
+
reasoning: {
|
|
54
|
+
...options.compat.reasoning,
|
|
55
|
+
...patch.reasoning,
|
|
56
|
+
},
|
|
57
|
+
},
|
|
58
|
+
});
|
|
59
|
+
}
|
|
60
|
+
return mergeProviderRequestOptions(options, { compat: patch });
|
|
61
|
+
}
|
|
62
|
+
/**
|
|
63
|
+
* Best-effort family inference from model metadata without a second options tree.
|
|
64
|
+
* Prefer an explicit family in hosts/use-case workers when the provider is known.
|
|
65
|
+
*
|
|
66
|
+
* Heuristics (ordered):
|
|
67
|
+
* 1. Existing `compat.thinking` object → `thinking_type`
|
|
68
|
+
* 2. Existing `compat.reasoning` → `openai_reasoning`
|
|
69
|
+
* 3. Existing `compat.reasoning_effort` → `reasoning_effort`
|
|
70
|
+
* 4. Provider id starting with `openai` → `openai_reasoning`
|
|
71
|
+
* 5. Provider id `neuralwatt` → `reasoning_effort`
|
|
72
|
+
* 6. `capabilities.reasoning` → `reasoning_effort` (portable string field)
|
|
73
|
+
* 7. Else `noop`
|
|
74
|
+
*/
|
|
75
|
+
export function thinkingFamilyForModel(model) {
|
|
76
|
+
const compat = model.compat ?? {};
|
|
77
|
+
if (compat.thinking != null && typeof compat.thinking === "object")
|
|
78
|
+
return "thinking_type";
|
|
79
|
+
if (compat.reasoning != null)
|
|
80
|
+
return "openai_reasoning";
|
|
81
|
+
if (compat.reasoning_effort != null)
|
|
82
|
+
return "reasoning_effort";
|
|
83
|
+
const provider = model.provider.trim().toLowerCase();
|
|
84
|
+
if (provider === "openai" || provider.startsWith("openai"))
|
|
85
|
+
return "openai_reasoning";
|
|
86
|
+
if (provider === "neuralwatt")
|
|
87
|
+
return "reasoning_effort";
|
|
88
|
+
if (model.capabilities?.reasoning)
|
|
89
|
+
return "reasoning_effort";
|
|
90
|
+
return "noop";
|
|
91
|
+
}
|
|
92
|
+
//# sourceMappingURL=thinking.js.map
|
package/dist/tools.js
CHANGED
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
import { isJsonObject } from "./config.js";
|
|
2
|
+
import { createId } from "./ids.js";
|
|
2
3
|
import { errorToErrorInfo, redactRunLedgerRecord, redactSecrets } from "./redaction.js";
|
|
3
4
|
import { assertCanRegister } from "./registry-options.js";
|
|
4
5
|
import { assertPermission } from "./security.js";
|
|
@@ -149,9 +150,7 @@ async function blocked(call, context, reason, error, options, startedAt) {
|
|
|
149
150
|
function toErrorInfo(value, secrets) {
|
|
150
151
|
return typeof value === "string" ? errorToErrorInfo(value, secrets) : redactSecrets(value, secrets);
|
|
151
152
|
}
|
|
152
|
-
|
|
153
|
-
return `${prefix}_${globalThis.crypto?.randomUUID?.() ?? Math.random().toString(36).slice(2)}`;
|
|
154
|
-
}
|
|
153
|
+
const randomId = createId;
|
|
155
154
|
function appendToolCallRecord(options, status, call, startedAt, fields) {
|
|
156
155
|
if (!options.ledger)
|
|
157
156
|
return undefined;
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import type { ModelConfig, ProviderRequestOptions } from "./contracts.js";
|
|
2
|
+
/**
|
|
3
|
+
* Host binding for a non-session LLM job (observational memory, LLM compaction,
|
|
4
|
+
* declarative agents, evals, etc.). Omitting `model` means "use the session model"
|
|
5
|
+
* when a session fallback is supplied to {@link resolveUseCaseModel}.
|
|
6
|
+
*
|
|
7
|
+
* Workers must not write `model_change` session entries; they resolve a model for
|
|
8
|
+
* their own provider calls only.
|
|
9
|
+
*/
|
|
10
|
+
export interface UseCaseModelBinding {
|
|
11
|
+
/** Explicit use-case model. When omitted, {@link resolveUseCaseModel} falls back to `sessionModel`. */
|
|
12
|
+
readonly model?: ModelConfig;
|
|
13
|
+
/**
|
|
14
|
+
* Optional provider id hint for docs / credential routing.
|
|
15
|
+
* When `model` is set, `model.provider` is authoritative.
|
|
16
|
+
*/
|
|
17
|
+
readonly provider?: string;
|
|
18
|
+
readonly providerOptions?: ProviderRequestOptions;
|
|
19
|
+
/** Portable thinking level; packages map via `applyThinkingLevel` into `compat`. */
|
|
20
|
+
readonly thinkingLevel?: string;
|
|
21
|
+
/**
|
|
22
|
+
* When true, do not fall back to `sessionModel` — leave resolution empty if
|
|
23
|
+
* `model` is omitted (preserves historical explicit-worker `missing_model` behavior).
|
|
24
|
+
*/
|
|
25
|
+
readonly requireExplicitModel?: boolean;
|
|
26
|
+
}
|
|
27
|
+
export interface ResolveUseCaseModelInput {
|
|
28
|
+
/** Explicit use-case model (or `binding.model`). */
|
|
29
|
+
readonly configured?: ModelConfig;
|
|
30
|
+
/** Active session / agent model used when `configured` is omitted. */
|
|
31
|
+
readonly sessionModel?: ModelConfig;
|
|
32
|
+
/** When true, skip session fallback (OM `missing_model` escape hatch). */
|
|
33
|
+
readonly requireExplicitModel?: boolean;
|
|
34
|
+
readonly providerOptions?: ProviderRequestOptions;
|
|
35
|
+
readonly thinkingLevel?: string;
|
|
36
|
+
}
|
|
37
|
+
export interface ResolvedUseCaseModel {
|
|
38
|
+
readonly model: ModelConfig;
|
|
39
|
+
/** Whether the model came from the use-case binding or session fallback. */
|
|
40
|
+
readonly source: "configured" | "session";
|
|
41
|
+
readonly providerOptions?: ProviderRequestOptions;
|
|
42
|
+
readonly thinkingLevel?: string;
|
|
43
|
+
}
|
|
44
|
+
/**
|
|
45
|
+
* Resolve the model for a non-session LLM job.
|
|
46
|
+
*
|
|
47
|
+
* Precedence:
|
|
48
|
+
* 1. `configured` → `source: "configured"`
|
|
49
|
+
* 2. Else `sessionModel` when `requireExplicitModel` is not set → `source: "session"`
|
|
50
|
+
* 3. Else `undefined` (caller skips / throws per package policy)
|
|
51
|
+
*
|
|
52
|
+
* O(1); no network. Does not mutate session history.
|
|
53
|
+
*/
|
|
54
|
+
export declare function resolveUseCaseModel(input: ResolveUseCaseModelInput): ResolvedUseCaseModel | undefined;
|
|
55
|
+
/**
|
|
56
|
+
* Resolve from a {@link UseCaseModelBinding} plus optional session fallback.
|
|
57
|
+
*/
|
|
58
|
+
export declare function resolveUseCaseModelBinding(binding: UseCaseModelBinding | undefined, sessionModel?: ModelConfig): ResolvedUseCaseModel | undefined;
|
|
59
|
+
/**
|
|
60
|
+
* Provider id for credential requests: always the **resolved** model's provider.
|
|
61
|
+
* Optional `binding.provider` is only a hint when no model resolved yet.
|
|
62
|
+
*/
|
|
63
|
+
export declare function useCaseCredentialProviderId(resolved: ResolvedUseCaseModel | undefined, binding?: Pick<UseCaseModelBinding, "provider">): string | undefined;
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Resolve the model for a non-session LLM job.
|
|
3
|
+
*
|
|
4
|
+
* Precedence:
|
|
5
|
+
* 1. `configured` → `source: "configured"`
|
|
6
|
+
* 2. Else `sessionModel` when `requireExplicitModel` is not set → `source: "session"`
|
|
7
|
+
* 3. Else `undefined` (caller skips / throws per package policy)
|
|
8
|
+
*
|
|
9
|
+
* O(1); no network. Does not mutate session history.
|
|
10
|
+
*/
|
|
11
|
+
export function resolveUseCaseModel(input) {
|
|
12
|
+
const { providerOptions, thinkingLevel } = input;
|
|
13
|
+
if (input.configured) {
|
|
14
|
+
return {
|
|
15
|
+
model: input.configured,
|
|
16
|
+
source: "configured",
|
|
17
|
+
providerOptions,
|
|
18
|
+
thinkingLevel,
|
|
19
|
+
};
|
|
20
|
+
}
|
|
21
|
+
if (input.requireExplicitModel)
|
|
22
|
+
return undefined;
|
|
23
|
+
if (input.sessionModel) {
|
|
24
|
+
return {
|
|
25
|
+
model: input.sessionModel,
|
|
26
|
+
source: "session",
|
|
27
|
+
providerOptions,
|
|
28
|
+
thinkingLevel,
|
|
29
|
+
};
|
|
30
|
+
}
|
|
31
|
+
return undefined;
|
|
32
|
+
}
|
|
33
|
+
/**
|
|
34
|
+
* Resolve from a {@link UseCaseModelBinding} plus optional session fallback.
|
|
35
|
+
*/
|
|
36
|
+
export function resolveUseCaseModelBinding(binding, sessionModel) {
|
|
37
|
+
return resolveUseCaseModel({
|
|
38
|
+
configured: binding?.model,
|
|
39
|
+
sessionModel,
|
|
40
|
+
requireExplicitModel: binding?.requireExplicitModel,
|
|
41
|
+
providerOptions: binding?.providerOptions,
|
|
42
|
+
thinkingLevel: binding?.thinkingLevel,
|
|
43
|
+
});
|
|
44
|
+
}
|
|
45
|
+
/**
|
|
46
|
+
* Provider id for credential requests: always the **resolved** model's provider.
|
|
47
|
+
* Optional `binding.provider` is only a hint when no model resolved yet.
|
|
48
|
+
*/
|
|
49
|
+
export function useCaseCredentialProviderId(resolved, binding) {
|
|
50
|
+
return resolved?.model.provider ?? binding?.provider;
|
|
51
|
+
}
|
|
52
|
+
//# sourceMappingURL=use-case-model.js.map
|
package/docs/a2a.md
CHANGED
|
@@ -21,7 +21,7 @@ Use it to expose one explicitly selected Prism agent at an A2A endpoint or call
|
|
|
21
21
|
|
|
22
22
|
## Outputs / response / events
|
|
23
23
|
|
|
24
|
-
The handler serves `GET /.well-known/agent-card.json` and its configured POST endpoint. JSON-RPC returns `{ result: { task } }` or a bounded error. Streaming returns backpressure-driven SSE task envelopes. Client `send()` maps a terminal remote task to `AgentRunResult`; `stream()` yields validated/redacted text artifacts.
|
|
24
|
+
The handler serves `GET /.well-known/agent-card.json` and its configured POST endpoint. JSON-RPC returns `{ result: { task } }` or a bounded error. Streaming returns backpressure-driven SSE task envelopes. Client `send()` maps a terminal remote task to `AgentRunResult`; `stream()` incrementally yields validated/redacted text artifacts. Client SSE accepts LF, CRLF, mixed blank-line separators, comments/unknown fields, and multiline `data:` joined with LF.
|
|
25
25
|
|
|
26
26
|
## Request/response example
|
|
27
27
|
|
|
@@ -59,7 +59,9 @@ Only `text` parts are accepted. File/data parts, push notifications, task persis
|
|
|
59
59
|
## Security and performance notes
|
|
60
60
|
|
|
61
61
|
- Endpoints and card URLs must be HTTPS and exactly origin-allow-listed before fetch; `redirect: "error"` prevents redirect SSRF.
|
|
62
|
-
- Treat every remote card, error, task, status, artifact, and SSE frame as untrusted. Shape/count/byte/time limits apply before mapping.
|
|
62
|
+
- Treat every remote card, error, task, status, artifact, and SSE frame as untrusted. Shape/count/byte/time limits apply before mapping. Streaming keeps raw stream bytes, current frame bytes, and event count as separate existing limits.
|
|
63
|
+
- One fatal streaming UTF-8 decoder is reused across every body chunk and flushed once at EOF. Split multibyte code points are preserved; malformed/truncated UTF-8 fails rather than inserting `U+FFFD` into JSON. A small coalesced line buffer keeps one-byte chunk handling incremental.
|
|
64
|
+
- SSE frames require a terminating blank line. A non-whitespace final partial frame, malformed JSON, missing terminal task, failed/canceled task, or any event after a completed task fails with bounded package-owned text. Existing request/response/event/stream/count/timeout hard caps are unchanged.
|
|
63
65
|
- Card verification pins `alg=ES256`, optional key ID, issue/expiry, optional maximum age, and canonical unsigned-card payload. Hosts provision trusted public keys; remote `jku` is never fetched automatically.
|
|
64
66
|
- Card discovery is public; extended-card and invoke methods call host authorization. Use TLS, rate limits, and replay controls at the host edge.
|
|
65
67
|
- Credentials remain in the client auth callback or server authorizer and never enter cards, messages, events, or metrics.
|
package/docs/agent-events.md
CHANGED
|
@@ -100,28 +100,23 @@ Artifact validation/refinement events (emitted only by `generateValidateReviseLo
|
|
|
100
100
|
| `artifact_validation_finished` | `sessionId`, `runId`, `turn`, `attempt`, `result: ArtifactValidation` |
|
|
101
101
|
| `artifact_revision_started` | `sessionId`, `runId`, `turn`, `attempt`, `failure: ArtifactValidation` |
|
|
102
102
|
| `artifact_finished` | `sessionId`, `runId`, `turn`, `attempt`, `result: ArtifactValidation` (loop ended successfully) |
|
|
103
|
-
| `artifact_failed` | `sessionId`, `runId`, `turn`, `attempt`, `result: ArtifactValidation` (budget exhausted) |
|
|
103
|
+
| `artifact_failed` | `sessionId`, `runId`, `turn`, `attempt`, `result: ArtifactValidation` (candidate budget exhausted, or `result.metadata.reason === "tool_round_limit"`) |
|
|
104
104
|
|
|
105
105
|
### Artifact event ordering
|
|
106
106
|
|
|
107
|
-
A `generateValidateReviseLoop`
|
|
107
|
+
A call-free candidate in `generateValidateReviseLoop` emits normal turn/message events then a strictly ordered artifact sequence, correlated by `runId` / `turn` / `attempt`:
|
|
108
108
|
|
|
109
109
|
```
|
|
110
|
-
turn_started
|
|
111
|
-
→
|
|
112
|
-
→
|
|
113
|
-
→ message_finished
|
|
114
|
-
→ turn_finished
|
|
115
|
-
→ artifact_validation_started
|
|
116
|
-
→ artifact_validation_finished
|
|
117
|
-
→ artifact_revision_started # when a revision will run next
|
|
118
|
-
| artifact_finished # loop ended successfully
|
|
119
|
-
| artifact_failed # budget exhausted (maxRevisions+1 attempts)
|
|
110
|
+
turn_started → message_started → message_delta* → message_finished → turn_finished
|
|
111
|
+
→ artifact_validation_started → artifact_validation_finished
|
|
112
|
+
→ artifact_revision_started | artifact_finished | artifact_failed
|
|
120
113
|
```
|
|
121
114
|
|
|
122
|
-
- `
|
|
115
|
+
With opt-in `toolCalls: "bounded"`, a provider turn containing calls emits its normal assistant envelope followed by existing `tool_execution_*` events and matching persisted tool results; it emits no validation event and the next provider turn consumes that transcript. A post-`maxToolRounds` call emits terminal `artifact_failed` directly after `turn_finished` and has no tool execution event.
|
|
116
|
+
|
|
117
|
+
- `attempt` is 1-indexed per call-free validation candidate. It can differ from provider `turn` when bounded tool calls occur.
|
|
123
118
|
- Single-shot runs emit zero artifact events.
|
|
124
|
-
- **Validation failure triggering a revision is recoverable and never an `error`.**
|
|
119
|
+
- **Validation failure triggering a revision is recoverable and never an `error`.** Terminal candidate-budget or `tool_round_limit` exhaustion emits `artifact_failed`; real failures remain on the `error` channel.
|
|
125
120
|
|
|
126
121
|
## Request/response example
|
|
127
122
|
|
|
@@ -193,7 +188,7 @@ for await (const event of session.stream("draft", { loop: { strategy: "generate-
|
|
|
193
188
|
- Slow consumers are bounded by `SubscribeOptions`. Use `RunLedger` or host storage for durable replay; do not rely on a live subscriber as a queue.
|
|
194
189
|
- Redaction is exact-string-match only and opt-in via `createSecretRedactor`; values not passed as known secrets are not redacted.
|
|
195
190
|
- `ArtifactValidation.errors[].message` and `metadata` may echo model text; `redactAgentEvent` walks arbitrary nesting and replaces cyclic references with `"[Circular]"` (WeakSet cycle guard), so secret values in `result`/`failure` are redacted without crashing.
|
|
196
|
-
- `artifact_*` events are bounded by `maxRevisions + 1`
|
|
191
|
+
- `artifact_*` validation events are bounded by `maxRevisions + 1` call-free candidates. With opt-in bounded artifact tools, provider turns are additionally bounded by run-global `maxToolRounds` (maximum `1 + maxRevisions + maxToolRounds`); a post-cap call emits exactly one terminal `artifact_failed` with `result.metadata.reason === "tool_round_limit"` and has no tool lifecycle event because it never dispatches.
|
|
197
192
|
- Runtime events contain messages/content only; do not put secrets in prompts, metadata, provider events, session entries, tool results, or artifact validation payloads.
|
|
198
193
|
|
|
199
194
|
## Related APIs
|
package/docs/agent-loops.md
CHANGED
|
@@ -16,7 +16,7 @@ The `Artifact*` contracts (`ArtifactValidation`, `ArtifactContext`, `ArtifactPar
|
|
|
16
16
|
|
|
17
17
|
Use the default `singleShotLoop` implicitly whenever you call `session.run()` — no configuration needed. Opt into `generateValidateReviseLoop` when a run should produce an artifact that must satisfy a host-supplied schema before it is considered complete (e.g. structured output, a validated JSON document, a generated file passing lint) and the host wants Prism to drive the revision turns.
|
|
18
18
|
|
|
19
|
-
Do not use a loop to re-implement provider calls, retry, abort, store, or event emission — those stay runtime-owned and are exposed to the loop only through `LoopContext`.
|
|
19
|
+
Do not use a loop to re-implement provider calls, retry, abort, store, or event emission — those stay runtime-owned and are exposed to the loop only through `LoopContext`. Artifact-loop tools stay disabled by default; opt into bounded calls only for host-registered, least-privilege lookup tools that must inform an artifact candidate.
|
|
20
20
|
|
|
21
21
|
## Inputs / request
|
|
22
22
|
|
|
@@ -57,6 +57,7 @@ await session.run(input, {
|
|
|
57
57
|
parser: hostParser, // optional; default treats assistant text as the value
|
|
58
58
|
repairer: hostRepairer, // optional; default stringifies validation.errors[].message
|
|
59
59
|
maxRevisions: 3, // optional; default 3
|
|
60
|
+
toolCalls: "bounded", // optional; default "disabled"; uses RunOptions.maxToolRounds
|
|
60
61
|
},
|
|
61
62
|
});
|
|
62
63
|
|
|
@@ -79,6 +80,8 @@ type AgentLoopOptions =
|
|
|
79
80
|
readonly parser?: ArtifactParser<unknown>;
|
|
80
81
|
readonly repairer?: ArtifactRepairer<unknown>;
|
|
81
82
|
readonly maxRevisions?: number;
|
|
83
|
+
/** Default "disabled". "bounded" dispatches sequentially up to RunOptions.maxToolRounds. */
|
|
84
|
+
readonly toolCalls?: "disabled" | "bounded";
|
|
82
85
|
};
|
|
83
86
|
```
|
|
84
87
|
|
|
@@ -112,7 +115,7 @@ Host callback contracts (all generic over host `T`):
|
|
|
112
115
|
|
|
113
116
|
Events during a loop run are the existing `AgentEvent`s (`turn_started`, `message_started`, `message_delta`, `message_finished`, `turn_finished`, tool-execution events when the loop dispatches tools, `error` on real failures). Both built-in loops emit `turn_started` before each provider turn, `message_finished` for every assistant draft, and `turn_finished` after the assistant draft is appended. First-turn input is appended to live history once, matching the already-persisted user message.
|
|
114
117
|
|
|
115
|
-
Validation-failure-triggering-a-revision is **not** an `error` event — it is recoverable, like `tool_execution_blocked`.
|
|
118
|
+
Validation-failure-triggering-a-revision is **not** an `error` event — it is recoverable, like `tool_execution_blocked`. In bounded artifact mode, a tool-calling provider response emits normal assistant/tool lifecycle events, skips artifact parsing/validation, then the next turn sees its persisted result. `generateValidateReviseLoop` emits artifact events only for call-free candidates: `artifact_validation_started` → `artifact_validation_finished` → (`artifact_revision_started`)* → `artifact_finished` | `artifact_failed`. A request beyond `maxToolRounds` executes nothing and emits terminal `artifact_failed` with `result.metadata.reason === "tool_round_limit"`; see [Agent events § Artifact event ordering](agent-events.md#artifact-event-ordering). `singleShotLoop` emits zero artifact events. Real failures stay on the `error` channel.
|
|
116
119
|
|
|
117
120
|
A loop has no path to credentials, provider objects, or unredacted secrets. `LoopContext.generate` receives the already-policy-applied, middleware-run, redacted request; `LoopContext.emit` runs through `redactAgentEvent` with the active `SecretRedactor`.
|
|
118
121
|
|
|
@@ -198,17 +201,17 @@ await session.run(input, { loop: twoShotLoop });
|
|
|
198
201
|
- `RunOptions.loop` wins over `AgentConfig.loop`; when neither is set the runtime uses `singleShotLoop`. This mirrors the other `RunOptions` overrides (`redactor`, `validate`, `activeSkills`).
|
|
199
202
|
- `{ strategy: "single-shot" }` resolves to the exported `singleShotLoop`; `{ strategy: "generate-validate-revise", ... }` is mapped by `resolveLoop()` to `generateValidateReviseLoop(opts)`. An unknown `strategy` throws before the first turn. Passing an `AgentLoopStrategy` instance bypasses the options form entirely (custom-loop escape hatch).
|
|
200
203
|
- The loop is resolved once per run inside `RuntimeAgentSession.run()`, after the usual setup (provider/skills/tools resolution, history rebuild, model-change entry, input append, auto-compaction). The runtime's outer try/catch/finally, run-exclusivity, abort bridging, and subscriber close remain in place around `loop.run(ctx)`.
|
|
201
|
-
- `LoopContext.assemble(nextInput, toolResults?)` accepts an optional tool-result accumulator so `singleShotLoop` can pass its loop-local
|
|
202
|
-
- `maxToolRounds` bounds `singleShotLoop`
|
|
203
|
-
- `maxRevisions` (default 3)
|
|
204
|
-
- A revision cycle appends one assistant draft and one repair user message per revision to the session store, so store entries reflect every attempted draft. The original user input is stored once by the runtime and pushed into loop history once on the first turn.
|
|
204
|
+
- `LoopContext.assemble(nextInput, toolResults?)` accepts an optional tool-result accumulator so `singleShotLoop` can pass its loop-local results. Bounded artifact tools append results directly to shared history, then assemble the next turn with empty new input; no second transcript path exists.
|
|
205
|
+
- `maxToolRounds` bounds both `singleShotLoop` and opt-in bounded artifact tool rounds across the whole run. Artifact mode always dispatches sequentially, regardless of `toolConcurrency`; all dispatches still use existing registry/filter/permission/validator/middleware/redactor/ledger guards.
|
|
206
|
+
- `maxRevisions` (default 3) counts only failed call-free artifact candidates. Bounded artifact runs make at most `1 + maxRevisions + maxToolRounds` provider turns. A tool-round limit is terminal and returns last usage after `artifact_failed`; it does not throw.
|
|
207
|
+
- A revision cycle appends one assistant draft and one repair user message per revision to the session store, so store entries reflect every attempted draft. The original user input is stored once by the runtime and pushed into loop history once on the first turn. Repair messages are assembled as the next provider `nextInput` and only pushed into live history after that revision request has been generated, so the model never receives a duplicated repair instruction.
|
|
205
208
|
|
|
206
209
|
## Security and performance notes
|
|
207
210
|
|
|
208
211
|
- Loops have no path to credentials, provider objects, or unredacted secrets. `LoopContext.generate` consumes an already-redacted request; `LoopContext.emit` runs through `redactAgentEvent` with the active `SecretRedactor`; `LoopContext.appendMessage` appends a redacted entry.
|
|
209
212
|
- `ArtifactValidation.errors[].message` may echo model text — `artifact_*` event payloads flow through the same `redactAgentEvent` path as other `AgentEvent`s (see [Agent events](agent-events.md)).
|
|
210
|
-
- `generateValidateReviseLoop` makes at most `maxRevisions +
|
|
211
|
-
-
|
|
213
|
+
- `generateValidateReviseLoop` makes at most `1 + maxRevisions + maxToolRounds` provider turns when bounded tools are enabled (otherwise `maxRevisions + 1`); it cannot loop forever. Each revision costs one provider turn plus one store append.
|
|
214
|
+
- Bounded artifact tool calls run sequentially through `dispatchToolCall` (permission + validation + execute); their assistant call and result are persisted before the next provider request. `singleShotLoop` retains its bounded parallel worker pool and original call-order transcript behavior.
|
|
212
215
|
- The loop is a plain object/factory; no class hierarchy, no background work, no extra dependencies. `LoopContext` is a single object literal of bound arrows built once per run.
|
|
213
216
|
- The Synapta-free boundary is guarded by tests: `src/` imports no `synapta*` package, and the `Artifact*`/`AgentLoop*`/`LoopContext` contracts contain no `workflow`/`node`/`step` field names. Hosts supply their own schema; no host domain type is imported by `src/`.
|
|
214
217
|
|
|
@@ -15,6 +15,8 @@
|
|
|
15
15
|
| `createAllTools(cwd, options?)` | Every tool the package provides (currently identical to `createCodingTools`). |
|
|
16
16
|
| `detectSupportedImageMimeType(buf)` / `detectSupportedImageMimeTypeFromFile(path)` | Magic-byte image MIME detection (PNG/JPEG/GIF/WebP/BMP) used by `read`. |
|
|
17
17
|
| `DEFAULT_MAX_IMAGE_BYTES` | Default `read` image size ceiling (10 MB). |
|
|
18
|
+
| `DEFAULT_*` / `HARD_*` coding limit constants | Published text-scan, image, write/edit, shell timeout, display, and total-output ceilings. |
|
|
19
|
+
| `ReadTextOptions` / `ReadTextResult` | Bounded text-page contract required by custom `ReadOperations`. |
|
|
18
20
|
| `TransformImage` / `TransformImageInput` | Types for the optional `read` `transformImage` callback. |
|
|
19
21
|
| `withFileMutationQueue(path, fn)` | Per-path serialization primitive re-exported for hosts. |
|
|
20
22
|
|
|
@@ -65,7 +67,7 @@ Run a shell command and return combined stdout+stderr.
|
|
|
65
67
|
| Field | Type | Purpose |
|
|
66
68
|
| --- | --- | --- |
|
|
67
69
|
| `command` | `string` | Shell command to execute (required). |
|
|
68
|
-
| `timeout` | `number` | Timeout in **seconds** (optional;
|
|
70
|
+
| `timeout` | `number` | Timeout in **seconds** (optional; defaults to 600, hard maximum 3600). |
|
|
69
71
|
|
|
70
72
|
**Outputs:** a `ToolResult` whose `content[0]` is a `TextContent` with the combined output. Non-zero exit is **not** a tool error: it is returned as a normal result with `[Command exited with code N]` appended to the content and `exitCode` in metadata. Timeout and abort are error results that still carry the partial output captured so far.
|
|
71
73
|
|
|
@@ -75,9 +77,11 @@ Run a shell command and return combined stdout+stderr.
|
|
|
75
77
|
| --- | --- | --- |
|
|
76
78
|
| `exitCode` | always | Process exit code, or `null` when the process was killed by timeout/abort. |
|
|
77
79
|
| `truncation` | always | `TruncationResult` from the bounded output accumulator. |
|
|
78
|
-
| `fullOutputPath?` | truncated only |
|
|
80
|
+
| `fullOutputPath?` | successful and truncated only | Host-owned path to retained output. Failed/aborted/timed-out/output-limited calls remove unpublished spills. |
|
|
81
|
+
| `totalOutputBytes` | shell executed | Raw bytes retained, never above `maxTotalOutputBytes`. |
|
|
82
|
+
| `outputLimitExceeded` / `outputStorageFailed` | shell executed | Attributable resource failure flags. |
|
|
79
83
|
|
|
80
|
-
Shell resolution honors `options.shellPath` → `SHELL` env → `/bin/bash` → `sh
|
|
84
|
+
Shell resolution honors `options.shellPath` → `SHELL` env → `/bin/bash` → `sh`. The process group is killed on timeout, caller abort, spill failure, or total-output overflow (`process.kill(-pid)` on Unix, `taskkill /F /T` on Windows). Combined output defaults to a 64 MiB total cap (1 GiB hard cap). Spill files use random exclusive creation and Unix mode `0600`; hosts own and must delete a successful result's `fullOutputPath` after consumption.
|
|
81
85
|
|
|
82
86
|
### `read`
|
|
83
87
|
|
|
@@ -91,7 +95,7 @@ Read a text or image file.
|
|
|
91
95
|
| `offset` | `number` | Line to start reading from (1-indexed). |
|
|
92
96
|
| `limit` | `number` | Maximum number of lines to read. |
|
|
93
97
|
|
|
94
|
-
**Outputs:** text files
|
|
98
|
+
**Outputs:** text files are scanned incrementally until one requested page, `maxLines`/`maxBytes`, EOF, or `maxScanBytes` (default 64 MiB scanned per call; 1 GiB hard cap). The default path never loads the complete file and returns a `Use offset=N to continue` footer when more remains. Exact total line count is reported only when EOF was already reached in the bounded scan. Image files (PNG/JPEG/GIF/WebP/BMP by **magic bytes**, not extension) become `[TextContent note, ImageContent]` with base64 `data` and `mimeType`. Oversize images are rejected by `stat` (when available) or `buffer.length` against `maxImageBytes` (default 10 MB) before base64 encoding. An optional `transformImage` callback lets hosts resize or re-encode images without adding image-processing dependencies to the base package. Read failures (missing file, offset beyond end, oversize image, abort) are error results.
|
|
95
99
|
|
|
96
100
|
`read` tool options (via `createReadTool(cwd, options)` or `ToolsOptions.read`):
|
|
97
101
|
|
|
@@ -100,8 +104,9 @@ Read a text or image file.
|
|
|
100
104
|
| `maxImageBytes` | `DEFAULT_MAX_IMAGE_BYTES` (10 MB) | Reject image reads larger than this many bytes. |
|
|
101
105
|
| `transformImage` | — | Host callback `( { buffer, mimeType } ) => Promise<Buffer>` run after read, before base64. |
|
|
102
106
|
| `autoResizeImages` | — | **Deprecated.** Ignored unless `transformImage` is also set (use `transformImage` instead). |
|
|
103
|
-
| `maxLines` / `maxBytes` | 2000 / 50
|
|
104
|
-
| `
|
|
107
|
+
| `maxLines` / `maxBytes` | 2000 / 50 KiB | Text page display limits (hard: 100,000 / 1 MiB). |
|
|
108
|
+
| `maxScanBytes` | 64 MiB | Raw bytes scanned to reach one page (hard: 1 GiB). |
|
|
109
|
+
| `operations` | local fs | Pluggable bounded `ReadOperations` backend. |
|
|
105
110
|
| `executionPolicy` | — | Structured pre-execution policy (see [Coding security](coding-security.md)). |
|
|
106
111
|
|
|
107
112
|
```ts
|
|
@@ -133,7 +138,7 @@ Create or overwrite a file, creating parent directories as needed.
|
|
|
133
138
|
| `path` | `string` | Path to the file to write (relative or absolute). Required. |
|
|
134
139
|
| `content` | `string` | Content to write (empty string creates an empty file). Required. |
|
|
135
140
|
|
|
136
|
-
**Outputs:** a `TextContent` confirmation naming the **absolute path** with UTF-8 byte and line counts (e.g. `Successfully wrote 42 bytes (3 lines) to /abs/path.txt`). Write failures and abort are error results. Empty `content` is valid.
|
|
141
|
+
**Outputs:** a `TextContent` confirmation naming the **absolute path** with UTF-8 byte and line counts (e.g. `Successfully wrote 42 bytes (3 lines) to /abs/path.txt`). `maxInputBytes` defaults to 8 MiB (64 MiB hard cap); oversized UTF-8 input fails before policy evaluation, directory creation, or write. Write failures and abort are error results. Empty `content` is valid.
|
|
137
142
|
|
|
138
143
|
`write` result `metadata`: `{ bytes, lines, path }` (absolute path). Concurrent writes to the same path serialize through `withFileMutationQueue`; writes to different paths run in parallel.
|
|
139
144
|
|
|
@@ -148,7 +153,7 @@ Precise text replacement in an existing file via exact-then-fuzzy matching.
|
|
|
148
153
|
| `path` | `string` | Path to the file to edit. Required. |
|
|
149
154
|
| `edits` | `Array<{ oldText: string, newText: string }>` | Targeted replacements, each matched against the **original** file (not incrementally). No overlapping/nested edits. Required, non-empty. |
|
|
150
155
|
|
|
151
|
-
Each `edits[].oldText` must match a unique, non-overlapping region of the original file. Matching is exact first, then fuzzy (unicode normalization / whitespace collapse). A BOM is stripped before matching and re-prepended on write; original line endings are restored.
|
|
156
|
+
Each `edits[].oldText` must match a unique, non-overlapping region of the original file. Matching is exact first, then fuzzy (unicode normalization / whitespace collapse). A BOM is stripped before matching and re-prepended on write; original line endings are restored. Defaults reject targets over 8 MiB, aggregate old/new UTF-8 input over 2 MiB, or more than 100 edits (hard caps: 64 MiB, 16 MiB, and 1,000). Stat and bounded read checks run before matching or mutation.
|
|
152
157
|
|
|
153
158
|
**Outputs:** a `TextContent` confirmation (`Successfully replaced N block(s) in {path}.`) plus `metadata`. Any failure — missing/unreadable file, no match, duplicate (non-unique) match, overlap, empty `oldText`, no-op edit, or abort — is an error result, and the file is left **unchanged** (the match runs before the write).
|
|
154
159
|
|
|
@@ -156,7 +161,7 @@ Each `edits[].oldText` must match a unique, non-overlapping region of the origin
|
|
|
156
161
|
|
|
157
162
|
## Outputs / response / events
|
|
158
163
|
|
|
159
|
-
Every tool returns a `ToolResult` with `toolCallId`, `name`, `content` (`readonly ContentBlock[]`), optional `error`, and optional `metadata`.
|
|
164
|
+
Every tool returns a `ToolResult` with `toolCallId`, `name`, `content` (`readonly ContentBlock[]`), optional `error`, and optional `metadata`. `write` and `edit` serialize per realpath through `withFileMutationQueue` so concurrent calls targeting one file do not interleave. `shell` is marked `exclusive`; tool dispatch serializes it at the turn level. The package emits no events of its own; hosts observe tool execution through the normal Prism `AgentEvent` stream via `dispatchToolCall`.
|
|
160
165
|
|
|
161
166
|
## Request/response example
|
|
162
167
|
|
|
@@ -208,6 +213,8 @@ const shell = createShellTool("/repo", {
|
|
|
208
213
|
shellPath: "/bin/bash",
|
|
209
214
|
commandPrefix: "set -euo pipefail",
|
|
210
215
|
maxLines: 500,
|
|
216
|
+
timeout: 600,
|
|
217
|
+
maxTotalOutputBytes: 64 * 1024 * 1024,
|
|
211
218
|
});
|
|
212
219
|
|
|
213
220
|
const remoteWrite = createWriteTool("/repo", {
|
|
@@ -220,8 +227,8 @@ const remoteWrite = createWriteTool("/repo", {
|
|
|
220
227
|
|
|
221
228
|
## Extension and configuration notes
|
|
222
229
|
|
|
223
|
-
- **Pluggable operation backends.** Every tool accepts an `operations` seam
|
|
224
|
-
- **Per-tool options.** `ShellToolOptions`
|
|
230
|
+
- **Pluggable operation backends.** Every tool accepts an `operations` seam. Custom `ReadOperations` must implement bounded `readText` plus `statFile`; custom `EditOperations` must implement `statFile`; read/write methods receive caps/signals. `BashOperations` must stream through `onData` and honor `signal`/`timeout`. A hostile custom backend can still violate its host-owned contract, so isolate it separately.
|
|
231
|
+
- **Per-tool options.** `ShellToolOptions` adds `timeout` and `maxTotalOutputBytes`; `ReadToolOptions` adds `maxScanBytes`; `WriteToolOptions` adds `maxInputBytes`; `EditToolOptions` adds `maxFileBytes`, `maxInputBytes`, and `maxEdits`. Invalid/non-finite/unsafe/above-hard-cap values throw during tool construction; request `timeout` errors before spawn.
|
|
225
232
|
- **Aggregator options.** `ToolsOptions` (`{ executionPolicy?, shell?, read?, write?, edit? }`) threads each sub-object to the matching tool. `createCodingTools()`, `createAllTools()`, and `createReadOnlyTools()` apply the shared policy unless that tool has an explicit per-tool override.
|
|
226
233
|
- **`ToolsOptions`** and the per-tool option types are exported from the package barrel for host configuration.
|
|
227
234
|
- No auto-discovery or manifest registration: import and register explicitly. This package registers no extensions and owns no globals (the mutation queue is a process-wide per-path map — see `ponytail:` note in the source).
|
|
@@ -230,10 +237,24 @@ const remoteWrite = createWriteTool("/repo", {
|
|
|
230
237
|
|
|
231
238
|
- **Host shell/filesystem access.** These tools run real commands and read/write real files. They provide **no sandbox**. Gate them with Prism `PermissionPolicy` / `ToolValidator` / trust policies before registering them for any provider turn. Shared `executionPolicy` applies to both full and read-only aggregators before filesystem/process side effects. See [Host security guide](host-security.md) and [Security/auth/trust](settings-auth-trust-security.md).
|
|
232
239
|
- **Non-zero exit is not an error.** A failing command is a normal `shell` result (exit code in metadata); only timeout/abort/spawn failures are error results. Do not assume `error == undefined` means the command succeeded.
|
|
233
|
-
- **Bounded
|
|
240
|
+
- **Bounded I/O.** `read` streams one page and bounds scan bytes; image/edit reads use stat plus a shared cap-enforcing reader; write/edit inputs are measured before mutation. `shell` retains only a rolling display tail and synchronously spills accepted raw chunks so stream backpressure cannot grow heap; wall time and total raw output remain finite.
|
|
234
241
|
- **Per-path serialization.** Concurrent mutations to the same file serialize; concurrent mutations to different files do not block each other. The queue is a process-wide map — across sessions in one process, same-path writes still serialize (upgrade path: scope per registry if throughput matters).
|
|
235
242
|
- **Bounded image reads.** `read` rejects images over `maxImageBytes` (default 10 MB) by `stat` before read when possible; MIME is detected from magic bytes only. Optional `transformImage` is host-owned — the base package has no image-processing dependency.
|
|
236
243
|
|
|
244
|
+
### Resource-limit defaults and hard caps
|
|
245
|
+
|
|
246
|
+
| Boundary | Default | Hard cap | Failure point |
|
|
247
|
+
| --- | ---: | ---: | --- |
|
|
248
|
+
| Display lines / bytes | 2,000 / 50 KiB | 100,000 / 1 MiB | tool construction |
|
|
249
|
+
| Text scan per read | 64 MiB | 1 GiB | bounded scan before more input is retained |
|
|
250
|
+
| Image | 10,000,000 bytes | 32 MiB | stat and bounded read before base64/transform result use |
|
|
251
|
+
| Write UTF-8 input | 8 MiB | 64 MiB | before policy/filesystem mutation |
|
|
252
|
+
| Edit target / input / count | 8 MiB / 2 MiB / 100 | 64 MiB / 16 MiB / 1,000 | before target read/matching/write |
|
|
253
|
+
| Shell wall time | 600 seconds | 3,600 seconds | process-tree kill |
|
|
254
|
+
| Shell total stdout+stderr | 64 MiB | 1 GiB | process-tree kill; spill removal |
|
|
255
|
+
|
|
256
|
+
Every configurable value is a positive safe integer; Prism rejects rather than clamps invalid values. Limits control resources, not authority: they do not replace root containment, approval, validation, or a sandbox.
|
|
257
|
+
|
|
237
258
|
## Related APIs
|
|
238
259
|
|
|
239
260
|
- [Tools](tools.md): the host-owned tool harness — `createToolRegistry`, `dispatchToolCall`, filtering, and the `ToolDefinition` contract these factories satisfy.
|
package/docs/coding-security.md
CHANGED
|
@@ -74,11 +74,11 @@ const tools = createCodingTools(workspaceRoot, {
|
|
|
74
74
|
|
|
75
75
|
Policies are ordinary host values: attach one globally through `createCodingTools()`/`createReadOnlyTools()` or per tool. A per-tool policy overrides the shared policy. `SandboxAdapter` is replaceable and host-owned; approval policy and sandboxing are separate layers.
|
|
76
76
|
|
|
77
|
-
Callback approval remains process-local. For approval that must survive restart, wrap the action in an opted-in workflow `toolNode({ approval: { reason, data?, resumeSchema? } })`. The workflow persists `suspended` state before any tool side effect. After explicit approve, it recomputes the action and invokes this package's current `ExecutionPolicy`; durable approval never populates or bypasses the process-local approval cache. Adapters should emit chunks through `request.onData` as they arrive and honor `request.signal`/`request.timeout`; buffering is unnecessary. Default caching is `none`; use run-scoped caching only when repeated approval within one run is desired, and session scope only when that wider lifecycle is intentional.
|
|
77
|
+
Callback approval remains process-local. For approval that must survive restart, wrap the action in an opted-in workflow `toolNode({ approval: { reason, data?, resumeSchema? } })`. The workflow persists `suspended` state before any tool side effect. After explicit approve, it recomputes the action and invokes this package's current `ExecutionPolicy`; durable approval never populates or bypasses the process-local approval cache. Adapters should emit chunks through `request.onData` as they arrive and honor `request.signal`/`request.timeout`; buffering is unnecessary. Coding-agent composes caller abort with its total-output controller, so ignoring the supplied signal defeats process termination even though Prism stops retaining output at the cap. Default caching is `none`; use run-scoped caching only when repeated approval within one run is desired, and session scope only when that wider lifecycle is intentional.
|
|
78
78
|
|
|
79
79
|
## Security and performance notes
|
|
80
80
|
|
|
81
|
-
Containment resolves symlinks and rejects paths outside roots. Command rules are not a shell parser; shell metacharacters require approval. Approval waits and subprocess execution honor abort/timeouts. Durable workflow denial/cancellation is terminal and attributable; approved resume still fails if roots, command rules, read-only mode, or other policy changed while suspended. Cache keys are fixed-size SHA-256 digests of selected identity plus action shape; caches remain process-local, retain at most 1,000 decisions with oldest-entry eviction, and have no default/global mode. Path checks and cache lookup are local; sandbox latency belongs to the supplied adapter.
|
|
81
|
+
Containment resolves symlinks and rejects paths outside roots. Command rules are not a shell parser; shell metacharacters require approval. Approval waits and subprocess execution honor abort/timeouts. Coding-agent resource ceilings independently bound text scans, image/edit target reads, write/edit payloads, edit counts, shell wall time, and retained/spilled output. Those ceilings reduce exhaustion risk but do not grant path/command authority or make an unsandboxed shell safe. Durable workflow denial/cancellation is terminal and attributable; approved resume still fails if roots, command rules, read-only mode, or other policy changed while suspended. Cache keys are fixed-size SHA-256 digests of selected identity plus action shape; caches remain process-local, retain at most 1,000 decisions with oldest-entry eviction, and have no default/global mode. Path checks and cache lookup are local; sandbox latency belongs to the supplied adapter.
|
|
82
82
|
|
|
83
83
|
## Related APIs
|
|
84
84
|
|
package/docs/compaction-llm.md
CHANGED
|
@@ -25,15 +25,16 @@ Key exports:
|
|
|
25
25
|
| Field | Purpose |
|
|
26
26
|
| --- | --- |
|
|
27
27
|
| `provider` / `summaryProvider` | Explicit `AIProvider`, or factory receiving a resolved credential. |
|
|
28
|
-
| `model` / `summaryModel` |
|
|
28
|
+
| `model` / `summaryModel` | Summary `ModelConfig`. `summaryModel` wins; `model` is the session/host fallback via `resolveUseCaseModel`. See [Use-case model selection](use-case-model-selection.md). |
|
|
29
29
|
| `credential`, `credentialRequest` | Optional per-call credential resolution for provider factories. |
|
|
30
30
|
| `providerOptions` | Generic `ProviderRequest.options`, including cache fields. |
|
|
31
31
|
| `providerRequestPolicies` | Optional Prism provider request policies applied before the summary call. |
|
|
32
32
|
| `customInstructions` | Additional summary focus appended to prompts. |
|
|
33
|
-
| `thinkingLevel` |
|
|
34
|
-
| `reserveTokens` | Output budget basis; defaults to `16384`. |
|
|
33
|
+
| `thinkingLevel` | Mapped into `ProviderRequest.options.compat` via `applyThinkingLevel` / `thinkingFamilyForModel` (not inert `extra.thinkingLevel`). See [Thinking and reasoning](thinking-and-reasoning.md). |
|
|
34
|
+
| `reserveTokens` | Output budget basis; defaults to `16384`, hard cap `131072`. |
|
|
35
35
|
| `keepRecentTokens` | Approximate recent-token budget; defaults to `20000`. |
|
|
36
|
-
| `maxSummaryTokens` / `maxOutputTokens` |
|
|
36
|
+
| `maxSummaryTokens` / `maxOutputTokens` | Summary retention/request ceiling; default `16384`, hard cap `131072`. `maxSummaryTokens` wins over the compatibility alias. The finite value is written to `model.parameters.maxTokens`; first-party providers map it to their wire field. |
|
|
37
|
+
| `maxErrorBytes` | Retained provider/factory/policy error detail; default `1024`, hard cap `8192`, UTF-8-safe and known-secret redacted. |
|
|
37
38
|
| `maxToolResultChars` | Tool-result JSON truncation limit; defaults to `2000`. |
|
|
38
39
|
| `trackFileOperations`, `includeFileOperations` | Control file path extraction and final summary blocks. |
|
|
39
40
|
| `secrets` | Exact strings to redact from serialized prompts and final summaries. |
|
|
@@ -64,7 +65,8 @@ const strategy = createLlmCompactionStrategy({
|
|
|
64
65
|
model: { provider: "openai", model: "gpt-4.1-mini" },
|
|
65
66
|
keepRecentTokens: 20_000,
|
|
66
67
|
reserveTokens: 16_384,
|
|
67
|
-
|
|
68
|
+
maxSummaryTokens: 800,
|
|
69
|
+
maxErrorBytes: 1_024,
|
|
68
70
|
providerOptions: { cacheRetention: "short" },
|
|
69
71
|
customInstructions: "Focus on current files and failing tests.",
|
|
70
72
|
});
|
|
@@ -102,10 +104,18 @@ const agent = createAgent({ model, provider, compaction: { strategy, thresholdEn
|
|
|
102
104
|
Registration only contributes an inert strategy. The host must resolve and pass it to runtime config.
|
|
103
105
|
|
|
104
106
|
## Security and performance notes
|
|
105
|
-
Preparation is O(n) over branch entries and uses only arrays, strings, and JSON serialization.
|
|
107
|
+
Preparation is O(n) over branch entries and uses only arrays, strings, and JSON serialization. Limit options must be positive safe integers at or below their hard caps and reject during strategy creation. Missing output options use a 16,384-token summary ceiling; reserve ratio/model metadata may narrow the provider request, never remove its finite `maxTokens`. A request policy that replaces `maxTokens` with NaN, Infinity, zero, an unsafe integer, or above-hard-cap input fails before provider generation.
|
|
108
|
+
|
|
109
|
+
Provider deltas are redacted while retained and stop at `maxSummaryTokens * 4` UTF-16 code units without splitting a surrogate pair. Provider iteration is closed/aborted on overflow. A derived finite event ceiling also stops endless empty/non-text deltas. Final history/turn/file composition receives the same cap. Provider error events, generator throws, provider-factory failures, and policy failures expose only bounded redacted detail; host abort remains authoritative.
|
|
110
|
+
|
|
111
|
+
The strategy makes only the needed provider call(s): one history summary plus one split-turn prefix summary when needed. It does not discover credentials, read files, start background jobs, or add provider SDK dependencies. Redaction is exact-string only; pass every known secret that may appear in history or provider output.
|
|
106
112
|
|
|
107
113
|
## Related APIs
|
|
108
|
-
|
|
114
|
+
|
|
115
|
+
- [Use-case model selection](use-case-model-selection.md): `summaryModel` vs session `model` fallback.
|
|
116
|
+
- [Thinking and reasoning](thinking-and-reasoning.md): `thinkingLevel` → `compat`.
|
|
117
|
+
- [Compaction and retry policies](compaction-and-retry.md): replaceable compaction strategy boundary and core compaction strategy surface.
|
|
118
|
+
- [Observational memory compaction package](compaction-observational-memory.md): source-backed memory workers with the same use-case binding pattern.
|
|
109
119
|
- [Agent/session runtime](agent-session-runtime.md): `AgentSession.compact()` and opt-in auto-compaction.
|
|
110
120
|
- [Provider layer](provider-layer.md): mock providers and provider request contracts.
|
|
111
121
|
- [Credentials and redaction](credentials-and-redaction.md): exact known-secret redaction behavior.
|