@arnilo/prism 0.0.7 → 0.0.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/CHANGELOG.md +50 -2
  2. package/README.md +3 -1
  3. package/dist/agent-loops.js +14 -8
  4. package/dist/agents.js +37 -3
  5. package/dist/contracts.d.ts +17 -0
  6. package/dist/index.d.ts +4 -2
  7. package/dist/index.js +3 -2
  8. package/dist/provider-events.d.ts +2 -0
  9. package/dist/provider-events.js +21 -13
  10. package/dist/providers/openai-compatible.js +8 -5
  11. package/dist/providers/transport.d.ts +10 -1
  12. package/dist/providers/transport.js +24 -8
  13. package/dist/run-ledger.d.ts +21 -0
  14. package/dist/run-ledger.js +115 -0
  15. package/dist/tools.js +2 -0
  16. package/docs/a2a.md +61 -42
  17. package/docs/agent-events.md +5 -4
  18. package/docs/agent-loops.md +2 -2
  19. package/docs/agent-session-runtime.md +1 -0
  20. package/docs/browser-automation.md +124 -0
  21. package/docs/coding-agent-tools.md +111 -14
  22. package/docs/coding-security.md +84 -11
  23. package/docs/credential-storage.md +9 -0
  24. package/docs/database-persistence.md +1 -1
  25. package/docs/evaluations.md +38 -4
  26. package/docs/guardrails.md +3 -2
  27. package/docs/host-security.md +29 -4
  28. package/docs/index.md +20 -15
  29. package/docs/mcp-tools.md +29 -5
  30. package/docs/migration.md +104 -0
  31. package/docs/observability.md +26 -14
  32. package/docs/performance.md +54 -0
  33. package/docs/postgres-persistence.md +1 -0
  34. package/docs/provider-conformance.md +1 -1
  35. package/docs/provider-primitives.md +7 -1
  36. package/docs/providers/kimi.md +16 -2
  37. package/docs/providers/opencode-go.md +43 -2
  38. package/docs/release-and-install.md +117 -62
  39. package/docs/resource-loading.md +4 -0
  40. package/docs/review-coverage-2026-07-19-phase-3.md +174 -0
  41. package/docs/review-coverage-2026-07-20-phase-4.md +175 -0
  42. package/docs/review-coverage-2026-07-21-phase-5.md +172 -0
  43. package/docs/run-ledger-conformance.md +1 -0
  44. package/docs/runs-and-usage.md +17 -2
  45. package/docs/sqlite-persistence.md +1 -0
  46. package/docs/structured-output.md +2 -2
  47. package/docs/supervisors.md +2 -2
  48. package/docs/tools.md +5 -1
  49. package/docs/web-tools.md +78 -0
  50. package/docs/workflows.md +2 -0
  51. package/package.json +6 -4
package/CHANGELOG.md CHANGED
@@ -1,10 +1,58 @@
1
1
  # Changelog
2
2
 
3
- All notable changes to this project will be documented in this file.
4
-
5
3
  The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
6
4
  and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
5
 
6
+ All notable changes to this project will be documented in this file.
7
+
8
+ ## [0.0.10] - 2026-07-21
9
+
10
+ ### Changed
11
+
12
+ - Coding harness workspace modes (Phase 5): required `workspaceMode` on `@arnilo/prism-coding-security` composition; sandbox mode unifies shell/FS on one disposable tree; host mode never claims containment; fail-closed mixed wiring + `allowMixedWorkspaceWiring` escape hatch; import/export tree identity; `scripts/benchmark-0.0.10.mjs` evidence.
13
+ - Versioned all 32 first-party manifests and exact internal ranges from the post-ship `0.0.96` graph to `0.0.10` for the roadmap Phase 5 release line.
14
+
15
+ ## [0.0.96] - 2026-07-21
16
+
17
+ ### Changed
18
+
19
+ - Package graph and runtime version pins bumped from 0.0.9 to 0.0.96 for a clean publish tag after the mistaken `v0.0.95` tag and TypeScript 7 / workspace-order CI fixes.
20
+
21
+ ## [0.0.9] - 2026-07-21
22
+
23
+ ### Added
24
+
25
+ - Production coding and browser execution for Release 0.0.9: disposable Docker sandbox, bounded native repository list/search, structured Git/named checks/PR handoff, durable coding-plan/checkpoint composition, and optional `@arnilo/prism-browser` with egress/side-effect/upload/download/screenshot policy.
26
+ - Versioned all 32 first-party manifests and exact internal ranges to 0.0.9 (adds `@arnilo/prism-browser` to the publishable graph; browser stays out of `@arnilo/prism-code` and activates only through explicit install or `@arnilo/prism-all`).
27
+ - Added network-free coding/browser adversarial evaluation fixtures, `scripts/benchmark-0.0.9.mjs`, and protected Docker/Playwright gates via `.github/workflows/sandbox-browser.yml`.
28
+ - Office execution remains outside Prism packaging by product decision (host-selected skills/instructions only).
29
+ - `tryParseJsonObjectArguments` and `toolCallFromArgumentsText` for recoverable streamed tool-call argument parsing.
30
+
31
+ ### Fixed
32
+
33
+ - Malformed streamed tool-call arguments (id+name present) become failed/`tool_execution_blocked` tool results (`invalid_arguments` / `invalid_json_arguments`) instead of terminal `ProviderTransportError`, so models can self-correct within existing turn budgets.
34
+ - Incomplete tool-call deltas (missing id/name) fail with typed `ProviderTransportError` / `ErrorInfo.code: "incomplete_delta"` instead of a bare `Error("Incomplete tool call delta...")`; openai-compatible streams no longer emit `done` alongside leftover incomplete deltas.
35
+ - Empty/whitespace-only call-free artifact candidates (including thinking-only output) are `parse_error` through the revision budget; `generate-validate-revise` session runs no longer resolve `succeeded` without `artifact_finished`.
36
+
37
+ ## [0.0.8] - 2026-07-20
38
+
39
+ ### Added
40
+
41
+ - Added OpenTelemetry GenAI agent/provider/tool hierarchy, context propagation, delegation/guardrail spans, bounded trace references, and evaluation linkage.
42
+ - Added bounded evaluation trace resolution, host model judges, deterministic pairwise reports, serialized artifacts, and CI threshold assertions.
43
+ - Added MCP resources/prompts/roots/sampling/elicitation plus principal-bound Streamable HTTP sessions on pinned SDK 1.29.0, and full A2A 1.0 durable task/rich-part/reconnect/push interoperability.
44
+ - Added immutable-revision CodeQL/dependency/SBOM/license/secret/attestation release gates, weekly dependency updates, and protected bounded provider/MCP/A2A/web live canaries.
45
+ - Added optional `@arnilo/prism-web-tools` with bounded host-selected Brave/Exa search, Firecrawl Markdown/schema extraction, stable citations, late credentials, and explicit untrusted-content results.
46
+ - Added optional `createBatchedRunLedger()` with bounded FIFO/backpressure, explicit durability/flush status, terminal acknowledgement, and documented buffered crash-loss semantics.
47
+ - Added one-leaf, one-second runtime session snapshot caching with mutation/checkout/resume invalidation and reproducible network-free 0.0.8 performance evidence.
48
+ - Versioned all 31 first-party manifests and exact internal ranges to 0.0.8; no tag or publication was created.
49
+
50
+ ### Fixed
51
+
52
+ - `generateValidateReviseLoop` routes artifact parse failures through the revision budget (`metadata.reason: "parse_error"`, repairer receives `value: undefined`) instead of returning silently after one provider turn.
53
+ - `@arnilo/prism-provider-opencode-go` Anthropic route sends provider-owned `x-api-key` and `anthropic-version: 2023-06-01` headers alongside Bearer, fixing HTTP 401 on MiniMax/Qwen models; `structuredOutput: "json_schema"` is no longer inferred from OpenAI routing alone (verified models only), fixing HTTP 400 on `deepseek-v4-pro`; both stream parsers require protocol completion evidence and fail truncated streams with a terminal `error` instead of a false `done`.
54
+ - `@arnilo/prism-provider-kimi` aligns with official contracts: featured Coding `k3` defaults `reasoning_effort: "high"`, 256K-class context windows use the exact `262_144`, the featured Moonshot catalog adds `kimi-k2.7-code-highspeed`/`kimi-k2.6`/`kimi-k2.5`, routing keys (`route`, `preserve_thinking`) no longer leak into wire bodies, the Coding route sends provider-owned `x-api-key`/`anthropic-version` headers, and both stream parsers fail truncated streams instead of emitting `done`.
55
+
8
56
  ## [0.0.7] - 2026-07-19
9
57
 
10
58
  ### Added
package/README.md CHANGED
@@ -53,6 +53,7 @@ npm install @arnilo/prism-sdk @arnilo/prism-provider-openai # application profi
53
53
  npm install @arnilo/prism-all # every first-party package
54
54
  npm install @arnilo/prism-server @arnilo/prism-workflows # optional Web API boundary
55
55
  npm install @arnilo/prism-supervisor # optional local delegation + A2A 1.0
56
+ npm install @arnilo/prism-web-tools # optional bounded Brave/Exa/Firecrawl research
56
57
  ```
57
58
 
58
59
  See [docs/release-and-install.md](docs/release-and-install.md) for install
@@ -157,6 +158,7 @@ printf '{"id":"1","command":"prompt","params":{"input":"Hi"}}\n' \
157
158
  | `@arnilo/prism-mcp` | MCP client/tool bridge |
158
159
  | `@arnilo/prism-workflows` | bounded DAG workflows, durable suspend/resume, schedules/background runs, composition/state/replay, and multi-process coordination |
159
160
  | `@arnilo/prism-supervisor` | bounded local child delegation and A2A 1.0 interoperability |
161
+ | `@arnilo/prism-web-tools` | host-selected bounded Brave/Exa search and Firecrawl Markdown/schema extraction |
160
162
  | `@arnilo/prism-observability-opentelemetry` | optional OpenTelemetry adapter |
161
163
  | `@arnilo/prism-credentials-node` | encrypted-file and keychain credentials |
162
164
  | `@arnilo/prism-session-store-sqlite` | SQLite persistence/checkpoints/leases/owned run feedback |
@@ -166,7 +168,7 @@ printf '{"id":"1","command":"prompt","params":{"input":"Hi"}}\n' \
166
168
  | `@arnilo/prism-base` | profile: core + compaction + JSON Schema validation |
167
169
  | `@arnilo/prism-code` | profile: base + coding tools/security + MCP |
168
170
  | `@arnilo/prism-sdk` | profile: base + workflows + MCP + credentials + OpenTelemetry |
169
- | `@arnilo/prism-all` | every first-party package, including both persistence adapters |
171
+ | `@arnilo/prism-all` | every first-party package, including both persistence adapters and web tools |
170
172
 
171
173
  ## Scripts
172
174
 
@@ -111,15 +111,21 @@ export function generateValidateReviseLoop(opts) {
111
111
  .filter((b) => b.type === "text")
112
112
  .map((b) => b.text)
113
113
  .join("");
114
- const parsed = opts.parser
115
- ? await opts.parser(text, artifactCtx)
116
- : { ok: true, value: text };
117
- // Parse failure ends the loop silently (terminal parse errors stay on `error`).
118
- if (!parsed.ok || parsed.value === undefined)
119
- return usage;
114
+ // Empty/whitespace-only call-free output is a parse failure (thinking-only
115
+ // models must not succeed with an empty artifact via the identity parser).
116
+ const parsed = text.trim() === ""
117
+ ? { ok: false, error: "no artifact text in model output" }
118
+ : opts.parser
119
+ ? await opts.parser(text, artifactCtx)
120
+ : { ok: true, value: text };
121
+ // Parse failure consumes revision budget like a validation failure; the
122
+ // repairer receives `undefined` value plus a synthetic parse issue.
123
+ const parseFailure = !parsed.ok || parsed.value === undefined
124
+ ? { ok: false, errors: [{ path: "$", message: parsed.error ?? "artifact parse failed" }], metadata: { reason: "parse_error" } }
125
+ : undefined;
120
126
  const attempt = ++attempts;
121
127
  ctx.emit({ type: "artifact_validation_started", sessionId: ctx.sessionId, runId: ctx.runId, turn, attempt });
122
- const result = await opts.validator(parsed.value, artifactCtx);
128
+ const result = parseFailure ?? await opts.validator(parsed.value, artifactCtx);
123
129
  ctx.emit({ type: "artifact_validation_finished", sessionId: ctx.sessionId, runId: ctx.runId, turn, attempt, result });
124
130
  if (result.ok) {
125
131
  ctx.emit({ type: "artifact_finished", sessionId: ctx.sessionId, runId: ctx.runId, turn, attempt, result });
@@ -130,7 +136,7 @@ export function generateValidateReviseLoop(opts) {
130
136
  return usage;
131
137
  }
132
138
  ctx.emit({ type: "artifact_revision_started", sessionId: ctx.sessionId, runId: ctx.runId, turn, attempt, failure: result });
133
- const repair = await repairer(parsed.value, result, artifactCtx);
139
+ const repair = await repairer(parseFailure ? undefined : parsed.value, result, artifactCtx);
134
140
  const repairMessages = inputMessages(repair).map((m) => ({ ...m, id: randomId("msg") }));
135
141
  for (const message of repairMessages)
136
142
  await ctx.appendMessage(message);
package/dist/agents.js CHANGED
@@ -12,6 +12,7 @@ import { errorToErrorInfo, redactAgentEvent, redactProviderRequest, redactRunLed
12
12
  import { composeSystemPrompt, mergeSystemPromptConfig } from "./system-prompts.js";
13
13
  import { createDefaultRetryPolicy, waitForRetry } from "./retry.js";
14
14
  import { createMemorySessionStore, createSessionEntry, getSessionBranchEntries, rebuildSessionContext } from "./session-stores.js";
15
+ import { isFlushableRunLedger } from "./run-ledger.js";
15
16
  import { createToolRegistry, dispatchToolCall } from "./tools.js";
16
17
  import { RunLimitError, RunLimitTracker, resolveRunLimits } from "./run-limits.js";
17
18
  import { agentFingerprint, initialAgentRunState, loadAgentRunState, publicState, saveAgentRunState, validateRunStateOptions } from "./agent-run-state.js";
@@ -103,6 +104,8 @@ class RuntimeAgentSession {
103
104
  activeLoopTurn = 1;
104
105
  ledgerChain = Promise.resolve();
105
106
  ledgerFailure;
107
+ snapshotGeneration = 0;
108
+ snapshotCache;
106
109
  constructor(config) {
107
110
  this.id = config.id ?? randomId("session");
108
111
  this.agent = config.agent;
@@ -168,6 +171,8 @@ class RuntimeAgentSession {
168
171
  this.activeIdempotencyKey = options.idempotencyKey ?? this.agent.config.idempotencyKey;
169
172
  this.activeGuardrails = mergeGuardrails(this.agent.config.guardrails, options.guardrails);
170
173
  this.activeDurable = resumed ?? (durableOptions ? { options: durableOptions, version: 0 } : undefined);
174
+ if (resumed)
175
+ this.invalidateSnapshot();
171
176
  const model = options.model ?? this.agent.config.model;
172
177
  const startedAt = new Date().toISOString();
173
178
  let runError;
@@ -269,6 +274,8 @@ class RuntimeAgentSession {
269
274
  };
270
275
  // ponytail: LoopContext binds existing private helpers; loop orchestrates only.
271
276
  let assembledTurn = false;
277
+ let artifactFinished = false;
278
+ let artifactFailedInfo;
272
279
  const ctx = {
273
280
  sessionId: this.id,
274
281
  runId,
@@ -363,6 +370,16 @@ class RuntimeAgentSession {
363
370
  emit: (event) => {
364
371
  if (event.type === "turn_started")
365
372
  this.activeLoopTurn = event.turn;
373
+ if (event.type === "artifact_finished")
374
+ artifactFinished = true;
375
+ if (event.type === "artifact_failed") {
376
+ const first = event.result.errors?.[0];
377
+ const reason = event.result.metadata?.reason;
378
+ artifactFailedInfo = {
379
+ message: first?.message ?? "artifact failed",
380
+ code: typeof reason === "string" || typeof reason === "number" ? reason : "artifact_failed",
381
+ };
382
+ }
366
383
  this.emit(event);
367
384
  },
368
385
  };
@@ -375,6 +392,9 @@ class RuntimeAgentSession {
375
392
  });
376
393
  }
377
394
  const loopUsage = await loop.run(ctx);
395
+ if (loop.name === "generate-validate-revise" && !artifactFinished) {
396
+ throw Object.assign(new Error(artifactFailedInfo?.message ?? "artifact loop ended without a validated artifact"), { name: "ArtifactFailed", code: artifactFailedInfo?.code ?? "artifact_failed" });
397
+ }
378
398
  usage = runUsage.value() ?? loopUsage;
379
399
  if (usage && this.activeLedger) {
380
400
  const usageRecord = {
@@ -442,6 +462,8 @@ class RuntimeAgentSession {
442
462
  ...this.activeOwnership,
443
463
  };
444
464
  await this.activeLedger.appendRun(redactRunLedgerRecord(finishRecord, this.activeRedactor));
465
+ if (isFlushableRunLedger(this.activeLedger) && this.activeLedger.durability === "flush_on_terminal")
466
+ await this.activeLedger.flush();
445
467
  }
446
468
  }
447
469
  finally {
@@ -570,6 +592,7 @@ class RuntimeAgentSession {
570
592
  }
571
593
  async checkout(leafId) {
572
594
  this.currentLeafId = leafId;
595
+ this.invalidateSnapshot();
573
596
  await this.rebuildHistory();
574
597
  }
575
598
  fork(options = {}) {
@@ -853,6 +876,11 @@ class RuntimeAgentSession {
853
876
  idempotencyKey: this.activeIdempotencyKey,
854
877
  });
855
878
  this.currentLeafId = redacted.id;
879
+ this.invalidateSnapshot();
880
+ }
881
+ invalidateSnapshot() {
882
+ this.snapshotGeneration += 1;
883
+ this.snapshotCache = undefined;
856
884
  }
857
885
  redact(value) {
858
886
  return this.activeRedactor?.redact(value) ?? value;
@@ -864,10 +892,16 @@ class RuntimeAgentSession {
864
892
  this.history = (await this.snapshot()).messages.slice();
865
893
  }
866
894
  async snapshot() {
895
+ const now = performance.now();
896
+ const cached = this.snapshotCache;
897
+ if (cached && cached.leafId === this.currentLeafId && cached.generation === this.snapshotGeneration && cached.expiresAt > now)
898
+ return cached.value;
867
899
  const reader = this.branchReader();
868
- return reader
869
- ? rebuildSessionContext(reader, { sessionId: this.id, leafId: this.currentLeafId })
900
+ const value = reader
901
+ ? await rebuildSessionContext(reader, { sessionId: this.id, leafId: this.currentLeafId })
870
902
  : rebuildSessionContext(await this.store.list(this.id), { leafId: this.currentLeafId });
903
+ this.snapshotCache = { leafId: this.currentLeafId, generation: this.snapshotGeneration, expiresAt: now + 1_000, value };
904
+ return value;
871
905
  }
872
906
  }
873
907
  class EventSubscriber {
@@ -980,7 +1014,7 @@ function policyList(policies) {
980
1014
  return "apply" in policies ? [policies] : policies;
981
1015
  }
982
1016
  function errorFromInfo(error) {
983
- return Object.assign(new Error(error.message), { name: error.name ?? "Error", cause: error.cause });
1017
+ return Object.assign(new Error(error.message), { name: error.name ?? "Error", cause: error.cause, code: error.code });
984
1018
  }
985
1019
  class ProviderTurnFailure extends Error {
986
1020
  info;
@@ -49,6 +49,8 @@ export interface ToolCallContent {
49
49
  readonly id: string;
50
50
  readonly name: string;
51
51
  readonly arguments: JsonObject;
52
+ /** Set when streamed arguments failed JSON parse; dispatch blocks without execute(). */
53
+ readonly argumentsError?: ErrorInfo;
52
54
  }
53
55
  export interface ToolResultContent {
54
56
  readonly type: "tool_result";
@@ -1278,6 +1280,21 @@ export interface RunLedger {
1278
1280
  }
1279
1281
  /** Union of records that may be handed to a {@link RunLedger}. */
1280
1282
  export type RunLedgerRecord = RunRecord | AgentEventRecord | ToolCallRecord | UsageRecord;
1283
+ export type RunLedgerDurability = "write_through" | "flush_on_terminal" | "buffered";
1284
+ export interface RunLedgerFlushResult {
1285
+ readonly accepted: number;
1286
+ readonly flushed: number;
1287
+ readonly buffered: number;
1288
+ }
1289
+ /** Optional durability seam implemented by bounded ledger adapters. */
1290
+ export interface FlushableRunLedger extends RunLedger {
1291
+ readonly durability: RunLedgerDurability;
1292
+ flush(): Promise<RunLedgerFlushResult>;
1293
+ status(): RunLedgerFlushResult;
1294
+ dispose(options?: {
1295
+ readonly flush?: boolean;
1296
+ }): Promise<void>;
1297
+ }
1281
1298
  /** Immutable human feedback linked to an existing owned run/trace and optional evaluations. */
1282
1299
  export interface RunFeedbackRecord extends OwnershipScope {
1283
1300
  readonly id: string;
package/dist/index.d.ts CHANGED
@@ -2,6 +2,8 @@ export type * from "./contracts.js";
2
2
  export type { RunLimitCounters, RunLimitName, SecureAgentOptions } from "./contracts.js";
3
3
  export { isSessionEntryKind, SESSION_APPEND_CONFLICT_CODE, SESSION_ENTRY_KINDS, SESSION_ENTRY_SCHEMA_VERSION, SessionAppendConflictError, isSessionAppendConflict, AgentRunError, AgentRunStateError } from "./contracts.js";
4
4
  export { createAgent, createAgentSession, resumeAgentRun } from "./agents.js";
5
+ export { createBatchedRunLedger, isFlushableRunLedger, DEFAULT_LEDGER_BATCH_ENTRIES, HARD_LEDGER_BATCH_ENTRIES, DEFAULT_LEDGER_BATCH_BYTES, HARD_LEDGER_BATCH_BYTES, DEFAULT_LEDGER_BATCH_DELAY_MS, HARD_LEDGER_BATCH_DELAY_MS, } from "./run-ledger.js";
6
+ export type { BatchedRunLedgerOptions } from "./run-ledger.js";
5
7
  export { createSecureAgent } from "./secure-agent.js";
6
8
  export { createMemoryRunFeedbackStore, prepareRunFeedback, requireRunFeedbackOwnership, runFeedbackPageLimit, RunFeedbackError, } from "./feedback.js";
7
9
  export type { MemoryRunFeedbackStoreOptions, PrepareRunFeedbackOptions, RunFeedbackLimits, RunFeedbackRun, RunFeedbackRunResolver, } from "./feedback.js";
@@ -58,7 +60,7 @@ export { createMockProvider } from "./mock-provider.js";
58
60
  export { createMemorySessionStore, createSessionEntry, getSessionBranchEntries, listSessionBranches, rebuildSessionContext } from "./session-stores.js";
59
61
  export type { CreateSessionEntryOptions, SessionBranch, SessionBranchOptions, SessionContextSnapshot } from "./session-stores.js";
60
62
  export type { MockProviderOptions } from "./mock-provider.js";
61
- export { providerContentDelta, providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, } from "./provider-events.js";
63
+ export { providerContentDelta, providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, toolCallFromArgumentsText, } from "./provider-events.js";
62
64
  export type { ProviderResolver } from "./contracts.js";
63
65
  export { createProviderRegistry, createProviderResolver } from "./providers.js";
64
66
  export type { ProviderRegistry, ProviderRegistryOptions } from "./providers.js";
@@ -81,5 +83,5 @@ export type { DispatchToolCallOptions, ToolArgumentValidationError, ToolArgument
81
83
  export type { DuplicateRegistrationOptions, DuplicateRegistrationPolicy } from "./registry-options.js";
82
84
  export { dispatchToolCallsInOrder, generateValidateReviseLoop, isAgentLoopOptions, resolveLoop, resolveToolConcurrency, singleShotLoop } from "./agent-loops.js";
83
85
  export declare const name = "prism";
84
- export declare const version = "0.0.7";
86
+ export declare const version = "0.0.10";
85
87
  export declare const description = "Agent harness for AI providers, agents, sessions, and tools.";
package/dist/index.js CHANGED
@@ -1,5 +1,6 @@
1
1
  export { isSessionEntryKind, SESSION_APPEND_CONFLICT_CODE, SESSION_ENTRY_KINDS, SESSION_ENTRY_SCHEMA_VERSION, SessionAppendConflictError, isSessionAppendConflict, AgentRunError, AgentRunStateError } from "./contracts.js";
2
2
  export { createAgent, createAgentSession, resumeAgentRun } from "./agents.js";
3
+ export { createBatchedRunLedger, isFlushableRunLedger, DEFAULT_LEDGER_BATCH_ENTRIES, HARD_LEDGER_BATCH_ENTRIES, DEFAULT_LEDGER_BATCH_BYTES, HARD_LEDGER_BATCH_BYTES, DEFAULT_LEDGER_BATCH_DELAY_MS, HARD_LEDGER_BATCH_DELAY_MS, } from "./run-ledger.js";
3
4
  export { createSecureAgent } from "./secure-agent.js";
4
5
  export { createMemoryRunFeedbackStore, prepareRunFeedback, requireRunFeedbackOwnership, runFeedbackPageLimit, RunFeedbackError, } from "./feedback.js";
5
6
  export { CHECKPOINT_CONFLICT_CODE, CheckpointConflictError, createMemoryCheckpointStore } from "./checkpoints.js";
@@ -32,7 +33,7 @@ export { createMiddlewareRegistry } from "./middleware.js";
32
33
  export { assembleProviderInput, createDefaultInputBuilder, createDefaultPromptBuilder, renderPromptTemplate, resolveContextProviders } from "./input.js";
33
34
  export { createMockProvider } from "./mock-provider.js";
34
35
  export { createMemorySessionStore, createSessionEntry, getSessionBranchEntries, listSessionBranches, rebuildSessionContext } from "./session-stores.js";
35
- export { providerContentDelta, providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, } from "./provider-events.js";
36
+ export { providerContentDelta, providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, toolCallFromArgumentsText, } from "./provider-events.js";
36
37
  export { createProviderRegistry, createProviderResolver } from "./providers.js";
37
38
  export { createSecretRedactor, errorToErrorInfo, redactAgentEvent, redactMessage, redactProviderRequest, redactRunLedgerRecord, redactSecrets, redactSessionEntry } from "./redaction.js";
38
39
  export { assertPermission, assertTrusted, checkPermission, createStaticPermissionPolicy, createStaticTrustPolicy, denialToErrorInfo, isTrusted, PermissionDeniedError, TrustDeniedError } from "./security.js";
@@ -44,6 +45,6 @@ export { assertGuardrailsAllowed, GuardrailError, MAX_GUARDRAIL_CONCURRENCY, run
44
45
  export { createRunLimitTracker, DEFAULT_RUN_LIMITS, HARD_MAX_RUN_COST, HARD_RUN_LIMITS, RunLimitError, RunLimitTracker, resolveRunLimits } from "./run-limits.js";
45
46
  export { dispatchToolCallsInOrder, generateValidateReviseLoop, isAgentLoopOptions, resolveLoop, resolveToolConcurrency, singleShotLoop } from "./agent-loops.js";
46
47
  export const name = "prism";
47
- export const version = "0.0.7";
48
+ export const version = "0.0.10";
48
49
  export const description = "Agent harness for AI providers, agents, sessions, and tools.";
49
50
  //# sourceMappingURL=index.js.map
@@ -15,3 +15,5 @@ export declare function providerUsage(usage: Usage): ProviderEvent;
15
15
  export declare function providerDone(usage?: Usage): ProviderEvent;
16
16
  export declare function providerError(error: unknown, secrets?: readonly (string | undefined)[]): ProviderEvent;
17
17
  export declare function toolCallContent(id: string, name: string, args?: JsonObject): ToolCallContent;
18
+ /** Build a tool call from streamed arguments text; malformed JSON becomes a blocked call (no throw). */
19
+ export declare function toolCallFromArgumentsText(id: string, name: string, argumentsText: string): ToolCallContent;
@@ -1,3 +1,4 @@
1
+ import { ProviderTransportError, tryParseJsonObjectArguments } from "./providers/transport.js";
1
2
  import { errorToErrorInfo } from "./redaction.js";
2
3
  export function providerTextDelta(text) {
3
4
  return { type: "content_delta", content: { type: "text", text } };
@@ -32,9 +33,10 @@ export function reconstructToolCallDeltas(events) {
32
33
  partials.set(event.index, partial);
33
34
  }
34
35
  return [...partials.entries()].sort(([a], [b]) => a - b).map(([index, partial]) => {
35
- if (!partial.id || !partial.name)
36
- throw new Error(`Incomplete tool call delta at index ${index}`);
37
- return toolCallContent(partial.id, partial.name, parseToolCallArguments(partial.argumentsText, index));
36
+ if (!partial.id || !partial.name) {
37
+ throw new ProviderTransportError("incomplete_delta", `Incomplete tool call delta at index ${index}`);
38
+ }
39
+ return toolCallFromArgumentsText(partial.id, partial.name, partial.argumentsText);
38
40
  });
39
41
  }
40
42
  export function providerUsage(usage) {
@@ -50,15 +52,21 @@ export function providerError(error, secrets = []) {
50
52
  export function toolCallContent(id, name, args = {}) {
51
53
  return { type: "tool_call", id, name, arguments: args };
52
54
  }
53
- function parseToolCallArguments(text, index) {
54
- try {
55
- const value = text ? JSON.parse(text) : {};
56
- if (!value || typeof value !== "object" || Array.isArray(value))
57
- throw new Error("not object");
58
- return value;
59
- }
60
- catch (error) {
61
- throw new Error(`Invalid tool call arguments at index ${index}: ${error instanceof Error ? error.message : String(error)}`);
62
- }
55
+ /** Build a tool call from streamed arguments text; malformed JSON becomes a blocked call (no throw). */
56
+ export function toolCallFromArgumentsText(id, name, argumentsText) {
57
+ const parsed = tryParseJsonObjectArguments(argumentsText, { toolName: name });
58
+ if (parsed.ok)
59
+ return toolCallContent(id, name, parsed.value);
60
+ return {
61
+ type: "tool_call",
62
+ id,
63
+ name,
64
+ arguments: {},
65
+ argumentsError: {
66
+ name: parsed.error.name,
67
+ message: parsed.error.message,
68
+ code: parsed.error.code,
69
+ },
70
+ };
63
71
  }
64
72
  //# sourceMappingURL=provider-events.js.map
@@ -1,7 +1,7 @@
1
1
  import { resolveCredentialValue } from "../credentials.js";
2
- import { providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallContent, } from "../provider-events.js";
2
+ import { providerDone, providerError, providerTextDelta, providerThinkingDelta, providerToolCall, providerToolCallDelta, providerUsage, toolCallFromArgumentsText, } from "../provider-events.js";
3
3
  import { assertOpenAIChatMessage, applyOpenAIChatStructuredOutput, mapOpenAIChatUsage, serializeOpenAIChatMessage, serializeOpenAITool, } from "./openai-primitives.js";
4
- import { parseJsonObjectArguments, readBoundedResponseText, readSseData, } from "./transport.js";
4
+ import { ProviderTransportError, readBoundedResponseText, readSseData, } from "./transport.js";
5
5
  import { assertStructuredOutputRequestSupported } from "../structured-output.js";
6
6
  export function createOpenAICompatibleProvider(options) {
7
7
  const providerId = options.id ?? "openai-compatible";
@@ -64,10 +64,13 @@ export function createOpenAICompatibleProvider(options) {
64
64
  }
65
65
  }
66
66
  }
67
+ const incomplete = [...tools.entries()].find(([, call]) => !call.id || !call.name);
68
+ if (incomplete) {
69
+ yield providerError(new ProviderTransportError("incomplete_delta", `Incomplete tool call delta at index ${incomplete[0]}`), secrets);
70
+ return;
71
+ }
67
72
  for (const call of tools.values()) {
68
- if (call.id && call.name) {
69
- yield providerToolCall(toolCallContent(call.id, call.name, parseJsonObjectArguments(call.argumentsText, { toolName: call.name })));
70
- }
73
+ yield providerToolCall(toolCallFromArgumentsText(call.id, call.name, call.argumentsText));
71
74
  }
72
75
  yield providerDone();
73
76
  }
@@ -14,7 +14,7 @@ export interface SseEvent {
14
14
  readonly data: string;
15
15
  readonly comments?: readonly string[];
16
16
  }
17
- export type ProviderTransportErrorCode = "sse_buffer_overflow" | "sse_event_overflow" | "response_body_overflow" | "aborted" | "invalid_json_arguments";
17
+ export type ProviderTransportErrorCode = "sse_buffer_overflow" | "sse_event_overflow" | "response_body_overflow" | "aborted" | "invalid_json_arguments" | "incomplete_delta";
18
18
  export declare class ProviderTransportError extends Error {
19
19
  readonly code: ProviderTransportErrorCode;
20
20
  readonly limitBytes?: number;
@@ -36,5 +36,14 @@ export declare function readSseEvents(body: ReadableStream<Uint8Array>, options?
36
36
  export declare function readSseData(body: ReadableStream<Uint8Array>, options?: ReadSseEventsOptions): AsyncGenerator<string>;
37
37
  /** Read a response body with a hard byte ceiling; redacts optional secrets. */
38
38
  export declare function readBoundedResponseText(response: Response, options?: ReadBoundedResponseTextOptions): Promise<string>;
39
+ export type ParseJsonObjectArgumentsResult = {
40
+ readonly ok: true;
41
+ readonly value: JsonObject;
42
+ } | {
43
+ readonly ok: false;
44
+ readonly error: ProviderTransportError;
45
+ };
46
+ /** Parse streamed tool arguments without throwing; prefer this for recoverable tool-call recovery. */
47
+ export declare function tryParseJsonObjectArguments(text: string, options?: ParseJsonObjectArgumentsOptions): ParseJsonObjectArgumentsResult;
39
48
  /** Parse streamed tool arguments as a JSON object; throws {@link ProviderTransportError} on invalid input. */
40
49
  export declare function parseJsonObjectArguments(text: string, options?: ParseJsonObjectArgumentsOptions): JsonObject;
@@ -197,25 +197,41 @@ export async function readBoundedResponseText(response, options) {
197
197
  }
198
198
  }
199
199
  }
200
- /** Parse streamed tool arguments as a JSON object; throws {@link ProviderTransportError} on invalid input. */
201
- export function parseJsonObjectArguments(text, options) {
200
+ /** Parse streamed tool arguments without throwing; prefer this for recoverable tool-call recovery. */
201
+ export function tryParseJsonObjectArguments(text, options) {
202
202
  const maxBytes = options?.maxBytes ?? DEFAULT_MAX_ARGUMENT_BYTES;
203
203
  const suffix = options?.toolName ? ` for tool ${options.toolName}` : "";
204
204
  if (!text)
205
- return {};
205
+ return { ok: true, value: {} };
206
206
  if (byteLength(text) > maxBytes) {
207
- throw new ProviderTransportError("invalid_json_arguments", `Tool arguments${suffix} exceeded ${maxBytes} bytes`, maxBytes);
207
+ return {
208
+ ok: false,
209
+ error: new ProviderTransportError("invalid_json_arguments", `Tool arguments${suffix} exceeded ${maxBytes} bytes`, maxBytes),
210
+ };
208
211
  }
209
212
  let parsed;
210
213
  try {
211
214
  parsed = JSON.parse(text);
212
215
  }
213
- catch (error) {
214
- throw new ProviderTransportError("invalid_json_arguments", `Invalid tool arguments JSON${suffix}`);
216
+ catch {
217
+ return {
218
+ ok: false,
219
+ error: new ProviderTransportError("invalid_json_arguments", `Invalid tool arguments JSON${suffix}`),
220
+ };
215
221
  }
216
222
  if (!parsed || typeof parsed !== "object" || Array.isArray(parsed)) {
217
- throw new ProviderTransportError("invalid_json_arguments", `Tool arguments${suffix} must be a JSON object`);
223
+ return {
224
+ ok: false,
225
+ error: new ProviderTransportError("invalid_json_arguments", `Tool arguments${suffix} must be a JSON object`),
226
+ };
218
227
  }
219
- return parsed;
228
+ return { ok: true, value: parsed };
229
+ }
230
+ /** Parse streamed tool arguments as a JSON object; throws {@link ProviderTransportError} on invalid input. */
231
+ export function parseJsonObjectArguments(text, options) {
232
+ const result = tryParseJsonObjectArguments(text, options);
233
+ if (!result.ok)
234
+ throw result.error;
235
+ return result.value;
220
236
  }
221
237
  //# sourceMappingURL=transport.js.map
@@ -0,0 +1,21 @@
1
+ import type { FlushableRunLedger, RunLedger, RunLedgerDurability } from "./contracts.js";
2
+ export declare const DEFAULT_LEDGER_BATCH_ENTRIES = 128;
3
+ export declare const HARD_LEDGER_BATCH_ENTRIES = 4096;
4
+ export declare const DEFAULT_LEDGER_BATCH_BYTES: number;
5
+ export declare const HARD_LEDGER_BATCH_BYTES: number;
6
+ export declare const DEFAULT_LEDGER_BATCH_DELAY_MS = 25;
7
+ export declare const HARD_LEDGER_BATCH_DELAY_MS = 60000;
8
+ export interface BatchedRunLedgerOptions {
9
+ readonly maxBatchEntries?: number;
10
+ readonly maxBatchBytes?: number;
11
+ readonly maxBufferedEntries?: number;
12
+ readonly maxBufferedBytes?: number;
13
+ readonly maxDelayMs?: number;
14
+ readonly durability?: RunLedgerDurability;
15
+ }
16
+ /**
17
+ * Wrap any RunLedger with one bounded FIFO. Inputs must already be redacted, as required by RunLedger.
18
+ * `buffered` may lose accepted records on process crash; call `flush()` for acknowledgement.
19
+ */
20
+ export declare function createBatchedRunLedger(target: RunLedger, options?: BatchedRunLedgerOptions): FlushableRunLedger;
21
+ export declare function isFlushableRunLedger(ledger: RunLedger): ledger is FlushableRunLedger;
@@ -0,0 +1,115 @@
1
+ export const DEFAULT_LEDGER_BATCH_ENTRIES = 128;
2
+ export const HARD_LEDGER_BATCH_ENTRIES = 4096;
3
+ export const DEFAULT_LEDGER_BATCH_BYTES = 512 * 1024;
4
+ export const HARD_LEDGER_BATCH_BYTES = 8 * 1024 * 1024;
5
+ export const DEFAULT_LEDGER_BATCH_DELAY_MS = 25;
6
+ export const HARD_LEDGER_BATCH_DELAY_MS = 60_000;
7
+ function integer(value, fallback, hard, name) {
8
+ const selected = value ?? fallback;
9
+ if (!Number.isInteger(selected) || selected < 1 || selected > hard)
10
+ throw new RangeError(`${name} must be an integer in [1, ${hard}]`);
11
+ return selected;
12
+ }
13
+ function terminal(record) {
14
+ return record.status !== undefined && record.status !== "queued" && record.status !== "running";
15
+ }
16
+ /**
17
+ * Wrap any RunLedger with one bounded FIFO. Inputs must already be redacted, as required by RunLedger.
18
+ * `buffered` may lose accepted records on process crash; call `flush()` for acknowledgement.
19
+ */
20
+ export function createBatchedRunLedger(target, options = {}) {
21
+ const maxBatchEntries = integer(options.maxBatchEntries, DEFAULT_LEDGER_BATCH_ENTRIES, HARD_LEDGER_BATCH_ENTRIES, "maxBatchEntries");
22
+ const maxBatchBytes = integer(options.maxBatchBytes, DEFAULT_LEDGER_BATCH_BYTES, HARD_LEDGER_BATCH_BYTES, "maxBatchBytes");
23
+ const maxBufferedEntries = integer(options.maxBufferedEntries, Math.min(HARD_LEDGER_BATCH_ENTRIES, maxBatchEntries * 2), HARD_LEDGER_BATCH_ENTRIES, "maxBufferedEntries");
24
+ const maxBufferedBytes = integer(options.maxBufferedBytes, Math.min(HARD_LEDGER_BATCH_BYTES, maxBatchBytes * 2), HARD_LEDGER_BATCH_BYTES, "maxBufferedBytes");
25
+ const maxDelayMs = integer(options.maxDelayMs, DEFAULT_LEDGER_BATCH_DELAY_MS, HARD_LEDGER_BATCH_DELAY_MS, "maxDelayMs");
26
+ const durability = options.durability ?? "flush_on_terminal";
27
+ const queue = [];
28
+ let bufferedBytes = 0;
29
+ let accepted = 0;
30
+ let flushed = 0;
31
+ let timer;
32
+ let flushChain = Promise.resolve();
33
+ let disposed = false;
34
+ const status = () => ({ accepted, flushed, buffered: queue.length });
35
+ const cancelTimer = () => { if (timer)
36
+ clearTimeout(timer); timer = undefined; };
37
+ const schedule = () => {
38
+ if (timer || disposed || queue.length === 0)
39
+ return;
40
+ timer = setTimeout(() => { timer = undefined; void flush().catch(() => undefined); }, maxDelayMs);
41
+ timer.unref?.();
42
+ };
43
+ const write = (item) => {
44
+ if (item.kind === "run")
45
+ return target.appendRun(item.record);
46
+ if (item.kind === "event")
47
+ return target.appendEvent(item.record);
48
+ if (item.kind === "tool")
49
+ return target.appendToolCall(item.record);
50
+ return target.appendUsage(item.record);
51
+ };
52
+ const flush = () => {
53
+ cancelTimer();
54
+ const operation = flushChain.then(async () => {
55
+ let entries = 0;
56
+ let bytes = 0;
57
+ while (queue.length) {
58
+ const item = queue[0];
59
+ if (entries && (entries >= maxBatchEntries || bytes + item.bytes > maxBatchBytes)) {
60
+ entries = 0;
61
+ bytes = 0;
62
+ }
63
+ await write(item);
64
+ queue.shift();
65
+ bufferedBytes -= item.bytes;
66
+ flushed += 1;
67
+ entries += 1;
68
+ bytes += item.bytes;
69
+ }
70
+ return status();
71
+ });
72
+ flushChain = operation.then(() => undefined, () => undefined);
73
+ return operation;
74
+ };
75
+ const enqueue = async (item) => {
76
+ if (disposed)
77
+ throw new Error("batched run ledger is disposed");
78
+ const bytes = Buffer.byteLength(JSON.stringify(item.record));
79
+ if (bytes > maxBatchBytes || bytes > maxBufferedBytes)
80
+ throw new RangeError("run ledger record exceeds byte limit");
81
+ if (queue.length >= maxBufferedEntries || bufferedBytes + bytes > maxBufferedBytes)
82
+ await flush();
83
+ queue.push({ ...item, bytes });
84
+ bufferedBytes += bytes;
85
+ accepted += 1;
86
+ if (durability === "write_through" || queue.length >= maxBatchEntries || bufferedBytes >= maxBatchBytes || (item.kind === "run" && terminal(item.record) && durability === "flush_on_terminal"))
87
+ await flush();
88
+ else
89
+ schedule();
90
+ };
91
+ return {
92
+ durability,
93
+ appendRun: (record) => enqueue({ kind: "run", record }),
94
+ appendEvent: (record) => enqueue({ kind: "event", record }),
95
+ appendToolCall: (record) => enqueue({ kind: "tool", record }),
96
+ appendUsage: (record) => enqueue({ kind: "usage", record }),
97
+ flush,
98
+ status,
99
+ async dispose(disposeOptions = {}) {
100
+ disposed = true;
101
+ cancelTimer();
102
+ if (disposeOptions.flush !== false)
103
+ await flush();
104
+ else {
105
+ await flushChain;
106
+ queue.length = 0;
107
+ bufferedBytes = 0;
108
+ }
109
+ },
110
+ };
111
+ }
112
+ export function isFlushableRunLedger(ledger) {
113
+ return "flush" in ledger && typeof ledger.flush === "function";
114
+ }
115
+ //# sourceMappingURL=run-ledger.js.map
package/dist/tools.js CHANGED
@@ -176,6 +176,8 @@ async function checkCall(call, options, startedAt) {
176
176
  return blocked(call, context, "unknown_tool", { message: `Unknown tool: ${call.name}` }, options, startedAt);
177
177
  if (filterTools([tool], options.filter).length === 0)
178
178
  return blocked(call, context, "tool_denied", { message: `Tool denied: ${call.name}` }, options, startedAt);
179
+ if (call.argumentsError)
180
+ return blocked(call, context, "invalid_arguments", call.argumentsError, options, startedAt);
179
181
  if (!isJsonObject(call.arguments))
180
182
  return blocked(call, context, "invalid_arguments", { message: "Tool arguments must be a JSON object" }, options, startedAt);
181
183
  return undefined;