@agentium/harness 4.1.0 → 4.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/CONVERSATIONAL-RUNS.md +173 -0
  2. package/README.md +13 -5
  3. package/dist/definition.d.cts +39 -0
  4. package/dist/{driver-Cr2BGG4h.cjs → driver-Bh_wLWCP.cjs} +285 -15
  5. package/dist/{driver-BvzvqvpB.js → driver-reTLBJ_o.js} +280 -16
  6. package/dist/drivers.d.cts +14 -0
  7. package/dist/drivers.d.ts.map +1 -1
  8. package/dist/index.cjs +103 -46
  9. package/dist/index.d.cts +19 -0
  10. package/dist/index.d.ts +4 -1
  11. package/dist/index.d.ts.map +1 -1
  12. package/dist/index.js +103 -48
  13. package/dist/input-tool.d.cts +7 -0
  14. package/dist/input-tool.d.ts +7 -0
  15. package/dist/input-tool.d.ts.map +1 -0
  16. package/dist/mcp-resources.d.cts +42 -0
  17. package/dist/policies.d.cts +31 -0
  18. package/dist/policies.d.ts +2 -0
  19. package/dist/policies.d.ts.map +1 -1
  20. package/dist/presets.d.cts +41 -0
  21. package/dist/runtime/context-policy.d.cts +21 -0
  22. package/dist/runtime/context-policy.d.ts +1 -0
  23. package/dist/runtime/context-policy.d.ts.map +1 -1
  24. package/dist/runtime/context.d.cts +15 -0
  25. package/dist/runtime/controller.d.cts +51 -0
  26. package/dist/runtime/driver.d.cts +157 -0
  27. package/dist/runtime/driver.d.ts +14 -3
  28. package/dist/runtime/driver.d.ts.map +1 -1
  29. package/dist/runtime/events.d.cts +116 -0
  30. package/dist/runtime/events.d.ts +35 -2
  31. package/dist/runtime/events.d.ts.map +1 -1
  32. package/dist/runtime/index.d.cts +9 -0
  33. package/dist/runtime/input.d.cts +46 -0
  34. package/dist/runtime/input.d.ts +46 -0
  35. package/dist/runtime/input.d.ts.map +1 -0
  36. package/dist/runtime/middleware.d.cts +10 -0
  37. package/dist/runtime/resolve.d.cts +20 -0
  38. package/dist/runtime/runtime-registry.d.cts +35 -0
  39. package/dist/runtime/session-bindings.d.cts +60 -0
  40. package/dist/runtime/types.d.cts +176 -0
  41. package/dist/testing.cjs +1 -1
  42. package/dist/testing.d.cts +7 -0
  43. package/dist/testing.js +1 -1
  44. package/dist/watch/definition.d.cts +11 -0
  45. package/dist/watch/gmail.d.cts +104 -0
  46. package/dist/watch/quiet-hours.d.cts +10 -0
  47. package/dist/watch/runtime.d.cts +42 -0
  48. package/dist/watch/types.d.cts +147 -0
  49. package/package.json +21 -10
@@ -0,0 +1,173 @@
1
+ # Conversational runs (4.6)
2
+
3
+ Questions, steering, public messages and compaction stay inside one live run.
4
+ The runtime retains the session lease, Agent instance, transcript, provider
5
+ continuations and remaining budgets. Waiting does not poll the model.
6
+
7
+ ## Ask and resume
8
+
9
+ ```ts
10
+ import { openai } from "@agentium/core";
11
+ import { agentDriver, HarnessRuntime, requestInputTool } from "@agentium/harness";
12
+
13
+ const runtime = new HarnessRuntime({
14
+ driver: agentDriver({
15
+ name: "analyst",
16
+ model: openai("gpt-6-sol"),
17
+ instructions: "Analyze the available data. Ask the user when a necessary detail is missing.",
18
+ }, { stream: true }),
19
+ tools: [requestInputTool()],
20
+ grants: { toolIds: ["request_input"], modelRoles: ["main"] },
21
+ budgets: { maxModelCalls: 20, maxToolCalls: 30, maxTokens: 100_000 },
22
+ activeTimeoutMs: 120_000,
23
+ inputTimeoutMs: 15 * 60_000,
24
+ });
25
+
26
+ const handle = runtime.start("Prepare a report", {
27
+ identity: { tenantId: verifiedTenantId, userId: verifiedUserId },
28
+ sessionId: conversationId,
29
+ });
30
+
31
+ // In the application's event consumer:
32
+ for await (const event of handle.events()) {
33
+ if (event.payload.type === "input.requested") {
34
+ showQuestion(event.payload.request); // Store its id with the form.
35
+ } else {
36
+ renderActivity(event);
37
+ }
38
+ }
39
+
40
+ // In a separate form submission handler:
41
+ await handle.reply(requestIdFromForm, userAnswer);
42
+ const result = await handle.result(); // Resolves once, after the live wait.
43
+ ```
44
+
45
+ The model chooses when to call the granted question tool. A custom tool or driver
46
+ can call `await services.requestInput({ question, choices?, timeoutMs? })`
47
+ directly. The returned `InputReply` contains `requestId` and `input`.
48
+ `choices` are presentation suggestions; the application may validate a richer
49
+ answer before submitting it. This API does not invent a form schema or force a
50
+ question on every run.
51
+
52
+ `handle.state` is `running`, `awaiting_input`, `cancelling`, or `finished`.
53
+ `handle.pendingInput` returns a cloned current question, useful when reconnecting
54
+ after an event-history gap. Only one question can be pending per run.
55
+
56
+ Replies claim the pending request synchronously. Duplicate, stale, wrong-run,
57
+ malformed and post-completion replies reject with `HarnessInputError.code`.
58
+ Messages are limited to 64KB; questions to 32KB. The application remains responsible
59
+ for authenticating the actor before exposing a handle or submitting a reply.
60
+ Events `input.requested`, `input.resolved`, and `run.resumed` share the request ID.
61
+
62
+ | Timeout | Includes waiting? | Expiry reason code |
63
+ | --- | --- | --- |
64
+ | start option `deadline` (absolute timestamp) | Yes | `deadline_exceeded` |
65
+ | runtime `activeTimeoutMs` | No | `active_timeout` |
66
+ | runtime `inputTimeoutMs`, overridden per question | Only the current wait | `input_timeout` |
67
+
68
+ Timeouts cancel the run cooperatively. `cancel()` also works during a question.
69
+ Cleanup waits for already-started work to settle; external effects are not rolled
70
+ back. Without an input timeout or deadline, a question can wait indefinitely.
71
+ No new model calls may start while a question is pending; concurrent work already
72
+ started before the question still has its usual cancellation/lifetime contract.
73
+
74
+ For minor-version compatibility, a custom driver that **returns** legacy
75
+ `status: "awaiting_input"`, or a completion policy returning `await_input`,
76
+ still produces the old terminal hand-back. That path cannot resume its stack.
77
+ Migrate live conversations to **awaiting `requestInput()`** inside the driver/tool;
78
+ the new `awaiting_input` handle state is nonterminal.
79
+
80
+ ## Steer an active Agent
81
+
82
+ ```ts
83
+ await handle.send("Only include last month", { mode: "steer" });
84
+ ```
85
+
86
+ The built-in Agent driver now supports steering in both streaming modes. It
87
+ appends input before the next model call, after the entire current tool-call/result
88
+ group. Steering during a final model call is incorporated at the next complete
89
+ boundary if budgets allow. Effects already completed are not repeated by the
90
+ runtime. The model still decides its subsequent actions.
91
+
92
+ `input.received` means queued. `input.applied` means inserted into the
93
+ conversation at a safe boundary; both carry `inputId` and `mode`. Inputs within
94
+ each mode preserve FIFO order. `follow_up` remains queued until the current
95
+ driver turn completes; `steer` and `follow_up` have different scheduling semantics.
96
+ The queue holds at most 32 inputs. Steering rejects once the Agent loop closes,
97
+ including while a completion policy is still evaluating. Child agents cannot
98
+ consume their parent's steering inbox.
99
+
100
+ ## Public communication
101
+
102
+ Render these typed events by their stable item IDs:
103
+
104
+ | Event | Meaning |
105
+ | --- | --- |
106
+ | `message.started` | Start an item; `phase: "pending"` is explicitly unclassified. |
107
+ | `message.delta` | Append text to that item. |
108
+ | `message.completed` | The item has phase `commentary`, `final`, or `reasoning_summary`. |
109
+ | `message.failed` | A partial item was interrupted or cancelled. |
110
+
111
+ A `final` message is a model answer, not a terminal runtime result: policies or
112
+ queued input may still keep the run open. Only `run.terminal` closes the run.
113
+ Standalone native commentary continues the same Agent loop under its existing
114
+ budgets. No update schedule, artificial thinking, or progress tool is introduced.
115
+
116
+ Large completed items use `artifactId` with an empty inline text field; the deltas
117
+ remain complete. Retrieve the full `PublicMessage` with
118
+ `runtime.getArtifact(identity, sessionId, artifactId)`. Event cursors and the
119
+ existing bounded event store retain their normal semantics.
120
+
121
+ Core consumers can use `RunOutput.publicMessages`, the `run.message` EventBus event,
122
+ or opt into `Agent.stream(input, { publicMessageEvents: true })`. The default
123
+ stream retains its existing text/finish sequence. Terminal usage and cost data
124
+ remain on the final finish chunk, including with public lifecycle chunks enabled.
125
+ Legacy `text.delta`, `thinking`, and `RunOutput.text` remain available; prefer
126
+ public items when distinguishing progress from answers.
127
+
128
+ `getCommunicationCapabilities(provider)` reports adapter support:
129
+
130
+ | Adapter | Message phases | Public summaries |
131
+ | --- | --- | --- |
132
+ | OpenAI Responses and compatible endpoints implementing these fields | Native when present; otherwise inferred | Documented reasoning summary text |
133
+ | Anthropic / AWS Claude with reasoning enabled | Inferred | Text requested with `display: "summarized"` |
134
+ | Google / Vertex Gemini | Inferred | Documented thought summary text |
135
+ | Other adapters without an explicit capability declaration | Inferred | Unsupported |
136
+
137
+ `conditional` means support depends on the selected model, endpoint and request
138
+ options. It is not a promise that the model emits a summary on every call.
139
+ For providers without native phases, a complete response with tools is commentary;
140
+ a complete response without tools is final. Unclassified streaming text remains
141
+ pending until the response is complete.
142
+
143
+ Raw reasoning, signatures, encrypted/redacted blocks and `providerExtras` are
144
+ never used as public summaries. Opaque data remains in canonical conversation
145
+ history for provider replay. Custom adapters should populate `publicMessages`,
146
+ `message.phase`, and the corresponding streaming metadata explicitly.
147
+
148
+ ## Compact completed tool rounds
149
+
150
+ ```ts
151
+ const contextPolicy = summaryContextPolicy({
152
+ modelRole: "summary",
153
+ maxContextTokens: 12_000,
154
+ grouping: "tool_roundtrip",
155
+ keepRecentTurns: 2, // Here: keep the two latest complete groups.
156
+ });
157
+ ```
158
+
159
+ Bind and grant the summary model role through `HarnessRuntime.models` and
160
+ `grants.modelRoles` as usual. The default grouping remains whole user turns.
161
+ The new option recognizes completed tool rounds within a long task, keeps the
162
+ latest user request and complete retained tool/replay groups, and summarizes
163
+ older groups through the existing summary policy. Applications do not insert
164
+ synthetic user turns. Canonical session history is unchanged.
165
+
166
+ `compaction.started`, `compaction.completed` and `compaction.failed` correlate by
167
+ `compactionId` and `policyId`. The summary call shares the run's model/token
168
+ budgets and cancellation signal. If an intact live group cannot fit, compaction
169
+ fails rather than dropping required continuation data.
170
+
171
+ Live continuation does not provide process-restart recovery. Events, pending
172
+ promises and run handles remain in memory. A future recovery adapter must persist
173
+ questions/checkpoints, establish ownership, and reconcile uncertain tool effects.
package/README.md CHANGED
@@ -271,11 +271,19 @@ events; reconnect with `events({ after: cursor })`. Older cursors receive
271
271
  `HarnessEventGapError`. Large terminal outputs become scoped artifact references;
272
272
  retrieve them with `runtime.getArtifact(identity, sessionId, artifactId)`.
273
273
 
274
- Built-in drivers support queued `follow_up` input. Custom drivers may declare
275
- `steer` and consume it at boundaries with `services.takeInput()`. Unsupported
276
- controls throw `HarnessUnsupportedError`. Interrupt-and-replace, pause/resume,
277
- remote policy coverage, and durable recovery are rejected by this local runtime.
278
- In-memory events, artifacts, and sessions are explicitly non-durable.
274
+ Built-in drivers support queued `follow_up` input. The built-in Agent driver also
275
+ supports `steer` at complete tool-roundtrip boundaries. Custom drivers may declare
276
+ `steer` and consume it with `services.takeInput()`. For live questions, use
277
+ `requestInputTool()` or `await services.requestInput()`, then answer with
278
+ `handle.reply(requestId, input)`. `handle.state === "awaiting_input"` is nonterminal;
279
+ `result()` stays pending and budgets are preserved. Returning the old terminal
280
+ `awaiting_input` status is a legacy hand-back, not live suspension.
281
+
282
+ See [conversational runs](./CONVERSATIONAL-RUNS.md) for clocks, input events,
283
+ public messages, provider capabilities, streaming compatibility and compaction.
284
+ Unsupported controls throw `HarnessUnsupportedError`. Interrupt-and-replace,
285
+ remote policy coverage, and durable recovery remain unsupported. In-memory
286
+ events, artifacts, sessions and pending questions are explicitly non-durable.
279
287
 
280
288
  Host grants are upper bounds. Controllers select only approved tools/model roles;
281
289
  omitting required tools or selecting unsupported model options fails closed.
@@ -0,0 +1,39 @@
1
+ import type { RunContext } from "@agentium/core";
2
+ import type { AbilityBinding } from "./runtime/index.cjs";
3
+ import { type AbilityFactory, createHarnessDefinition, describeHarnessDefinition, exportHarnessManifest, extendHarnessDefinition, hashHarnessManifest, type JsonObject, type LocalAbilityUse, loadHarnessManifest } from "./runtime/index.cjs";
4
+ export interface AbilityDescription {
5
+ toolNames: readonly string[];
6
+ requirements: readonly string[];
7
+ runtimeDependent?: boolean;
8
+ }
9
+ export interface AbilityDefinition<Options> {
10
+ type: string;
11
+ version?: number;
12
+ /** Pure validation/snapshotting. Preserve caller-owned service references; never freeze clients. */
13
+ validate: (options: Options) => Options;
14
+ describe: (options: Options) => AbilityDescription;
15
+ bind: (options: Options, ctx: RunContext) => AbilityBinding | Promise<AbilityBinding>;
16
+ /** Explicit trusted mapping. Omit for local callbacks/services that cannot be serialized. */
17
+ portable?: {
18
+ validateOptions: (options: JsonObject) => JsonObject;
19
+ toOptions: (options: JsonObject) => Options;
20
+ toJSON: (options: Options) => JsonObject;
21
+ };
22
+ }
23
+ export interface Ability<Options> {
24
+ (options: Options, config?: {
25
+ instanceId?: string;
26
+ }): LocalAbilityUse;
27
+ readonly type: string;
28
+ readonly version: number;
29
+ readonly factory?: AbilityFactory;
30
+ }
31
+ /** Capture typed local options; erase only the wrapper so heterogeneous abilities compose without casts. */
32
+ export declare function defineAbility<Options>(definition: AbilityDefinition<Options>): Ability<Options>;
33
+ export declare const defineHarness: typeof createHarnessDefinition;
34
+ export declare const extendHarness: typeof extendHarnessDefinition;
35
+ export declare const describeHarness: typeof describeHarnessDefinition;
36
+ export declare const exportManifest: typeof exportHarnessManifest;
37
+ export declare const loadManifest: typeof loadHarnessManifest;
38
+ export declare const hashManifest: typeof hashHarnessManifest;
39
+ //# sourceMappingURL=definition.d.ts.map