@agentium/harness 4.1.0 → 4.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CONVERSATIONAL-RUNS.md +173 -0
- package/README.md +13 -5
- package/dist/definition.d.cts +39 -0
- package/dist/{driver-Cr2BGG4h.cjs → driver-Bh_wLWCP.cjs} +285 -15
- package/dist/{driver-BvzvqvpB.js → driver-reTLBJ_o.js} +280 -16
- package/dist/drivers.d.cts +14 -0
- package/dist/drivers.d.ts.map +1 -1
- package/dist/index.cjs +103 -46
- package/dist/index.d.cts +19 -0
- package/dist/index.d.ts +4 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +103 -48
- package/dist/input-tool.d.cts +7 -0
- package/dist/input-tool.d.ts +7 -0
- package/dist/input-tool.d.ts.map +1 -0
- package/dist/mcp-resources.d.cts +42 -0
- package/dist/policies.d.cts +31 -0
- package/dist/policies.d.ts +2 -0
- package/dist/policies.d.ts.map +1 -1
- package/dist/presets.d.cts +41 -0
- package/dist/runtime/context-policy.d.cts +21 -0
- package/dist/runtime/context-policy.d.ts +1 -0
- package/dist/runtime/context-policy.d.ts.map +1 -1
- package/dist/runtime/context.d.cts +15 -0
- package/dist/runtime/controller.d.cts +51 -0
- package/dist/runtime/driver.d.cts +157 -0
- package/dist/runtime/driver.d.ts +14 -3
- package/dist/runtime/driver.d.ts.map +1 -1
- package/dist/runtime/events.d.cts +116 -0
- package/dist/runtime/events.d.ts +35 -2
- package/dist/runtime/events.d.ts.map +1 -1
- package/dist/runtime/index.d.cts +9 -0
- package/dist/runtime/input.d.cts +46 -0
- package/dist/runtime/input.d.ts +46 -0
- package/dist/runtime/input.d.ts.map +1 -0
- package/dist/runtime/middleware.d.cts +10 -0
- package/dist/runtime/resolve.d.cts +20 -0
- package/dist/runtime/runtime-registry.d.cts +35 -0
- package/dist/runtime/session-bindings.d.cts +60 -0
- package/dist/runtime/types.d.cts +176 -0
- package/dist/testing.cjs +1 -1
- package/dist/testing.d.cts +7 -0
- package/dist/testing.js +1 -1
- package/dist/watch/definition.d.cts +11 -0
- package/dist/watch/gmail.d.cts +104 -0
- package/dist/watch/quiet-hours.d.cts +10 -0
- package/dist/watch/runtime.d.cts +42 -0
- package/dist/watch/types.d.cts +147 -0
- package/package.json +21 -10
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
# Conversational runs (4.6)
|
|
2
|
+
|
|
3
|
+
Questions, steering, public messages and compaction stay inside one live run.
|
|
4
|
+
The runtime retains the session lease, Agent instance, transcript, provider
|
|
5
|
+
continuations and remaining budgets. Waiting does not poll the model.
|
|
6
|
+
|
|
7
|
+
## Ask and resume
|
|
8
|
+
|
|
9
|
+
```ts
|
|
10
|
+
import { openai } from "@agentium/core";
|
|
11
|
+
import { agentDriver, HarnessRuntime, requestInputTool } from "@agentium/harness";
|
|
12
|
+
|
|
13
|
+
const runtime = new HarnessRuntime({
|
|
14
|
+
driver: agentDriver({
|
|
15
|
+
name: "analyst",
|
|
16
|
+
model: openai("gpt-6-sol"),
|
|
17
|
+
instructions: "Analyze the available data. Ask the user when a necessary detail is missing.",
|
|
18
|
+
}, { stream: true }),
|
|
19
|
+
tools: [requestInputTool()],
|
|
20
|
+
grants: { toolIds: ["request_input"], modelRoles: ["main"] },
|
|
21
|
+
budgets: { maxModelCalls: 20, maxToolCalls: 30, maxTokens: 100_000 },
|
|
22
|
+
activeTimeoutMs: 120_000,
|
|
23
|
+
inputTimeoutMs: 15 * 60_000,
|
|
24
|
+
});
|
|
25
|
+
|
|
26
|
+
const handle = runtime.start("Prepare a report", {
|
|
27
|
+
identity: { tenantId: verifiedTenantId, userId: verifiedUserId },
|
|
28
|
+
sessionId: conversationId,
|
|
29
|
+
});
|
|
30
|
+
|
|
31
|
+
// In the application's event consumer:
|
|
32
|
+
for await (const event of handle.events()) {
|
|
33
|
+
if (event.payload.type === "input.requested") {
|
|
34
|
+
showQuestion(event.payload.request); // Store its id with the form.
|
|
35
|
+
} else {
|
|
36
|
+
renderActivity(event);
|
|
37
|
+
}
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
// In a separate form submission handler:
|
|
41
|
+
await handle.reply(requestIdFromForm, userAnswer);
|
|
42
|
+
const result = await handle.result(); // Resolves once, after the live wait.
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
The model chooses when to call the granted question tool. A custom tool or driver
|
|
46
|
+
can call `await services.requestInput({ question, choices?, timeoutMs? })`
|
|
47
|
+
directly. The returned `InputReply` contains `requestId` and `input`.
|
|
48
|
+
`choices` are presentation suggestions; the application may validate a richer
|
|
49
|
+
answer before submitting it. This API does not invent a form schema or force a
|
|
50
|
+
question on every run.
|
|
51
|
+
|
|
52
|
+
`handle.state` is `running`, `awaiting_input`, `cancelling`, or `finished`.
|
|
53
|
+
`handle.pendingInput` returns a cloned current question, useful when reconnecting
|
|
54
|
+
after an event-history gap. Only one question can be pending per run.
|
|
55
|
+
|
|
56
|
+
Replies claim the pending request synchronously. Duplicate, stale, wrong-run,
|
|
57
|
+
malformed and post-completion replies reject with `HarnessInputError.code`.
|
|
58
|
+
Messages are limited to 64KB; questions to 32KB. The application remains responsible
|
|
59
|
+
for authenticating the actor before exposing a handle or submitting a reply.
|
|
60
|
+
Events `input.requested`, `input.resolved`, and `run.resumed` share the request ID.
|
|
61
|
+
|
|
62
|
+
| Timeout | Includes waiting? | Expiry reason code |
|
|
63
|
+
| --- | --- | --- |
|
|
64
|
+
| start option `deadline` (absolute timestamp) | Yes | `deadline_exceeded` |
|
|
65
|
+
| runtime `activeTimeoutMs` | No | `active_timeout` |
|
|
66
|
+
| runtime `inputTimeoutMs`, overridden per question | Only the current wait | `input_timeout` |
|
|
67
|
+
|
|
68
|
+
Timeouts cancel the run cooperatively. `cancel()` also works during a question.
|
|
69
|
+
Cleanup waits for already-started work to settle; external effects are not rolled
|
|
70
|
+
back. Without an input timeout or deadline, a question can wait indefinitely.
|
|
71
|
+
No new model calls may start while a question is pending; concurrent work already
|
|
72
|
+
started before the question still has its usual cancellation/lifetime contract.
|
|
73
|
+
|
|
74
|
+
For minor-version compatibility, a custom driver that **returns** legacy
|
|
75
|
+
`status: "awaiting_input"`, or a completion policy returning `await_input`,
|
|
76
|
+
still produces the old terminal hand-back. That path cannot resume its stack.
|
|
77
|
+
Migrate live conversations to **awaiting `requestInput()`** inside the driver/tool;
|
|
78
|
+
the new `awaiting_input` handle state is nonterminal.
|
|
79
|
+
|
|
80
|
+
## Steer an active Agent
|
|
81
|
+
|
|
82
|
+
```ts
|
|
83
|
+
await handle.send("Only include last month", { mode: "steer" });
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
The built-in Agent driver now supports steering in both streaming modes. It
|
|
87
|
+
appends input before the next model call, after the entire current tool-call/result
|
|
88
|
+
group. Steering during a final model call is incorporated at the next complete
|
|
89
|
+
boundary if budgets allow. Effects already completed are not repeated by the
|
|
90
|
+
runtime. The model still decides its subsequent actions.
|
|
91
|
+
|
|
92
|
+
`input.received` means queued. `input.applied` means inserted into the
|
|
93
|
+
conversation at a safe boundary; both carry `inputId` and `mode`. Inputs within
|
|
94
|
+
each mode preserve FIFO order. `follow_up` remains queued until the current
|
|
95
|
+
driver turn completes; `steer` and `follow_up` have different scheduling semantics.
|
|
96
|
+
The queue holds at most 32 inputs. Steering rejects once the Agent loop closes,
|
|
97
|
+
including while a completion policy is still evaluating. Child agents cannot
|
|
98
|
+
consume their parent's steering inbox.
|
|
99
|
+
|
|
100
|
+
## Public communication
|
|
101
|
+
|
|
102
|
+
Render these typed events by their stable item IDs:
|
|
103
|
+
|
|
104
|
+
| Event | Meaning |
|
|
105
|
+
| --- | --- |
|
|
106
|
+
| `message.started` | Start an item; `phase: "pending"` is explicitly unclassified. |
|
|
107
|
+
| `message.delta` | Append text to that item. |
|
|
108
|
+
| `message.completed` | The item has phase `commentary`, `final`, or `reasoning_summary`. |
|
|
109
|
+
| `message.failed` | A partial item was interrupted or cancelled. |
|
|
110
|
+
|
|
111
|
+
A `final` message is a model answer, not a terminal runtime result: policies or
|
|
112
|
+
queued input may still keep the run open. Only `run.terminal` closes the run.
|
|
113
|
+
Standalone native commentary continues the same Agent loop under its existing
|
|
114
|
+
budgets. No update schedule, artificial thinking, or progress tool is introduced.
|
|
115
|
+
|
|
116
|
+
Large completed items use `artifactId` with an empty inline text field; the deltas
|
|
117
|
+
remain complete. Retrieve the full `PublicMessage` with
|
|
118
|
+
`runtime.getArtifact(identity, sessionId, artifactId)`. Event cursors and the
|
|
119
|
+
existing bounded event store retain their normal semantics.
|
|
120
|
+
|
|
121
|
+
Core consumers can use `RunOutput.publicMessages`, the `run.message` EventBus event,
|
|
122
|
+
or opt into `Agent.stream(input, { publicMessageEvents: true })`. The default
|
|
123
|
+
stream retains its existing text/finish sequence. Terminal usage and cost data
|
|
124
|
+
remain on the final finish chunk, including with public lifecycle chunks enabled.
|
|
125
|
+
Legacy `text.delta`, `thinking`, and `RunOutput.text` remain available; prefer
|
|
126
|
+
public items when distinguishing progress from answers.
|
|
127
|
+
|
|
128
|
+
`getCommunicationCapabilities(provider)` reports adapter support:
|
|
129
|
+
|
|
130
|
+
| Adapter | Message phases | Public summaries |
|
|
131
|
+
| --- | --- | --- |
|
|
132
|
+
| OpenAI Responses and compatible endpoints implementing these fields | Native when present; otherwise inferred | Documented reasoning summary text |
|
|
133
|
+
| Anthropic / AWS Claude with reasoning enabled | Inferred | Text requested with `display: "summarized"` |
|
|
134
|
+
| Google / Vertex Gemini | Inferred | Documented thought summary text |
|
|
135
|
+
| Other adapters without an explicit capability declaration | Inferred | Unsupported |
|
|
136
|
+
|
|
137
|
+
`conditional` means support depends on the selected model, endpoint and request
|
|
138
|
+
options. It is not a promise that the model emits a summary on every call.
|
|
139
|
+
For providers without native phases, a complete response with tools is commentary;
|
|
140
|
+
a complete response without tools is final. Unclassified streaming text remains
|
|
141
|
+
pending until the response is complete.
|
|
142
|
+
|
|
143
|
+
Raw reasoning, signatures, encrypted/redacted blocks and `providerExtras` are
|
|
144
|
+
never used as public summaries. Opaque data remains in canonical conversation
|
|
145
|
+
history for provider replay. Custom adapters should populate `publicMessages`,
|
|
146
|
+
`message.phase`, and the corresponding streaming metadata explicitly.
|
|
147
|
+
|
|
148
|
+
## Compact completed tool rounds
|
|
149
|
+
|
|
150
|
+
```ts
|
|
151
|
+
const contextPolicy = summaryContextPolicy({
|
|
152
|
+
modelRole: "summary",
|
|
153
|
+
maxContextTokens: 12_000,
|
|
154
|
+
grouping: "tool_roundtrip",
|
|
155
|
+
keepRecentTurns: 2, // Here: keep the two latest complete groups.
|
|
156
|
+
});
|
|
157
|
+
```
|
|
158
|
+
|
|
159
|
+
Bind and grant the summary model role through `HarnessRuntime.models` and
|
|
160
|
+
`grants.modelRoles` as usual. The default grouping remains whole user turns.
|
|
161
|
+
The new option recognizes completed tool rounds within a long task, keeps the
|
|
162
|
+
latest user request and complete retained tool/replay groups, and summarizes
|
|
163
|
+
older groups through the existing summary policy. Applications do not insert
|
|
164
|
+
synthetic user turns. Canonical session history is unchanged.
|
|
165
|
+
|
|
166
|
+
`compaction.started`, `compaction.completed` and `compaction.failed` correlate by
|
|
167
|
+
`compactionId` and `policyId`. The summary call shares the run's model/token
|
|
168
|
+
budgets and cancellation signal. If an intact live group cannot fit, compaction
|
|
169
|
+
fails rather than dropping required continuation data.
|
|
170
|
+
|
|
171
|
+
Live continuation does not provide process-restart recovery. Events, pending
|
|
172
|
+
promises and run handles remain in memory. A future recovery adapter must persist
|
|
173
|
+
questions/checkpoints, establish ownership, and reconcile uncertain tool effects.
|
package/README.md
CHANGED
|
@@ -271,11 +271,19 @@ events; reconnect with `events({ after: cursor })`. Older cursors receive
|
|
|
271
271
|
`HarnessEventGapError`. Large terminal outputs become scoped artifact references;
|
|
272
272
|
retrieve them with `runtime.getArtifact(identity, sessionId, artifactId)`.
|
|
273
273
|
|
|
274
|
-
Built-in drivers support queued `follow_up` input.
|
|
275
|
-
`steer`
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
274
|
+
Built-in drivers support queued `follow_up` input. The built-in Agent driver also
|
|
275
|
+
supports `steer` at complete tool-roundtrip boundaries. Custom drivers may declare
|
|
276
|
+
`steer` and consume it with `services.takeInput()`. For live questions, use
|
|
277
|
+
`requestInputTool()` or `await services.requestInput()`, then answer with
|
|
278
|
+
`handle.reply(requestId, input)`. `handle.state === "awaiting_input"` is nonterminal;
|
|
279
|
+
`result()` stays pending and budgets are preserved. Returning the old terminal
|
|
280
|
+
`awaiting_input` status is a legacy hand-back, not live suspension.
|
|
281
|
+
|
|
282
|
+
See [conversational runs](./CONVERSATIONAL-RUNS.md) for clocks, input events,
|
|
283
|
+
public messages, provider capabilities, streaming compatibility and compaction.
|
|
284
|
+
Unsupported controls throw `HarnessUnsupportedError`. Interrupt-and-replace,
|
|
285
|
+
remote policy coverage, and durable recovery remain unsupported. In-memory
|
|
286
|
+
events, artifacts, sessions and pending questions are explicitly non-durable.
|
|
279
287
|
|
|
280
288
|
Host grants are upper bounds. Controllers select only approved tools/model roles;
|
|
281
289
|
omitting required tools or selecting unsupported model options fails closed.
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import type { RunContext } from "@agentium/core";
|
|
2
|
+
import type { AbilityBinding } from "./runtime/index.cjs";
|
|
3
|
+
import { type AbilityFactory, createHarnessDefinition, describeHarnessDefinition, exportHarnessManifest, extendHarnessDefinition, hashHarnessManifest, type JsonObject, type LocalAbilityUse, loadHarnessManifest } from "./runtime/index.cjs";
|
|
4
|
+
export interface AbilityDescription {
|
|
5
|
+
toolNames: readonly string[];
|
|
6
|
+
requirements: readonly string[];
|
|
7
|
+
runtimeDependent?: boolean;
|
|
8
|
+
}
|
|
9
|
+
export interface AbilityDefinition<Options> {
|
|
10
|
+
type: string;
|
|
11
|
+
version?: number;
|
|
12
|
+
/** Pure validation/snapshotting. Preserve caller-owned service references; never freeze clients. */
|
|
13
|
+
validate: (options: Options) => Options;
|
|
14
|
+
describe: (options: Options) => AbilityDescription;
|
|
15
|
+
bind: (options: Options, ctx: RunContext) => AbilityBinding | Promise<AbilityBinding>;
|
|
16
|
+
/** Explicit trusted mapping. Omit for local callbacks/services that cannot be serialized. */
|
|
17
|
+
portable?: {
|
|
18
|
+
validateOptions: (options: JsonObject) => JsonObject;
|
|
19
|
+
toOptions: (options: JsonObject) => Options;
|
|
20
|
+
toJSON: (options: Options) => JsonObject;
|
|
21
|
+
};
|
|
22
|
+
}
|
|
23
|
+
export interface Ability<Options> {
|
|
24
|
+
(options: Options, config?: {
|
|
25
|
+
instanceId?: string;
|
|
26
|
+
}): LocalAbilityUse;
|
|
27
|
+
readonly type: string;
|
|
28
|
+
readonly version: number;
|
|
29
|
+
readonly factory?: AbilityFactory;
|
|
30
|
+
}
|
|
31
|
+
/** Capture typed local options; erase only the wrapper so heterogeneous abilities compose without casts. */
|
|
32
|
+
export declare function defineAbility<Options>(definition: AbilityDefinition<Options>): Ability<Options>;
|
|
33
|
+
export declare const defineHarness: typeof createHarnessDefinition;
|
|
34
|
+
export declare const extendHarness: typeof extendHarnessDefinition;
|
|
35
|
+
export declare const describeHarness: typeof describeHarnessDefinition;
|
|
36
|
+
export declare const exportManifest: typeof exportHarnessManifest;
|
|
37
|
+
export declare const loadManifest: typeof loadHarnessManifest;
|
|
38
|
+
export declare const hashManifest: typeof hashHarnessManifest;
|
|
39
|
+
//# sourceMappingURL=definition.d.ts.map
|