@owlmeans/llm 0.1.18-rc.21 → 0.1.18-rc.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/agent-meta/manifest.json +9 -2
- package/agent-meta/skills/inquiry/SKILL.md +199 -0
- package/agent-meta/skills/llm/SKILL.md +9 -1
- package/build/consts.d.ts +4 -0
- package/build/consts.d.ts.map +1 -1
- package/build/consts.js +4 -0
- package/build/consts.js.map +1 -1
- package/build/execution/service.d.ts.map +1 -1
- package/build/execution/service.js +15 -1
- package/build/execution/service.js.map +1 -1
- package/build/execution/types.d.ts +19 -1
- package/build/execution/types.d.ts.map +1 -1
- package/build/index.d.ts +1 -0
- package/build/index.d.ts.map +1 -1
- package/build/index.js +1 -0
- package/build/index.js.map +1 -1
- package/build/inquiry/bridge.d.ts +22 -0
- package/build/inquiry/bridge.d.ts.map +1 -0
- package/build/inquiry/bridge.js +29 -0
- package/build/inquiry/bridge.js.map +1 -0
- package/build/inquiry/errors.d.ts +29 -0
- package/build/inquiry/errors.d.ts.map +1 -0
- package/build/inquiry/errors.js +40 -0
- package/build/inquiry/errors.js.map +1 -0
- package/build/inquiry/index.d.ts +4 -0
- package/build/inquiry/index.d.ts.map +1 -0
- package/build/inquiry/index.js +4 -0
- package/build/inquiry/index.js.map +1 -0
- package/build/inquiry/transport.d.ts +12 -0
- package/build/inquiry/transport.d.ts.map +1 -0
- package/build/inquiry/transport.js +35 -0
- package/build/inquiry/transport.js.map +1 -0
- package/build/model.d.ts.map +1 -1
- package/build/model.js +177 -150
- package/build/model.js.map +1 -1
- package/build/types.d.ts +12 -0
- package/build/types.d.ts.map +1 -1
- package/package.json +6 -6
- package/src/consts.ts +4 -0
- package/src/execution/service.ts +17 -1
- package/src/execution/types.ts +22 -2
- package/src/index.ts +1 -0
- package/src/inquiry/bridge.ts +33 -0
- package/src/inquiry/errors.ts +46 -0
- package/src/inquiry/index.ts +3 -0
- package/src/inquiry/transport.ts +42 -0
- package/src/model.ts +42 -16
- package/src/types.ts +13 -0
- package/tests/inquiry.spec.ts +200 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@owlmeans/llm",
|
|
3
|
-
"version": "0.1.18-rc.
|
|
3
|
+
"version": "0.1.18-rc.22",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"scripts": {
|
|
@@ -47,7 +47,7 @@
|
|
|
47
47
|
"@langchain/core": "^1.2.9",
|
|
48
48
|
"@langchain/openai": "^1.5.10",
|
|
49
49
|
"@owlmeans/dep-config": "workspace:*",
|
|
50
|
-
"@owlmeans/test": "^0.1.18-rc.
|
|
50
|
+
"@owlmeans/test": "^0.1.18-rc.17",
|
|
51
51
|
"@types/bun": "^1.4.0",
|
|
52
52
|
"@types/node": "^26.1.0",
|
|
53
53
|
"nodemon": "^3.1.14",
|
|
@@ -55,10 +55,10 @@
|
|
|
55
55
|
},
|
|
56
56
|
"dependencies": {
|
|
57
57
|
"@anthropic-ai/sdk": "^0.78.0",
|
|
58
|
-
"@owlmeans/basic-ids": "^0.1.18-rc.
|
|
59
|
-
"@owlmeans/context": "^0.1.18-rc.
|
|
60
|
-
"@owlmeans/error": "^0.1.18-rc.
|
|
61
|
-
"@owlmeans/llm-common": "^0.1.18-rc.
|
|
58
|
+
"@owlmeans/basic-ids": "^0.1.18-rc.18",
|
|
59
|
+
"@owlmeans/context": "^0.1.18-rc.17",
|
|
60
|
+
"@owlmeans/error": "^0.1.18-rc.17",
|
|
61
|
+
"@owlmeans/llm-common": "^0.1.18-rc.21",
|
|
62
62
|
"ajv": "^8.17.1"
|
|
63
63
|
},
|
|
64
64
|
"publishConfig": {
|
package/src/consts.ts
CHANGED
|
@@ -129,6 +129,10 @@ export const EFFORT_TABLE: Record<ExecutionEffort, ModelConfigPatch> = {
|
|
|
129
129
|
*
|
|
130
130
|
* `state` is in the list because a `TaskExecution` carries its own composed state —
|
|
131
131
|
* without excluding it every `derive`/`escalate`/`withPurpose` would nest another copy.
|
|
132
|
+
*
|
|
133
|
+
* `inquiry` is deliberately ABSENT: how a run may put a question to a person is state, and a run
|
|
134
|
+
* resumed from a snapshot must ask through the same channel under the same policy. Listing it
|
|
135
|
+
* here would leave a resumed run silently unable to ask anything.
|
|
132
136
|
*/
|
|
133
137
|
export const COLLABORATOR_KEYS: string[] = [
|
|
134
138
|
'state', 'models', 'model', 'temperatureFactory', 'outputErrors', 'files', 'prompts',
|
package/src/execution/service.ts
CHANGED
|
@@ -1,8 +1,12 @@
|
|
|
1
1
|
import { createService } from '@owlmeans/context'
|
|
2
2
|
import type { BasicConfig, BasicContext } from '@owlmeans/context'
|
|
3
|
-
import {
|
|
3
|
+
import {
|
|
4
|
+
capAnswer, defaultAnswerFor, ExecutionEffort, ExecutionLevel, InquiryPolicy, UTILITY_ROLE,
|
|
5
|
+
} from '@owlmeans/llm-common'
|
|
4
6
|
import type { ExecutionState, ModelPolicy, TaskExecutionState } from '@owlmeans/llm-common'
|
|
5
7
|
import { COLLABORATOR_KEYS, EXECUTION_SERVICE } from '../consts.js'
|
|
8
|
+
import { InquiryDeclined } from '../inquiry/errors.js'
|
|
9
|
+
import { inquiryTransportFor } from '../inquiry/transport.js'
|
|
6
10
|
import type { TemperatureFactory } from '../types.js'
|
|
7
11
|
import type {
|
|
8
12
|
Execution, ExecutionPlugin, ExecutionService, ExecutionServiceOptions, ExecutionShape,
|
|
@@ -52,6 +56,9 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
|
|
|
52
56
|
purpose: { ...input.purpose },
|
|
53
57
|
policy: { ...input.policy },
|
|
54
58
|
...(input.prompt != null ? { prompt: { ...input.prompt } } : {}),
|
|
59
|
+
// Copied like every other piece of state, and NOT listed in `COLLABORATOR_KEYS`: a resumed
|
|
60
|
+
// run has to ask through the channel and policy it was started with.
|
|
61
|
+
...(input.inquiry != null ? { inquiry: { ...input.inquiry } } : {}),
|
|
55
62
|
}) as S['project'],
|
|
56
63
|
|
|
57
64
|
forTask: (parent, input) => {
|
|
@@ -180,6 +187,15 @@ export const executionServiceApi = <S extends ExecutionShape = ExecutionShape>(
|
|
|
180
187
|
return null
|
|
181
188
|
},
|
|
182
189
|
|
|
190
|
+
ask: async (exec, inquiry, signal) => {
|
|
191
|
+
const policy = exec.inquiry?.policy ?? InquiryPolicy.Default
|
|
192
|
+
// No channel was ever configured, so there is nobody to wait for: assume and carry on.
|
|
193
|
+
if (policy === InquiryPolicy.Default) return defaultAnswerFor(inquiry)
|
|
194
|
+
if (policy === InquiryPolicy.Refuse) throw new InquiryDeclined(inquiry.id)
|
|
195
|
+
|
|
196
|
+
return capAnswer(await inquiryTransportFor(exec.inquiry?.transport).ask(inquiry, signal))
|
|
197
|
+
},
|
|
198
|
+
|
|
183
199
|
snapshot: exec => {
|
|
184
200
|
if (exec.level === ExecutionLevel.Task) {
|
|
185
201
|
return freeze({ ...(exec as unknown as TaskExecution).state })
|
package/src/execution/types.ts
CHANGED
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
|
|
2
2
|
import type { InitializedService } from '@owlmeans/context'
|
|
3
3
|
import type {
|
|
4
|
-
ExecutionEffort, ExecutionLevel, ExecutionState, FileProviderRef,
|
|
5
|
-
ModelConfigOverride, ModelPolicy, ModelRole, PromptPolicy,
|
|
4
|
+
ExecutionEffort, ExecutionLevel, ExecutionState, FileProviderRef, Inquiry, InquiryAnswer,
|
|
5
|
+
InquiryConfig, LlmPurpose, ModelConfigOverride, ModelPolicy, ModelRole, PromptPolicy,
|
|
6
|
+
TaskExecutionState,
|
|
6
7
|
} from '@owlmeans/llm-common'
|
|
7
8
|
import type { LlmService, TemperatureFactory } from '../types.js'
|
|
8
9
|
import type { PromptService } from '../prompt/types.js'
|
|
@@ -53,6 +54,11 @@ export interface ProjectExecutionInput {
|
|
|
53
54
|
purpose: LlmPurpose
|
|
54
55
|
/** Baseline role and skills for the whole run. */
|
|
55
56
|
prompt?: PromptPolicy
|
|
57
|
+
/**
|
|
58
|
+
* How this run may put a question to a person (see {@link ExecutionService.ask}). State, not a
|
|
59
|
+
* collaborator — it travels into every snapshot, so a resumed run asks the same way.
|
|
60
|
+
*/
|
|
61
|
+
inquiry?: InquiryConfig
|
|
56
62
|
outputErrors?: boolean
|
|
57
63
|
captureNull?: boolean
|
|
58
64
|
}
|
|
@@ -200,6 +206,20 @@ export interface ExecutionService<S extends ExecutionShape = ExecutionShape> ext
|
|
|
200
206
|
snapshot: (exec: S['exec']) => ExecutionState
|
|
201
207
|
restore: (state: ExecutionState, collaborators?: S['collaborators']) => S['exec']
|
|
202
208
|
|
|
209
|
+
/**
|
|
210
|
+
* Put a question to whoever is behind this execution, under its own policy.
|
|
211
|
+
*
|
|
212
|
+
* `ask` → the seated transport (and the answer comes back capped to
|
|
213
|
+
* `DEFAULT_INQUIRY_ANSWER_CHARS`); `default` → the question's own default, or a decline, which
|
|
214
|
+
* the caller is expected to RECORD as an assumption; `refuse` → `InquiryDeclined`. No config at
|
|
215
|
+
* all means `default`: a run that was never given a channel must never block on one.
|
|
216
|
+
*
|
|
217
|
+
* Throws `InquiryUnavailable` when the policy is `ask` and nothing is seated under the
|
|
218
|
+
* execution's transport key. Wire it through `executionInquiry` to read that as "nobody is
|
|
219
|
+
* there" instead.
|
|
220
|
+
*/
|
|
221
|
+
ask: (exec: S['exec'], inquiry: Inquiry, signal?: AbortSignal) => Promise<InquiryAnswer>
|
|
222
|
+
|
|
203
223
|
/**
|
|
204
224
|
* Ask the registered plugins about the project this execution runs in. Plugins are
|
|
205
225
|
* consulted in registration order and the FIRST usable answer wins, so one advisor owns
|
package/src/index.ts
CHANGED
|
@@ -6,6 +6,7 @@ export * from './model.js'
|
|
|
6
6
|
export * from './service.js'
|
|
7
7
|
export * from './helpers/index.js'
|
|
8
8
|
export * from './execution/index.js'
|
|
9
|
+
export * from './inquiry/index.js'
|
|
9
10
|
export * from './prompt/index.js'
|
|
10
11
|
export type * from './plugins/types.js'
|
|
11
12
|
export { plugins, registerLlmPlugin, pluginOf, pluginFor, resolvePlugin } from './plugins/index.js'
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
import type { Inquiry, InquiryAnswer } from '@owlmeans/llm-common'
|
|
2
|
+
import { InquiryUnavailable } from './errors.js'
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* The ONE adapter between an execution and whatever asks questions through it — a pipeline
|
|
6
|
+
* runner's `inquiry.ask`, an agent plugin's channel.
|
|
7
|
+
*
|
|
8
|
+
* It exists so "nobody is there" is decided in a single place: {@link InquiryUnavailable} becomes
|
|
9
|
+
* `null`, which every consumer reads as "carry on without an answer", and everything else — a
|
|
10
|
+
* decline included — escapes untouched. A consumer that mapped the two together would turn a
|
|
11
|
+
* person saying "I will not decide" into a run that parks forever, or the reverse.
|
|
12
|
+
*
|
|
13
|
+
* It asks for the one METHOD it calls rather than for an `ExecutionService<S>`, and infers the
|
|
14
|
+
* execution type from the execution it is handed. A shape generic would be unusable here: every
|
|
15
|
+
* member of `ExecutionService<S>` mentions `S` only through an indexed access (`S['exec']`,
|
|
16
|
+
* `S['projectInput']`, …), which is not an inference site, so `S` fell back to the bare
|
|
17
|
+
* `ExecutionShape` and then refused every real service contravariantly on `root`. The narrow
|
|
18
|
+
* parameter also lets anything that can answer a question stand in — a service, a facade, a
|
|
19
|
+
* test double — which is the whole point of an adapter.
|
|
20
|
+
*/
|
|
21
|
+
export const executionInquiry = <E>(
|
|
22
|
+
service: { ask: (exec: E, inquiry: Inquiry, signal?: AbortSignal) => Promise<InquiryAnswer> },
|
|
23
|
+
exec: E
|
|
24
|
+
): ((inquiry: Inquiry, signal?: AbortSignal) => Promise<InquiryAnswer | null>) =>
|
|
25
|
+
async (inquiry, signal) => {
|
|
26
|
+
try {
|
|
27
|
+
return await service.ask(exec, inquiry, signal)
|
|
28
|
+
} catch (e) {
|
|
29
|
+
if (e instanceof InquiryUnavailable) return null
|
|
30
|
+
|
|
31
|
+
throw e
|
|
32
|
+
}
|
|
33
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
import { ResilientError } from '@owlmeans/error'
|
|
2
|
+
|
|
3
|
+
export class InquiryError extends ResilientError {
|
|
4
|
+
public static override typeName = `Inquiry${ResilientError.typeName}`
|
|
5
|
+
|
|
6
|
+
constructor(message: string = 'error') {
|
|
7
|
+
super(InquiryError.typeName, `inquiry:${message}`)
|
|
8
|
+
}
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
/**
|
|
12
|
+
* No transport is seated under this key, or the one that was has gone.
|
|
13
|
+
*
|
|
14
|
+
* FATAL, and registered as such beside the throw. Nothing above it can improve by trying again:
|
|
15
|
+
* every retry ladder in the stack exists for a performer that answered badly, and there is nobody
|
|
16
|
+
* to ask. Left unrecognised, an absent channel costs an agent its whole turn budget on a tool it
|
|
17
|
+
* will never get an answer from.
|
|
18
|
+
*/
|
|
19
|
+
export class InquiryUnavailable extends InquiryError {
|
|
20
|
+
public static override typeName = `Unavailable${InquiryError.typeName}`
|
|
21
|
+
|
|
22
|
+
constructor(message: string = 'error') {
|
|
23
|
+
super(`unavailable:${message}`)
|
|
24
|
+
this.type = InquiryUnavailable.typeName
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* Nobody may be asked (policy `refuse`), or the answerer refused to decide.
|
|
30
|
+
*
|
|
31
|
+
* NOT retryable either, and deliberately a different class from {@link InquiryUnavailable}: a
|
|
32
|
+
* channel that answered "I will not decide this" is working. Whoever asked must decide itself and
|
|
33
|
+
* record the assumption, rather than treat the run as broken.
|
|
34
|
+
*/
|
|
35
|
+
export class InquiryDeclined extends InquiryError {
|
|
36
|
+
public static override typeName = `Declined${InquiryError.typeName}`
|
|
37
|
+
|
|
38
|
+
constructor(message: string = 'error') {
|
|
39
|
+
super(`declined:${message}`)
|
|
40
|
+
this.type = InquiryDeclined.typeName
|
|
41
|
+
}
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
ResilientError.registerErrorClass(InquiryError)
|
|
45
|
+
ResilientError.registerErrorClass(InquiryUnavailable)
|
|
46
|
+
ResilientError.registerErrorClass(InquiryDeclined)
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
import type { InquiryTransport } from '@owlmeans/llm-common'
|
|
2
|
+
import { registerFatalError } from '../helpers/retry.js'
|
|
3
|
+
import { InquiryUnavailable } from './errors.js'
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Which channel reaches which person.
|
|
7
|
+
*
|
|
8
|
+
* Module-level and keyed by string, exactly like the delegate-transport registry beside it: a
|
|
9
|
+
* process holds many at once — one per connected agent, one per open browser session — and an
|
|
10
|
+
* execution names the one its run belongs to. Seating is what an application does when a channel
|
|
11
|
+
* attaches; releasing is what it does when one goes away, and a question that arrives afterwards
|
|
12
|
+
* fails immediately rather than hanging on nobody.
|
|
13
|
+
*/
|
|
14
|
+
const transports: Record<string, InquiryTransport> = {}
|
|
15
|
+
|
|
16
|
+
export const registerInquiryTransport = (key: string, transport: InquiryTransport): void => {
|
|
17
|
+
transports[key] = transport
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export const releaseInquiryTransport = (key: string): void => {
|
|
21
|
+
delete transports[key]
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
export const hasInquiryTransport = (key: string): boolean => transports[key] != null
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* The transport for a key, or a fatal refusal — never a wait.
|
|
28
|
+
*
|
|
29
|
+
* Named `inquiryTransportFor` rather than `transportFor`: `@owlmeans/llm` and
|
|
30
|
+
* `@owlmeans/llm-delegate` are re-exported into one namespace by `@owlmeans/viable`.
|
|
31
|
+
*/
|
|
32
|
+
export const inquiryTransportFor = (key: string | undefined): InquiryTransport => {
|
|
33
|
+
if (key == null || transports[key] == null) {
|
|
34
|
+
throw new InquiryUnavailable(key ?? 'unkeyed')
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
return transports[key]
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
// Beside the throw, so no caller has to remember: an absent channel aborts every retry loop at
|
|
41
|
+
// once instead of being spent through as if the answer might come next time.
|
|
42
|
+
registerFatalError(e => e instanceof InquiryUnavailable ? e : null)
|
package/src/model.ts
CHANGED
|
@@ -184,6 +184,24 @@ export const makeLlmModel = ({
|
|
|
184
184
|
}
|
|
185
185
|
}
|
|
186
186
|
|
|
187
|
+
/**
|
|
188
|
+
* Give a spectator one best-effort terminal-error observation without allowing that
|
|
189
|
+
* observation to replace the work's real failure. Keeping this outside `withRetry` means
|
|
190
|
+
* transient attempts stay private to the retry ladder.
|
|
191
|
+
*/
|
|
192
|
+
const observeFailure = async <T>(action: string, work: () => Promise<T>): Promise<T> => {
|
|
193
|
+
try {
|
|
194
|
+
return await work()
|
|
195
|
+
} catch (error) {
|
|
196
|
+
try {
|
|
197
|
+
await spectator.error?.({ action, error })
|
|
198
|
+
} catch (observerError) {
|
|
199
|
+
console.warn('[MODEL-ERROR] spectator observation failed', observerError)
|
|
200
|
+
}
|
|
201
|
+
throw error
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
|
|
187
205
|
/**
|
|
188
206
|
* Record the diagnostics of a call that produced nothing usable and build the
|
|
189
207
|
* retryable error describing it. The caller throws it, so control flow stays visible.
|
|
@@ -316,9 +334,10 @@ export const makeLlmModel = ({
|
|
|
316
334
|
escalation, fatal,
|
|
317
335
|
}: LlmAskOptions
|
|
318
336
|
) => {
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
337
|
+
return await observeFailure(action, async () => {
|
|
338
|
+
const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
|
|
339
|
+
const seed = ladderSeed(escalation)
|
|
340
|
+
return await withRetry({ retries, outputErrors, fatal }, async i => {
|
|
322
341
|
const refined = refineModel(seed + i)
|
|
323
342
|
console.log('Use model to ask: ', refined.getName(), refined.lc_kwargs.model)
|
|
324
343
|
const startedAt = Date.now()
|
|
@@ -378,6 +397,7 @@ export const makeLlmModel = ({
|
|
|
378
397
|
|
|
379
398
|
notifyRef(ref, message)
|
|
380
399
|
return output
|
|
400
|
+
})
|
|
381
401
|
})
|
|
382
402
|
},
|
|
383
403
|
|
|
@@ -388,9 +408,10 @@ export const makeLlmModel = ({
|
|
|
388
408
|
escalation, fatal,
|
|
389
409
|
}: LlmTalkOptions
|
|
390
410
|
) => {
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
411
|
+
return await observeFailure(action, async () => {
|
|
412
|
+
const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
|
|
413
|
+
const seed = ladderSeed(escalation)
|
|
414
|
+
return await withRetry({ retries, outputErrors, fatal }, async i => {
|
|
394
415
|
const refined = refineModel(seed + i)
|
|
395
416
|
console.log('Use model to talk: ', refined.getName(), refined.lc_kwargs.model)
|
|
396
417
|
const startedAt = Date.now()
|
|
@@ -417,6 +438,7 @@ export const makeLlmModel = ({
|
|
|
417
438
|
|
|
418
439
|
notifyRef(ref, message)
|
|
419
440
|
return message
|
|
441
|
+
})
|
|
420
442
|
})
|
|
421
443
|
},
|
|
422
444
|
|
|
@@ -428,12 +450,13 @@ export const makeLlmModel = ({
|
|
|
428
450
|
skills, escalation, fatal,
|
|
429
451
|
}: LlmInvokeOptions<T>
|
|
430
452
|
) => {
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
453
|
+
return await observeFailure(action, async () => {
|
|
454
|
+
const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
|
|
455
|
+
const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
|
|
456
|
+
const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
|
|
434
457
|
|
|
435
|
-
|
|
436
|
-
|
|
458
|
+
const seed = ladderSeed(escalation)
|
|
459
|
+
return await withRetry({ retries, outputErrors, fatal }, async i => {
|
|
437
460
|
const refined = refineModel(seed + i, temperature)
|
|
438
461
|
console.log('Use model invoke: ', refined.getName(), refined.lc_kwargs.model)
|
|
439
462
|
const startedAt = Date.now()
|
|
@@ -469,6 +492,7 @@ export const makeLlmModel = ({
|
|
|
469
492
|
|
|
470
493
|
notifyRef(ref, message)
|
|
471
494
|
return result as T
|
|
495
|
+
})
|
|
472
496
|
})
|
|
473
497
|
},
|
|
474
498
|
|
|
@@ -480,12 +504,13 @@ export const makeLlmModel = ({
|
|
|
480
504
|
escalation, fatal,
|
|
481
505
|
}: LlmRequestOptions
|
|
482
506
|
) => {
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
507
|
+
return await observeFailure(action, async () => {
|
|
508
|
+
const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
|
|
509
|
+
const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
|
|
510
|
+
const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
|
|
486
511
|
|
|
487
|
-
|
|
488
|
-
|
|
512
|
+
const seed = ladderSeed(escalation)
|
|
513
|
+
return await withRetry({ retries, outputErrors, fatal }, async i => {
|
|
489
514
|
const refined = refineModel(seed + i)
|
|
490
515
|
console.log('Use model request: ', refined.getName(), refined.lc_kwargs.model)
|
|
491
516
|
const startedAt = Date.now()
|
|
@@ -525,6 +550,7 @@ export const makeLlmModel = ({
|
|
|
525
550
|
|
|
526
551
|
notifyRef(ref, message)
|
|
527
552
|
return message
|
|
553
|
+
})
|
|
528
554
|
})
|
|
529
555
|
},
|
|
530
556
|
}
|
package/src/types.ts
CHANGED
|
@@ -163,6 +163,19 @@ export interface LlmSpectator {
|
|
|
163
163
|
log: (arg: SpectatorArgument) => Promise<SpectatorEntryLogged>
|
|
164
164
|
/** Optional sink for full diagnostics of a call that returned nothing usable. */
|
|
165
165
|
captureNull?: (capture: NullCapture) => Promise<void>
|
|
166
|
+
/**
|
|
167
|
+
* Optional observer for a call that failed permanently after its retry policy finished.
|
|
168
|
+
*
|
|
169
|
+
* Observers are diagnostics only: the model preserves the original failure even when one
|
|
170
|
+
* cannot receive it. This is deliberately terminal rather than per-attempt so consumers do
|
|
171
|
+
* not turn one exhausted budget into a notification storm.
|
|
172
|
+
*/
|
|
173
|
+
error?: (event: LlmSpectatorError) => Promise<void>
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
export interface LlmSpectatorError {
|
|
177
|
+
action: string
|
|
178
|
+
error: unknown
|
|
166
179
|
}
|
|
167
180
|
|
|
168
181
|
/** Resolves a model of the same role at a different temperature. */
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
import { afterEach, beforeEach, describe, expect, test } from 'bun:test'
|
|
2
|
+
import { DEFAULT_INQUIRY_ANSWER_CHARS, InquiryKind, InquiryPolicy } from '@owlmeans/llm-common'
|
|
3
|
+
import type { Inquiry, InquiryAnswer } from '@owlmeans/llm-common'
|
|
4
|
+
import {
|
|
5
|
+
DEFAULT_EFFORT, executionInquiry, hasInquiryTransport, inquiryTransportFor, InquiryDeclined,
|
|
6
|
+
InquiryUnavailable, isFatalError, makeExecutionService, makeLlmService,
|
|
7
|
+
registerInquiryTransport, releaseInquiryTransport,
|
|
8
|
+
} from '@owlmeans/llm'
|
|
9
|
+
import type {
|
|
10
|
+
Execution, ExecutionService, ExecutionShape, ProjectExecution, ProjectExecutionInput,
|
|
11
|
+
} from '@owlmeans/llm'
|
|
12
|
+
import { offlineConfigs } from './context.js'
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* The runtime half of the inquiry primitive: who answers, under which policy, and what a caller
|
|
16
|
+
* above it is allowed to conclude from a failure. The distinction the specs are here to hold is
|
|
17
|
+
* that "nobody is there" and "the person declined" are different outcomes — one is a channel
|
|
18
|
+
* fault no retry can fix, the other is a decision the run carries on from.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
const KEY = 'spec-inquiry'
|
|
22
|
+
|
|
23
|
+
/**
|
|
24
|
+
* A consumer's OWN shape, every member narrowed — the way `@owlmeans/viable` declares
|
|
25
|
+
* `ViableExecutionShape` and instantiates the service with it.
|
|
26
|
+
*
|
|
27
|
+
* The bridge must take a service built from one without a type argument. Being generic over the
|
|
28
|
+
* SHAPE cannot do that: `S` appears in `ExecutionService<S>` only through indexed accesses
|
|
29
|
+
* (`S['exec']`, `S['projectInput']`, …), which is not an inference site, so it fell back to the
|
|
30
|
+
* bare `ExecutionShape` and then refused every real service contravariantly on `root`. Nothing
|
|
31
|
+
* written against the DEFAULT shape can see that, which is why this spec declares its own.
|
|
32
|
+
*/
|
|
33
|
+
interface SpecExecution extends Execution {
|
|
34
|
+
origin: string
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
interface SpecProjectExecution extends ProjectExecution {
|
|
38
|
+
origin: string
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
interface SpecProjectInput extends ProjectExecutionInput {
|
|
42
|
+
origin: string
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
interface SpecShape extends ExecutionShape {
|
|
46
|
+
exec: SpecExecution
|
|
47
|
+
project: SpecProjectExecution
|
|
48
|
+
projectInput: SpecProjectInput
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
const inquiry = (patch: Partial<Inquiry> = {}): Inquiry => ({
|
|
52
|
+
id: 'q1',
|
|
53
|
+
kind: InquiryKind.Choice,
|
|
54
|
+
question: 'Which database does the origin use?',
|
|
55
|
+
options: [
|
|
56
|
+
{ value: 'postgres', label: 'PostgreSQL' },
|
|
57
|
+
{ value: 'mongo', label: 'MongoDB' },
|
|
58
|
+
],
|
|
59
|
+
...patch,
|
|
60
|
+
})
|
|
61
|
+
|
|
62
|
+
let service: ExecutionService
|
|
63
|
+
let asked: Inquiry[]
|
|
64
|
+
|
|
65
|
+
/** A channel that records what it was asked and answers what the spec told it to. */
|
|
66
|
+
const seat = (answer: (asked: Inquiry) => InquiryAnswer | Promise<InquiryAnswer>): void => {
|
|
67
|
+
registerInquiryTransport(KEY, {
|
|
68
|
+
ask: async question => {
|
|
69
|
+
asked.push(question)
|
|
70
|
+
|
|
71
|
+
return await answer(question)
|
|
72
|
+
},
|
|
73
|
+
})
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const root = (transport?: string, policy?: InquiryPolicy): ProjectExecution => service.root({
|
|
77
|
+
models: () => makeLlmService({ models: offlineConfigs }, `spec-inquiry-llm-${asked.length}`),
|
|
78
|
+
policy: { effort: DEFAULT_EFFORT },
|
|
79
|
+
purpose: { type: 'spec' },
|
|
80
|
+
...(policy != null ? { inquiry: { transport, policy } } : {}),
|
|
81
|
+
})
|
|
82
|
+
|
|
83
|
+
beforeEach(() => {
|
|
84
|
+
asked = []
|
|
85
|
+
service = makeExecutionService(`spec-inquiry-${Math.trunc(performance.now() * 1000)}`)
|
|
86
|
+
})
|
|
87
|
+
|
|
88
|
+
afterEach(() => releaseInquiryTransport(KEY))
|
|
89
|
+
|
|
90
|
+
describe('@owlmeans/llm — the inquiry transport registry', () => {
|
|
91
|
+
test('a channel is seated under a key and released again', () => {
|
|
92
|
+
expect(hasInquiryTransport(KEY)).toBe(false)
|
|
93
|
+
seat(() => ({ inquiryId: 'q1', value: 'mongo' }))
|
|
94
|
+
expect(hasInquiryTransport(KEY)).toBe(true)
|
|
95
|
+
expect(inquiryTransportFor(KEY)).toBeDefined()
|
|
96
|
+
releaseInquiryTransport(KEY)
|
|
97
|
+
expect(hasInquiryTransport(KEY)).toBe(false)
|
|
98
|
+
})
|
|
99
|
+
|
|
100
|
+
test('an unseated key refuses at once rather than waiting for one to arrive', () => {
|
|
101
|
+
expect(() => inquiryTransportFor(KEY)).toThrow(InquiryUnavailable)
|
|
102
|
+
expect(() => inquiryTransportFor(undefined)).toThrow(InquiryUnavailable)
|
|
103
|
+
})
|
|
104
|
+
|
|
105
|
+
test('an absent channel is fatal, so no retry ladder spends itself on it', () => {
|
|
106
|
+
expect(isFatalError(new InquiryUnavailable('x'))).not.toBeNull()
|
|
107
|
+
// A decline is an answer, not a fault: nothing above may abort a run over one.
|
|
108
|
+
expect(isFatalError(new InquiryDeclined('q1'))).toBeNull()
|
|
109
|
+
})
|
|
110
|
+
})
|
|
111
|
+
|
|
112
|
+
describe('@owlmeans/llm — ExecutionService.ask policy matrix', () => {
|
|
113
|
+
test('`ask` reaches the seated channel and caps what comes back', async () => {
|
|
114
|
+
seat(() => ({ inquiryId: 'q1', value: 'mongo', text: 'x'.repeat(10) }))
|
|
115
|
+
const answer = await service.ask(root(KEY, InquiryPolicy.Ask), inquiry())
|
|
116
|
+
expect(answer.value).toBe('mongo')
|
|
117
|
+
expect(answer.truncated).toBeUndefined()
|
|
118
|
+
expect(asked).toHaveLength(1)
|
|
119
|
+
})
|
|
120
|
+
|
|
121
|
+
test('an answer over the ceiling is cut here, and the cut is reported', async () => {
|
|
122
|
+
// The cap belongs to the service, not to the channel: nothing seated by an application is
|
|
123
|
+
// obliged to know the ceiling, and an answer that reached a pipeline state uncut is prose in
|
|
124
|
+
// a state made of keys.
|
|
125
|
+
seat(() => ({ inquiryId: 'q1', text: 'x'.repeat(DEFAULT_INQUIRY_ANSWER_CHARS + 5) }))
|
|
126
|
+
const answer = await service.ask(root(KEY, InquiryPolicy.Ask), inquiry({ kind: InquiryKind.Text }))
|
|
127
|
+
expect(answer.text).toHaveLength(DEFAULT_INQUIRY_ANSWER_CHARS)
|
|
128
|
+
expect(answer.truncated).toBe(true)
|
|
129
|
+
})
|
|
130
|
+
|
|
131
|
+
test('`ask` with nothing seated is an unavailable channel, never a decline', async () => {
|
|
132
|
+
await expect(service.ask(root(KEY, InquiryPolicy.Ask), inquiry()))
|
|
133
|
+
.rejects.toThrow(InquiryUnavailable)
|
|
134
|
+
})
|
|
135
|
+
|
|
136
|
+
test('`default` assumes the default the question carries, or declines — and asks nobody', async () => {
|
|
137
|
+
const exec = root(KEY, InquiryPolicy.Default)
|
|
138
|
+
seat(() => ({ inquiryId: 'q1', value: 'mongo' }))
|
|
139
|
+
expect(await service.ask(exec, inquiry({ default: 'postgres' })))
|
|
140
|
+
.toEqual({ inquiryId: 'q1', value: 'postgres' })
|
|
141
|
+
expect(await service.ask(exec, inquiry())).toEqual({ inquiryId: 'q1', declined: true })
|
|
142
|
+
expect(asked).toHaveLength(0)
|
|
143
|
+
})
|
|
144
|
+
|
|
145
|
+
test('`refuse` declines every question outright', async () => {
|
|
146
|
+
seat(() => ({ inquiryId: 'q1', value: 'mongo' }))
|
|
147
|
+
await expect(service.ask(root(KEY, InquiryPolicy.Refuse), inquiry({ default: 'postgres' })))
|
|
148
|
+
.rejects.toThrow(InquiryDeclined)
|
|
149
|
+
expect(asked).toHaveLength(0)
|
|
150
|
+
})
|
|
151
|
+
|
|
152
|
+
test('no configuration at all behaves as `default` — a run given no channel never blocks', async () => {
|
|
153
|
+
seat(() => ({ inquiryId: 'q1', value: 'mongo' }))
|
|
154
|
+
expect(await service.ask(root(), inquiry({ default: 'postgres' })))
|
|
155
|
+
.toEqual({ inquiryId: 'q1', value: 'postgres' })
|
|
156
|
+
expect(asked).toHaveLength(0)
|
|
157
|
+
})
|
|
158
|
+
})
|
|
159
|
+
|
|
160
|
+
describe('@owlmeans/llm — the executionInquiry bridge', () => {
|
|
161
|
+
test('an unavailable channel reads as "nobody is there"', async () => {
|
|
162
|
+
const ask = executionInquiry(service, root(KEY, InquiryPolicy.Ask))
|
|
163
|
+
expect(await ask(inquiry())).toBeNull()
|
|
164
|
+
})
|
|
165
|
+
|
|
166
|
+
test('a decline escapes — a person who would not decide is not an absent person', async () => {
|
|
167
|
+
const ask = executionInquiry(service, root(KEY, InquiryPolicy.Refuse))
|
|
168
|
+
await expect(ask(inquiry())).rejects.toThrow(InquiryDeclined)
|
|
169
|
+
})
|
|
170
|
+
|
|
171
|
+
test('an answered question comes back whole', async () => {
|
|
172
|
+
seat(() => ({ inquiryId: 'q1', value: 'postgres' }))
|
|
173
|
+
const ask = executionInquiry(service, root(KEY, InquiryPolicy.Ask))
|
|
174
|
+
expect(await ask(inquiry())).toEqual({ inquiryId: 'q1', value: 'postgres' })
|
|
175
|
+
})
|
|
176
|
+
|
|
177
|
+
test('a consumer\'s own execution shape wires through unannotated', async () => {
|
|
178
|
+
seat(() => ({ inquiryId: 'q1', value: 'postgres' }))
|
|
179
|
+
const scoped = makeExecutionService<SpecShape>(`spec-inquiry-shape-${asked.length}`)
|
|
180
|
+
const exec = scoped.root({
|
|
181
|
+
models: () => makeLlmService({ models: offlineConfigs }, 'spec-inquiry-shape-llm'),
|
|
182
|
+
policy: { effort: DEFAULT_EFFORT },
|
|
183
|
+
purpose: { type: 'spec' },
|
|
184
|
+
origin: 'legacy-app',
|
|
185
|
+
inquiry: { transport: KEY, policy: InquiryPolicy.Ask },
|
|
186
|
+
})
|
|
187
|
+
|
|
188
|
+
expect(exec.origin).toBe('legacy-app')
|
|
189
|
+
expect(await executionInquiry(scoped, exec)(inquiry()))
|
|
190
|
+
.toEqual({ inquiryId: 'q1', value: 'postgres' })
|
|
191
|
+
})
|
|
192
|
+
})
|
|
193
|
+
|
|
194
|
+
describe('@owlmeans/llm — the channel survives a snapshot', () => {
|
|
195
|
+
test('a resumed execution asks through the same channel under the same policy', () => {
|
|
196
|
+
const state = service.snapshot(root(KEY, InquiryPolicy.Ask))
|
|
197
|
+
expect(state.inquiry).toEqual({ transport: KEY, policy: InquiryPolicy.Ask })
|
|
198
|
+
expect(service.restore(state).inquiry).toEqual({ transport: KEY, policy: InquiryPolicy.Ask })
|
|
199
|
+
})
|
|
200
|
+
})
|