@livx.cc/agentx 0.99.35 → 0.99.37
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{Agent-USp7SIao.d.ts → Agent-B--fZNJy.d.ts} +51 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +75 -4
- package/dist/cli.js.map +1 -1
- package/dist/index.d.ts +2 -2
- package/dist/index.js +75 -4
- package/dist/index.js.map +1 -1
- package/dist/tools.shell.js.map +1 -1
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { a as AgentOptions, H as Hooks, i as RunResult, A as Agent } from './Agent-
|
|
2
|
-
export { C as ChatFragment, b as CompactResult, D as DEFAULT_MUTATING, c as Decision, P as PermissionOptions, d as PermissionPolicy, e as PermissionRule, f as PreToolUseDecision, R as ReasoningEffort, g as RecordingHooks, h as RecordingLifecycle, T as ToolUse, j as ToolUseMeta, k as composeHooks, l as estimateTokens, p as planMode, r as reasoningToChatFragment } from './Agent-
|
|
1
|
+
import { a as AgentOptions, H as Hooks, i as RunResult, A as Agent } from './Agent-B--fZNJy.js';
|
|
2
|
+
export { C as ChatFragment, b as CompactResult, D as DEFAULT_MUTATING, c as Decision, P as PermissionOptions, d as PermissionPolicy, e as PermissionRule, f as PreToolUseDecision, R as ReasoningEffort, g as RecordingHooks, h as RecordingLifecycle, T as ToolUse, j as ToolUseMeta, k as composeHooks, l as estimateTokens, p as planMode, r as reasoningToChatFragment } from './Agent-B--fZNJy.js';
|
|
3
3
|
export { MODEL_ALIASES, ModelSwitch, ResolveModelSwitchOpts, modelShortLabel, resolveModelAlias, resolveModelSwitch } from './models.js';
|
|
4
4
|
import { IFilesystem, FileMetadata } from '@livx.cc/wcli/core';
|
|
5
5
|
export { CommandExecutor, FileMetadata, IFilesystem, IndexedDbFilesystem, MemFilesystem, registerHeadlessCommands } from '@livx.cc/wcli/core';
|
package/dist/index.js
CHANGED
|
@@ -263,10 +263,10 @@ function stripContent(content, stub) {
|
|
|
263
263
|
function endsOnAnnouncedAction(text) {
|
|
264
264
|
const sentences = text.split(/(?<=[.!?])\s+|\n+/).map((s) => s.trim()).filter(Boolean);
|
|
265
265
|
const last = sentences[sentences.length - 1];
|
|
266
|
-
if (!last || NEXT_STEP.test(last)) return false;
|
|
266
|
+
if (!last || NEXT_STEP.test(last) || SIGN_OFF.test(last)) return false;
|
|
267
267
|
return ANNOUNCED_ACTION.test(last);
|
|
268
268
|
}
|
|
269
|
-
var log2, p_sep, IMAGE_ELIDE_STUB, IMAGE_REF_SCHEME, REF_REGISTRY_MAX, refRegistry, isImageRefUrl, bearsImage, ANNOUNCED_ACTION, NEXT_STEP;
|
|
269
|
+
var log2, p_sep, IMAGE_ELIDE_STUB, IMAGE_REF_SCHEME, REF_REGISTRY_MAX, refRegistry, isImageRefUrl, bearsImage, ANNOUNCED_ACTION, NEXT_STEP, SIGN_OFF;
|
|
270
270
|
var init_llm = __esm({
|
|
271
271
|
"src/llm.ts"() {
|
|
272
272
|
"use strict";
|
|
@@ -281,6 +281,7 @@ var init_llm = __esm({
|
|
|
281
281
|
bearsImage = (m) => messageHasImage(m.content) || toolImageRef(m.content) != null;
|
|
282
282
|
ANNOUNCED_ACTION = /^(?:also\s+|now\s+|then\s+)*(?:let me\b|i'?ll\b|i will\b|i'?m going to\b|(?:re-?)?(?:check|search|scan|read|verify|run|inspect|expand|look|examin|continu|proceed|try|fetch|load|open|test|build|grep|find|review|trac|dig|explor)\w*ing\b)/i;
|
|
283
283
|
NEXT_STEP = /^next (?:step|run|up)\b/i;
|
|
284
|
+
SIGN_OFF = /^let me know\b/i;
|
|
284
285
|
}
|
|
285
286
|
});
|
|
286
287
|
|
|
@@ -3692,6 +3693,13 @@ var AgentOptions = class {
|
|
|
3692
3693
|
* permission-asking instead of answers (see the nudge text in runLoop).
|
|
3693
3694
|
* Fires at most once per run and only for delegated runtimes (native turns never emit the trigger). */
|
|
3694
3695
|
closeDelegatedTurns = true;
|
|
3696
|
+
/** Catch a FABRICATED tool use: a turn that emits no tool call at all, yet whose text names an advertised
|
|
3697
|
+
* tool this run has never actually invoked — the model announcing `look_at_app` and then writing its own
|
|
3698
|
+
* "it didn't respond in time" result. No call-reactive layer can see it (there IS no call), and to every
|
|
3699
|
+
* surface it is indistinguishable from a real failure. The discriminator is the per-run invocation ledger,
|
|
3700
|
+
* not prose: a named tool with zero invocations is provable. On detection the loop nudges ONCE for a real
|
|
3701
|
+
* call, and records the names in `RunResult.fabricatedToolClaims` either way. */
|
|
3702
|
+
catchFabricatedToolUse = true;
|
|
3695
3703
|
/** Fold the dropped middle of an over-long transcript into a synthetic summary (edge-safe, no LLM). Off => drop-oldest. */
|
|
3696
3704
|
compaction;
|
|
3697
3705
|
/** Add `Checkpoint`/`Rollback` tools (requires the fs to be an OverlayFilesystem). */
|
|
@@ -3758,6 +3766,11 @@ var Agent = class _Agent {
|
|
|
3758
3766
|
lastTrimNotified = 0;
|
|
3759
3767
|
// last auto-trim drop count surfaced via host.notify (dedup)
|
|
3760
3768
|
activeTools = [];
|
|
3769
|
+
/** Per-run ground truth: every tool name actually INVOKED this run (native dispatch + delegated
|
|
3770
|
+
* runtime activity). The ledger that makes a fabricated tool claim provable. Reset per run. */
|
|
3771
|
+
invokedTools = /* @__PURE__ */ new Set();
|
|
3772
|
+
/** Advertised tools this run's text claimed to have used while the ledger showed zero calls. */
|
|
3773
|
+
fabricatedClaims = /* @__PURE__ */ new Set();
|
|
3761
3774
|
activeHooks;
|
|
3762
3775
|
// composed: user hooks + plan-mode + permissions
|
|
3763
3776
|
prepared = false;
|
|
@@ -3949,6 +3962,8 @@ var Agent = class _Agent {
|
|
|
3949
3962
|
await this.ensureFs();
|
|
3950
3963
|
this.prepared = false;
|
|
3951
3964
|
this.started = false;
|
|
3965
|
+
this.invokedTools.clear();
|
|
3966
|
+
this.fabricatedClaims.clear();
|
|
3952
3967
|
const systemPrompt = await this.prepare(contentText(task));
|
|
3953
3968
|
const startCtx = await this.fireSessionStart();
|
|
3954
3969
|
const userContent = await this.applyPromptSubmit(task);
|
|
@@ -4049,7 +4064,7 @@ var Agent = class _Agent {
|
|
|
4049
4064
|
const kill = (finishReason) => {
|
|
4050
4065
|
log6.warn(`kill-switch: ${finishReason} (steps=${steps}, tokens=${usage.totalTokens}, budgetTokens=${Math.round(usage.totalTokens - 0.9 * usage.cacheReadTokens)}, ms=${Date.now() - start - this.parkedMs}${finishReason === "timeout" ? `, idle=${Date.now() - lastProgressAt - this.parkedMs}ms` : ""}${this.parkedMs ? ` +${this.parkedMs} parked` : ""})`);
|
|
4051
4066
|
this.ctx.jobs?.killAll();
|
|
4052
|
-
return { text: lastAssistantText(this.transcript), steps, finishReason, messages: this.transcript, usage, usageEstimated };
|
|
4067
|
+
return { text: lastAssistantText(this.transcript), steps, finishReason, messages: this.transcript, usage, usageEstimated, ...this.fabricatedClaims.size ? { fabricatedToolClaims: [...this.fabricatedClaims] } : {} };
|
|
4053
4068
|
};
|
|
4054
4069
|
while (true) {
|
|
4055
4070
|
if (o.signal?.aborted) return kill("aborted");
|
|
@@ -4188,6 +4203,17 @@ var Agent = class _Agent {
|
|
|
4188
4203
|
}
|
|
4189
4204
|
if (toolCalls.length === 0) {
|
|
4190
4205
|
if (this.drainInjections()) continue;
|
|
4206
|
+
const fabricated = o.catchFabricatedToolUse && !closureNudged ? this.fabricatedToolNames(contentText(res.content ?? "")) : [];
|
|
4207
|
+
for (const n of fabricated) this.fabricatedClaims.add(n);
|
|
4208
|
+
if (fabricated.length && steps < o.maxSteps) {
|
|
4209
|
+
closureNudged = true;
|
|
4210
|
+
log6.warn(`fabricated tool use: text names ${fabricated.join(", ")} but the run invoked ${this.invokedTools.size ? [...this.invokedTools].join(", ") : "nothing"} \u2014 demanding the real call`);
|
|
4211
|
+
this.transcript.push({
|
|
4212
|
+
role: "user",
|
|
4213
|
+
content: `You referred to ${fabricated.map((n) => `\`${n}\``).join(", ")} in your reply, but you never actually called ${fabricated.length > 1 ? "those tools" : "that tool"} \u2014 no tool call was made, so any result you reported for ${fabricated.length > 1 ? "them" : "it"} was invented. Either call ${fabricated.length > 1 ? "them" : "it"} for real now, or answer without ${fabricated.length > 1 ? "them" : "it"} and do NOT claim any outcome \u2014 including a failure or a timeout \u2014 for a call you did not make.`
|
|
4214
|
+
});
|
|
4215
|
+
continue;
|
|
4216
|
+
}
|
|
4191
4217
|
if (o.closeDelegatedTurns && res.endedWithoutClosing && !closureNudged && steps < o.maxSteps) {
|
|
4192
4218
|
closureNudged = true;
|
|
4193
4219
|
log6.verbose("delegated turn went silent after a tool \u2014 nudging a closing answer");
|
|
@@ -4201,7 +4227,7 @@ var Agent = class _Agent {
|
|
|
4201
4227
|
await this.ctx.jobs?.drain();
|
|
4202
4228
|
this.releaseStaleImages();
|
|
4203
4229
|
await this.activeHooks?.onStop?.(res.content ?? "");
|
|
4204
|
-
return { text: res.content ?? "", steps, finishReason: "stop", messages: this.transcript, usage, usageEstimated };
|
|
4230
|
+
return { text: res.content ?? "", steps, finishReason: "stop", messages: this.transcript, usage, usageEstimated, ...this.fabricatedClaims.size ? { fabricatedToolClaims: [...this.fabricatedClaims] } : {} };
|
|
4205
4231
|
}
|
|
4206
4232
|
const fp = toolCalls.map((tc) => tc.function.name + ":" + (tc.function.arguments ?? "")).join("|");
|
|
4207
4233
|
repeats = fp === lastFp ? repeats + 1 : 1;
|
|
@@ -4317,6 +4343,7 @@ var Agent = class _Agent {
|
|
|
4317
4343
|
prev.status = a.status;
|
|
4318
4344
|
}
|
|
4319
4345
|
sawTool = true;
|
|
4346
|
+
this.invokedTools.add(a.name);
|
|
4320
4347
|
trailing = "";
|
|
4321
4348
|
}
|
|
4322
4349
|
if (chunk.content) {
|
|
@@ -4332,7 +4359,51 @@ var Agent = class _Agent {
|
|
|
4332
4359
|
const endedWithoutClosing = sawTool && (!trailing.trim() || endsOnAnnouncedAction(trailing));
|
|
4333
4360
|
return { content, ...finishReason ? { finishReason } : {}, ...toolCalls ? { toolCalls } : {}, ...usage ? { usage } : {}, ...delegatedTools.length ? { delegatedTools } : {}, ...endedWithoutClosing ? { endedWithoutClosing: true } : {} };
|
|
4334
4361
|
}
|
|
4362
|
+
/**
|
|
4363
|
+
* FABRICATED TOOL USE — the failure no call-reactive layer can see, because there IS no call.
|
|
4364
|
+
*
|
|
4365
|
+
* The model announces a tool ("I'll take a look with look_at_app"), emits no tool_use block at all, and
|
|
4366
|
+
* then writes a plausible RESULT for the call it never made ("the app view didn't respond in time"). To
|
|
4367
|
+
* every surface that turn is indistinguishable from a genuine timeout: the user reads a credible failure
|
|
4368
|
+
* sentence, the logs show no tool call because none was made, and nothing errors. Malformed-call handling
|
|
4369
|
+
* (dispatch's earlyError) cannot reach it — that path needs a tool_use block to exist.
|
|
4370
|
+
*
|
|
4371
|
+
* The discriminator is NOT prose. It is the per-run invocation ledger, which is ground truth: an advertised
|
|
4372
|
+
* tool NAMED in the model's own text while `invokedTools` holds zero calls to it is a provable claim about
|
|
4373
|
+
* an event that did not happen, not a guess about phrasing. Matching is on the exact registered tool name
|
|
4374
|
+
* at word boundaries, so it cannot fire on ordinary English the way `endsOnAnnouncedAction` can.
|
|
4375
|
+
*
|
|
4376
|
+
* Deliberately narrow, because a false positive costs a metered step:
|
|
4377
|
+
* - only tools with a distinctive name (>= 4 chars, containing `_`, `-` or a capital) — a tool named `run`
|
|
4378
|
+
* or `web` would match half of English prose;
|
|
4379
|
+
* - only against the ledger for the WHOLE run, so referring back to a call it genuinely made earlier
|
|
4380
|
+
* ("the screenshot showed an empty grid") is honest and stays silent;
|
|
4381
|
+
* - only on a turn that made no tool calls of its own — a turn that DID call tools is closing normally.
|
|
4382
|
+
* (That is the enclosing `toolCalls.length === 0` branch. It is NOT extended to delegated activity:
|
|
4383
|
+
* a `cursor/*` turn runs its own tools and then fabricates in the SAME response, which is the exact
|
|
4384
|
+
* reported defect — see the call site.)
|
|
4385
|
+
*
|
|
4386
|
+
* What it deliberately does NOT catch, both verified:
|
|
4387
|
+
* - a fabrication that names no tool at all ("Let me take a look at the running app. The app view did
|
|
4388
|
+
* not respond in time.") — there is nothing to prove it against. Only prose matching could reach it,
|
|
4389
|
+
* and ungating `endsOnAnnouncedAction` for that was tried and reverted (it fires on ordinary English).
|
|
4390
|
+
* - a SECOND invented result for a tool the run genuinely called earlier. The ledger is per-run by
|
|
4391
|
+
* design, because that is the same fact that keeps an honest back-reference ("the screenshot showed
|
|
4392
|
+
* an empty grid") silent. Distinguishing them would need per-claim positional accounting, which buys
|
|
4393
|
+
* a rarer case at the cost of the common honest one.
|
|
4394
|
+
*/
|
|
4395
|
+
fabricatedToolNames(text) {
|
|
4396
|
+
const out = [];
|
|
4397
|
+
for (const t of this.activeTools) {
|
|
4398
|
+
if (this.invokedTools.has(t.name)) continue;
|
|
4399
|
+
if (t.name.length < 4 || !/[_\-A-Z]/.test(t.name)) continue;
|
|
4400
|
+
const re = new RegExp(`(?:^|[^\\w])${t.name.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}(?:[^\\w]|$)`);
|
|
4401
|
+
if (re.test(text)) out.push(t.name);
|
|
4402
|
+
}
|
|
4403
|
+
return out;
|
|
4404
|
+
}
|
|
4335
4405
|
async dispatch(tc) {
|
|
4406
|
+
this.invokedTools.add(tc.function.name);
|
|
4336
4407
|
const tool = this.activeTools.find((t) => t.name === tc.function.name);
|
|
4337
4408
|
let args = {};
|
|
4338
4409
|
let earlyError;
|