@livx.cc/agentx 0.99.35 → 0.99.36
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/{Agent-USp7SIao.d.ts → Agent-Fr8uwN_l.d.ts} +39 -0
- package/dist/cli.d.ts +1 -1
- package/dist/cli.js +63 -4
- package/dist/cli.js.map +1 -1
- package/dist/index.d.ts +2 -2
- package/dist/index.js +63 -4
- package/dist/index.js.map +1 -1
- package/dist/tools.shell.js.map +1 -1
- package/package.json +1 -1
|
@@ -199,6 +199,11 @@ interface RunResult {
|
|
|
199
199
|
};
|
|
200
200
|
/** True if ANY turn's usage was estimated (provider gave none) rather than exact — lets the UI mark cost `~`. */
|
|
201
201
|
usageEstimated?: boolean;
|
|
202
|
+
/** Advertised tools the model NAMED in its own prose while the run's invocation ledger shows zero calls
|
|
203
|
+
* to them — a fabricated tool use, proven against ground truth rather than guessed from phrasing.
|
|
204
|
+
* Present even when the recovery nudge succeeded (it records that the turn needed rescuing); a host
|
|
205
|
+
* can use it to tell the user plainly instead of shipping the model's invented result. */
|
|
206
|
+
fabricatedToolClaims?: string[];
|
|
202
207
|
error?: unknown;
|
|
203
208
|
}
|
|
204
209
|
/** What a manual `/compact` actually freed (token estimates, ~4 chars/token). */
|
|
@@ -333,6 +338,13 @@ declare class AgentOptions {
|
|
|
333
338
|
* permission-asking instead of answers (see the nudge text in runLoop).
|
|
334
339
|
* Fires at most once per run and only for delegated runtimes (native turns never emit the trigger). */
|
|
335
340
|
closeDelegatedTurns: boolean;
|
|
341
|
+
/** Catch a FABRICATED tool use: a turn that emits no tool call at all, yet whose text names an advertised
|
|
342
|
+
* tool this run has never actually invoked — the model announcing `look_at_app` and then writing its own
|
|
343
|
+
* "it didn't respond in time" result. No call-reactive layer can see it (there IS no call), and to every
|
|
344
|
+
* surface it is indistinguishable from a real failure. The discriminator is the per-run invocation ledger,
|
|
345
|
+
* not prose: a named tool with zero invocations is provable. On detection the loop nudges ONCE for a real
|
|
346
|
+
* call, and records the names in `RunResult.fabricatedToolClaims` either way. */
|
|
347
|
+
catchFabricatedToolUse: boolean;
|
|
336
348
|
/** Fold the dropped middle of an over-long transcript into a synthetic summary (edge-safe, no LLM). Off => drop-oldest. */
|
|
337
349
|
compaction?: {
|
|
338
350
|
maxMessages: number;
|
|
@@ -389,6 +401,11 @@ declare class Agent {
|
|
|
389
401
|
private ctx;
|
|
390
402
|
private lastTrimNotified;
|
|
391
403
|
private activeTools;
|
|
404
|
+
/** Per-run ground truth: every tool name actually INVOKED this run (native dispatch + delegated
|
|
405
|
+
* runtime activity). The ledger that makes a fabricated tool claim provable. Reset per run. */
|
|
406
|
+
private invokedTools;
|
|
407
|
+
/** Advertised tools this run's text claimed to have used while the ledger showed zero calls. */
|
|
408
|
+
private fabricatedClaims;
|
|
392
409
|
private activeHooks?;
|
|
393
410
|
private prepared;
|
|
394
411
|
private systemPromptCache;
|
|
@@ -495,6 +512,28 @@ declare class Agent {
|
|
|
495
512
|
* non-result the model produced by going silent. Callers retry it rather than report "done". */
|
|
496
513
|
private isEmptyStop;
|
|
497
514
|
private consumeStream;
|
|
515
|
+
/**
|
|
516
|
+
* FABRICATED TOOL USE — the failure no call-reactive layer can see, because there IS no call.
|
|
517
|
+
*
|
|
518
|
+
* The model announces a tool ("I'll take a look with look_at_app"), emits no tool_use block at all, and
|
|
519
|
+
* then writes a plausible RESULT for the call it never made ("the app view didn't respond in time"). To
|
|
520
|
+
* every surface that turn is indistinguishable from a genuine timeout: the user reads a credible failure
|
|
521
|
+
* sentence, the logs show no tool call because none was made, and nothing errors. Malformed-call handling
|
|
522
|
+
* (dispatch's earlyError) cannot reach it — that path needs a tool_use block to exist.
|
|
523
|
+
*
|
|
524
|
+
* The discriminator is NOT prose. It is the per-run invocation ledger, which is ground truth: an advertised
|
|
525
|
+
* tool NAMED in the model's own text while `invokedTools` holds zero calls to it is a provable claim about
|
|
526
|
+
* an event that did not happen, not a guess about phrasing. Matching is on the exact registered tool name
|
|
527
|
+
* at word boundaries, so it cannot fire on ordinary English the way `endsOnAnnouncedAction` can.
|
|
528
|
+
*
|
|
529
|
+
* Deliberately narrow, because a false positive costs a metered step:
|
|
530
|
+
* - only tools with a distinctive name (>= 4 chars, containing `_`, `-` or a capital) — a tool named `run`
|
|
531
|
+
* or `web` would match half of English prose;
|
|
532
|
+
* - only against the ledger for the WHOLE run, so referring back to a call it genuinely made earlier
|
|
533
|
+
* ("the screenshot showed an empty grid") is honest and stays silent;
|
|
534
|
+
* - only on a turn that made no tool calls of its own — a turn that DID call tools is closing normally.
|
|
535
|
+
*/
|
|
536
|
+
private fabricatedToolNames;
|
|
498
537
|
private dispatch;
|
|
499
538
|
/** `/compact` aims to land the transcript at this fraction of the context window. */
|
|
500
539
|
private static readonly COMPACT_TARGET;
|
package/dist/cli.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env bun
|
|
2
|
-
import { H as Hooks, i as RunResult, R as ReasoningEffort, A as Agent } from './Agent-
|
|
2
|
+
import { H as Hooks, i as RunResult, R as ReasoningEffort, A as Agent } from './Agent-Fr8uwN_l.js';
|
|
3
3
|
import { IFilesystem } from '@livx.cc/wcli/core';
|
|
4
4
|
import { M as Message, U as UserQuestion, H as HostBridge, c as ContentPart, g as MessageContent } from './tools-Byyh_bae.js';
|
|
5
5
|
|
package/dist/cli.js
CHANGED
|
@@ -261,10 +261,10 @@ function stripContent(content, stub) {
|
|
|
261
261
|
function endsOnAnnouncedAction(text) {
|
|
262
262
|
const sentences = text.split(/(?<=[.!?])\s+|\n+/).map((s) => s.trim()).filter(Boolean);
|
|
263
263
|
const last = sentences[sentences.length - 1];
|
|
264
|
-
if (!last || NEXT_STEP.test(last)) return false;
|
|
264
|
+
if (!last || NEXT_STEP.test(last) || SIGN_OFF.test(last)) return false;
|
|
265
265
|
return ANNOUNCED_ACTION.test(last);
|
|
266
266
|
}
|
|
267
|
-
var log2, p_sep, IMAGE_ELIDE_STUB, IMAGE_REF_SCHEME, REF_REGISTRY_MAX, refRegistry, isImageRefUrl, bearsImage, ANNOUNCED_ACTION, NEXT_STEP;
|
|
267
|
+
var log2, p_sep, IMAGE_ELIDE_STUB, IMAGE_REF_SCHEME, REF_REGISTRY_MAX, refRegistry, isImageRefUrl, bearsImage, ANNOUNCED_ACTION, NEXT_STEP, SIGN_OFF;
|
|
268
268
|
var init_llm = __esm({
|
|
269
269
|
"src/llm.ts"() {
|
|
270
270
|
"use strict";
|
|
@@ -279,6 +279,7 @@ var init_llm = __esm({
|
|
|
279
279
|
bearsImage = (m) => messageHasImage(m.content) || toolImageRef(m.content) != null;
|
|
280
280
|
ANNOUNCED_ACTION = /^(?:also\s+|now\s+|then\s+)*(?:let me\b|i'?ll\b|i will\b|i'?m going to\b|(?:re-?)?(?:check|search|scan|read|verify|run|inspect|expand|look|examin|continu|proceed|try|fetch|load|open|test|build|grep|find|review|trac|dig|explor)\w*ing\b)/i;
|
|
281
281
|
NEXT_STEP = /^next (?:step|run|up)\b/i;
|
|
282
|
+
SIGN_OFF = /^let me know\b/i;
|
|
282
283
|
}
|
|
283
284
|
});
|
|
284
285
|
|
|
@@ -3727,6 +3728,13 @@ var AgentOptions = class {
|
|
|
3727
3728
|
* permission-asking instead of answers (see the nudge text in runLoop).
|
|
3728
3729
|
* Fires at most once per run and only for delegated runtimes (native turns never emit the trigger). */
|
|
3729
3730
|
closeDelegatedTurns = true;
|
|
3731
|
+
/** Catch a FABRICATED tool use: a turn that emits no tool call at all, yet whose text names an advertised
|
|
3732
|
+
* tool this run has never actually invoked — the model announcing `look_at_app` and then writing its own
|
|
3733
|
+
* "it didn't respond in time" result. No call-reactive layer can see it (there IS no call), and to every
|
|
3734
|
+
* surface it is indistinguishable from a real failure. The discriminator is the per-run invocation ledger,
|
|
3735
|
+
* not prose: a named tool with zero invocations is provable. On detection the loop nudges ONCE for a real
|
|
3736
|
+
* call, and records the names in `RunResult.fabricatedToolClaims` either way. */
|
|
3737
|
+
catchFabricatedToolUse = true;
|
|
3730
3738
|
/** Fold the dropped middle of an over-long transcript into a synthetic summary (edge-safe, no LLM). Off => drop-oldest. */
|
|
3731
3739
|
compaction;
|
|
3732
3740
|
/** Add `Checkpoint`/`Rollback` tools (requires the fs to be an OverlayFilesystem). */
|
|
@@ -3793,6 +3801,11 @@ var Agent = class _Agent {
|
|
|
3793
3801
|
lastTrimNotified = 0;
|
|
3794
3802
|
// last auto-trim drop count surfaced via host.notify (dedup)
|
|
3795
3803
|
activeTools = [];
|
|
3804
|
+
/** Per-run ground truth: every tool name actually INVOKED this run (native dispatch + delegated
|
|
3805
|
+
* runtime activity). The ledger that makes a fabricated tool claim provable. Reset per run. */
|
|
3806
|
+
invokedTools = /* @__PURE__ */ new Set();
|
|
3807
|
+
/** Advertised tools this run's text claimed to have used while the ledger showed zero calls. */
|
|
3808
|
+
fabricatedClaims = /* @__PURE__ */ new Set();
|
|
3796
3809
|
activeHooks;
|
|
3797
3810
|
// composed: user hooks + plan-mode + permissions
|
|
3798
3811
|
prepared = false;
|
|
@@ -3984,6 +3997,8 @@ var Agent = class _Agent {
|
|
|
3984
3997
|
await this.ensureFs();
|
|
3985
3998
|
this.prepared = false;
|
|
3986
3999
|
this.started = false;
|
|
4000
|
+
this.invokedTools.clear();
|
|
4001
|
+
this.fabricatedClaims.clear();
|
|
3987
4002
|
const systemPrompt = await this.prepare(contentText(task));
|
|
3988
4003
|
const startCtx = await this.fireSessionStart();
|
|
3989
4004
|
const userContent = await this.applyPromptSubmit(task);
|
|
@@ -4084,7 +4099,7 @@ var Agent = class _Agent {
|
|
|
4084
4099
|
const kill = (finishReason) => {
|
|
4085
4100
|
log6.warn(`kill-switch: ${finishReason} (steps=${steps}, tokens=${usage.totalTokens}, budgetTokens=${Math.round(usage.totalTokens - 0.9 * usage.cacheReadTokens)}, ms=${Date.now() - start - this.parkedMs}${finishReason === "timeout" ? `, idle=${Date.now() - lastProgressAt - this.parkedMs}ms` : ""}${this.parkedMs ? ` +${this.parkedMs} parked` : ""})`);
|
|
4086
4101
|
this.ctx.jobs?.killAll();
|
|
4087
|
-
return { text: lastAssistantText(this.transcript), steps, finishReason, messages: this.transcript, usage, usageEstimated };
|
|
4102
|
+
return { text: lastAssistantText(this.transcript), steps, finishReason, messages: this.transcript, usage, usageEstimated, ...this.fabricatedClaims.size ? { fabricatedToolClaims: [...this.fabricatedClaims] } : {} };
|
|
4088
4103
|
};
|
|
4089
4104
|
while (true) {
|
|
4090
4105
|
if (o.signal?.aborted) return kill("aborted");
|
|
@@ -4223,6 +4238,17 @@ var Agent = class _Agent {
|
|
|
4223
4238
|
}
|
|
4224
4239
|
if (toolCalls.length === 0) {
|
|
4225
4240
|
if (this.drainInjections()) continue;
|
|
4241
|
+
const fabricated = o.catchFabricatedToolUse && !closureNudged && delegatedTools.length === 0 ? this.fabricatedToolNames(contentText(res.content ?? "")) : [];
|
|
4242
|
+
for (const n of fabricated) this.fabricatedClaims.add(n);
|
|
4243
|
+
if (fabricated.length && steps < o.maxSteps) {
|
|
4244
|
+
closureNudged = true;
|
|
4245
|
+
log6.warn(`fabricated tool use: text names ${fabricated.join(", ")} but the run invoked ${this.invokedTools.size ? [...this.invokedTools].join(", ") : "nothing"} \u2014 demanding the real call`);
|
|
4246
|
+
this.transcript.push({
|
|
4247
|
+
role: "user",
|
|
4248
|
+
content: `You referred to ${fabricated.map((n) => `\`${n}\``).join(", ")} in your reply, but you never actually called ${fabricated.length > 1 ? "those tools" : "that tool"} \u2014 no tool call was made, so any result you reported for ${fabricated.length > 1 ? "them" : "it"} was invented. Either call ${fabricated.length > 1 ? "them" : "it"} for real now, or answer without ${fabricated.length > 1 ? "them" : "it"} and do NOT claim any outcome \u2014 including a failure or a timeout \u2014 for a call you did not make.`
|
|
4249
|
+
});
|
|
4250
|
+
continue;
|
|
4251
|
+
}
|
|
4226
4252
|
if (o.closeDelegatedTurns && res.endedWithoutClosing && !closureNudged && steps < o.maxSteps) {
|
|
4227
4253
|
closureNudged = true;
|
|
4228
4254
|
log6.verbose("delegated turn went silent after a tool \u2014 nudging a closing answer");
|
|
@@ -4236,7 +4262,7 @@ var Agent = class _Agent {
|
|
|
4236
4262
|
await this.ctx.jobs?.drain();
|
|
4237
4263
|
this.releaseStaleImages();
|
|
4238
4264
|
await this.activeHooks?.onStop?.(res.content ?? "");
|
|
4239
|
-
return { text: res.content ?? "", steps, finishReason: "stop", messages: this.transcript, usage, usageEstimated };
|
|
4265
|
+
return { text: res.content ?? "", steps, finishReason: "stop", messages: this.transcript, usage, usageEstimated, ...this.fabricatedClaims.size ? { fabricatedToolClaims: [...this.fabricatedClaims] } : {} };
|
|
4240
4266
|
}
|
|
4241
4267
|
const fp = toolCalls.map((tc) => tc.function.name + ":" + (tc.function.arguments ?? "")).join("|");
|
|
4242
4268
|
repeats = fp === lastFp ? repeats + 1 : 1;
|
|
@@ -4352,6 +4378,7 @@ var Agent = class _Agent {
|
|
|
4352
4378
|
prev.status = a.status;
|
|
4353
4379
|
}
|
|
4354
4380
|
sawTool = true;
|
|
4381
|
+
this.invokedTools.add(a.name);
|
|
4355
4382
|
trailing = "";
|
|
4356
4383
|
}
|
|
4357
4384
|
if (chunk.content) {
|
|
@@ -4367,7 +4394,39 @@ var Agent = class _Agent {
|
|
|
4367
4394
|
const endedWithoutClosing = sawTool && (!trailing.trim() || endsOnAnnouncedAction(trailing));
|
|
4368
4395
|
return { content, ...finishReason ? { finishReason } : {}, ...toolCalls ? { toolCalls } : {}, ...usage ? { usage } : {}, ...delegatedTools.length ? { delegatedTools } : {}, ...endedWithoutClosing ? { endedWithoutClosing: true } : {} };
|
|
4369
4396
|
}
|
|
4397
|
+
/**
|
|
4398
|
+
* FABRICATED TOOL USE — the failure no call-reactive layer can see, because there IS no call.
|
|
4399
|
+
*
|
|
4400
|
+
* The model announces a tool ("I'll take a look with look_at_app"), emits no tool_use block at all, and
|
|
4401
|
+
* then writes a plausible RESULT for the call it never made ("the app view didn't respond in time"). To
|
|
4402
|
+
* every surface that turn is indistinguishable from a genuine timeout: the user reads a credible failure
|
|
4403
|
+
* sentence, the logs show no tool call because none was made, and nothing errors. Malformed-call handling
|
|
4404
|
+
* (dispatch's earlyError) cannot reach it — that path needs a tool_use block to exist.
|
|
4405
|
+
*
|
|
4406
|
+
* The discriminator is NOT prose. It is the per-run invocation ledger, which is ground truth: an advertised
|
|
4407
|
+
* tool NAMED in the model's own text while `invokedTools` holds zero calls to it is a provable claim about
|
|
4408
|
+
* an event that did not happen, not a guess about phrasing. Matching is on the exact registered tool name
|
|
4409
|
+
* at word boundaries, so it cannot fire on ordinary English the way `endsOnAnnouncedAction` can.
|
|
4410
|
+
*
|
|
4411
|
+
* Deliberately narrow, because a false positive costs a metered step:
|
|
4412
|
+
* - only tools with a distinctive name (>= 4 chars, containing `_`, `-` or a capital) — a tool named `run`
|
|
4413
|
+
* or `web` would match half of English prose;
|
|
4414
|
+
* - only against the ledger for the WHOLE run, so referring back to a call it genuinely made earlier
|
|
4415
|
+
* ("the screenshot showed an empty grid") is honest and stays silent;
|
|
4416
|
+
* - only on a turn that made no tool calls of its own — a turn that DID call tools is closing normally.
|
|
4417
|
+
*/
|
|
4418
|
+
fabricatedToolNames(text) {
|
|
4419
|
+
const out = [];
|
|
4420
|
+
for (const t of this.activeTools) {
|
|
4421
|
+
if (this.invokedTools.has(t.name)) continue;
|
|
4422
|
+
if (t.name.length < 4 || !/[_\-A-Z]/.test(t.name)) continue;
|
|
4423
|
+
const re = new RegExp(`(?:^|[^\\w])${t.name.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")}(?:[^\\w]|$)`);
|
|
4424
|
+
if (re.test(text)) out.push(t.name);
|
|
4425
|
+
}
|
|
4426
|
+
return out;
|
|
4427
|
+
}
|
|
4370
4428
|
async dispatch(tc) {
|
|
4429
|
+
this.invokedTools.add(tc.function.name);
|
|
4371
4430
|
const tool = this.activeTools.find((t) => t.name === tc.function.name);
|
|
4372
4431
|
let args = {};
|
|
4373
4432
|
let earlyError;
|