@livx.cc/agentx 0.99.36 → 0.99.38

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -199,10 +199,15 @@ interface RunResult {
199
199
  };
200
200
  /** True if ANY turn's usage was estimated (provider gave none) rather than exact — lets the UI mark cost `~`. */
201
201
  usageEstimated?: boolean;
202
- /** Advertised tools the model NAMED in its own prose while the run's invocation ledger shows zero calls
203
- * to them — a fabricated tool use, proven against ground truth rather than guessed from phrasing.
204
- * Present even when the recovery nudge succeeded (it records that the turn needed rescuing); a host
205
- * can use it to tell the user plainly instead of shipping the model's invented result. */
202
+ /** Advertised tools the model NAMED in the reply it is handing back while the run's invocation ledger
203
+ * shows zero calls to them — a fabricated tool use, proven against ground truth rather than guessed
204
+ * from phrasing. A host can use it to tell the user plainly instead of shipping the invented result.
205
+ *
206
+ * Re-evaluated at RETURN time, not frozen at detection: a claim the recovery nudge rescued (the model
207
+ * went and made the real call) is dropped. It used to survive, on the reading that the field recorded
208
+ * "this turn needed rescuing" — but the only consumer turns it into a sentence for the user ("I never
209
+ * ran it"), and on the guard's own success path that sentence is false. Whether a turn needed rescuing
210
+ * is a telemetry question, and the `log.warn` at the nudge already answers it. */
206
211
  fabricatedToolClaims?: string[];
207
212
  error?: unknown;
208
213
  }
@@ -532,8 +537,25 @@ declare class Agent {
532
537
  * - only against the ledger for the WHOLE run, so referring back to a call it genuinely made earlier
533
538
  * ("the screenshot showed an empty grid") is honest and stays silent;
534
539
  * - only on a turn that made no tool calls of its own — a turn that DID call tools is closing normally.
540
+ * (That is the enclosing `toolCalls.length === 0` branch. It is NOT extended to delegated activity:
541
+ * a `cursor/*` turn runs its own tools and then fabricates in the SAME response, which is the exact
542
+ * reported defect — see the call site.)
543
+ *
544
+ * What it deliberately does NOT catch, both verified:
545
+ * - a fabrication that names no tool at all ("Let me take a look at the running app. The app view did
546
+ * not respond in time.") — there is nothing to prove it against. Only prose matching could reach it,
547
+ * and ungating `endsOnAnnouncedAction` for that was tried and reverted (it fires on ordinary English).
548
+ * - a SECOND invented result for a tool the run genuinely called earlier. The ledger is per-run by
549
+ * design, because that is the same fact that keeps an honest back-reference ("the screenshot showed
550
+ * an empty grid") silent. Distinguishing them would need per-claim positional accounting, which buys
551
+ * a rarer case at the cost of the common honest one.
535
552
  */
536
553
  private fabricatedToolNames;
554
+ /** The claims still standing at return time: named in this run's prose, and STILL absent from the
555
+ * invocation ledger. Re-checked here rather than trusted from detection because the ledger moves —
556
+ * the recovery nudge can succeed, and then the host would tell the user "I never ran it" about a
557
+ * call that demonstrably ran. See `RunResult.fabricatedToolClaims`. */
558
+ private survivingClaims;
537
559
  private dispatch;
538
560
  /** `/compact` aims to land the transcript at this fraction of the context window. */
539
561
  private static readonly COMPACT_TARGET;
package/dist/cli.d.ts CHANGED
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env bun
2
- import { H as Hooks, i as RunResult, R as ReasoningEffort, A as Agent } from './Agent-Fr8uwN_l.js';
2
+ import { H as Hooks, i as RunResult, R as ReasoningEffort, A as Agent } from './Agent-DJwg1g8V.js';
3
3
  import { IFilesystem } from '@livx.cc/wcli/core';
4
4
  import { M as Message, U as UserQuestion, H as HostBridge, c as ContentPart, g as MessageContent } from './tools-Byyh_bae.js';
5
5
 
package/dist/cli.js CHANGED
@@ -4099,7 +4099,8 @@ var Agent = class _Agent {
4099
4099
  const kill = (finishReason) => {
4100
4100
  log6.warn(`kill-switch: ${finishReason} (steps=${steps}, tokens=${usage.totalTokens}, budgetTokens=${Math.round(usage.totalTokens - 0.9 * usage.cacheReadTokens)}, ms=${Date.now() - start - this.parkedMs}${finishReason === "timeout" ? `, idle=${Date.now() - lastProgressAt - this.parkedMs}ms` : ""}${this.parkedMs ? ` +${this.parkedMs} parked` : ""})`);
4101
4101
  this.ctx.jobs?.killAll();
4102
- return { text: lastAssistantText(this.transcript), steps, finishReason, messages: this.transcript, usage, usageEstimated, ...this.fabricatedClaims.size ? { fabricatedToolClaims: [...this.fabricatedClaims] } : {} };
4102
+ const claims = this.survivingClaims();
4103
+ return { text: lastAssistantText(this.transcript), steps, finishReason, messages: this.transcript, usage, usageEstimated, ...claims.length ? { fabricatedToolClaims: claims } : {} };
4103
4104
  };
4104
4105
  while (true) {
4105
4106
  if (o.signal?.aborted) return kill("aborted");
@@ -4238,7 +4239,7 @@ var Agent = class _Agent {
4238
4239
  }
4239
4240
  if (toolCalls.length === 0) {
4240
4241
  if (this.drainInjections()) continue;
4241
- const fabricated = o.catchFabricatedToolUse && !closureNudged && delegatedTools.length === 0 ? this.fabricatedToolNames(contentText(res.content ?? "")) : [];
4242
+ const fabricated = o.catchFabricatedToolUse && !closureNudged ? this.fabricatedToolNames(contentText(res.content ?? "")) : [];
4242
4243
  for (const n of fabricated) this.fabricatedClaims.add(n);
4243
4244
  if (fabricated.length && steps < o.maxSteps) {
4244
4245
  closureNudged = true;
@@ -4262,7 +4263,8 @@ var Agent = class _Agent {
4262
4263
  await this.ctx.jobs?.drain();
4263
4264
  this.releaseStaleImages();
4264
4265
  await this.activeHooks?.onStop?.(res.content ?? "");
4265
- return { text: res.content ?? "", steps, finishReason: "stop", messages: this.transcript, usage, usageEstimated, ...this.fabricatedClaims.size ? { fabricatedToolClaims: [...this.fabricatedClaims] } : {} };
4266
+ const claims = this.survivingClaims();
4267
+ return { text: res.content ?? "", steps, finishReason: "stop", messages: this.transcript, usage, usageEstimated, ...claims.length ? { fabricatedToolClaims: claims } : {} };
4266
4268
  }
4267
4269
  const fp = toolCalls.map((tc) => tc.function.name + ":" + (tc.function.arguments ?? "")).join("|");
4268
4270
  repeats = fp === lastFp ? repeats + 1 : 1;
@@ -4414,6 +4416,18 @@ var Agent = class _Agent {
4414
4416
  * - only against the ledger for the WHOLE run, so referring back to a call it genuinely made earlier
4415
4417
  * ("the screenshot showed an empty grid") is honest and stays silent;
4416
4418
  * - only on a turn that made no tool calls of its own — a turn that DID call tools is closing normally.
4419
+ * (That is the enclosing `toolCalls.length === 0` branch. It is NOT extended to delegated activity:
4420
+ * a `cursor/*` turn runs its own tools and then fabricates in the SAME response, which is the exact
4421
+ * reported defect — see the call site.)
4422
+ *
4423
+ * What it deliberately does NOT catch, both verified:
4424
+ * - a fabrication that names no tool at all ("Let me take a look at the running app. The app view did
4425
+ * not respond in time.") — there is nothing to prove it against. Only prose matching could reach it,
4426
+ * and ungating `endsOnAnnouncedAction` for that was tried and reverted (it fires on ordinary English).
4427
+ * - a SECOND invented result for a tool the run genuinely called earlier. The ledger is per-run by
4428
+ * design, because that is the same fact that keeps an honest back-reference ("the screenshot showed
4429
+ * an empty grid") silent. Distinguishing them would need per-claim positional accounting, which buys
4430
+ * a rarer case at the cost of the common honest one.
4417
4431
  */
4418
4432
  fabricatedToolNames(text) {
4419
4433
  const out = [];
@@ -4425,6 +4439,14 @@ var Agent = class _Agent {
4425
4439
  }
4426
4440
  return out;
4427
4441
  }
4442
+ /** The claims still standing at return time: named in this run's prose, and STILL absent from the
4443
+ * invocation ledger. Re-checked here rather than trusted from detection because the ledger moves —
4444
+ * the recovery nudge can succeed, and then the host would tell the user "I never ran it" about a
4445
+ * call that demonstrably ran. See `RunResult.fabricatedToolClaims`. */
4446
+ survivingClaims() {
4447
+ if (!this.fabricatedClaims.size) return [];
4448
+ return [...this.fabricatedClaims].filter((n) => !this.invokedTools.has(n));
4449
+ }
4428
4450
  async dispatch(tc) {
4429
4451
  this.invokedTools.add(tc.function.name);
4430
4452
  const tool = this.activeTools.find((t) => t.name === tc.function.name);