@hicaru/pi-rlm 0.3.18 → 0.3.20

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -43,24 +43,16 @@ models, recursively. Same Pi session, same tools, same keys: `/rlm` and go. Read
43
43
 
44
44
  ## Benchmarks
45
45
 
46
- **OOLONG (oolong-synth)** — paper-tier long-context suite; latest journal per model,
47
- cost per task from real `costUsd` (older journals estimated at OpenRouter list prices):
46
+ **OOLONG (oolong-synth)** — paper-tier long-context suite (the only suite); latest
47
+ journal per model, cost per task from real `costUsd`:
48
48
 
49
49
  | Model | Score | Avg. cost/task |
50
50
  |-------|-------|----------------|
51
- | `qwen/qwen3.8-27b` | **100%** | $0.0127 |
52
- | `google/gemma-3-27b-it` | 83.3% | $0.0013 |
53
- | `qwen/qwen3-30b-a3b-instruct-2507` | 66.7% | $0.0009 |
54
- | `mistralai/mistral-small-3.2-24b-instruct` | 66.7% | $0.0025 |
51
+ | `zai/glm-4.7` | **83%** | $0.0000 * |
52
+ | `qwen/qwen3.8-27b` | 49% | $0.1038 |
53
+ | `inception/mercury-2.5` | 38% | $0.0052 |
55
54
 
56
- Lite suite `needle` multi-needle recall, `codeqa` repo-QA, `coding` fix task
57
- (7 tasks × 2 passes per model, deterministic graders, no LLM-as-judge):
58
-
59
- | Model | Score | Accuracy |
60
- |-------|-------|----------|
61
- | `qwen/qwen3-30b-a3b-instruct-2507` | **14/14** | **100%** |
62
- | `google/gemma-3-27b-it` | 12/14 | 86% |
63
- | `mistralai/mistral-small-3.2-24b-instruct` | 12/14 | 86% |
55
+ \* glm-4.7 runs on Z.ai's coding-plan endpoint subscription billing, `costUsd` stays $0.
64
56
 
65
57
  Raw per-task rows (correct, recall, latency, tokens, cost) live in
66
58
  `bench/runs/*.jsonl` — one JSONL row per task, committed as history.
@@ -68,17 +60,16 @@ Raw per-task rows (correct, recall, latency, tokens, cost) live in
68
60
  ### Run the benchmarks
69
61
 
70
62
  ```bash
71
- export OPENROUTER_API_KEY=sk-or-... # required env vars are the only key transport
63
+ export OPENROUTER_API_KEY=sk-or-... # required for openrouter/* models
64
+ export ZAI_API_KEY=... # required for zai/* models (coding endpoint)
72
65
 
73
- bun run bench # lite suite: needle + codeqa + coding
74
- bun run bench --suite needle --limit 1 # one suite, first task only
75
- bun run bench --model openrouter/qwen/qwen3-30b-a3b-instruct-2507
66
+ bun run bench # oolong suite, default model (qwen3.8-27b)
67
+ bun run bench --model zai/glm-4.7
68
+ bun run bench --model openrouter/inception/mercury-2.5
76
69
  bun run bench --list # print tasks, no engine / no key
77
- bun run bench --suite paper # paper tier: s_niah, oolong, browsecomp, codeqa_lb (downloads datasets)
78
70
  ```
79
71
 
80
- Suites: `all` (lite, default) · `needle` · `codeqa` · `coding` · `paper` · `s_niah` ·
81
- `oolong` · `browsecomp` · `codeqa_lb`. Regenerate the hero chart:
72
+ One suite (`oolong`). Regenerate the hero chart:
82
73
  `python3 bench/hero.py` (needs `matplotlib`).
83
74
 
84
75
  ## How it works
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@hicaru/pi-rlm",
3
- "version": "0.3.18",
3
+ "version": "0.3.20",
4
4
  "author": "hicaru",
5
5
  "repository": {
6
6
  "type": "git",
@@ -83,6 +83,9 @@ export async function complete1(
83
83
  onThrottlePark: hooks?.onThrottlePark,
84
84
  onThrottleRelease: hooks?.onThrottleRelease,
85
85
  signal: deps.signal,
86
+ // Long-context providers (zai bigmodel TTFB ~1 min per 10k ctx chars) need the
87
+ // per-request wall cap raised from the pi-ai default.
88
+ timeoutMs: config.requestTimeoutMs,
86
89
  }),
87
90
  );
88
91
  inv.limits.addUsage(res.usage);
@@ -54,12 +54,10 @@ export async function emitting<T>(
54
54
  inv.emitter.emitSubcallUpdated({ id, ...u });
55
55
  };
56
56
 
57
- let costUsd = 0;
58
57
  let tokens = 0;
59
58
  let tokensIn = 0;
60
59
  let tokensOut = 0;
61
60
  const track = (u: Usage): void => {
62
- costUsd += u.cost.total;
63
61
  tokens += u.totalTokens;
64
62
  tokensIn += u.input;
65
63
  tokensOut += u.output;
@@ -72,7 +70,6 @@ export async function emitting<T>(
72
70
  id,
73
71
  status: summary.error !== undefined ? "error" : "done",
74
72
  resultPreview: summary.preview,
75
- costUsd,
76
73
  tokens,
77
74
  tokensIn,
78
75
  tokensOut,
@@ -87,7 +84,6 @@ export async function emitting<T>(
87
84
  id,
88
85
  status: "error",
89
86
  resultPreview: msg,
90
- costUsd,
91
87
  tokens,
92
88
  tokensIn,
93
89
  tokensOut,
@@ -24,10 +24,10 @@ function emptyResult(answer: string): RlmResult {
24
24
  return {
25
25
  answer,
26
26
  iterations: 0,
27
- costUsd: 0,
28
27
  inputTokens: 0,
29
28
  outputTokens: 0,
30
29
  durationMs: 0,
30
+ lastStdout: "",
31
31
  };
32
32
  }
33
33
 
@@ -149,8 +149,8 @@ async function childRun(
149
149
 
150
150
  try {
151
151
  const res = await deps.gates.rlm.at(childDepth).run(() => run(input, inv));
152
- inv.limits.addRaw(res.costUsd, res.inputTokens, res.outputTokens);
153
- deps.onChildUsage?.(res.costUsd, res.inputTokens, res.outputTokens);
152
+ inv.limits.addRaw(res.inputTokens, res.outputTokens);
153
+ deps.onChildUsage?.(res.inputTokens, res.outputTokens);
154
154
  if (ledger !== undefined && claimKey !== undefined) ledger.finish(claimKey, res.answer);
155
155
  inv.emitter.emitSubcallUpdated({
156
156
  id: subId,
@@ -54,17 +54,19 @@ interface Waiter {
54
54
  timer?: ReturnType<typeof setTimeout>;
55
55
  }
56
56
 
57
- function notify(waiters: Map<string, Waiter>, taskId: string, entry: TaskEntry): void {
58
- const w = waiters.get(taskId);
59
- if (w === undefined) return;
60
- if (w.timer !== undefined) clearTimeout(w.timer);
61
- w.resolve(entry);
57
+ function notify(waiters: Map<string, Waiter[]>, taskId: string, entry: TaskEntry): void {
58
+ const list = waiters.get(taskId);
59
+ if (list === undefined) return;
62
60
  waiters.delete(taskId);
61
+ for (const w of list) {
62
+ if (w.timer !== undefined) clearTimeout(w.timer);
63
+ w.resolve(entry);
64
+ }
63
65
  }
64
66
 
65
67
  export function createTaskRegistry(): TaskRegistry {
66
68
  const tasks = new Map<string, TaskEntry>();
67
- const waiters = new Map<string, Waiter>();
69
+ const waiters = new Map<string, Waiter[]>();
68
70
  let counter = 0;
69
71
 
70
72
  const spawnDeps: SpawnDeps = {
@@ -92,7 +94,9 @@ export function createTaskRegistry(): TaskRegistry {
92
94
  },
93
95
  resolve(taskId, result) {
94
96
  const entry = tasks.get(taskId);
95
- if (entry === undefined) return;
97
+ // Settle-once: a late resolve after a timeout/reject must not flip status or
98
+ // double-notify; the entry is terminal the moment it leaves "pending".
99
+ if (entry === undefined || entry.status !== "pending") return;
96
100
  entry.status = "done";
97
101
  if (typeof result === "string") {
98
102
  entry.result = result;
@@ -103,14 +107,17 @@ export function createTaskRegistry(): TaskRegistry {
103
107
  },
104
108
  reject(taskId, error) {
105
109
  const entry = tasks.get(taskId);
106
- if (entry === undefined) return;
110
+ // Settle-once (mirror of resolve): reject after resolve/timeout is a no-op.
111
+ if (entry === undefined || entry.status !== "pending") return;
107
112
  entry.status = "error";
108
113
  entry.error = error;
109
- const w = waiters.get(taskId);
110
- if (w !== undefined) {
111
- if (w.timer !== undefined) clearTimeout(w.timer);
112
- w.reject(new Error(error));
114
+ const list = waiters.get(taskId);
115
+ if (list !== undefined) {
113
116
  waiters.delete(taskId);
117
+ for (const w of list) {
118
+ if (w.timer !== undefined) clearTimeout(w.timer);
119
+ w.reject(new Error(error));
120
+ }
114
121
  }
115
122
  },
116
123
  };
@@ -124,19 +131,39 @@ export function createTaskRegistry(): TaskRegistry {
124
131
  resolve(entry);
125
132
  return;
126
133
  }
127
- const timer =
134
+ const w: Waiter = { resolve, reject };
135
+ w.timer =
128
136
  timeoutMs !== undefined
129
137
  ? setTimeout(() => {
130
- waiters.delete(taskId);
131
- const e = tasks.get(taskId);
132
- if (e !== undefined && e.status === "pending") {
133
- e.status = "timeout";
134
- e.error = `Timeout after ${timeoutMs}ms`;
138
+ w.timer = undefined;
139
+ // Remove only THIS waiter — a sibling wait() on the same task stays parked.
140
+ const list = waiters.get(taskId);
141
+ let lastWaiter = true;
142
+ if (list !== undefined) {
143
+ const i = list.indexOf(w);
144
+ if (i >= 0) {
145
+ list.splice(i, 1);
146
+ lastWaiter = list.length === 0;
147
+ if (lastWaiter) waiters.delete(taskId);
148
+ }
149
+ }
150
+ // Timeout is a per-waiter event, not a task property: only when the LAST
151
+ // waiter gives up does the shared entry record "timeout" (settle-once then
152
+ // keeps a late resolve from resurrecting it). While a sibling stays parked,
153
+ // the entry remains pending and resolve() still wakes it.
154
+ if (lastWaiter) {
155
+ const e = tasks.get(taskId);
156
+ if (e !== undefined && e.status === "pending") {
157
+ e.status = "timeout";
158
+ e.error = `Timeout after ${timeoutMs}ms`;
159
+ }
135
160
  }
136
161
  reject(new Error(`Timeout waiting for task ${taskId}`));
137
162
  }, timeoutMs)
138
163
  : undefined;
139
- waiters.set(taskId, { resolve, reject, timer });
164
+ const list = waiters.get(taskId);
165
+ if (list === undefined) waiters.set(taskId, [w]);
166
+ else list.push(w);
140
167
  });
141
168
  },
142
169
  unawaitedIds: () => {
@@ -57,7 +57,7 @@ export interface FinishResult {
57
57
  export interface InvocationLimits {
58
58
  remainingTimeoutMs(): number | undefined;
59
59
  addUsage(usage: Usage): void;
60
- addRaw(costUsd: number, inputTokens: number, outputTokens: number): void;
60
+ addRaw(inputTokens: number, outputTokens: number): void;
61
61
  }
62
62
 
63
63
  export function limitsFromRemaining(
@@ -66,7 +66,7 @@ export function limitsFromRemaining(
66
66
  return Object.freeze({
67
67
  remainingTimeoutMs: () => remaining?.().timeoutMs,
68
68
  addUsage: (_usage: Usage) => {},
69
- addRaw: (_costUsd: number, _inputTokens: number, _outputTokens: number) => {},
69
+ addRaw: (_inputTokens: number, _outputTokens: number) => {},
70
70
  });
71
71
  }
72
72
 
@@ -86,6 +86,9 @@ export interface SubcallConfig {
86
86
  readonly maxDepth: number;
87
87
  readonly subSampling?: Sampling;
88
88
  readonly subSystemPrompt?: string;
89
+ /** Per-request wall cap (ms) for the llm tier — long-context providers (zai bigmodel TTFB
90
+ * ~1 min per 10k ctx chars) abort under the pi-ai default without it. */
91
+ readonly requestTimeoutMs: number;
89
92
  /** v5 TaskLedger: claim/coalesce/echo gates (optional — unwired callers keep ledger off). */
90
93
  readonly enableLedger?: boolean;
91
94
  /** v5: real rlm spawns before demotion to llm (0 = never demote). */
@@ -114,7 +117,7 @@ export interface SubcallHandlerDeps {
114
117
  readonly getChildContext?: () => unknown;
115
118
  readonly getModel?: () => Model<Api>;
116
119
  readonly degrade?: (prompt: string, depth: number) => Promise<string>;
117
- readonly onChildUsage?: (costUsd: number, inputTokens: number, outputTokens: number) => void;
120
+ readonly onChildUsage?: (inputTokens: number, outputTokens: number) => void;
118
121
  readonly trackDetached?: <T>(run: () => Promise<T>) => Promise<T>;
119
122
  /** v5 TaskLedger blackboard shared across the whole run tree (claim/coalesce/echo/demote). */
120
123
  readonly ledger?: TaskLedger;
@@ -28,6 +28,9 @@ export interface CompleteOptions {
28
28
  readonly temperature?: number;
29
29
  readonly reasoning?: ThinkingLevel;
30
30
  readonly signal?: AbortSignal;
31
+ /** Wall-clock cap for ONE provider request (ms); omitted = pi-ai default. Long-context
32
+ * providers (zai bigmodel: ~1 min per 10k ctx chars TTFB) need this raised per config. */
33
+ readonly timeoutMs?: number;
31
34
  /** Retry + adaptive throttle for transient 429/5xx; defaults apply when omitted. */
32
35
  readonly retry?: RetryPolicy;
33
36
  /** v5.1 UX: fired while parked on the rate-limit cooldown ("queued") / when released. */
@@ -116,6 +119,7 @@ export async function modelComplete(messages: readonly ChatMsg[], opts: Complete
116
119
  temperature: opts.temperature,
117
120
  reasoning: effectiveReasoning(opts.model, opts.reasoning),
118
121
  signal: opts.signal,
122
+ timeoutMs: opts.timeoutMs,
119
123
  onResponse: (res) => { note(res.status, res.headers); },
120
124
  },
121
125
  );
@@ -9,7 +9,13 @@ const DEFAULT_SUB_SYSTEM_PROMPT =
9
9
  export const DEFAULT_CONFIG: Readonly<RlmConfig> = Object.freeze({
10
10
  enabled: true,
11
11
  maxDepth: 4,
12
- maxIterations: 30,
12
+ // Max-long runs: the engine may keep iterating until budget/compaction walls hit. The budget
13
+ // cascade and compactionThresholdPct are the real length controls; this ceiling only stops
14
+ // truly runaway loops. Was 30 — capped long tasks prematurely.
15
+ // Strongly oversized on purpose: runs end on FINAL answer / errors / wall-clock long
16
+ // before this bites. History: 30 capped long tasks; the bench's 16 all-failed oolong
17
+ // mid-retrieval. 1000 ≈ 5× the old interactive default, effectively a runaway backstop.
18
+ maxIterations: 1_000,
13
19
  execTimeoutS: 120,
14
20
  requestTimeoutMs: 15 * 60_000,
15
21
  // Session-wide, not per-batch: spawn() puts many requests on the wire at once, so this is
@@ -34,7 +40,11 @@ export const DEFAULT_CONFIG: Readonly<RlmConfig> = Object.freeze({
34
40
  maxErrors: 5,
35
41
  orchestrator: true,
36
42
  compaction: true,
37
- compactionThresholdPct: 0.65,
43
+ // Compact only near the hard ceiling: ≈125K of the default 128K window (0.976). Earlier
44
+ // compaction (0.65) amputated usable working memory long before it was needed.
45
+ // DEPRECATED: ignored since the absolute 256k compaction ceiling (limits.ts); kept so old
46
+ // rlm.json files still load. Do not read this value in new code.
47
+ compactionThresholdPct: 0.976,
38
48
  python: "python3",
39
49
  sandboxInitTimeoutMs: 30_000,
40
50
  contextLoader: true,
@@ -46,9 +56,11 @@ export const DEFAULT_CONFIG: Readonly<RlmConfig> = Object.freeze({
46
56
  enableTokenBudget: true,
47
57
  budgetShare: 0.25,
48
58
  budgetSoftFrac: 0.8,
49
- budgetTaskCap: 400_000,
59
+ budgetTaskCap: 1_000_000,
50
60
  budgetMaxContinuations: 2,
51
- budgetHandoffChars: 4_000,
61
+ // 24K: the handoff must carry Σ + findings + query verbatim — a 4K skeleton is what made
62
+ // long research runs "lose context" on hard-budget continuation (amputated, not lost).
63
+ budgetHandoffChars: 24_000,
52
64
  // v5 TaskLedger blackboard
53
65
  enableLedger: true,
54
66
  rlmBudget: 8,
@@ -197,11 +197,9 @@ export function validateConfig(raw: unknown): Partial<RlmConfig> {
197
197
  }
198
198
  if (typeof r.rootSampling === "object" && r.rootSampling !== null) {
199
199
  const rs = r.rootSampling as Record<string, unknown>;
200
- const rootSampling: { maxTokens?: number; temperature?: number; reasoning?: ThinkingLevel } = {};
200
+ const rootSampling: { maxTokens?: number; reasoning?: ThinkingLevel } = {};
201
201
  const rsMaxTokens = validateNumber(rs.maxTokens, 1);
202
202
  if (rsMaxTokens !== undefined) rootSampling.maxTokens = rsMaxTokens;
203
- const rsTemperature = validateNumber(rs.temperature, 0);
204
- if (rsTemperature !== undefined) rootSampling.temperature = rsTemperature;
205
203
  const rsReasoning = validateThinkingLevel(rs.reasoning);
206
204
  if (rsReasoning !== undefined) rootSampling.reasoning = rsReasoning;
207
205
  out.rootSampling = Object.freeze(rootSampling);
@@ -37,12 +37,19 @@ export function filterContextByPaths(context: unknown, prefixes: readonly string
37
37
  for (let i = 0; i < context.length; i++) {
38
38
  const entry: unknown = context[i];
39
39
  if (!isContextFile(entry)) continue;
40
+ let matched = false;
40
41
  for (let p = 0; p < prefixes.length; p++) {
41
- if (!entry.path.startsWith(prefixes[p])) continue;
42
+ // Path-boundary match: exact file, or a directory prefix ending at a separator.
43
+ // Bare startsWith would let "src/cont" match "src/context/x.ts" (sibling leak).
44
+ const prefix = prefixes[p];
45
+ const bounded = prefix.endsWith("/") ? prefix : `${prefix}/`;
46
+ if (entry.path !== prefix && !entry.path.startsWith(bounded)) continue;
42
47
  hit[p] = true;
43
- out[n++] = entry;
44
- break;
48
+ matched = true;
45
49
  }
50
+ // Emit once even when several prefixes matched the same file (exact + its dir),
51
+ // but every matching prefix still counts as "matched" for the unmatched report.
52
+ if (matched) out[n++] = entry;
46
53
  }
47
54
  out.length = n;
48
55
  const unmatched = new Array<string>(prefixes.length); // pre-allocated, no .push()
@@ -126,8 +126,12 @@ export function namespaceContextFiles(
126
126
  return namespaceContextFilesWithChars(payload, sourceId).files;
127
127
  }
128
128
 
129
+ /** Regex SOURCE for the `ctx/<id>/` namespace — single literal; CTX_PREFIX_RE and the
130
+ * sandbox exec-code (Python, refresh.ts) both derive from it. Never inline a copy. */
131
+ export const CTX_PREFIX_PATTERN_SOURCE = "ctx/[^/]+/";
132
+
129
133
  /** The one `ctx/<id>/` matcher. Never re-declare this regex; use the helpers below. */
130
- const CTX_PREFIX_RE = /^(ctx\/[^/]+\/)/;
134
+ const CTX_PREFIX_RE = new RegExp(`^(${CTX_PREFIX_PATTERN_SOURCE})`);
131
135
 
132
136
  /** Narrow an unknown context entry to a ContextFile. Type guard, never a cast. */
133
137
  export function isContextFile(entry: unknown): entry is ContextFile {
@@ -142,7 +146,7 @@ export function contextEntryPath(entry: unknown): string | undefined {
142
146
  }
143
147
 
144
148
  /** The `ctx/<id>/` prefix owning this path, or undefined. Skips the legacy catch-all. */
145
- function ctxPrefixOf(path: string): string | undefined {
149
+ export function ctxPrefixOf(path: string): string | undefined {
146
150
  const prefix = CTX_PREFIX_RE.exec(path)?.[1];
147
151
  // `ctx/unknown/` is the legacy catch-all: never treat it as an identity.
148
152
  return prefix === undefined || prefix === LEGACY_UNKNOWN_PREFIX ? undefined : prefix;
@@ -7,6 +7,7 @@
7
7
  import { readFile } from "node:fs/promises";
8
8
  import { isAbsolute, relative, resolve } from "node:path";
9
9
  import { estimateTokens } from "../text/tokens.ts";
10
+ import { ctxPrefixOf } from "./namespace.ts";
10
11
  import type { ContextFile } from "./types.ts";
11
12
 
12
13
  /** Paths that look like tool file targets. */
@@ -38,8 +39,12 @@ function pathMatches(entryPath: string, target: string, cwd: string): boolean {
38
39
  const a = normalizeContextPath(entryPath, cwd);
39
40
  const b = normalizeContextPath(target, cwd);
40
41
  if (a === b) return true;
41
- // suffix match for namespaced entries
42
- return entryPath.endsWith("/" + target) || entryPath.endsWith(target);
42
+ // Namespaced entries (`ctx/<id>/…`) must match on the FULL remainder after the
43
+ // namespace — never a bare suffix: "/src/a.ts" also ends "ctx/A/other/src/a.ts",
44
+ // which would refresh an unrelated deeper file (wrong-entry overwrite). The prefix
45
+ // regex itself is owned by namespace.ts (single matcher, DRY).
46
+ const ns = ctxPrefixOf(entryPath);
47
+ return ns !== undefined && entryPath.slice(ns.length) === b;
43
48
  }
44
49
 
45
50
  /**
@@ -72,8 +77,12 @@ export function upsertContextFile(
72
77
  typeof (item).path === "string" &&
73
78
  pathMatches((item as { path: string }).path, path, cwd)
74
79
  ) {
75
- next[n++] = entry;
76
- replaced = true;
80
+ // First match replaces; later duplicates of the same logical file are absorbed,
81
+ // otherwise two matching payload entries would emit the replacement twice.
82
+ if (!replaced) {
83
+ next[n++] = entry;
84
+ replaced = true;
85
+ }
77
86
  } else if (
78
87
  item !== null &&
79
88
  typeof item === "object" &&
@@ -123,15 +132,27 @@ _tokens = ${tokens}
123
132
  _old = context if isinstance(context, list) else []
124
133
  _next = []
125
134
  _found = False
135
+ def _ns_matches(p, t):
136
+ # Anchored namespace match, mirrors host ctxPrefixOf: full remainder after
137
+ # the ctx/<id>/ prefix must EQUAL the target — suffix endswith also hits
138
+ # deeper paths like ctx/A/other/src/a.ts for target src/a.ts (wrong-entry).
139
+ if not p.startswith("ctx/"):
140
+ return False
141
+ rest = p[4:]
142
+ i = rest.find("/")
143
+ if i < 0:
144
+ return False
145
+ return rest[i + 1:] == t
126
146
  for _e in _old:
127
147
  if isinstance(_e, dict) and str(_e.get("path", "")) in (_path, _path.replace("\\\\", "/")):
128
- _next.append({"path": _path, "content": _content, "tokens": _tokens})
129
- _found = True
130
- elif isinstance(_e, dict) and (
131
- str(_e.get("path", "")).endswith("/" + _path) or str(_e.get("path", "")).endswith(_path)
132
- ):
133
- _next.append({"path": str(_e.get("path")), "content": _content, "tokens": _tokens})
134
- _found = True
148
+ if not _found:
149
+ _next.append({"path": _path, "content": _content, "tokens": _tokens})
150
+ _found = True
151
+ elif isinstance(_e, dict) and _ns_matches(str(_e.get("path", "")), _path):
152
+ # First match replaces; later duplicates are absorbed (dedup mirrors host upsert).
153
+ if not _found:
154
+ _next.append({"path": str(_e.get("path")), "content": _content, "tokens": _tokens})
155
+ _found = True
135
156
  else:
136
157
  _next.append(_e)
137
158
  if not _found:
@@ -21,6 +21,21 @@ export function latestAnswerContentOf(results: readonly ReplResult[]): string |
21
21
  return null;
22
22
  }
23
23
 
24
+ /** Cap for the recovered-stdout fallback (P2 §3.4) — it rides `RlmResult`, not history. */
25
+ const LAST_STDOUT_CAP = 4_000;
26
+
27
+ /** Last non-empty stdout across a turn's blocks, capped. P2 §3.4: a run that ends without an
28
+ * `answer[...]` frame still printed its winning value, and re-running the whole task to get it
29
+ * is a waste (and non-deterministic). The bench recovers from here and marks the row
30
+ * `recovered: true`; the engine itself never treats stdout as an answer. */
31
+ export function latestStdoutOf(results: readonly ReplResult[]): string {
32
+ for (let i = results.length - 1; i >= 0; i--) {
33
+ const out = results[i]?.stdout.trim();
34
+ if (out) return out.length > LAST_STDOUT_CAP ? out.slice(-LAST_STDOUT_CAP) : out;
35
+ }
36
+ return "";
37
+ }
38
+
24
39
  /** True if any block in the turn raised an exception. Plain stderr does not count. */
25
40
  export function turnHadError(results: readonly ReplResult[]): boolean {
26
41
  return results.some((r) => r.raised);
@@ -1,10 +1,13 @@
1
1
  /**
2
2
  * Token budget cascade (port of the v4/v5 `budget.py` engine).
3
3
  *
4
- * The budget is the PRIMARY run-length control: cap = budgetShare × model context window,
5
- * one soft wrap-up turn at `softFrac` of the cap, and at the hard cap a deterministic
6
- * handoff (`distillTrajectory`) is handed to a fresh continuation run chain-capped at
7
- * `maxContinuations`. Wall-clock timeouts stay only as hang backstops.
4
+ * The budget is an OUTLIER CEILING, not a progress control (progress = maxIterations):
5
+ * windows at/below COMPACTION_CEILING_TOKENS are never metered; above it the cap is
6
+ * max(ceiling, budgetShare × model context window) the share can only stretch the working
7
+ * budget further out, never cut under the ceiling. One soft wrap-up turn at `softFrac` of
8
+ * the cap, and at the hard cap a deterministic handoff (`distillTrajectory`) is handed to
9
+ * a fresh continuation run — chain-capped at `maxContinuations`. Wall-clock timeouts stay
10
+ * only as hang backstops.
8
11
  *
9
12
  * v5 counts the whole tree (root turns + sub-LLM usage) against the cap; the engine feeds
10
13
  * the run's LimitGuard totals in via `observeTotal` after every turn. Each continuation
@@ -16,6 +19,7 @@ import type { ChatMsg } from "../bridge/model.ts";
16
19
  import type { RlmConfig } from "./types.ts";
17
20
  import type { RunState } from "./run-state.ts";
18
21
  import { compactJSON } from "./run-state.ts";
22
+ import { COMPACTION_CEILING_TOKENS } from "./limits.ts";
19
23
 
20
24
  interface TokenBudgetOptions {
21
25
  readonly softFrac?: number;
@@ -27,22 +31,25 @@ type BudgetState = "" | "soft" | "hard";
27
31
 
28
32
  /** v5 verbatim: the soft wrap-up note prepended to the single turn after crossing soft. */
29
33
  export const WRAP_UP_BUDGET: string =
30
- "[budget] ~80% of your token cap — ONE turn left. If the task is answerable NOW, finalize " +
31
- '(set answer["ready"] = True). Otherwise print a compact findings dump: what is confirmed, ' +
32
- "current file/line or search position, and the exact next step — a fresh continuation picks " +
33
- "it up. Do not start new exploration.";
34
+ "[budget] ~80% of this run's outlier cap — ONE turn left. If the task is answerable NOW, " +
35
+ 'finalize (set answer["ready"] = True). Otherwise print a compact findings dump IN THE ' +
36
+ "NOTES: what is confirmed, current file/line or search position, and the exact next " +
37
+ "step — a continuation picks it up. The task, the packed context and the ledger carry " +
38
+ "over; only repl variables are re-derived. Do not start new exploration.";
34
39
 
35
40
  export const DEFAULT_NEXT_STEP: string =
36
41
  "continue the probing that was in flight, then finalize";
37
42
 
38
43
  /** v5 verbatim template (adapting the finalize spelling to this plugin's REPL). */
39
44
  const HANDOFF_TEMPLATE: string =
40
- "A prior RLM run hit its token cap mid-task.\n" +
45
+ "A prior RLM run hit its outlier token ceiling mid-task.\n" +
41
46
  "You are its continuation — pick up EXACTLY where it stopped.\n\n" +
42
47
  "ORIGINAL TASK:\n{query}\n\n" +
43
48
  "CONFIRMED FINDINGS SO FAR:\n{findings}\n\n" +
44
49
  "CURRENT STATE / LAST ACTIONS:\n{state}\n\n" +
45
50
  "NEXT STEP: {next}\n" +
51
+ "NOTE: repl variables are re-derived, but the task, the packed context and the ledger " +
52
+ "carry over. `add_context()` the same external sources again if you still need them.\n" +
46
53
  "Do not re-do confirmed work; continue from the NEXT STEP and finalize as\n" +
47
54
  'soon as the task is answerable (answer["ready"] = True).';
48
55
 
@@ -110,14 +117,13 @@ export class TokenBudget {
110
117
  /**
111
118
  * Minimum context window (tokens) for the token-budget cascade to engage at all.
112
119
  *
113
- * The formula (window × budgetShare) assumes the window is large enough that a fraction of it
114
- * is a meaningful working budget. Below this floor the derived cap shrinks below a task's FIXED
115
- * overhead (system prompt + per-turn history re-send + sub-LLM calls) and strangles the run
116
- * a 32k window would cap a task at 8k tokens, less than the protocol scaffolding alone.
117
- * So for smaller windows the rule does not apply: the budget is effectively unbounded and runs
118
- * stay bounded by maxIterations / maxErrors / wall-clock instead.
120
+ * LO rule (2025-09-09): windows at/below COMPACTION_CEILING_TOKENS (256k) are never
121
+ * budget-amputated the derived share would shrink below a task's FIXED overhead (system
122
+ * prompt + per-turn history re-send + sub-LLM calls); a 32k window would cap a task at 8k
123
+ * tokens, less than the protocol scaffolding alone. Windows above the ceiling are budgeted
124
+ * AT the ceiling, never below it. Unbounded runs stay bounded by
125
+ * maxIterations / maxErrors / wall-clock instead.
119
126
  */
120
- export const BUDGET_WINDOW_FLOOR = 250_000;
121
127
 
122
128
  /** One TokenBudget construction shape — the cap varies, the policy knobs never do (DRY). */
123
129
  function makeBudget(config: RlmConfig, cap: number): TokenBudget {
@@ -135,9 +141,14 @@ function unboundedBudget(config: RlmConfig): TokenBudget {
135
141
 
136
142
  export function resolveBudget(contextWindow: number | undefined, config: RlmConfig): TokenBudget {
137
143
  const ctx = contextWindow !== undefined && contextWindow > 0 ? contextWindow : 32_000;
138
- if (ctx < BUDGET_WINDOW_FLOOR) return unboundedBudget(config);
139
- const shareCap = Math.floor(ctx * config.budgetShare);
144
+ // The share only stretches the budget BEYOND the absolute ceiling — never under it.
145
+ const shareCap = Math.max(COMPACTION_CEILING_TOKENS, Math.floor(ctx * config.budgetShare));
140
146
  const cap = config.budgetTaskCap > 0 ? Math.min(shareCap, config.budgetTaskCap) : shareCap;
147
+ // LO rule (2025-09-10): small windows are never budget-amputated — but an EXPLICIT
148
+ // budgetTaskCap that actually binds (below shareCap) must still be honored. The old
149
+ // early-return swallowed the explicit cap on windows ≤ the ceiling (task-cap bug).
150
+ const userCapped = config.budgetTaskCap > 0 && config.budgetTaskCap < shareCap;
151
+ if (!userCapped && ctx <= COMPACTION_CEILING_TOKENS) return unboundedBudget(config);
141
152
  return makeBudget(config, Math.max(cap, 1));
142
153
  }
143
154
 
@@ -147,17 +158,27 @@ export function resolveBudget(contextWindow: number | undefined, config: RlmConf
147
158
  */
148
159
  export function truncateMid(text: string, maxChars: number): string {
149
160
  if (text.length <= maxChars) return text;
150
- const half = Math.max(0, maxChars - ELISION_MARK.length) >> 1;
161
+ // Reserve for the WORST-CASE rendered mark, not the shortest (FINDING-5): the elision
162
+ // count is substituted at render time, so a 2+ digit count grows the mark past the
163
+ // length `half` was budgeted from. digits(text.length) upper-bounds digits(elided) —
164
+ // one pass, and the output can never exceed maxChars.
165
+ const reserve = maxChars - (ELISION_MARK.length - 1 + String(text.length).length);
166
+ if (reserve <= 0) {
167
+ // Cap smaller than even a mark-only render: head-truncate to keep the exact
168
+ // ≤ maxChars guarantee instead of emitting an oversized degenerate mark.
169
+ return text.slice(0, maxChars);
170
+ }
171
+ const half = reserve >> 1;
151
172
  const elided = text.length - (half * 2);
152
173
  return text.slice(0, half) + ELISION_MARK.replace("N", String(elided)) + text.slice(text.length - half);
153
174
  }
154
175
 
155
176
  /** Digest/handoff section caps — ONE source: budget.ts's handoff distillation and the root
156
177
  * digest (core/root-digest.ts) must never drift apart on the same trajectory heuristics. */
157
- export const FINDINGS_MAX = 6;
178
+ export const FINDINGS_MAX = 12; // aligns with RUN_STATE_LIMITS.findings — Σ and handoff agree
158
179
  export const FINDINGS_MIN_CHARS = 20;
159
180
  export const STATE_MAX = 8;
160
- const QUERY_CHARS = 800;
181
+ const QUERY_CHARS = 4_000; // full task statement fits; 800 forced the model to "forget" its own goal
161
182
  const STATE_NEEDLE = "REPL stdout";
162
183
  /** Next-step probe shared by the engine handoff and the root digest (one wording source). */
163
184
  export const NEXT_STEP_RE = /next|then|will |todo/i;