@yagni-app/code-staging 1.2.3-staging.1632.1 → 1.2.3-staging.1661.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -10,8 +10,8 @@
10
10
  *
11
11
  * The turn cap is enforced here via aggregated `usage.turns` (not a pi --max-turns
12
12
  * flag, which is unverified in the pi flag set; that is on the upstream-pi
13
- * wishlist). Ceilings are generous safety bounds, not tight budgets, so a normal
14
- * run never trips them.
13
+ * wishlist). Ceilings are generous safety bounds, not tight budgets: they catch
14
+ * a runaway, not a normal run.
15
15
  */
16
16
  import type { ReviewRound, StageResult } from "./types.js";
17
17
  /** Aggregated usage across a whole run. */
@@ -39,11 +39,21 @@ export interface RunBudget {
39
39
  deadlineMs?: number;
40
40
  }
41
41
  /**
42
- * Generous default ceilings: a 3-round /go fans out review across 3 lenses on top
43
- * of the build half, so dozens of turns and millions of tokens are normal; these
44
- * only catch a runaway. Tunable per lane.
42
+ * Generous default ceilings for interactive /go: review fans out across 3 lenses
43
+ * on top of the build half, so hundreds of turns and millions of tokens are
44
+ * normal; these only catch a runaway. Tunable per lane.
45
45
  */
46
46
  export declare const DEFAULT_RUN_BUDGET: RunBudget;
47
+ /**
48
+ * The sandbox lane's ceilings (a run with a deadline). A sandbox build runs
49
+ * unattended through build, review, fix and review again, and production builds
50
+ * took up to 532 turns on that normal path (2026-10), so 400 stopped them before
51
+ * their last review. 1,200 is over twice the worst seen. Cost and tokens keep the
52
+ * default ceilings, and the build's window and its model capability's request
53
+ * limit still bound the run; the turn ceiling stays a hard stop for a model that
54
+ * loops.
55
+ */
56
+ export declare const SANDBOX_RUN_BUDGET: RunBudget;
47
57
  export declare const EMPTY_RUN_USAGE: RunUsage;
48
58
  export declare function addRunUsage(a: RunUsage, b: RunUsage): RunUsage;
49
59
  /** Sum usage across the build/fix stages AND every review lens result. */
@@ -10,19 +10,32 @@
10
10
  *
11
11
  * The turn cap is enforced here via aggregated `usage.turns` (not a pi --max-turns
12
12
  * flag, which is unverified in the pi flag set; that is on the upstream-pi
13
- * wishlist). Ceilings are generous safety bounds, not tight budgets, so a normal
14
- * run never trips them.
13
+ * wishlist). Ceilings are generous safety bounds, not tight budgets: they catch
14
+ * a runaway, not a normal run.
15
15
  */
16
16
  /**
17
- * Generous default ceilings: a 3-round /go fans out review across 3 lenses on top
18
- * of the build half, so dozens of turns and millions of tokens are normal; these
19
- * only catch a runaway. Tunable per lane.
17
+ * Generous default ceilings for interactive /go: review fans out across 3 lenses
18
+ * on top of the build half, so hundreds of turns and millions of tokens are
19
+ * normal; these only catch a runaway. Tunable per lane.
20
20
  */
21
21
  export const DEFAULT_RUN_BUDGET = {
22
22
  maxTurns: 400,
23
23
  maxCost: 50,
24
24
  maxTokens: 8_000_000,
25
25
  };
26
+ /**
27
+ * The sandbox lane's ceilings (a run with a deadline). A sandbox build runs
28
+ * unattended through build, review, fix and review again, and production builds
29
+ * took up to 532 turns on that normal path (2026-10), so 400 stopped them before
30
+ * their last review. 1,200 is over twice the worst seen. Cost and tokens keep the
31
+ * default ceilings, and the build's window and its model capability's request
32
+ * limit still bound the run; the turn ceiling stays a hard stop for a model that
33
+ * loops.
34
+ */
35
+ export const SANDBOX_RUN_BUDGET = {
36
+ ...DEFAULT_RUN_BUDGET,
37
+ maxTurns: 1_200,
38
+ };
26
39
  export const EMPTY_RUN_USAGE = { turns: 0, cost: 0, input: 0, output: 0 };
27
40
  export function addRunUsage(a, b) {
28
41
  return {
@@ -55,7 +55,8 @@
55
55
  * way and pins the implement diamond for a benchmark lane (see fanout.ts).
56
56
  * `YAGNI_PIPELINE_TIME_LEFT_MS` (the sandbox host) is the run's time left,
57
57
  * anchored to this process's clock at start: the run then stops `budget_exceeded` before a review round or fix it
58
- * cannot finish, rather than being killed partway through one (see budget.ts).
58
+ * cannot finish, rather than being killed partway through one (see budget.ts). Such a run gets the sandbox lane's
59
+ * ceilings (`SANDBOX_RUN_BUDGET`), not interactive /go's.
59
60
  */
60
61
  import { type TreeSnapshotDeps } from "./treeSnapshot.js";
61
62
  import { runPipeline as defaultRunPipeline } from "./orchestrator.js";
@@ -55,7 +55,8 @@
55
55
  * way and pins the implement diamond for a benchmark lane (see fanout.ts).
56
56
  * `YAGNI_PIPELINE_TIME_LEFT_MS` (the sandbox host) is the run's time left,
57
57
  * anchored to this process's clock at start: the run then stops `budget_exceeded` before a review round or fix it
58
- * cannot finish, rather than being killed partway through one (see budget.ts).
58
+ * cannot finish, rather than being killed partway through one (see budget.ts). Such a run gets the sandbox lane's
59
+ * ceilings (`SANDBOX_RUN_BUDGET`), not interactive /go's.
59
60
  */
60
61
  import { readFileSync } from "node:fs";
61
62
  import { continuationResumePlan, parseContinuation } from "./continuation.js";
@@ -67,7 +68,7 @@ import { runPipeline as defaultRunPipeline } from "./orchestrator.js";
67
68
  import { loadGroundingEnabled } from "../grounding.js";
68
69
  import { codeStateHome } from "../stateHome.js";
69
70
  import { parseTierCap, TIER_CAP_ENV } from "./tierCap.js";
70
- import { DEFAULT_RUN_BUDGET, parsePipelineTimeLeft, PIPELINE_TIME_LEFT_ENV } from "./budget.js";
71
+ import { parsePipelineTimeLeft, PIPELINE_TIME_LEFT_ENV, SANDBOX_RUN_BUDGET } from "./budget.js";
71
72
  /** One-line usage copy, shared by every argument error. */
72
73
  export const HEADLESS_GO_USAGE = "Usage: yagni go --headless --ticket-file <path> [--plan-file <path>] [--memo-file <path>] [--continue-file <path>] [--run-id <id>] [--cwd <path>] [--json]";
73
74
  /**
@@ -272,8 +273,9 @@ export async function runHeadlessGo(argv, deps = {}) {
272
273
  writeErr(`Ignoring ${TIER_CAP_ENV}="${rawCap}": not a tier (peak, advanced, standard, efficient). Running uncapped.`);
273
274
  }
274
275
  // The sandbox host's time left for this run, anchored to THIS process's clock
275
- // (the host's may be skewed from the VM's). Absent (every local run, and a
276
- // host that predates it) leaves the budget exactly as it was.
276
+ // (the host's may be skewed from the VM's). Its presence is what makes this a
277
+ // sandbox run, so it also brings the sandbox lane's ceilings. Absent (every
278
+ // local run, and a host that predates it) leaves the budget exactly as it was.
277
279
  const rawTimeLeft = env[PIPELINE_TIME_LEFT_ENV];
278
280
  const timeLeftMs = parsePipelineTimeLeft(rawTimeLeft);
279
281
  if (rawTimeLeft?.trim() && timeLeftMs === undefined) {
@@ -282,7 +284,7 @@ export async function runHeadlessGo(argv, deps = {}) {
282
284
  const deadlineMs = timeLeftMs === undefined ? undefined : (deps.now ?? Date.now)() + timeLeftMs;
283
285
  const budget = deadlineMs === undefined
284
286
  ? deps.budget
285
- : { ...(deps.budget ?? DEFAULT_RUN_BUDGET), deadlineMs: deps.budget?.deadlineMs ?? deadlineMs };
287
+ : { ...(deps.budget ?? SANDBOX_RUN_BUDGET), deadlineMs: deps.budget?.deadlineMs ?? deadlineMs };
286
288
  const cwd = args.cwd ?? deps.cwd ?? process.cwd();
287
289
  const emit = (obj) => write(JSON.stringify(obj));
288
290
  const stream = args.json;
@@ -128,6 +128,14 @@ function incompleteHandoff(r) {
128
128
  * against stderr/errorMessage only — never a stage's real output.
129
129
  */
130
130
  const CREDIT_EXHAUSTION_RE = /out of credits|insufficient_quota|payment required|\b402\b/;
131
+ /**
132
+ * The run-scoped budget shapes: a Run's own approved budget or model-request
133
+ * limit, which the proxy refuses with a 429 typed `insufficient_quota`. The
134
+ * workspace can have plenty of credits when this fires, so it is checked first
135
+ * and never reported as the workspace's credits (production 2026-10-08: a $25
136
+ * run budget read as "the workspace ran out of credits").
137
+ */
138
+ const RUN_BUDGET_EXHAUSTION_RE = /run_budget_exhausted|run_request_limit_exhausted|This Run has reached its model-request or approved packet budget|This Run has used its \d+ model requests/;
131
139
  /**
132
140
  * The silent-402 landmine: when the workspace runs out of credits mid-run, the
133
141
  * proxy 402s every model request, the child pi exits 0 with an EMPTY transcript,
@@ -135,12 +143,16 @@ const CREDIT_EXHAUSTION_RE = /out of credits|insufficient_quota|payment required
135
143
  * "clean" stop with the real cause buried in stderr. Detect it and stop with an
136
144
  * actionable message instead. Gated on the stage having produced no real output
137
145
  * (or having failed outright) so a healthy stage whose stderr happens to contain
138
- * a matching token can never trip it.
146
+ * a matching token can never trip it. A Run that spent its own budget stops the
147
+ * same way, with its own message.
139
148
  */
140
149
  function creditExhaustionReason(r) {
141
150
  if (r.finalOutput.trim().length > 0 && !isFailed(r))
142
151
  return undefined;
143
152
  const hay = `${r.errorMessage ?? ""}\n${r.stderr}`;
153
+ if (RUN_BUDGET_EXHAUSTION_RE.test(hay)) {
154
+ return "this run reached its approved budget (the model proxy refused further calls). Raise the Team's run budget or revise the plan, then re-run /go.";
155
+ }
144
156
  if (!CREDIT_EXHAUSTION_RE.test(hay))
145
157
  return undefined;
146
158
  return "the workspace is out of YAGNI credits (the model proxy returned 402). Top up credits, then re-run /go.";
@@ -1069,6 +1081,13 @@ export async function runPipeline(ticket, deps) {
1069
1081
  log("review_degraded", { round, degradedLenses });
1070
1082
  progress({ kind: "stage_done", stageId: "review", round });
1071
1083
  observedMs.review = clock() - reviewStartedAt;
1084
+ // The verdict first: a clean review is clean and a capped one is capped, even
1085
+ // past a ceiling, since neither spends anything more.
1086
+ const decision = shouldStop(findings, round);
1087
+ if (decision.stop && decision.reason) {
1088
+ stopReason = decision.reason;
1089
+ break;
1090
+ }
1072
1091
  // R3-b: stop honestly if the review fan-out pushed the run over budget, before
1073
1092
  // spending another fix + round. The round's findings are already recorded.
1074
1093
  const overAfterReview = overBudget();
@@ -1077,11 +1096,6 @@ export async function runPipeline(ticket, deps) {
1077
1096
  log("pipeline_stop", { reason: "budget_exceeded", phase: "review", round, detail: overAfterReview });
1078
1097
  break;
1079
1098
  }
1080
- const decision = shouldStop(findings, round);
1081
- if (decision.stop && decision.reason) {
1082
- stopReason = decision.reason;
1083
- break;
1084
- }
1085
1099
  // The review's findings are recorded; a fix that cannot finish in time would
1086
1100
  // only leave them half-applied.
1087
1101
  const fixWontFit = wontFit("fix");
@@ -1132,13 +1146,10 @@ export async function runPipeline(ticket, deps) {
1132
1146
  }
1133
1147
  progress({ kind: "stage_done", stageId: "fix", round });
1134
1148
  stageEvent("fix", "finish", { round });
1135
- // R3-b: stop honestly if this round crossed the budget, before the next round.
1136
- const overAfterFix = overBudget();
1137
- if (overAfterFix) {
1138
- stopReason = "budget_exceeded";
1139
- log("pipeline_stop", { reason: "budget_exceeded", phase: "fix", round, detail: overAfterFix });
1140
- break;
1141
- }
1149
+ // No usage check here: a fix always gets its review, or the run would hand
1150
+ // over a change no review saw. The bound holds anyway: past a ceiling, that
1151
+ // review is the last stage, since the check after it stops the loop (and the
1152
+ // review's own time is still checked at the top of the loop).
1142
1153
  // This whole round (review + fix) completed and the loop will continue, so a
1143
1154
  // crash now can resume at the NEXT round against this fix's output. Snapshot
1144
1155
  // the post-fix tree as the resume consistency key.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@yagni-app/code-staging",
3
- "version": "1.2.3-staging.1632.1",
3
+ "version": "1.2.3-staging.1661.1",
4
4
  "description": "YAGNI Code: a terminal coding agent that already knows your company. One YAGNI login routes the model and grounds the agent in your team's context.",
5
5
  "license": "SEE LICENSE IN LICENSE.md",
6
6
  "author": "YAGNI, Inc. <jack@yagni.app> (https://yagni.app)",
@@ -58,5 +58,5 @@
58
58
  "turndown": "^7.2.4",
59
59
  "typebox": "^1.3.15"
60
60
  },
61
- "yagniSourceSha": "3275eb7ef2a6e502e47bd2e835aa38bd4fb28b37"
61
+ "yagniSourceSha": "78db828d4bed8121dd01039b1ee8ef846f2136a7"
62
62
  }