@yagni-app/code-staging 1.2.3-staging.1659.1 → 1.2.3-staging.1686.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/extension/pipeline/budget.d.ts +23 -8
- package/dist/extension/pipeline/budget.js +27 -9
- package/dist/extension/pipeline/headlessGo.d.ts +2 -1
- package/dist/extension/pipeline/headlessGo.js +7 -5
- package/dist/extension/pipeline/orchestrator.js +11 -12
- package/dist/extension/pipeline/types.d.ts +4 -3
- package/dist/extension/pipeline/types.js +5 -4
- package/dist/extension/pipeline/verify.d.ts +6 -2
- package/dist/extension/pipeline/verify.js +8 -4
- package/package.json +2 -2
|
@@ -10,8 +10,8 @@
|
|
|
10
10
|
*
|
|
11
11
|
* The turn cap is enforced here via aggregated `usage.turns` (not a pi --max-turns
|
|
12
12
|
* flag, which is unverified in the pi flag set; that is on the upstream-pi
|
|
13
|
-
* wishlist). Ceilings are generous safety bounds, not tight budgets
|
|
14
|
-
*
|
|
13
|
+
* wishlist). Ceilings are generous safety bounds, not tight budgets: they catch
|
|
14
|
+
* a runaway, not a normal run.
|
|
15
15
|
*/
|
|
16
16
|
import type { ReviewRound, StageResult } from "./types.js";
|
|
17
17
|
/** Aggregated usage across a whole run. */
|
|
@@ -39,11 +39,21 @@ export interface RunBudget {
|
|
|
39
39
|
deadlineMs?: number;
|
|
40
40
|
}
|
|
41
41
|
/**
|
|
42
|
-
* Generous default ceilings
|
|
43
|
-
* of the build half, so
|
|
44
|
-
* only catch a runaway. Tunable per lane.
|
|
42
|
+
* Generous default ceilings for interactive /go: review fans out across 3 lenses
|
|
43
|
+
* on top of the build half, so hundreds of turns and millions of tokens are
|
|
44
|
+
* normal; these only catch a runaway. Tunable per lane.
|
|
45
45
|
*/
|
|
46
46
|
export declare const DEFAULT_RUN_BUDGET: RunBudget;
|
|
47
|
+
/**
|
|
48
|
+
* The sandbox lane's ceilings (a run with a deadline). A sandbox build runs
|
|
49
|
+
* unattended through build, review, fix and review again, and production builds
|
|
50
|
+
* took up to 532 turns on that normal path (2026-10), so 400 stopped them before
|
|
51
|
+
* their last review. 1,200 is over twice the worst seen. Cost and tokens keep the
|
|
52
|
+
* default ceilings, and the build's window and its model capability's request
|
|
53
|
+
* limit still bound the run; the turn ceiling stays a hard stop for a model that
|
|
54
|
+
* loops.
|
|
55
|
+
*/
|
|
56
|
+
export declare const SANDBOX_RUN_BUDGET: RunBudget;
|
|
47
57
|
export declare const EMPTY_RUN_USAGE: RunUsage;
|
|
48
58
|
export declare function addRunUsage(a: RunUsage, b: RunUsage): RunUsage;
|
|
49
59
|
/** Sum usage across the build/fix stages AND every review lens result. */
|
|
@@ -84,12 +94,17 @@ export declare function stageWontFit(budget: RunBudget, kind: TimedStageKind, ex
|
|
|
84
94
|
* than the time actually left, since the host stops the run at the deadline.
|
|
85
95
|
*/
|
|
86
96
|
export declare const STAGE_WALL_FLOOR_MS: number;
|
|
87
|
-
/**
|
|
97
|
+
/**
|
|
98
|
+
* Time the verify gate is allowed after the last write, on top of the stage
|
|
99
|
+
* estimates: the verify run's own ceiling (`VERIFY_TIMEOUT_MS`, twenty
|
|
100
|
+
* minutes). Five until 2026-10-09, when verify could run ten, so a long stage
|
|
101
|
+
* could leave the gate less time than it may take.
|
|
102
|
+
*/
|
|
88
103
|
export declare const VERIFY_ALLOWANCE_MS: number;
|
|
89
104
|
/**
|
|
90
105
|
* What must still run after a stage stops: a review round, one fix, and the
|
|
91
|
-
* verify gate. Held back from the stage's wall clock
|
|
92
|
-
* leaves the run time to review what it wrote.
|
|
106
|
+
* verify gate (4 + 6 + 20 = 30 minutes). Held back from the stage's wall clock
|
|
107
|
+
* so a long stage still leaves the run time to review and verify what it wrote.
|
|
93
108
|
*/
|
|
94
109
|
export declare const STAGE_WALL_RESERVE_MS: number;
|
|
95
110
|
/**
|
|
@@ -10,19 +10,32 @@
|
|
|
10
10
|
*
|
|
11
11
|
* The turn cap is enforced here via aggregated `usage.turns` (not a pi --max-turns
|
|
12
12
|
* flag, which is unverified in the pi flag set; that is on the upstream-pi
|
|
13
|
-
* wishlist). Ceilings are generous safety bounds, not tight budgets
|
|
14
|
-
*
|
|
13
|
+
* wishlist). Ceilings are generous safety bounds, not tight budgets: they catch
|
|
14
|
+
* a runaway, not a normal run.
|
|
15
15
|
*/
|
|
16
16
|
/**
|
|
17
|
-
* Generous default ceilings
|
|
18
|
-
* of the build half, so
|
|
19
|
-
* only catch a runaway. Tunable per lane.
|
|
17
|
+
* Generous default ceilings for interactive /go: review fans out across 3 lenses
|
|
18
|
+
* on top of the build half, so hundreds of turns and millions of tokens are
|
|
19
|
+
* normal; these only catch a runaway. Tunable per lane.
|
|
20
20
|
*/
|
|
21
21
|
export const DEFAULT_RUN_BUDGET = {
|
|
22
22
|
maxTurns: 400,
|
|
23
23
|
maxCost: 50,
|
|
24
24
|
maxTokens: 8_000_000,
|
|
25
25
|
};
|
|
26
|
+
/**
|
|
27
|
+
* The sandbox lane's ceilings (a run with a deadline). A sandbox build runs
|
|
28
|
+
* unattended through build, review, fix and review again, and production builds
|
|
29
|
+
* took up to 532 turns on that normal path (2026-10), so 400 stopped them before
|
|
30
|
+
* their last review. 1,200 is over twice the worst seen. Cost and tokens keep the
|
|
31
|
+
* default ceilings, and the build's window and its model capability's request
|
|
32
|
+
* limit still bound the run; the turn ceiling stays a hard stop for a model that
|
|
33
|
+
* loops.
|
|
34
|
+
*/
|
|
35
|
+
export const SANDBOX_RUN_BUDGET = {
|
|
36
|
+
...DEFAULT_RUN_BUDGET,
|
|
37
|
+
maxTurns: 1_200,
|
|
38
|
+
};
|
|
26
39
|
export const EMPTY_RUN_USAGE = { turns: 0, cost: 0, input: 0, output: 0 };
|
|
27
40
|
export function addRunUsage(a, b) {
|
|
28
41
|
return {
|
|
@@ -113,12 +126,17 @@ export function stageWontFit(budget, kind, expectedMs, nowMs = Date.now()) {
|
|
|
113
126
|
* than the time actually left, since the host stops the run at the deadline.
|
|
114
127
|
*/
|
|
115
128
|
export const STAGE_WALL_FLOOR_MS = 20 * 60_000;
|
|
116
|
-
/**
|
|
117
|
-
|
|
129
|
+
/**
|
|
130
|
+
* Time the verify gate is allowed after the last write, on top of the stage
|
|
131
|
+
* estimates: the verify run's own ceiling (`VERIFY_TIMEOUT_MS`, twenty
|
|
132
|
+
* minutes). Five until 2026-10-09, when verify could run ten, so a long stage
|
|
133
|
+
* could leave the gate less time than it may take.
|
|
134
|
+
*/
|
|
135
|
+
export const VERIFY_ALLOWANCE_MS = 20 * 60_000;
|
|
118
136
|
/**
|
|
119
137
|
* What must still run after a stage stops: a review round, one fix, and the
|
|
120
|
-
* verify gate. Held back from the stage's wall clock
|
|
121
|
-
* leaves the run time to review what it wrote.
|
|
138
|
+
* verify gate (4 + 6 + 20 = 30 minutes). Held back from the stage's wall clock
|
|
139
|
+
* so a long stage still leaves the run time to review and verify what it wrote.
|
|
122
140
|
*/
|
|
123
141
|
export const STAGE_WALL_RESERVE_MS = DEFAULT_STAGE_ESTIMATE_MS.review + DEFAULT_STAGE_ESTIMATE_MS.fix + VERIFY_ALLOWANCE_MS;
|
|
124
142
|
/**
|
|
@@ -55,7 +55,8 @@
|
|
|
55
55
|
* way and pins the implement diamond for a benchmark lane (see fanout.ts).
|
|
56
56
|
* `YAGNI_PIPELINE_TIME_LEFT_MS` (the sandbox host) is the run's time left,
|
|
57
57
|
* anchored to this process's clock at start: the run then stops `budget_exceeded` before a review round or fix it
|
|
58
|
-
* cannot finish, rather than being killed partway through one (see budget.ts).
|
|
58
|
+
* cannot finish, rather than being killed partway through one (see budget.ts). Such a run gets the sandbox lane's
|
|
59
|
+
* ceilings (`SANDBOX_RUN_BUDGET`), not interactive /go's.
|
|
59
60
|
*/
|
|
60
61
|
import { type TreeSnapshotDeps } from "./treeSnapshot.js";
|
|
61
62
|
import { runPipeline as defaultRunPipeline } from "./orchestrator.js";
|
|
@@ -55,7 +55,8 @@
|
|
|
55
55
|
* way and pins the implement diamond for a benchmark lane (see fanout.ts).
|
|
56
56
|
* `YAGNI_PIPELINE_TIME_LEFT_MS` (the sandbox host) is the run's time left,
|
|
57
57
|
* anchored to this process's clock at start: the run then stops `budget_exceeded` before a review round or fix it
|
|
58
|
-
* cannot finish, rather than being killed partway through one (see budget.ts).
|
|
58
|
+
* cannot finish, rather than being killed partway through one (see budget.ts). Such a run gets the sandbox lane's
|
|
59
|
+
* ceilings (`SANDBOX_RUN_BUDGET`), not interactive /go's.
|
|
59
60
|
*/
|
|
60
61
|
import { readFileSync } from "node:fs";
|
|
61
62
|
import { continuationResumePlan, parseContinuation } from "./continuation.js";
|
|
@@ -67,7 +68,7 @@ import { runPipeline as defaultRunPipeline } from "./orchestrator.js";
|
|
|
67
68
|
import { loadGroundingEnabled } from "../grounding.js";
|
|
68
69
|
import { codeStateHome } from "../stateHome.js";
|
|
69
70
|
import { parseTierCap, TIER_CAP_ENV } from "./tierCap.js";
|
|
70
|
-
import {
|
|
71
|
+
import { parsePipelineTimeLeft, PIPELINE_TIME_LEFT_ENV, SANDBOX_RUN_BUDGET } from "./budget.js";
|
|
71
72
|
/** One-line usage copy, shared by every argument error. */
|
|
72
73
|
export const HEADLESS_GO_USAGE = "Usage: yagni go --headless --ticket-file <path> [--plan-file <path>] [--memo-file <path>] [--continue-file <path>] [--run-id <id>] [--cwd <path>] [--json]";
|
|
73
74
|
/**
|
|
@@ -272,8 +273,9 @@ export async function runHeadlessGo(argv, deps = {}) {
|
|
|
272
273
|
writeErr(`Ignoring ${TIER_CAP_ENV}="${rawCap}": not a tier (peak, advanced, standard, efficient). Running uncapped.`);
|
|
273
274
|
}
|
|
274
275
|
// The sandbox host's time left for this run, anchored to THIS process's clock
|
|
275
|
-
// (the host's may be skewed from the VM's).
|
|
276
|
-
//
|
|
276
|
+
// (the host's may be skewed from the VM's). Its presence is what makes this a
|
|
277
|
+
// sandbox run, so it also brings the sandbox lane's ceilings. Absent (every
|
|
278
|
+
// local run, and a host that predates it) leaves the budget exactly as it was.
|
|
277
279
|
const rawTimeLeft = env[PIPELINE_TIME_LEFT_ENV];
|
|
278
280
|
const timeLeftMs = parsePipelineTimeLeft(rawTimeLeft);
|
|
279
281
|
if (rawTimeLeft?.trim() && timeLeftMs === undefined) {
|
|
@@ -282,7 +284,7 @@ export async function runHeadlessGo(argv, deps = {}) {
|
|
|
282
284
|
const deadlineMs = timeLeftMs === undefined ? undefined : (deps.now ?? Date.now)() + timeLeftMs;
|
|
283
285
|
const budget = deadlineMs === undefined
|
|
284
286
|
? deps.budget
|
|
285
|
-
: { ...(deps.budget ??
|
|
287
|
+
: { ...(deps.budget ?? SANDBOX_RUN_BUDGET), deadlineMs: deps.budget?.deadlineMs ?? deadlineMs };
|
|
286
288
|
const cwd = args.cwd ?? deps.cwd ?? process.cwd();
|
|
287
289
|
const emit = (obj) => write(JSON.stringify(obj));
|
|
288
290
|
const stream = args.json;
|
|
@@ -1081,6 +1081,13 @@ export async function runPipeline(ticket, deps) {
|
|
|
1081
1081
|
log("review_degraded", { round, degradedLenses });
|
|
1082
1082
|
progress({ kind: "stage_done", stageId: "review", round });
|
|
1083
1083
|
observedMs.review = clock() - reviewStartedAt;
|
|
1084
|
+
// The verdict first: a clean review is clean and a capped one is capped, even
|
|
1085
|
+
// past a ceiling, since neither spends anything more.
|
|
1086
|
+
const decision = shouldStop(findings, round);
|
|
1087
|
+
if (decision.stop && decision.reason) {
|
|
1088
|
+
stopReason = decision.reason;
|
|
1089
|
+
break;
|
|
1090
|
+
}
|
|
1084
1091
|
// R3-b: stop honestly if the review fan-out pushed the run over budget, before
|
|
1085
1092
|
// spending another fix + round. The round's findings are already recorded.
|
|
1086
1093
|
const overAfterReview = overBudget();
|
|
@@ -1089,11 +1096,6 @@ export async function runPipeline(ticket, deps) {
|
|
|
1089
1096
|
log("pipeline_stop", { reason: "budget_exceeded", phase: "review", round, detail: overAfterReview });
|
|
1090
1097
|
break;
|
|
1091
1098
|
}
|
|
1092
|
-
const decision = shouldStop(findings, round);
|
|
1093
|
-
if (decision.stop && decision.reason) {
|
|
1094
|
-
stopReason = decision.reason;
|
|
1095
|
-
break;
|
|
1096
|
-
}
|
|
1097
1099
|
// The review's findings are recorded; a fix that cannot finish in time would
|
|
1098
1100
|
// only leave them half-applied.
|
|
1099
1101
|
const fixWontFit = wontFit("fix");
|
|
@@ -1144,13 +1146,10 @@ export async function runPipeline(ticket, deps) {
|
|
|
1144
1146
|
}
|
|
1145
1147
|
progress({ kind: "stage_done", stageId: "fix", round });
|
|
1146
1148
|
stageEvent("fix", "finish", { round });
|
|
1147
|
-
//
|
|
1148
|
-
|
|
1149
|
-
|
|
1150
|
-
|
|
1151
|
-
log("pipeline_stop", { reason: "budget_exceeded", phase: "fix", round, detail: overAfterFix });
|
|
1152
|
-
break;
|
|
1153
|
-
}
|
|
1149
|
+
// No usage check here: a fix always gets its review, or the run would hand
|
|
1150
|
+
// over a change no review saw. The bound holds anyway: past a ceiling, that
|
|
1151
|
+
// review is the last stage, since the check after it stops the loop (and the
|
|
1152
|
+
// review's own time is still checked at the top of the loop).
|
|
1154
1153
|
// This whole round (review + fix) completed and the loop will continue, so a
|
|
1155
1154
|
// crash now can resume at the NEXT round against this fix's output. Snapshot
|
|
1156
1155
|
// the post-fix tree as the resume consistency key.
|
|
@@ -544,9 +544,10 @@ export interface ResiliencePolicy {
|
|
|
544
544
|
}
|
|
545
545
|
/**
|
|
546
546
|
* Default resilience policy. Generous timeouts so a legitimately long but live
|
|
547
|
-
* stage is never killed: the idle window
|
|
548
|
-
*
|
|
549
|
-
*
|
|
547
|
+
* stage is never killed: the idle window is the check for a stuck model (four
|
|
548
|
+
* minutes until 2026-10-09; fifteen now, since a long test run or a slow
|
|
549
|
+
* model turn emits nothing for minutes and is not stuck), and the wall window
|
|
550
|
+
* only bounds a runaway that keeps emitting events. The wall is 60 minutes for an interactive /go, where a
|
|
550
551
|
* person can abort; a run with a deadline replaces it per stage with the time
|
|
551
552
|
* left (budget.ts#stageWall). Three attempts with short, jittered
|
|
552
553
|
* backoff so a single transient blip across the up-to-8 child spawns no longer
|
|
@@ -21,16 +21,17 @@ export const MAX_REVIEW_ROUNDS = 2;
|
|
|
21
21
|
export const MAX_FIX_TURNS = 3;
|
|
22
22
|
/**
|
|
23
23
|
* Default resilience policy. Generous timeouts so a legitimately long but live
|
|
24
|
-
* stage is never killed: the idle window
|
|
25
|
-
*
|
|
26
|
-
*
|
|
24
|
+
* stage is never killed: the idle window is the check for a stuck model (four
|
|
25
|
+
* minutes until 2026-10-09; fifteen now, since a long test run or a slow
|
|
26
|
+
* model turn emits nothing for minutes and is not stuck), and the wall window
|
|
27
|
+
* only bounds a runaway that keeps emitting events. The wall is 60 minutes for an interactive /go, where a
|
|
27
28
|
* person can abort; a run with a deadline replaces it per stage with the time
|
|
28
29
|
* left (budget.ts#stageWall). Three attempts with short, jittered
|
|
29
30
|
* backoff so a single transient blip across the up-to-8 child spawns no longer
|
|
30
31
|
* terminates the whole run.
|
|
31
32
|
*/
|
|
32
33
|
export const DEFAULT_RESILIENCE_POLICY = {
|
|
33
|
-
idleTimeoutMs:
|
|
34
|
+
idleTimeoutMs: 15 * 60_000,
|
|
34
35
|
wallTimeoutMs: 60 * 60_000,
|
|
35
36
|
maxAttempts: 3,
|
|
36
37
|
backoffBaseMs: 1_000,
|
|
@@ -47,9 +47,13 @@
|
|
|
47
47
|
* findings, handoffs, or run records (§0.9).
|
|
48
48
|
*/
|
|
49
49
|
import type { Finding } from "./types.js";
|
|
50
|
-
/**
|
|
50
|
+
/**
|
|
51
|
+
* Total wall-clock budget for a single verify run before it is abandoned
|
|
52
|
+
* (fail-open). Ten minutes until 2026-10-09; twenty now, matching the run
|
|
53
|
+
* budget's `VERIFY_ALLOWANCE_MS`.
|
|
54
|
+
*/
|
|
51
55
|
export declare const VERIFY_TIMEOUT_MS: number;
|
|
52
|
-
/** Per-PACKAGE wall-clock budget for the test half (spec §3d); overrun fails open. */
|
|
56
|
+
/** Per-PACKAGE wall-clock budget for the test half (spec §3d); overrun fails open. Ten minutes until 2026-10-09. */
|
|
53
57
|
export declare const VERIFY_TEST_TIMEOUT_MS: number;
|
|
54
58
|
/** A resolved verify command plus how it was found. */
|
|
55
59
|
export interface VerifyCommand {
|
|
@@ -59,10 +59,14 @@ import { claimCovers } from "./fanout.js";
|
|
|
59
59
|
import { composeAbortSignal } from "./resilience.js";
|
|
60
60
|
import { scrubSecrets } from "./scrubSecrets.js";
|
|
61
61
|
import { snapshotWorkspace } from "./workspace.js";
|
|
62
|
-
/**
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
62
|
+
/**
|
|
63
|
+
* Total wall-clock budget for a single verify run before it is abandoned
|
|
64
|
+
* (fail-open). Ten minutes until 2026-10-09; twenty now, matching the run
|
|
65
|
+
* budget's `VERIFY_ALLOWANCE_MS`.
|
|
66
|
+
*/
|
|
67
|
+
export const VERIFY_TIMEOUT_MS = 20 * 60_000;
|
|
68
|
+
/** Per-PACKAGE wall-clock budget for the test half (spec §3d); overrun fails open. Ten minutes until 2026-10-09. */
|
|
69
|
+
export const VERIFY_TEST_TIMEOUT_MS = 20 * 60_000;
|
|
66
70
|
/** Cap on located findings parsed from one failing run, and on the opaque-tail message. */
|
|
67
71
|
const MAX_LOCATED_FINDINGS = 25;
|
|
68
72
|
const MAX_TAIL_CHARS = 1000;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@yagni-app/code-staging",
|
|
3
|
-
"version": "1.2.3-staging.
|
|
3
|
+
"version": "1.2.3-staging.1686.1",
|
|
4
4
|
"description": "YAGNI Code: a terminal coding agent that already knows your company. One YAGNI login routes the model and grounds the agent in your team's context.",
|
|
5
5
|
"license": "SEE LICENSE IN LICENSE.md",
|
|
6
6
|
"author": "YAGNI, Inc. <jack@yagni.app> (https://yagni.app)",
|
|
@@ -58,5 +58,5 @@
|
|
|
58
58
|
"turndown": "^7.2.4",
|
|
59
59
|
"typebox": "^1.3.15"
|
|
60
60
|
},
|
|
61
|
-
"yagniSourceSha": "
|
|
61
|
+
"yagniSourceSha": "44055f7635d957d2f2f19a547cb945a6e942de62"
|
|
62
62
|
}
|