tickmarkr 2.1.4 → 2.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/compile/collateral.d.ts +23 -2
- package/dist/compile/collateral.js +126 -11
- package/dist/compile/native.js +35 -3
- package/dist/gates/baseline.d.ts +23 -3
- package/dist/gates/baseline.js +75 -16
- package/dist/gates/llm.d.ts +1 -0
- package/dist/gates/llm.js +1 -0
- package/dist/gates/review.d.ts +9 -2
- package/dist/gates/review.js +51 -10
- package/dist/gates/run-gates.js +12 -1
- package/dist/run/daemon.js +110 -14
- package/dist/run/git.d.ts +50 -0
- package/dist/run/git.js +56 -2
- package/dist/run/journal.d.ts +23 -0
- package/dist/run/journal.js +96 -14
- package/dist/run/merge.js +13 -3
- package/package.json +1 -1
- package/skills/tickmarkr-overseer/SKILL.md +108 -2
- package/skills/tickmarkr-overseer/scripts/watch-context.sh +63 -5
package/dist/gates/review.js
CHANGED
|
@@ -170,7 +170,8 @@ function reviewPreferIndex(c, prefer) {
|
|
|
170
170
|
return i === -1 ? prefer.length : i;
|
|
171
171
|
}
|
|
172
172
|
export function pickReviewer(author, channels, exclude = [], // v1.1 failover: reviewer channels that already produced garbage for this task
|
|
173
|
-
prefer = []
|
|
173
|
+
prefer = [], // v1.53 T2: review.prefer — reorders eligible channels, never changes eligibility
|
|
174
|
+
floor) {
|
|
174
175
|
// FLEET-05 success criterion 2: an author not resolvable in the channel list yields NO reviewer.
|
|
175
176
|
// The old `?? author.adapter` fallback compared an adapter id to vendor names, matched nothing, and
|
|
176
177
|
// admitted every reviewer — including the author's own channel (fail-OPEN). null lands on reviewGate's
|
|
@@ -182,9 +183,25 @@ prefer = []) {
|
|
|
182
183
|
// two independent axes: different vendor AND different base-model identity (ADDED TO the vendor
|
|
183
184
|
// rule, never replacing it — a future edit can't silently drop either). The diversity filter runs
|
|
184
185
|
// BEFORE preference ranking: prefer sorts survivors only, so no entry can resurrect an excluded channel.
|
|
185
|
-
.filter((c) => c.vendor !== authorChannel.vendor
|
|
186
|
+
.filter((c) => c.vendor !== authorChannel.vendor
|
|
187
|
+
&& modelId(c.model) !== modelId(author.model)
|
|
188
|
+
&& !exclude.includes(channelKey(c))
|
|
189
|
+
&& (floor === undefined || TIER_RANK[c.tier] >= TIER_RANK[floor]))
|
|
186
190
|
.sort((a, b) => reviewPreferIndex(a, prefer) - reviewPreferIndex(b, prefer) || TIER_RANK[b.tier] - TIER_RANK[a.tier] || marginalCostRank(a) - marginalCostRank(b))[0] ?? null);
|
|
187
191
|
}
|
|
192
|
+
/**
|
|
193
|
+
* This shows the reviewer what the task DECLARED, never what the diff may actually reach. The diff
|
|
194
|
+
* remains a separate stated input, so whether its touched paths fit the declaration stays a reviewer
|
|
195
|
+
* judgement rather than a guarantee made by this renderer.
|
|
196
|
+
*/
|
|
197
|
+
export function renderDeclaredWriteScope(files) {
|
|
198
|
+
if (files.length === 0) {
|
|
199
|
+
return "## Declared write scope\nUnrestricted: this task declared no write-scope patterns.";
|
|
200
|
+
}
|
|
201
|
+
return `## Declared write scope
|
|
202
|
+
The task DECLARED these write-scope patterns:
|
|
203
|
+
${files.map((path) => `- ${path}`).join("\n")}`;
|
|
204
|
+
}
|
|
188
205
|
export async function reviewGate(task, worktree, baseRef, author, channels, adapters, cfg, via, excludeReviewers,
|
|
189
206
|
// OBS-196: run dir for raw-output persistence on an unparseable verdict; absent (older callers,
|
|
190
207
|
// direct tests) skips persistence and changes nothing else.
|
|
@@ -257,13 +274,19 @@ artifactDir) {
|
|
|
257
274
|
policy: "full",
|
|
258
275
|
...(promotedBy ? { promotedFrom: declaredPolicy, promotedBy } : {}),
|
|
259
276
|
};
|
|
260
|
-
|
|
277
|
+
// A reviewer floor is opt-in at the task. Applying cfg.routing.floors here would change the
|
|
278
|
+
// historical seat for every task that never asked for review-tier coupling.
|
|
279
|
+
const reviewerFloor = task.routingHints?.floor;
|
|
280
|
+
const reviewer = pickReviewer(author, channels, excludeReviewers ?? [], cfg.review.prefer ?? [], reviewerFloor);
|
|
261
281
|
if (!reviewer) {
|
|
262
282
|
// meta.noEligibleReviewer lets run-gates' review-retry keep the ORIGINAL unparseable result when
|
|
263
283
|
// the retry finds no second seat — a truthful cause beats a synthetic no-reviewer failure.
|
|
284
|
+
const reason = reviewerFloor
|
|
285
|
+
? `no cross-vendor reviewer available at or above task-declared ${reviewerFloor} floor (diversity rule)`
|
|
286
|
+
: "no cross-vendor reviewer available (diversity rule)";
|
|
264
287
|
return cfg.review.required
|
|
265
|
-
? { gate: "review", pass: false, details:
|
|
266
|
-
: { gate: "review", pass: true, details:
|
|
288
|
+
? { gate: "review", pass: false, details: `${reason}; set review.required:false to waive`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } }
|
|
289
|
+
: { gate: "review", pass: true, details: `WARNING: ${reason} — review waived by config`, meta: { noEligibleReviewer: true, ...(reviewerFloor ? { reviewerFloor } : {}) } };
|
|
267
290
|
}
|
|
268
291
|
const measuredDiff = await fetchTaskDiff(worktree, baseRef);
|
|
269
292
|
// Keep the reader payload identical to the text charged to the strict cap:
|
|
@@ -284,6 +307,8 @@ ${COMPLETION_FAKING_CHECKLIST}
|
|
|
284
307
|
## Acceptance criteria
|
|
285
308
|
${task.acceptance.map((a) => `- ${renderAcceptanceItem(a)}`).join("\n")}
|
|
286
309
|
|
|
310
|
+
${renderDeclaredWriteScope(task.files)}
|
|
311
|
+
|
|
287
312
|
## Diff
|
|
288
313
|
\`\`\`diff
|
|
289
314
|
${diff}
|
|
@@ -301,7 +326,15 @@ Respond with ONLY this JSON:
|
|
|
301
326
|
Approve iff no material finding remains; an empty findings list is a clean approval.
|
|
302
327
|
The top-level comments array is optional. Use it only for actionable line-anchored feedback.
|
|
303
328
|
`;
|
|
304
|
-
|
|
329
|
+
let concludedOnInactivity = false;
|
|
330
|
+
const raw = await runLlm(getAdapter(reviewer.adapter, adapters), reviewer.model, prompt, worktree, via ? {
|
|
331
|
+
driver: via.driver,
|
|
332
|
+
keep: via.keep,
|
|
333
|
+
onSlot: via.onSlot,
|
|
334
|
+
name: via.nameFor("review", reviewer.adapter),
|
|
335
|
+
label: via.labelFor("review"),
|
|
336
|
+
onInactivity: () => { concludedOnInactivity = true; },
|
|
337
|
+
} : undefined,
|
|
305
338
|
// frontier reviewers routinely need >5min on a configured-cap-sized diff, and `claude -p` buffers all
|
|
306
339
|
// output until completion — runLlm's 300s default killed reviews mid-flight, returning empty
|
|
307
340
|
// stdout that read as "unparseable" and escalated to re-implementation of green code
|
|
@@ -325,14 +358,22 @@ The top-level comments array is optional. Use it only for actionable line-anchor
|
|
|
325
358
|
saved = undefined; // persistence is evidence, not a gate input — never fail the gate on it
|
|
326
359
|
}
|
|
327
360
|
}
|
|
328
|
-
const failure =
|
|
329
|
-
? "review output unparseable"
|
|
330
|
-
:
|
|
361
|
+
const failure = concludedOnInactivity
|
|
362
|
+
? "review dispatch concluded on the inactivity policy without a structurally valid nonce-bound response; output unparseable"
|
|
363
|
+
: cause === "malformed-verdict"
|
|
364
|
+
? "review output unparseable"
|
|
365
|
+
: "review dispatch failed — no structurally valid nonce-bound response; output unparseable";
|
|
331
366
|
return {
|
|
332
367
|
gate: "review",
|
|
333
368
|
pass: false,
|
|
334
369
|
details: `${failure} (reviewer ${reviewer.adapter}:${reviewer.model}; cause: ${cause}${saved ? `; raw saved: ${saved}` : ""}) — failing closed`,
|
|
335
|
-
meta: {
|
|
370
|
+
meta: {
|
|
371
|
+
...policyMeta,
|
|
372
|
+
reviewer: channelKey(reviewer),
|
|
373
|
+
unparseable: true,
|
|
374
|
+
cause,
|
|
375
|
+
...(concludedOnInactivity ? { classification: "infra", infra: true } : {}),
|
|
376
|
+
},
|
|
336
377
|
};
|
|
337
378
|
}
|
|
338
379
|
const decided = findings !== null
|
package/dist/gates/run-gates.js
CHANGED
|
@@ -589,7 +589,18 @@ export async function runGates(task, ctx) {
|
|
|
589
589
|
const second = await dispatch((adapters) => reviewGate(task, ctx.worktree, ctx.baseRef, ctx.author, ctx.channels, adapters, ctx.cfg, retryVia, [...(ctx.excludeReviewers ?? []), flaked], ctx.artifactDir));
|
|
590
590
|
if (second.meta?.noEligibleReviewer !== true) {
|
|
591
591
|
const retried = typeof second.meta?.reviewer === "string" ? second.meta.reviewer : "none";
|
|
592
|
-
rv = {
|
|
592
|
+
rv = {
|
|
593
|
+
...second,
|
|
594
|
+
// `details` is lifted onto the journal's gate-result row; meta.reviewRetry is not. Keep the
|
|
595
|
+
// re-route visible in the result text a reader actually opens, including on a red retry.
|
|
596
|
+
details: `review re-route: ${flaked} produced no parseable verdict; replaced by ${retried}\n${second.details}`,
|
|
597
|
+
meta: { ...second.meta, reviewRetry: { flaked, retried } },
|
|
598
|
+
};
|
|
599
|
+
}
|
|
600
|
+
else if (task.routingHints?.floor) {
|
|
601
|
+
// Preserve the original no-answer cause when no replacement exists, but name the declared
|
|
602
|
+
// floor that correctly refused a lower-tier fallback.
|
|
603
|
+
rv = { ...rv, details: `${rv.details}\nreview re-route refused: ${second.details}` };
|
|
593
604
|
}
|
|
594
605
|
}
|
|
595
606
|
return invocations.length ? { ...rv, meta: { ...rv.meta, invocations } } : rv;
|
package/dist/run/daemon.js
CHANGED
|
@@ -19,9 +19,9 @@ import { addEvidence, attributeBlocked, blockedTasks, getTask, graphDefinitionHa
|
|
|
19
19
|
import { GATE_NAMES } from "../graph/schema.js";
|
|
20
20
|
import { augmentRetryBrief, consult, renderRetryGuidance } from "./consult.js";
|
|
21
21
|
import { runEnvironment } from "./environment.js";
|
|
22
|
-
import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, runWithForkBudget, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
22
|
+
import { cleanupRunWorktrees, gitHead, linkNodeModules, npmDependencyInstallCommand, npmDependencyManifestChanged, preserveWorktree, resolvedCapacity, runWithForkBudget, sameCapacity, sh, shGit, WORKTREE_LAYOUT_CONTRACT, worktreePath } from "./git.js";
|
|
23
23
|
import { runInteractiveSeed } from "./interactive-seed.js";
|
|
24
|
-
import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingReviewFindings, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
24
|
+
import { activeRetryBan, classifyTaskFailure, classifyWorkerResultCause, deferredReviewFindings, engagementComparable, formatPriorFindingEvidence, GATE_FINGERPRINT_CAP, GATE_SATISFIED_RELEASE, identicalGateFailures, isDeferredFinding, journaledFailureBrief, Journal, loadRoutingProfile, newRunId, normalizeGateFailure, outstandingReviewFindings, pendingRepairFindings, phaseForGate, readPriorRunEvidence, recordedTaskFailureKind, renderStructuredReviewFinding, repairsSinceApproval, reviewRoundsSinceApproval, runHasEnded, structuredFindings, upheldFeedbackByTask } from "./journal.js";
|
|
25
25
|
import { isDiffCapPark } from "../gates/review.js";
|
|
26
26
|
import { acquireApprovalSerialization, acquireRunLock, releaseRunLock } from "./lock.js";
|
|
27
27
|
import { ensureIntegration, integrationBranch, integrationHead, mergeTask, verifyIntegrationTip } from "./merge.js";
|
|
@@ -242,6 +242,10 @@ function repairBrief(findings, diff, baseRef) {
|
|
|
242
242
|
// v1.70 T5: default request-changes rounds a task may draw before it parks. OBS-419 keeps this as the
|
|
243
243
|
// no-ceiling behavior; an operator may narrow only the next engagement on the approval that releases it.
|
|
244
244
|
const REVIEW_ROUND_CAP = 2;
|
|
245
|
+
// T2: one heading per fact. The first is owed a passing review; the second already drew one and was
|
|
246
|
+
// accepted with the reviewer's own rationale, which travels beside (but never changes) its identity.
|
|
247
|
+
const OUTSTANDING_FINDINGS_HEADING = "## Outstanding review findings — a review has NOT passed on these yet";
|
|
248
|
+
const DEFERRED_FINDINGS_HEADING = "## Deferred review findings — a reviewer ACCEPTED these with a rationale and did NOT block on them; do not re-litigate, fix only if your change touches them";
|
|
245
249
|
// OBS-419: the newest approval starts the current engagement, so it is also the sole authority for
|
|
246
250
|
// that engagement's optional ceiling. Stop at the newest approval even when the field is absent: a
|
|
247
251
|
// later ordinary release restores the module default instead of inheriting an older operator limit.
|
|
@@ -392,7 +396,7 @@ function lastVerifyCycle(events) {
|
|
|
392
396
|
if (e.event === "tip-verify-start") {
|
|
393
397
|
const { tip, cmdHash } = e.data;
|
|
394
398
|
cur = typeof tip === "string" && typeof cmdHash === "string"
|
|
395
|
-
? { tip, cmdHash, gates: new Set(), failed: false, forgiven: false }
|
|
399
|
+
? { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [e.data.capacity] }
|
|
396
400
|
: undefined;
|
|
397
401
|
afterRunEnd = false;
|
|
398
402
|
continue;
|
|
@@ -410,9 +414,14 @@ function lastVerifyCycle(events) {
|
|
|
410
414
|
continue;
|
|
411
415
|
}
|
|
412
416
|
if (!cur || afterRunEnd || cur.tip !== tip || cur.cmdHash !== cmdHash) {
|
|
413
|
-
cur = { tip, cmdHash, gates: new Set(), failed: false, forgiven: false };
|
|
417
|
+
cur = { tip, cmdHash, gates: new Set(), failed: false, forgiven: false, capacities: [] };
|
|
414
418
|
}
|
|
415
419
|
afterRunEnd = false;
|
|
420
|
+
// T7: EVERY verdict row's own capacity, not just the start row's. The start row is a statement of
|
|
421
|
+
// intent written before a single command ran; the green this cache would carry forward lives on
|
|
422
|
+
// these rows, so a row whose capacity differs from the session's — or which is malformed — has to
|
|
423
|
+
// be able to sink the cycle by itself.
|
|
424
|
+
cur.capacities.push(e.data.capacity);
|
|
416
425
|
if (e.event === "tip-verify-failed")
|
|
417
426
|
cur.failed = true;
|
|
418
427
|
else {
|
|
@@ -440,23 +449,31 @@ function lastVerifyCycle(events) {
|
|
|
440
449
|
export async function verifyIntegrationTipCached(intWt, commands, journal, opts = {}) {
|
|
441
450
|
const cmdHash = commandsHash(commands);
|
|
442
451
|
const tip = await gitHead(intWt);
|
|
452
|
+
// T7: the capacity this session's verify children WOULD run under — the third thing a carried
|
|
453
|
+
// green must match, beside the tip and the command set. A cached verdict is the one place a green
|
|
454
|
+
// crosses a session boundary with nothing re-run, and a session resumed at a different concurrency
|
|
455
|
+
// divides the machine by a different number: that green was established in another world, so it is
|
|
456
|
+
// not carried forward and the commands run again. A pre-T7 cycle records no capacity and keeps
|
|
457
|
+
// exactly the behaviour it has today.
|
|
458
|
+
const capacity = resolvedCapacity();
|
|
443
459
|
const porcelain = await shGit("git status --porcelain", intWt);
|
|
444
460
|
const clean = porcelain.code === 0 && porcelain.stdout.trim() === "";
|
|
445
461
|
const last = lastVerifyCycle(journal.read());
|
|
446
462
|
const cached = last !== undefined && !last.failed && !last.forgiven && last.tip === tip && last.cmdHash === cmdHash
|
|
447
|
-
&& Object.keys(commands).every((g) => last.gates.has(g))
|
|
463
|
+
&& Object.keys(commands).every((g) => last.gates.has(g))
|
|
464
|
+
&& last.capacities.every((recorded) => sameCapacity(recorded, capacity));
|
|
448
465
|
// A pair can be verified red and then green without either SHA or command hash changing (for
|
|
449
466
|
// example, an external service or ignored fixture recovers). Delimit attempts explicitly so that
|
|
450
467
|
// the earlier red cannot remain latched into the later complete green cycle.
|
|
451
|
-
journal.append("tip-verify-start", undefined, { tip, cmdHash, gates: Object.keys(commands), cached: clean && cached });
|
|
468
|
+
journal.append("tip-verify-start", undefined, { tip, cmdHash, capacity, gates: Object.keys(commands), cached: clean && cached });
|
|
452
469
|
if (clean && cached) {
|
|
453
|
-
journal.append("tip-verify-cached", undefined, { tip, cmdHash, gates: Object.keys(commands) });
|
|
470
|
+
journal.append("tip-verify-cached", undefined, { tip, cmdHash, capacity, gates: Object.keys(commands) });
|
|
454
471
|
// The skip must not read as a red. Every surface derives the tip's verdict from this cycle's
|
|
455
472
|
// `tip-verify` events (cockpit derive.ts tipVerificationPassed: a run-end claiming "passed" with
|
|
456
473
|
// ZERO events is fail-closed to FALSE), so a carried-forward green still journals its per-gate
|
|
457
474
|
// pass — `cached: true` keeps it honest about not having re-run the command.
|
|
458
475
|
for (const gate of Object.keys(commands)) {
|
|
459
|
-
journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash });
|
|
476
|
+
journal.append("tip-verify", undefined, { gate, cmd: commands[gate], pass: true, exitCode: 0, cached: true, tip, cmdHash, capacity });
|
|
460
477
|
}
|
|
461
478
|
return false;
|
|
462
479
|
}
|
|
@@ -464,7 +481,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
|
|
|
464
481
|
for (const r of await verifyIntegrationTip(intWt, commands, journal.dir, opts.baseline)) {
|
|
465
482
|
if (r.pass) {
|
|
466
483
|
// Q121s: a forgiven pass journals its fingerprints — honest about what was carried, never a silent green.
|
|
467
|
-
journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash });
|
|
484
|
+
journal.append("tip-verify", undefined, { gate: r.gate, cmd: r.cmd, pass: true, exitCode: r.exitCode, details: r.details, ...(r.forgiven ? { forgiven: true, fingerprints: r.fingerprints } : {}), tip, cmdHash, capacity });
|
|
468
485
|
}
|
|
469
486
|
else {
|
|
470
487
|
journal.append("tip-verify-failed", undefined, {
|
|
@@ -476,6 +493,7 @@ export async function verifyIntegrationTipCached(intWt, commands, journal, opts
|
|
|
476
493
|
lastMergedTask: opts.lastMergedTask,
|
|
477
494
|
tip,
|
|
478
495
|
cmdHash,
|
|
496
|
+
capacity,
|
|
479
497
|
});
|
|
480
498
|
tipFailed = true;
|
|
481
499
|
}
|
|
@@ -898,6 +916,10 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
898
916
|
const replayedGateResults = resumeLifecycleOpen
|
|
899
917
|
? journal.replayCurrentAttemptGateResults()
|
|
900
918
|
: new Map();
|
|
919
|
+
// T7: the capacity THIS session resolved — read inside the run's fork budget, so it is the number
|
|
920
|
+
// every shell this run spawns will divide the machine by. Recorded evidence from another session
|
|
921
|
+
// is only reusable against this.
|
|
922
|
+
const sessionCapacity = resolvedCapacity();
|
|
901
923
|
const replayedExclusions = opts.resume ? journal.replayExcludedChannels() : new Set();
|
|
902
924
|
if (opts.resume) {
|
|
903
925
|
// v1.53 T5: a superseded run is dead — resuming it beside its successor is the exact
|
|
@@ -1221,6 +1243,15 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1221
1243
|
let gateSubject;
|
|
1222
1244
|
const journalGateResult = (g) => {
|
|
1223
1245
|
const blocking = gateFailed(g) && (g.gate === "review" || g.gate === "acceptance");
|
|
1246
|
+
// T2: a review that PASSED while DEFERRING a concern still recorded a defect — the prompt
|
|
1247
|
+
// promises the deferral is recorded and never dropped, and a details string is not a record a
|
|
1248
|
+
// later round can match. The blocking projection above is the only writer of structured
|
|
1249
|
+
// findings today, so on this row it writes nothing and every structured reader goes blind.
|
|
1250
|
+
// Same shape, same identity, on the passing row: the verdict is untouched (`pass` stays true),
|
|
1251
|
+
// only the projection widens to the rows the reviewer itself classified as deferred.
|
|
1252
|
+
const deferred = !blocking && g.gate === "review" && g.pass === true
|
|
1253
|
+
? deferredReviewFindings(g.details)
|
|
1254
|
+
: [];
|
|
1224
1255
|
// R3 (OBS-186): a gate that DECLINED has no verdict to state, and this row is the ONE seam every
|
|
1225
1256
|
// fold outside this file shares. Writing `pass: false` for a decline is what turned a skip into
|
|
1226
1257
|
// a failure at all of them at once — the engagement round budget (reviewRoundsSinceApproval,
|
|
@@ -1265,6 +1296,9 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1265
1296
|
...(blocking ? {
|
|
1266
1297
|
taskContentDigest: contentDigest,
|
|
1267
1298
|
findings: structuredFindings(g.gate, g.details),
|
|
1299
|
+
} : deferred.length > 0 ? {
|
|
1300
|
+
taskContentDigest: contentDigest,
|
|
1301
|
+
findings: deferred,
|
|
1268
1302
|
} : {}),
|
|
1269
1303
|
// v2.0 T2 (OBS-554): the gate's OWN measurement, lifted verbatim from the meta run-gates
|
|
1270
1304
|
// stamped WHERE THE GATE RAN. Nothing here re-derives a duration by subtracting journal
|
|
@@ -1275,6 +1309,27 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1275
1309
|
// every recalibration this telemetry funds, a gap is honest and a zero is a lie. The
|
|
1276
1310
|
// seven-gate closed set is asserted end-to-end in tests/run/gate-telemetry.test.ts.
|
|
1277
1311
|
...gateMeasurement(g.meta),
|
|
1312
|
+
// T7: the capacity the gate's own command child ran under, lifted verbatim from the result
|
|
1313
|
+
// the battery produced — read where the shell built that child's environment, never
|
|
1314
|
+
// re-derived from the run's own budget, which would answer a different number than the
|
|
1315
|
+
// operator's export did. It is this row's only COMPARABLE identity: the two load samples
|
|
1316
|
+
// beside it are endpoint reads of a gate whose interior neither of them saw, so no reader
|
|
1317
|
+
// compares them, and matching capacity never claims the machine was calm. A gate that ran
|
|
1318
|
+
// no command carries nothing — it divided nothing.
|
|
1319
|
+
//
|
|
1320
|
+
// `dirtiedBy` is the one hole in that lift: run-gates REPLACES a green battery verdict with
|
|
1321
|
+
// a refusal when the command left the worktree dirty (run-gates.ts, three sites: the legacy
|
|
1322
|
+
// batch, the per-command loop and the merge-candidate full suite), and the refusal is a
|
|
1323
|
+
// fresh verdict object carrying nothing off the result it replaced. That command's child DID
|
|
1324
|
+
// run, so its row still owes the world it ran in. `sessionCapacity` is that world and not a
|
|
1325
|
+
// re-derivation of it: it applies the same precedence `shell` does — an operator export
|
|
1326
|
+
// first — and was read inside the same fork budget every gate child of this run is spawned
|
|
1327
|
+
// under, so it is by construction the number that child received. The flag is set only where
|
|
1328
|
+
// a command of THIS gate ran and dirtied the tree, so the round-entry refusal (no command
|
|
1329
|
+
// ran) and the round-end withdrawal (lands on a gate that runs no command) still carry
|
|
1330
|
+
// nothing.
|
|
1331
|
+
...(g.capacity ? { capacity: g.capacity }
|
|
1332
|
+
: g.meta?.dirtiedBy === g.gate ? { capacity: sessionCapacity } : {}),
|
|
1278
1333
|
});
|
|
1279
1334
|
};
|
|
1280
1335
|
// R3 (OBS-186): judge ‖ review are launched together and publish in COMPLETION order
|
|
@@ -1555,7 +1610,21 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1555
1610
|
const canonicalCurrentCommit = currentTaskSubject === replayedGates.commit;
|
|
1556
1611
|
const recreatedLegacyCommit = priorTaskTip === replayedGates.commit
|
|
1557
1612
|
&& priorTaskSubject === currentTaskSubject;
|
|
1558
|
-
|
|
1613
|
+
// T7: the commit says the gates would inspect the same TREE; it says nothing about the
|
|
1614
|
+
// machine they measured it on. A resume is a new session and may have resolved a different
|
|
1615
|
+
// concurrency, so the rows behind a replayed green must also have been measured under the
|
|
1616
|
+
// capacity this session resolved — otherwise a contiguous green prefix spans two worlds.
|
|
1617
|
+
// Rows from before this stamp carry no capacity and replay exactly as they do today.
|
|
1618
|
+
const replayedCapacities = priorEvents
|
|
1619
|
+
.filter((e) => e.event === "gate-result" && e.taskId === t.id && e.data.commit === replayedGates.commit)
|
|
1620
|
+
.map((e) => e.data.capacity);
|
|
1621
|
+
const sameWorld = replayedCapacities.every((recorded) => sameCapacity(recorded, sessionCapacity));
|
|
1622
|
+
if (!sameWorld) {
|
|
1623
|
+
journal.append("gate-replay-capacity-changed", t.id, {
|
|
1624
|
+
commit: replayedGates.commit, recorded: replayedCapacities, resolved: sessionCapacity,
|
|
1625
|
+
});
|
|
1626
|
+
}
|
|
1627
|
+
let reusable = sameWorld && (exactCurrentCommit || canonicalCurrentCommit || recreatedLegacyCommit);
|
|
1559
1628
|
const reused = [];
|
|
1560
1629
|
const declaredGates = GATE_NAMES.filter((gate) => t.gates.includes(gate));
|
|
1561
1630
|
for (const gate of declaredGates) {
|
|
@@ -1768,10 +1837,37 @@ export async function runDaemon(repoRoot, opts = {}) {
|
|
|
1768
1837
|
// task (journal.ts `outstandingReviewFindings`). Appended row-wise, because this round's own
|
|
1769
1838
|
// feedback or a repair brief may already quote a finding and repeating it helps no worker.
|
|
1770
1839
|
const outstandingFindings = outstandingReviewFindings(journaledSoFar, t.id);
|
|
1771
|
-
|
|
1772
|
-
|
|
1773
|
-
|
|
1774
|
-
|
|
1840
|
+
// T2: the two are different facts about the work and no one heading is true of both. A finding
|
|
1841
|
+
// the reviewer DEFERRED was accepted with a rationale by a review that did not block on it; a
|
|
1842
|
+
// blocking one is still waiting for a review to pass. Filing the deferral under the blocking
|
|
1843
|
+
// heading tells the next worker a passing review is owed on a concern that already drew one —
|
|
1844
|
+
// the exact falsehood this carry exists to remove, restated in the brief that carries it.
|
|
1845
|
+
//
|
|
1846
|
+
// The de-dup below is BLOCKING-ONLY on purpose. A review round's raw bytes quote every finding
|
|
1847
|
+
// it recorded, deferrals included, and those bytes ride into the very next dispatch under the
|
|
1848
|
+
// repair brief's "fix ONLY what these findings name" — so on the ordinary immediate retry the
|
|
1849
|
+
// deferral is already stated, and stated AS BLOCKING. Suppressing its heading there because it
|
|
1850
|
+
// is "already quoted" leaves exactly the falsehood. A quoted BLOCKING finding is quoted
|
|
1851
|
+
// truthfully, so that one still de-dups; a deferral is instead CUT from the raw bytes and
|
|
1852
|
+
// restated once, under the only heading true of it.
|
|
1853
|
+
const deferredRows = outstandingFindings.filter(isDeferredFinding);
|
|
1854
|
+
const withoutDeferrals = (text) => deferredRows.reduce((brief, finding) => brief.replaceAll(renderStructuredReviewFinding(finding), ""), text).replace(/\n{3,}/g, "\n\n").trim();
|
|
1855
|
+
feedback = withoutDeferrals(feedback);
|
|
1856
|
+
if (repairFindings !== undefined)
|
|
1857
|
+
repairFindings = withoutDeferrals(repairFindings);
|
|
1858
|
+
const briefs = [
|
|
1859
|
+
[OUTSTANDING_FINDINGS_HEADING, outstandingFindings.filter((f) => !isDeferredFinding(f) && !feedback.includes(f.note))],
|
|
1860
|
+
[DEFERRED_FINDINGS_HEADING, deferredRows],
|
|
1861
|
+
];
|
|
1862
|
+
for (const [heading, rows] of briefs) {
|
|
1863
|
+
if (rows.length === 0)
|
|
1864
|
+
continue;
|
|
1865
|
+
const brief = [heading, ...rows.map((f) => {
|
|
1866
|
+
const rationale = f.rationale === undefined
|
|
1867
|
+
? ""
|
|
1868
|
+
: `\n Rationale: ${f.rationale}`;
|
|
1869
|
+
return `- ${f.path}: ${f.note}${rationale}`;
|
|
1870
|
+
})].join("\n");
|
|
1775
1871
|
feedback = feedback ? `${feedback}\n\n${brief}` : brief;
|
|
1776
1872
|
}
|
|
1777
1873
|
retryMode = repairFindings
|
package/dist/run/git.d.ts
CHANGED
|
@@ -34,6 +34,54 @@ export declare const deriveForkCap: (concurrency: number, cores?: number) => num
|
|
|
34
34
|
export declare const runWithForkBudget: <T>(concurrency: number, fn: () => Promise<T>) => Promise<T>;
|
|
35
35
|
/** The cap owned by the run on this async context; the standalone default outside one. */
|
|
36
36
|
export declare const resolvedForkCap: () => string;
|
|
37
|
+
/**
|
|
38
|
+
* T7: the CAPACITY a suite verdict was measured under — the fork cap the command's child actually
|
|
39
|
+
* received, and the core count that cap was divided from. Two verdicts are comparable only when both
|
|
40
|
+
* numbers match: a run resumed at a different concurrency divides the same machine by a different
|
|
41
|
+
* number, so a green measured in that other world is not evidence about this one.
|
|
42
|
+
*
|
|
43
|
+
* This pair is the WHOLE comparable identity, and the load averages a gate row already carries beside
|
|
44
|
+
* it are deliberately NOT part of it — no reader below ever feeds a load sample into the comparison.
|
|
45
|
+
* The capacity is deterministic and resolved here, where the child's environment is built. The load
|
|
46
|
+
* endpoints are neither: they are two samples taken at a gate's boundaries, and a gate's INTERIOR is
|
|
47
|
+
* invisible to them — on this milestone's own run a gate's interior reached well over twice what
|
|
48
|
+
* either of its own endpoints saw. So matching capacity establishes only that two measurements
|
|
49
|
+
* divided the same machine by the same number. It says nothing about whether the machine was calm.
|
|
50
|
+
*/
|
|
51
|
+
export interface RunCapacity {
|
|
52
|
+
forkCap: number;
|
|
53
|
+
cores: number;
|
|
54
|
+
}
|
|
55
|
+
export type CapacityRead = {
|
|
56
|
+
state: "present";
|
|
57
|
+
capacity: RunCapacity;
|
|
58
|
+
} | {
|
|
59
|
+
state: "absent";
|
|
60
|
+
} | {
|
|
61
|
+
state: "malformed";
|
|
62
|
+
};
|
|
63
|
+
/**
|
|
64
|
+
* Three states, never two. A record carrying NO capacity is an older record from before this stamp
|
|
65
|
+
* existed: it keeps exactly the verdict it has today. A record carrying a capacity it cannot state —
|
|
66
|
+
* half the pair, an empty container, a zero, a negative, an unparseable value — is a NEWER record
|
|
67
|
+
* that is malformed, and reading it as an older one is how a fail-closed guard stops firing silently.
|
|
68
|
+
*/
|
|
69
|
+
export declare function readCapacity(value: unknown): CapacityRead;
|
|
70
|
+
/**
|
|
71
|
+
* May a verdict recorded under `recorded` be reused — forgiven, cached, replayed — by a session
|
|
72
|
+
* running under `current`? Absent → yes, unchanged. Present and identical → yes. Malformed, a
|
|
73
|
+
* different capacity, or a current capacity the caller could not state → no.
|
|
74
|
+
*/
|
|
75
|
+
export declare function sameCapacity(recorded: unknown, current: RunCapacity | undefined): boolean;
|
|
76
|
+
export declare const describeCapacity: (value: unknown) => string;
|
|
77
|
+
/**
|
|
78
|
+
* The capacity a child spawned on THIS async context would receive: the same precedence `shell`
|
|
79
|
+
* applies below — an operator export of the cap wins over the run's own derived value — beside the
|
|
80
|
+
* cores it was divided from. A caller holding a command's own result reads the capacity off THAT
|
|
81
|
+
* result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
|
|
82
|
+
* decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
|
|
83
|
+
*/
|
|
84
|
+
export declare const resolvedCapacity: () => RunCapacity;
|
|
37
85
|
/** The shipped shell ceiling: the fallback every caller gets when nothing measured a better one. */
|
|
38
86
|
export declare const DEFAULT_SHELL_TIMEOUT_MS = 600000;
|
|
39
87
|
export interface ShResult {
|
|
@@ -42,6 +90,8 @@ export interface ShResult {
|
|
|
42
90
|
stderr: string;
|
|
43
91
|
timedOut?: boolean;
|
|
44
92
|
durationMs?: number;
|
|
93
|
+
/** T7: the capacity THIS child ran under, stamped where its environment was built (see `shell`). */
|
|
94
|
+
capacity?: RunCapacity;
|
|
45
95
|
}
|
|
46
96
|
export declare const setSpawnForTests: (fn: typeof spawn) => void;
|
|
47
97
|
export declare const resetSpawnForTests: () => void;
|
package/dist/run/git.js
CHANGED
|
@@ -63,6 +63,53 @@ export const deriveForkCap = (concurrency, cores = availableParallelism()) => Ma
|
|
|
63
63
|
export const runWithForkBudget = (concurrency, fn) => forkBudget.run(String(deriveForkCap(concurrency)), fn);
|
|
64
64
|
/** The cap owned by the run on this async context; the standalone default outside one. */
|
|
65
65
|
export const resolvedForkCap = () => forkBudget.getStore() ?? DEFAULT_FORK_CAP;
|
|
66
|
+
const positiveInt = (v) => typeof v === "number" && Number.isInteger(v) && v > 0;
|
|
67
|
+
/**
|
|
68
|
+
* Three states, never two. A record carrying NO capacity is an older record from before this stamp
|
|
69
|
+
* existed: it keeps exactly the verdict it has today. A record carrying a capacity it cannot state —
|
|
70
|
+
* half the pair, an empty container, a zero, a negative, an unparseable value — is a NEWER record
|
|
71
|
+
* that is malformed, and reading it as an older one is how a fail-closed guard stops firing silently.
|
|
72
|
+
*/
|
|
73
|
+
export function readCapacity(value) {
|
|
74
|
+
if (value === undefined)
|
|
75
|
+
return { state: "absent" };
|
|
76
|
+
if (value === null || typeof value !== "object")
|
|
77
|
+
return { state: "malformed" };
|
|
78
|
+
const { forkCap, cores } = value;
|
|
79
|
+
return positiveInt(forkCap) && positiveInt(cores)
|
|
80
|
+
? { state: "present", capacity: { forkCap, cores } }
|
|
81
|
+
: { state: "malformed" };
|
|
82
|
+
}
|
|
83
|
+
/**
|
|
84
|
+
* May a verdict recorded under `recorded` be reused — forgiven, cached, replayed — by a session
|
|
85
|
+
* running under `current`? Absent → yes, unchanged. Present and identical → yes. Malformed, a
|
|
86
|
+
* different capacity, or a current capacity the caller could not state → no.
|
|
87
|
+
*/
|
|
88
|
+
export function sameCapacity(recorded, current) {
|
|
89
|
+
const read = readCapacity(recorded);
|
|
90
|
+
if (read.state === "absent")
|
|
91
|
+
return true;
|
|
92
|
+
if (read.state === "malformed" || current === undefined)
|
|
93
|
+
return false;
|
|
94
|
+
return read.capacity.forkCap === current.forkCap && read.capacity.cores === current.cores;
|
|
95
|
+
}
|
|
96
|
+
export const describeCapacity = (value) => {
|
|
97
|
+
const read = readCapacity(value);
|
|
98
|
+
return read.state === "present"
|
|
99
|
+
? `fork cap ${read.capacity.forkCap} of ${read.capacity.cores} cores`
|
|
100
|
+
: read.state === "absent" ? "an unrecorded capacity" : "a malformed capacity";
|
|
101
|
+
};
|
|
102
|
+
/**
|
|
103
|
+
* The capacity a child spawned on THIS async context would receive: the same precedence `shell`
|
|
104
|
+
* applies below — an operator export of the cap wins over the run's own derived value — beside the
|
|
105
|
+
* cores it was divided from. A caller holding a command's own result reads the capacity off THAT
|
|
106
|
+
* result (`ShResult.capacity`, stamped where the child's environment was built); this is for the
|
|
107
|
+
* decisions taken BEFORE any child exists — a cache hit, a reuse predicate.
|
|
108
|
+
*/
|
|
109
|
+
export const resolvedCapacity = () => ({
|
|
110
|
+
forkCap: Number(FORK_CAP_ENV in process.env ? process.env[FORK_CAP_ENV] : resolvedForkCap()),
|
|
111
|
+
cores: availableParallelism(),
|
|
112
|
+
});
|
|
66
113
|
/** The shipped shell ceiling: the fallback every caller gets when nothing measured a better one. */
|
|
67
114
|
export const DEFAULT_SHELL_TIMEOUT_MS = 600000;
|
|
68
115
|
/**
|
|
@@ -103,6 +150,13 @@ function shell(cmd, cwd, timeoutMs, login) {
|
|
|
103
150
|
// OBS-110: apply the run's own fork cap only when the operator has not already set one.
|
|
104
151
|
if (!(FORK_CAP_ENV in env))
|
|
105
152
|
env[FORK_CAP_ENV] = resolvedForkCap();
|
|
153
|
+
// T7: the capacity every result of this shell carries, read HERE — off the environment the child
|
|
154
|
+
// is about to receive, after the precedence above has settled. An operator export is already in
|
|
155
|
+
// `env`, so what gets recorded is the operator's number, which is the case a release was re-taken
|
|
156
|
+
// for; re-deriving the run's own budget after the command returned would stamp a cap no child ran
|
|
157
|
+
// under. `Number` of an unparseable export is NaN, which every reader treats as malformed and
|
|
158
|
+
// therefore fails closed — the honest direction when the cap in play cannot be stated.
|
|
159
|
+
const capacity = { forkCap: Number(env[FORK_CAP_ENV]), cores: availableParallelism() };
|
|
106
160
|
const attempt = () => new Promise((resolve) => {
|
|
107
161
|
const startedAt = Date.now();
|
|
108
162
|
// detached: bash gets its own process group so a timeout can kill the whole tree —
|
|
@@ -120,7 +174,7 @@ function shell(cmd, cwd, timeoutMs, login) {
|
|
|
120
174
|
clearTimeout(timer);
|
|
121
175
|
stdout += stdoutDecoder.end();
|
|
122
176
|
stderr += stderrDecoder.end();
|
|
123
|
-
resolve({ code, stdout, stderr: err ?? stderr, timedOut, durationMs: Date.now() - startedAt });
|
|
177
|
+
resolve({ code, stdout, stderr: err ?? stderr, timedOut, durationMs: Date.now() - startedAt, capacity });
|
|
124
178
|
};
|
|
125
179
|
const timer = setTimeout(() => {
|
|
126
180
|
timedOut = true;
|
|
@@ -170,7 +224,7 @@ function shell(cmd, cwd, timeoutMs, login) {
|
|
|
170
224
|
// Bounded, and the bound is what makes a persisting shortage a REPORTED failure rather than a
|
|
171
225
|
// wedged daemon: past it the caller gets the refusal's own text under exit 127, as before.
|
|
172
226
|
if (n >= SPAWN_ATTEMPT_LIMIT) {
|
|
173
|
-
return { code: 127, stdout: "", stderr: String(r.refused), durationMs: Date.now() - startedAt };
|
|
227
|
+
return { code: 127, stdout: "", stderr: String(r.refused), durationMs: Date.now() - startedAt, capacity };
|
|
174
228
|
}
|
|
175
229
|
await new Promise((wake) => setTimeout(wake, SPAWN_RETRY_BACKOFF_MS * n));
|
|
176
230
|
}
|
package/dist/run/journal.d.ts
CHANGED
|
@@ -36,6 +36,7 @@ export interface StructuredFinding {
|
|
|
36
36
|
path: string;
|
|
37
37
|
symbol: string;
|
|
38
38
|
note: string;
|
|
39
|
+
rationale?: string;
|
|
39
40
|
fingerprint: string;
|
|
40
41
|
}
|
|
41
42
|
export declare const UNIDENTIFIED = "<unidentified>";
|
|
@@ -52,6 +53,15 @@ export declare const UNIDENTIFIED = "<unidentified>";
|
|
|
52
53
|
* normalized words (see toFinding).
|
|
53
54
|
*/
|
|
54
55
|
export declare function structuredFindings(gate: string, details: string, _scopeFiles?: string[]): StructuredFinding[];
|
|
56
|
+
export declare function isDeferredFinding(finding: StructuredFinding): boolean;
|
|
57
|
+
/**
|
|
58
|
+
* The findings a PASSING review DEFERRED — the rows a blocking-only projection drops on the floor.
|
|
59
|
+
* A passing review's details are prose; without this the deferral has no identity a later round can
|
|
60
|
+
* match, and every structured reader of the journal is blind to a defect the reviewer itself named.
|
|
61
|
+
*/
|
|
62
|
+
export declare function deferredReviewFindings(details: string): StructuredFinding[];
|
|
63
|
+
/** The exact review.ts details fragment represented by a structured review finding. */
|
|
64
|
+
export declare function renderStructuredReviewFinding(finding: StructuredFinding): string;
|
|
55
65
|
export interface PriorRunJournal {
|
|
56
66
|
runId: string;
|
|
57
67
|
events: JournalEvent[];
|
|
@@ -126,6 +136,19 @@ export declare function journaledFailureBrief(events: JournalEvent[], taskId: st
|
|
|
126
136
|
* finding was dropped at the exact moment the operator paid for another attempt to fix it. A review
|
|
127
137
|
* that DECLINED (`skipped`) is not a verdict and neither adds nor retires — fail closed. Findings are
|
|
128
138
|
* keyed by fingerprint, so a reviewer restating one across rounds carries it once, not once per round.
|
|
139
|
+
*
|
|
140
|
+
* v2.1.5 T2: a passing review settles the findings it BLOCKED on. It does not settle the ones it
|
|
141
|
+
* DEFERRED — those it saw, declined to block on, and recorded a rationale for, and nothing has fixed
|
|
142
|
+
* them. So a pass retires the blocking set and re-seats its own deferrals, and the two retirements
|
|
143
|
+
* stay distinguishable: the blocking finding is gone, the deferral travels on as accepted work.
|
|
144
|
+
*
|
|
145
|
+
* A deferral's bound is the SAME single release as a blocking finding's — the operator accepting the
|
|
146
|
+
* review gate itself (`GATE_SATISFIED_RELEASE` stamped `gate: "review"`), the one approval in which a
|
|
147
|
+
* human actually looked at what the reviewer waved through. It is deliberately NOT bounded by a round
|
|
148
|
+
* count or by a time window: both retire a finding by arithmetic nobody read, which is the silent drop
|
|
149
|
+
* this fold exists to refuse. Nor can it accumulate — a reviewer restating the same path/note round
|
|
150
|
+
* after round re-seats ONE fingerprint, and a revised rationale replaces the prior rationale on that
|
|
151
|
+
* row. N rounds of the same concern therefore carry the newest accepted explanation once, not N rows.
|
|
129
152
|
*/
|
|
130
153
|
export declare function outstandingReviewFindings(events: JournalEvent[], taskId: string): StructuredFinding[];
|
|
131
154
|
/** The findings a funded repair must carry into the next dispatch, or undefined if none is pending. */
|