@coreplane/switchboard 1.207.1 → 1.209.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/config/config.example.yaml +11 -11
- package/dist/assets/package-lock.json +3 -3
- package/dist/assets/package.json +1 -1
- package/dist/assets/source.json +3 -3
- package/dist/assets/src/agents/registry.ts +34 -16
- package/dist/assets/src/core/coordinator/contract.ts +11 -0
- package/dist/assets/src/core/coordinator/driver.ts +22 -1
- package/dist/assets/src/core/reviewVerdict.ts +40 -0
- package/dist/assets/src/core/runEvents.ts +29 -1
- package/dist/assets/src/core/runFriction.ts +5 -4
- package/dist/assets/src/core/runRecord.ts +10 -1
- package/dist/assets/src/core/ship/contract.ts +9 -0
- package/dist/assets/src/core/ship/coordinator.ts +17 -12
- package/dist/assets/src/core/trace/streamSpans.ts +3 -0
- package/dist/assets/src/core/trace/workerTrace.ts +3 -0
- package/dist/assets/web/dist/.vite/manifest.json +18 -18
- package/dist/assets/web/dist/assets/{ResidentDetailPage-D3P21yeI.js → ResidentDetailPage-CV3wfAIP.js} +1 -1
- package/dist/assets/web/dist/assets/{ResidentsIndexPage-DOYqnZ1q.js → ResidentsIndexPage-CLkWc50b.js} +1 -1
- package/dist/assets/web/dist/assets/{RunRoutePage-OmvrvPXY.js → RunRoutePage-CQYRfQ_B.js} +4 -4
- package/dist/assets/web/dist/assets/{RunsIndexPage-DWbSQtL4.js → RunsIndexPage-BLRPp_gk.js} +1 -1
- package/dist/assets/web/dist/assets/{ScheduledPage-CPKfJ4mR.js → ScheduledPage-C-VO4Ddl.js} +1 -1
- package/dist/assets/web/dist/assets/{StatusDot-COr8jTyM.js → StatusDot-CIAoBB5Y.js} +1 -1
- package/dist/assets/web/dist/assets/{Tooltip-fOqTZkNT.js → Tooltip-DEL1ic4g.js} +1 -1
- package/dist/assets/web/dist/assets/{dist-BVjAWgkb.js → dist-CawBR4t8.js} +1 -1
- package/dist/assets/web/dist/assets/{main-DZbJaqUb.js → main-CctUbVOl.js} +2 -2
- package/dist/cli.js +6570 -6602
- package/package.json +1 -1
|
@@ -278,20 +278,20 @@ workspaceDir: ./workspaces
|
|
|
278
278
|
# tokenEnv: MEMORY_TOKEN # env var holding the Worker's bearer (default)
|
|
279
279
|
|
|
280
280
|
# agent:ship pipeline caps (docs/reference/specs/agent-ship.md). `agent:ship in owner/repo:
|
|
281
|
-
# <task>`
|
|
281
|
+
# <task>` hands the coding → review → fix loop to LGTM to the plan runner — the
|
|
282
|
+
# ShipCoordinator Workflow in the bot Worker, whose rounds are child runs — which
|
|
283
|
+
# needs the `coordinator` ingress entry, its grants, run history on the state
|
|
284
|
+
# Worker and PUBLIC_BASE_URL (docs/how-to/turn-features-on-and-off.md); without
|
|
285
|
+
# them the request is refused naming what is missing. Per unit: at most
|
|
282
286
|
# `maxRounds` review rounds and `maxMinutes` minutes of wall clock — whichever
|
|
283
|
-
# hits first ends the
|
|
284
|
-
# remaining
|
|
285
|
-
#
|
|
286
|
-
#
|
|
287
|
-
#
|
|
287
|
+
# hits first ends the unit, and each child round's own budget is clipped to the
|
|
288
|
+
# remaining time. `maxMinutes` is the ship preset's declared budget: a channel or
|
|
289
|
+
# user `boundary.maxMinutes` and a per-message `budget:` directive clip it like
|
|
290
|
+
# any preset's, and the card says so. Defaults shown; both must be integers >= 1;
|
|
291
|
+
# any other key under `ship` fails the load by name.
|
|
288
292
|
# ship:
|
|
289
|
-
# maxRounds: 3 # review rounds per
|
|
293
|
+
# maxRounds: 3 # review rounds per unit
|
|
290
294
|
# maxMinutes: 120 # the ship preset's wall-clock budget in minutes (the registry's default)
|
|
291
|
-
# coordinator: false # true hands every agent:ship request to the plan runner — the ShipCoordinator
|
|
292
|
-
# # Workflow in the bot Worker — instead of the in-process round loop; needs the
|
|
293
|
-
# # `coordinator` ingress entry, its grant, run history on the state Worker and
|
|
294
|
-
# # PUBLIC_BASE_URL (see docs/how-to/turn-features-on-and-off.md)
|
|
295
295
|
|
|
296
296
|
# The fan-out cap a spawning run meets (docs/reference/specs/agent-conductor.md).
|
|
297
297
|
# `agent:conductor` starts child runs as the person who asked — each an
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "switchboard",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.209.0",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "switchboard",
|
|
9
|
-
"version": "1.
|
|
9
|
+
"version": "1.209.0",
|
|
10
10
|
"license": "Apache-2.0",
|
|
11
11
|
"workspaces": [
|
|
12
12
|
"web",
|
|
@@ -18999,7 +18999,7 @@
|
|
|
18999
18999
|
},
|
|
19000
19000
|
"packages/switchboard": {
|
|
19001
19001
|
"name": "@coreplane/switchboard",
|
|
19002
|
-
"version": "1.
|
|
19002
|
+
"version": "1.209.0",
|
|
19003
19003
|
"license": "Apache-2.0",
|
|
19004
19004
|
"dependencies": {
|
|
19005
19005
|
"@anthropic-ai/sdk": "^0.124.0",
|
package/dist/assets/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "switchboard",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.209.0",
|
|
4
4
|
"private": true,
|
|
5
5
|
"description": "Mention it in Slack and an agent reviews the PR, ships the fix, or answers the question — on the model you choose, with its tools running where you decide.",
|
|
6
6
|
"license": "Apache-2.0",
|
package/dist/assets/source.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
// Agent definitions. An agent is a system prompt + toolset + machine class +
|
|
1
|
+
// Agent definitions. An agent is a system prompt + toolset + machine class + wall-clock budget.
|
|
2
2
|
import type { Effort } from "../effort.js";
|
|
3
3
|
import type { CacheTtl } from "../providers/types.js";
|
|
4
4
|
import { BASH_TIMEOUT_MAX_MS } from "../execution/bashTimeout.js";
|
|
5
|
-
import { CONTRACT_HEADING, CONTRACT_SECTION_HEADINGS } from "../core/ship/contract.js";
|
|
5
|
+
import { CONTRACT_HEADING, CONTRACT_SECTION_HEADINGS, PR_TITLE_GUARD } from "../core/ship/contract.js";
|
|
6
6
|
// Which model runs it is resolved separately by the config layers, so any
|
|
7
7
|
// agent can run on any configured provider/model.
|
|
8
8
|
|
|
@@ -42,13 +42,37 @@ export function machineNeedsRepo(machine: MachineClass): boolean {
|
|
|
42
42
|
export const IDENTITIES = ["none", "read", "write"] as const;
|
|
43
43
|
export type Identity = (typeof IDENTITIES)[number];
|
|
44
44
|
|
|
45
|
+
/** The pace that marks a run as looping rather than working: a model turn
|
|
46
|
+
* every ten seconds, sustained for the whole wall clock. A busy run takes
|
|
47
|
+
* 20–40 s a turn (a model think plus a tool call), so a run that averages six
|
|
48
|
+
* a minute from start to end is re-issuing calls, not making progress — and
|
|
49
|
+
* its turn cap ends it before the wall clock would, with a write-up that
|
|
50
|
+
* says so (docs/reference/specs/run-loop.md item 1). */
|
|
51
|
+
export const RUNAWAY_TURNS_PER_MINUTE = 6;
|
|
52
|
+
|
|
53
|
+
/** The turn cap a wall clock implies: `maxMinutes × RUNAWAY_TURNS_PER_MINUTE`.
|
|
54
|
+
* Every preset that runs the loop derives its `maxTurns` from this, so the
|
|
55
|
+
* cap is never a number a good run reaches — the minutes are the budget. */
|
|
56
|
+
export function runawayTurnCap(maxMinutes: number): number {
|
|
57
|
+
return maxMinutes * RUNAWAY_TURNS_PER_MINUTE;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** A loop-running preset's budget as one fact: the wall clock, and the runaway
|
|
61
|
+
* guard derived from it. */
|
|
62
|
+
function loopBudget(maxMinutes: number): Pick<AgentDef, "maxMinutes" | "maxTurns"> {
|
|
63
|
+
return { maxMinutes, maxTurns: runawayTurnCap(maxMinutes) };
|
|
64
|
+
}
|
|
65
|
+
|
|
45
66
|
export interface AgentDef {
|
|
46
67
|
name: string;
|
|
47
68
|
description: string;
|
|
48
69
|
system: string;
|
|
49
70
|
/** key into TOOLSETS: "full" | "readonly" | "web" | "assistant" | "explore" | "conductor" | "none" */
|
|
50
71
|
toolset: "full" | "readonly" | "web" | "assistant" | "explore" | "conductor" | "none";
|
|
51
|
-
/**
|
|
72
|
+
/** The runaway guard, not a budget: `runawayTurnCap(maxMinutes)` for every
|
|
73
|
+
* preset that runs the loop (`loopBudget`). The wall clock below is the
|
|
74
|
+
* budget; a run that reaches this cap first was pacing like a loop, and its
|
|
75
|
+
* write-up says so. The proxy refuses model calls past it too. */
|
|
52
76
|
maxTurns: number;
|
|
53
77
|
maxTokens: number;
|
|
54
78
|
/** hard wall-clock budget for the tool loop; at the deadline the agent is
|
|
@@ -97,7 +121,7 @@ export interface AgentDef {
|
|
|
97
121
|
// and rendered against the head sha at render time, so a repush is a
|
|
98
122
|
// re-render by Switchboard — the agent only resubmits when the CONTENT (line
|
|
99
123
|
// numbers included) changed.
|
|
100
|
-
const PR_DESCRIPTION_TEMPLATE = `PR description — submit it with the submit_pr_description tool for EVERY PR (this is the default, not something to wait to be asked for). Switchboard renders the GitHub body from the object you submit, so never author PR-body markdown yourself. Content contract per field (each renders as its own section): prose is unwrapped — no hard line breaks inside a paragraph. Always hyperlink the triggering issue/request. Never fabricate validation — state exactly what you ran and the real result. Keep each field concise, not padded.
|
|
124
|
+
const PR_DESCRIPTION_TEMPLATE = `PR description — submit it with the submit_pr_description tool for EVERY PR (this is the default, not something to wait to be asked for). Switchboard renders the GitHub body from the object you submit, so never author PR-body markdown yourself. Before submitting, judge your title with the ${PR_TITLE_GUARD} gate — \`npm run check:pr-title -- "<title>"\` — and submit only a title it accepts; the same gate refuses the PR in CI. Content contract per field (each renders as its own section): prose is unwrapped — no hard line breaks inside a paragraph. Always hyperlink the triggering issue/request. Never fabricate validation — state exactly what you ran and the real result. Keep each field concise, not padded.
|
|
101
125
|
EVERY PR includes one that already exists when you push — opened by a person, by dependabot, or by an earlier run. After EVERY push to such a PR: read its current title and body (\`github_issue_get\` with the PR number works for pull requests; \`gh pr view\` where gh exists), judge them against the change as it now stands at the pushed head, and submit the object that describes the PR as it is NOW — carry forward what the existing body says that is still true (a dependency bump's release notes belong in whatWhy), add what you changed, and anchor the Tour at the new head. Switchboard replaces the PR's title and body with your rendering. A description that describes an earlier state of its branch is a bug; "it is someone else's PR" is never a reason to leave it.
|
|
102
126
|
- **title**: the PR title — one line naming the change, specific enough to pick out of a PR list.
|
|
103
127
|
- **TL;DR** (\`tldr\`, rendered first): two sentences for a naive reader with zero context — what this PR does and why it matters.
|
|
@@ -401,9 +425,8 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
401
425
|
// and mints no credential of its own.
|
|
402
426
|
machine: "none",
|
|
403
427
|
identity: "none",
|
|
404
|
-
maxTurns: 8, // a repo read is 2-3 calls (repos → tree → file); an issue action 1-2; still fast
|
|
405
428
|
maxTokens: 16000,
|
|
406
|
-
|
|
429
|
+
...loopBudget(5),
|
|
407
430
|
},
|
|
408
431
|
coding: {
|
|
409
432
|
name: "coding",
|
|
@@ -411,9 +434,8 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
411
434
|
system: CODING_SYSTEM,
|
|
412
435
|
residentSystem: CODING_SYSTEM_RESIDENT,
|
|
413
436
|
toolset: "full",
|
|
414
|
-
maxTurns: 60, // scoping is capped at ~5 calls by the prompt; this is implementation room
|
|
415
437
|
maxTokens: 64000,
|
|
416
|
-
|
|
438
|
+
...loopBudget(45),
|
|
417
439
|
// Coding steps run long: a single model turn can take 5-6 minutes and
|
|
418
440
|
// installs/tests add more — a 5m cache entry would expire between
|
|
419
441
|
// requests, so the 2× write buys reads for the whole run.
|
|
@@ -431,9 +453,8 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
431
453
|
toolset: "readonly",
|
|
432
454
|
machine: "repo-resident",
|
|
433
455
|
identity: "read", // a read-scoped token and a read-only worktree: it cannot post or push from inside
|
|
434
|
-
maxTurns: 30, // backstop only; wall clock is the real budget (12 bound at ~4 min in practice)
|
|
435
456
|
maxTokens: 64000,
|
|
436
|
-
|
|
457
|
+
...loopBudget(25), // a safety net — typical reviews land in ~5 minutes
|
|
437
458
|
effort: "medium", // fast turns; one big-context pass does the deep work
|
|
438
459
|
},
|
|
439
460
|
ship: {
|
|
@@ -468,9 +489,8 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
468
489
|
toolset: "web",
|
|
469
490
|
machine: "none", // web I/O only; no workspace is provisioned
|
|
470
491
|
identity: "none",
|
|
471
|
-
maxTurns: 12,
|
|
472
492
|
maxTokens: 24000,
|
|
473
|
-
|
|
493
|
+
...loopBudget(8),
|
|
474
494
|
effort: "medium",
|
|
475
495
|
},
|
|
476
496
|
explore: {
|
|
@@ -483,9 +503,8 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
483
503
|
// review depends on: a two-hour job shares no container with anyone.
|
|
484
504
|
machine: "repo-cold",
|
|
485
505
|
identity: "read", // a read-scoped token: it can clone and read, never push — whatever the caller holds
|
|
486
|
-
maxTurns: 150, // a backstop for a two-hour loop of batched checks; the wall clock is the budget
|
|
487
506
|
maxTokens: 64000,
|
|
488
|
-
|
|
507
|
+
...loopBudget(120),
|
|
489
508
|
// A detached job polled across calls makes long steps: a 5m cache entry
|
|
490
509
|
// would expire between them, so the 2× write buys reads for the whole run.
|
|
491
510
|
cacheTtl: "1h",
|
|
@@ -501,9 +520,8 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
501
520
|
// dispatcher, the GitHub reads are REST in the bot process.
|
|
502
521
|
machine: "none",
|
|
503
522
|
identity: "none",
|
|
504
|
-
maxTurns: 40, // a spawn, then a poll per child every few minutes; the wall clock is the budget
|
|
505
523
|
maxTokens: 32000,
|
|
506
|
-
|
|
524
|
+
...loopBudget(120), // long enough to outlast a coding child; every child is capped by what remains of it
|
|
507
525
|
// No built-in effort: the deployment decides, as for coding.
|
|
508
526
|
},
|
|
509
527
|
};
|
|
@@ -145,6 +145,11 @@ export interface CoordinatorUnit {
|
|
|
145
145
|
/** The unit's board issue in the repository, when one titled by the unit id exists — the handoff's destination. */
|
|
146
146
|
issue?: number;
|
|
147
147
|
pr?: { number: number; url: string };
|
|
148
|
+
/** Resume at review (agent-ship item 10): the open pull request of ship's own
|
|
149
|
+
* the requester named, so the unit's pipeline opens at its first review round
|
|
150
|
+
* — no pre-check, no branch, no round 0. A task string's row only; written by
|
|
151
|
+
* the hand-off, read by the driver into the machine's input. */
|
|
152
|
+
resume?: { pr: number; headSha?: string; url?: string };
|
|
148
153
|
/** The round boundaries the coordinator reported, oldest first (the `ship_round` vocabulary). */
|
|
149
154
|
rounds: Array<{ index: number; agent: string; outcome: string; at: number }>;
|
|
150
155
|
/** How the unit ended: the ending's kind and the thread's report, when it has. */
|
|
@@ -163,6 +168,11 @@ const isOptionalText = (v: unknown): boolean => v === undefined || isText(v);
|
|
|
163
168
|
const isFinite = (v: unknown): v is number => typeof v === "number" && Number.isFinite(v);
|
|
164
169
|
const isObject = (v: unknown): v is Record<string, unknown> => typeof v === "object" && v !== null;
|
|
165
170
|
const isPr = (v: unknown): boolean => isObject(v) && isFinite(v.number) && isText(v.url, 2048);
|
|
171
|
+
const isResume = (v: unknown): boolean =>
|
|
172
|
+
isObject(v) &&
|
|
173
|
+
isFinite(v.pr) &&
|
|
174
|
+
(v.headSha === undefined || isText(v.headSha)) &&
|
|
175
|
+
(v.url === undefined || isText(v.url, 2048));
|
|
166
176
|
|
|
167
177
|
/** Structural check on a record from outside the process (a Worker response, an HTTP body). */
|
|
168
178
|
export function isCoordinatorInstance(v: unknown): v is CoordinatorInstance {
|
|
@@ -194,6 +204,7 @@ export function isCoordinatorUnit(v: unknown): v is CoordinatorUnit {
|
|
|
194
204
|
if (!isOptionalText(r.threadKey) || !isOptionalText(r.sourceUrl)) return false;
|
|
195
205
|
if (r.issue !== undefined && !isFinite(r.issue)) return false;
|
|
196
206
|
if (r.pr !== undefined && !isPr(r.pr)) return false;
|
|
207
|
+
if (r.resume !== undefined && !isResume(r.resume)) return false;
|
|
197
208
|
if (
|
|
198
209
|
!Array.isArray(r.rounds) ||
|
|
199
210
|
r.rounds.length > MAX_ROUNDS ||
|
|
@@ -233,7 +233,18 @@ function readRecordReturn(step: string, a: BotAnswer): StepReturn {
|
|
|
233
233
|
// The typed artifacts as the bot's record carries them — shape-checked where
|
|
234
234
|
// they were written (the run record's validator), read here as they are.
|
|
235
235
|
const facts = run as unknown as Omit<Extract<ChildFacts, { finished: true }>, "finished" | "status">;
|
|
236
|
-
const {
|
|
236
|
+
const {
|
|
237
|
+
finalReply,
|
|
238
|
+
pr,
|
|
239
|
+
headSha,
|
|
240
|
+
description,
|
|
241
|
+
verdict,
|
|
242
|
+
reviewPosted,
|
|
243
|
+
reviewPostReason,
|
|
244
|
+
reviewHead,
|
|
245
|
+
dispositions,
|
|
246
|
+
handoff,
|
|
247
|
+
} = facts;
|
|
237
248
|
return {
|
|
238
249
|
type: "read-record",
|
|
239
250
|
step,
|
|
@@ -246,6 +257,7 @@ function readRecordReturn(step: string, a: BotAnswer): StepReturn {
|
|
|
246
257
|
...(description !== undefined ? { description } : {}),
|
|
247
258
|
...(verdict !== undefined ? { verdict } : {}),
|
|
248
259
|
...(reviewPosted !== undefined ? { reviewPosted } : {}),
|
|
260
|
+
...(reviewPostReason !== undefined ? { reviewPostReason } : {}),
|
|
249
261
|
...(reviewHead !== undefined ? { reviewHead } : {}),
|
|
250
262
|
...(dispositions !== undefined ? { dispositions } : {}),
|
|
251
263
|
...(handoff !== undefined ? { handoff } : {}),
|
|
@@ -399,6 +411,10 @@ async function runUnit(
|
|
|
399
411
|
const start = readUnitStart(
|
|
400
412
|
answerOf("unit-start", await step.do(`${unit}/start`, STEP_CONFIG, () => call(bot, "unit-start", tag))),
|
|
401
413
|
);
|
|
414
|
+
// A resume at review (agent-ship item 10) rides the unit's row: the pull
|
|
415
|
+
// request of ship's own the requester named opens the pipeline at its first
|
|
416
|
+
// review round, with no pre-check, no branch and no round 0.
|
|
417
|
+
const resume = plan.units.find((u) => u.unit === unit)?.resume;
|
|
402
418
|
let state: UnitPipelineState = openUnitPipeline(
|
|
403
419
|
{
|
|
404
420
|
unit: { id: unit, branch: node.branch },
|
|
@@ -411,6 +427,7 @@ async function runUnit(
|
|
|
411
427
|
// approved at its head and the checks are green; any other branch — a
|
|
412
428
|
// task string's ship branch — waits for a person.
|
|
413
429
|
merge: parsePlanBranch(node.branch) !== undefined ? "runner" : "person",
|
|
430
|
+
...(resume !== undefined ? { resume } : {}),
|
|
414
431
|
},
|
|
415
432
|
start.at,
|
|
416
433
|
);
|
|
@@ -425,10 +442,14 @@ async function runUnit(
|
|
|
425
442
|
const body = { ...tag, index: note.index, agent: note.agent, outcome: note.outcome };
|
|
426
443
|
await step.do(`${unit}/note/${++notes}`, STEP_CONFIG, () => call(bot, "round", body));
|
|
427
444
|
} else {
|
|
445
|
+
// The last coding child's run is named so the bot can put its handoff
|
|
446
|
+
// — the deviations it recorded — on the unit's board issue beside the
|
|
447
|
+
// ending (agent-ship item 14).
|
|
428
448
|
const body = {
|
|
429
449
|
...tag,
|
|
430
450
|
ending: { kind: note.ending.kind, report: renderUnitReport(state) },
|
|
431
451
|
...(state.pr !== undefined ? { pr: state.pr } : {}),
|
|
452
|
+
...(state.lastCodingRunId !== undefined ? { codingRunId: state.lastCodingRunId } : {}),
|
|
432
453
|
};
|
|
433
454
|
await step.do(`${unit}/end`, STEP_CONFIG, () => call(bot, "unit-end", body));
|
|
434
455
|
}
|
|
@@ -259,6 +259,46 @@ export function isReviewVerdictShape(v: unknown): v is ReviewVerdict {
|
|
|
259
259
|
return true;
|
|
260
260
|
}
|
|
261
261
|
|
|
262
|
+
/** How a review run's post-step ended, as the run's record carries it
|
|
263
|
+
* (docs/reference/specs/agent-review.md item 18; run-history item 2): the verdict
|
|
264
|
+
* landed on a named pull request pinned to `head` (the verdict kind rides when
|
|
265
|
+
* one was submitted — a review that posted without a verdict posts the
|
|
266
|
+
* no-verdict line), or nothing landed and `reason` says why — a guard's
|
|
267
|
+
* refusal, an opt-out, no pull request, GitHub's own error. A coordinator's
|
|
268
|
+
* `read-record` answers `reviewPosted` from this before it asks GitHub, whose
|
|
269
|
+
* review list can lag a post it accepted a second ago. */
|
|
270
|
+
export type ReviewPost =
|
|
271
|
+
| { posted: true; target: { repo: string; number: number }; head: string; verdict?: ReviewVerdictKind }
|
|
272
|
+
| { posted: false; reason: string };
|
|
273
|
+
|
|
274
|
+
const REVIEW_POST_HEAD = /^[0-9a-f]{7,40}$/;
|
|
275
|
+
|
|
276
|
+
/** Structural check on a review post read back from a stored record: a posted
|
|
277
|
+
* outcome names its pull request and a 7-to-40-hex head, its verdict (when
|
|
278
|
+
* present) a known kind; a skipped one carries a string reason. */
|
|
279
|
+
export function isReviewPostShape(v: unknown): v is ReviewPost {
|
|
280
|
+
if (!isRecordLike(v)) return false;
|
|
281
|
+
if (v.posted === false) return typeof v.reason === "string";
|
|
282
|
+
if (v.posted !== true) return false;
|
|
283
|
+
const target = v.target;
|
|
284
|
+
if (
|
|
285
|
+
!isRecordLike(target) ||
|
|
286
|
+
typeof target.repo !== "string" ||
|
|
287
|
+
typeof target.number !== "number" ||
|
|
288
|
+
!Number.isInteger(target.number) ||
|
|
289
|
+
target.number <= 0
|
|
290
|
+
)
|
|
291
|
+
return false;
|
|
292
|
+
if (typeof v.head !== "string" || !REVIEW_POST_HEAD.test(v.head)) return false;
|
|
293
|
+
return v.verdict === undefined || VERDICT_KINDS.includes(v.verdict as string);
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
/** The skip's reason through the redaction seam (it may carry GitHub's own
|
|
297
|
+
* words); a posted outcome has no free text and is returned as it is. */
|
|
298
|
+
export function redactReviewPost(post: ReviewPost, redact: (s: string) => string = redactSecrets): ReviewPost {
|
|
299
|
+
return post.posted ? post : { posted: false, reason: redact(post.reason) };
|
|
300
|
+
}
|
|
301
|
+
|
|
262
302
|
/** Structural check on a disposition set read back from a stored record. */
|
|
263
303
|
export function isFindingDispositionsShape(v: unknown): v is FindingDisposition[] {
|
|
264
304
|
return (
|
|
@@ -110,7 +110,15 @@ export type RunNoteKind =
|
|
|
110
110
|
* request would target (docs/reference/specs/pr-description.md item 5) —
|
|
111
111
|
* the summary names the branch. Published by the post-step, so a unit
|
|
112
112
|
* that ends without a pull request says why on the record and the card. */
|
|
113
|
-
| "pr_not_opened"
|
|
113
|
+
| "pr_not_opened"
|
|
114
|
+
/** A review run's post-step posted nothing to the pull request — a guard's
|
|
115
|
+
* refusal, an opt-out, no pull request resolved, GitHub's own error — and
|
|
116
|
+
* the summary names the pull request (when one was resolved) and the
|
|
117
|
+
* reason (docs/reference/specs/agent-review.md item 18). Published by the
|
|
118
|
+
* post-step beside the thread's Slack-only note, so the record says the
|
|
119
|
+
* verdict is Slack-only and a coordinator reading it never asks GitHub
|
|
120
|
+
* for a review that was never sent. */
|
|
121
|
+
| "review_not_posted";
|
|
114
122
|
|
|
115
123
|
/** Every `RunNoteKind`, as a value (a reader that filters notes by kind uses
|
|
116
124
|
* this; adding a kind to the union without adding it here is a type error). */
|
|
@@ -130,6 +138,7 @@ export const RUN_NOTE_KINDS = [
|
|
|
130
138
|
"description_turn",
|
|
131
139
|
"cold_sandbox",
|
|
132
140
|
"pr_not_opened",
|
|
141
|
+
"review_not_posted",
|
|
133
142
|
] as const satisfies readonly RunNoteKind[];
|
|
134
143
|
type _EveryKindListed = [RunNoteKind] extends [(typeof RUN_NOTE_KINDS)[number]] ? true : never;
|
|
135
144
|
const _everyKindListed: _EveryKindListed = true;
|
|
@@ -438,6 +447,25 @@ export type RunEvent =
|
|
|
438
447
|
* run record carries the PR URL as a fact of the run rather than only the
|
|
439
448
|
* channel reply's projection of it. Additive: unknown → ignored. */
|
|
440
449
|
| { type: "pr_opened"; url: string; number: number; created: boolean; seq?: number; at?: number }
|
|
450
|
+
/** The review post-step's outcome when the verdict landed
|
|
451
|
+
* (docs/reference/specs/agent-review.md item 18): the pull request it was
|
|
452
|
+
* posted to, the head it was pinned to (the carried head after a rebase,
|
|
453
|
+
* item 12) and the verdict kind when one was submitted. Published by the
|
|
454
|
+
* post-step straight to the registry BEFORE the stream finishes — the
|
|
455
|
+
* post-step runs inside the run loop, like the coding one — so the record
|
|
456
|
+
* carries the post as a fact of the run and a coordinator woken by the
|
|
457
|
+
* finish reads it there instead of asking GitHub, whose review list can
|
|
458
|
+
* lag the post it just accepted. A post that did not land is a
|
|
459
|
+
* `review_not_posted` note. Additive: unknown → ignored. */
|
|
460
|
+
| {
|
|
461
|
+
type: "review_posted";
|
|
462
|
+
repo: string;
|
|
463
|
+
number: number;
|
|
464
|
+
head: string;
|
|
465
|
+
verdict?: "approve" | "request_changes";
|
|
466
|
+
seq?: number;
|
|
467
|
+
at?: number;
|
|
468
|
+
}
|
|
441
469
|
/** One `agent:ship` round boundary (docs/reference/specs/agent-ship.md item 12): the
|
|
442
470
|
* pipeline publishes a `started` event when a round's child is dispatched
|
|
443
471
|
* and one settle event when its outcome is known (`ShipRoundOutcome`), so
|
|
@@ -370,7 +370,7 @@ export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOp
|
|
|
370
370
|
// final answer (`answer`) — are the run's story, not its steps: none counts
|
|
371
371
|
// toward `eventCount`.
|
|
372
372
|
let narrativeEvents = 0;
|
|
373
|
-
let sideFactEvents = 0; // skill_use / review_artifact / pr_description / pr_opened / ship_round: facts about the run, not steps
|
|
373
|
+
let sideFactEvents = 0; // skill_use / review_artifact / pr_description / pr_opened / review_posted / ship_round: facts about the run, not steps
|
|
374
374
|
let spanEvents = 0; // span_start / span_end (docs/reference/specs/tracing.md): timing records, not steps
|
|
375
375
|
let wrapUp: { index: number; at?: number } | undefined;
|
|
376
376
|
events.forEach((ev, index) => {
|
|
@@ -385,14 +385,15 @@ export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOp
|
|
|
385
385
|
}
|
|
386
386
|
// Side facts about the run, not steps: skill_use rides beside a use_skill
|
|
387
387
|
// call that already produced its own tool pair; review_artifact,
|
|
388
|
-
// pr_description, pr_opened and the ship_round boundaries
|
|
389
|
-
// by the dispatcher/pipeline outside the model loop
|
|
390
|
-
// any of them would distort the story.
|
|
388
|
+
// pr_description, pr_opened, review_posted and the ship_round boundaries
|
|
389
|
+
// are published by the dispatcher/pipeline outside the model loop
|
|
390
|
+
// entirely. Counting any of them would distort the story.
|
|
391
391
|
if (
|
|
392
392
|
ev.type === "skill_use" ||
|
|
393
393
|
ev.type === "review_artifact" ||
|
|
394
394
|
ev.type === "pr_description" ||
|
|
395
395
|
ev.type === "pr_opened" ||
|
|
396
|
+
ev.type === "review_posted" ||
|
|
396
397
|
ev.type === "ship_round"
|
|
397
398
|
) {
|
|
398
399
|
sideFactEvents++;
|
|
@@ -4,9 +4,11 @@ import type { RunEvent } from "./runEvents.js";
|
|
|
4
4
|
import { isHeadMaterial, isSpanRecord } from "./runEvents.js";
|
|
5
5
|
import { isHandoffShape, type Handoff } from "./ship/handoff.js";
|
|
6
6
|
import {
|
|
7
|
+
type FindingDisposition,
|
|
7
8
|
isFindingDispositionsShape,
|
|
9
|
+
isReviewPostShape,
|
|
8
10
|
isReviewVerdictShape,
|
|
9
|
-
type
|
|
11
|
+
type ReviewPost,
|
|
10
12
|
type ReviewVerdict,
|
|
11
13
|
} from "./reviewVerdict.js";
|
|
12
14
|
import { IDEMPOTENCY_KEY_PATTERN, INSTANCE_ID_PATTERN } from "./coordinator/contract.js";
|
|
@@ -114,6 +116,12 @@ export interface RunRecord {
|
|
|
114
116
|
/** The head a review run reviewed and posted against (7 to 40 lowercase hex),
|
|
115
117
|
* after the settle; present only on a review run that pinned one. */
|
|
116
118
|
reviewHead?: string;
|
|
119
|
+
/** How the review run's post-step ended (docs/reference/specs/agent-review.md
|
|
120
|
+
* item 18): the verdict posted to a named pull request at a pinned head, or
|
|
121
|
+
* not posted with the reason — what a coordinator's `read-record` answers
|
|
122
|
+
* `reviewPosted` from before it asks GitHub. Present only on a review run
|
|
123
|
+
* that reached its post-step; absent on records written before it existed. */
|
|
124
|
+
reviewPost?: ReviewPost;
|
|
117
125
|
/** The dispositions a fix round submitted through `submit_dispositions`
|
|
118
126
|
* (docs/reference/specs/agent-ship.md item 6), the last call's set, redacted;
|
|
119
127
|
* present only on a coding run dispatched as a fix round that submitted one. */
|
|
@@ -502,6 +510,7 @@ export function isRunRecord(v: unknown): v is RunRecord {
|
|
|
502
510
|
if (r.reviewHead !== undefined && (typeof r.reviewHead !== "string" || !REVIEW_HEAD_PATTERN.test(r.reviewHead)))
|
|
503
511
|
return false;
|
|
504
512
|
if (r.dispositions !== undefined && !isFindingDispositionsShape(r.dispositions)) return false;
|
|
513
|
+
if (r.reviewPost !== undefined && !isReviewPostShape(r.reviewPost)) return false;
|
|
505
514
|
if (r.profile !== undefined && !isRunProfileRecord(r.profile)) return false;
|
|
506
515
|
// A parent is named by a run id (item 46): the same shape as the record's own.
|
|
507
516
|
if (r.parentRunId !== undefined && (typeof r.parentRunId !== "string" || !RUN_ID_PATTERN.test(r.parentRunId)))
|
|
@@ -79,6 +79,10 @@ export interface ChildContract {
|
|
|
79
79
|
issue: { repo: string; number: number } | undefined;
|
|
80
80
|
}
|
|
81
81
|
|
|
82
|
+
/** The PR-title gate's name, exported so another module can name the same
|
|
83
|
+
* gate without a copied string (`npm run check:pr-title`; CI's `title` check). */
|
|
84
|
+
export const PR_TITLE_GUARD = "check:pr-title";
|
|
85
|
+
|
|
82
86
|
/** The guards a child may not weaken (AGENTS.md's Commands table), one line each on what they refuse. */
|
|
83
87
|
export const GUARDS: readonly Guard[] = [
|
|
84
88
|
{
|
|
@@ -96,6 +100,11 @@ export const GUARDS: readonly Guard[] = [
|
|
|
96
100
|
refuses:
|
|
97
101
|
"a new imprint in the public tree (a company, a person, a tracker reference, a plan id, a platform id, a date); the recorded list only shrinks",
|
|
98
102
|
},
|
|
103
|
+
{
|
|
104
|
+
name: PR_TITLE_GUARD,
|
|
105
|
+
refuses:
|
|
106
|
+
"a title whose type, scope or grammar is not the changelog line, the scope being one of the code map's Areas",
|
|
107
|
+
},
|
|
99
108
|
{
|
|
100
109
|
name: "decisions:check",
|
|
101
110
|
refuses:
|
|
@@ -4,11 +4,10 @@
|
|
|
4
4
|
// Workflow instance in the bot's shim Worker with no model turn and no
|
|
5
5
|
// credential — decides between the steps it asks the bot for. The Workflow
|
|
6
6
|
// asks `nextAction`, performs it (a bot route, a `waitForEvent`, a sleep) and
|
|
7
|
-
// feeds the answer to `applyReturn`; everything the
|
|
8
|
-
//
|
|
9
|
-
//
|
|
10
|
-
//
|
|
11
|
-
// with the loop's process gone.
|
|
7
|
+
// feeds the answer to `applyReturn`; everything the pipeline decides — which
|
|
8
|
+
// round is next, what a child's end means, when a cap ends the pipeline, when
|
|
9
|
+
// the pull request is merge-ready — is decided here over the step returns, so
|
|
10
|
+
// the endings hold with no process of the pipeline's own to die.
|
|
12
11
|
//
|
|
13
12
|
// Two machines, both pure. The plan cursor walks a plan record's unit graph:
|
|
14
13
|
// which units are ready (their dependencies merged), which one a failure
|
|
@@ -345,8 +344,12 @@ export type ChildFacts =
|
|
|
345
344
|
description?: boolean;
|
|
346
345
|
/** A review child's verdict. */
|
|
347
346
|
verdict?: { verdict: ReviewVerdictKind; summary?: string; findings: Finding[] };
|
|
348
|
-
/** Whether the review child's verdict landed on the pull request
|
|
347
|
+
/** Whether the review child's verdict landed on the pull request — the
|
|
348
|
+
* child's own record of its post, or GitHub's review list when the
|
|
349
|
+
* record is silent (http-ingress.md item 9). */
|
|
349
350
|
reviewPosted?: boolean;
|
|
351
|
+
/** Why the child recorded no post, when it recorded one it chose or failed. */
|
|
352
|
+
reviewPostReason?: string;
|
|
350
353
|
reviewHead?: string;
|
|
351
354
|
/** A fix child's dispositions. */
|
|
352
355
|
dispositions?: FindingDisposition[];
|
|
@@ -375,7 +378,7 @@ export type StepReturn =
|
|
|
375
378
|
| { type: "merge"; step: string; outcome: "pending" | "refused"; reason: string; at: number }
|
|
376
379
|
| { type: "sleep"; step: string };
|
|
377
380
|
|
|
378
|
-
/** How one unit's pipeline ended — the truthful vocabulary the
|
|
381
|
+
/** How one unit's pipeline ended — the truthful vocabulary the ship pipeline
|
|
379
382
|
* has, plus the merge's own: `merged` by the runner, or found merged (`by:
|
|
380
383
|
* other` — a person's merge, or an earlier attempt's that died after it, so
|
|
381
384
|
* the runner merged nothing), `merge_ready` for a person, `merge_refused` by
|
|
@@ -512,7 +515,7 @@ function waitSliceMs(clock: number, until: number): number {
|
|
|
512
515
|
}
|
|
513
516
|
|
|
514
517
|
/** The child's budget: its preset's own, clipped to the pipeline's remaining
|
|
515
|
-
* wall clock (
|
|
518
|
+
* wall clock (agent-ship item 8's clip), never under the two minutes a
|
|
516
519
|
* spawn accepts. */
|
|
517
520
|
function budgetMinutesFor(s: UnitPipelineState, preset: ChildPreset): number {
|
|
518
521
|
return Math.max(2, Math.min(s.input.childMinutes[preset], Math.floor(remainingMs(s) / MIN)));
|
|
@@ -606,7 +609,7 @@ const roundNote = (round: RoundRef, outcome: ShipRoundOutcome): CoordinatorNote
|
|
|
606
609
|
outcome,
|
|
607
610
|
});
|
|
608
611
|
|
|
609
|
-
/** Start a round if the reservation holds (
|
|
612
|
+
/** Start a round if the reservation holds (agent-ship item 8's check: a
|
|
610
613
|
* child clipped under the reserve cannot do useful work). */
|
|
611
614
|
function enterRound(s: UnitPipelineState, round: RoundRef, notes: CoordinatorNote[] = []): Transition {
|
|
612
615
|
const remaining = remainingMs(s);
|
|
@@ -720,13 +723,15 @@ function settleReview(
|
|
|
720
723
|
const notes = [roundNote(round, verdict.verdict)];
|
|
721
724
|
if (verdict.verdict === "approve") {
|
|
722
725
|
// Merge-ready stands on the POSTED approval: an approve whose post did
|
|
723
|
-
// not land left no approving review on the pull request.
|
|
726
|
+
// not land left no approving review on the pull request. The reason is
|
|
727
|
+
// the child's own when it recorded one; how to continue is the report's
|
|
728
|
+
// re-issue line, in the runner's words (`renderUnitReport`).
|
|
724
729
|
if (facts.reviewPosted === false)
|
|
725
730
|
return end(
|
|
726
731
|
next,
|
|
727
732
|
{
|
|
728
733
|
kind: "aborted",
|
|
729
|
-
reason: `⚠️ The review approved, but the approval could not be posted
|
|
734
|
+
reason: `⚠️ The review approved, but the approval could not be posted${facts.reviewPostReason !== undefined ? ` (${facts.reviewPostReason})` : ""} — the pull request carries no approving review.`,
|
|
730
735
|
round,
|
|
731
736
|
reviewRounds: next.reviewRounds,
|
|
732
737
|
},
|
|
@@ -1030,7 +1035,7 @@ function splitReport(s: UnitPipelineState): string {
|
|
|
1030
1035
|
return lines.join("\n");
|
|
1031
1036
|
}
|
|
1032
1037
|
|
|
1033
|
-
/** The thread's report for a unit's ending — the
|
|
1038
|
+
/** The thread's report for a unit's ending — the ship pipeline's own words
|
|
1034
1039
|
* for the endings it has, and the merge's for the ones it gains. */
|
|
1035
1040
|
export function renderUnitReport(s: UnitPipelineState): string {
|
|
1036
1041
|
const e = s.ending;
|
|
@@ -45,6 +45,7 @@ export const STREAMED_SPANS = [
|
|
|
45
45
|
"run.description_turn",
|
|
46
46
|
"run.observe_workspace",
|
|
47
47
|
"run.pr_post_step",
|
|
48
|
+
"run.review_post_step",
|
|
48
49
|
"run.reading_diff_join",
|
|
49
50
|
"run.pr_description_join",
|
|
50
51
|
"model.turn",
|
|
@@ -92,6 +93,7 @@ const GETTING_READY: ReadonlySet<string> = new Set([
|
|
|
92
93
|
const FINISHING_UP: ReadonlySet<string> = new Set([
|
|
93
94
|
"run.observe_workspace",
|
|
94
95
|
"run.pr_post_step",
|
|
96
|
+
"run.review_post_step",
|
|
95
97
|
"run.reading_diff_join",
|
|
96
98
|
"run.pr_description_join",
|
|
97
99
|
]);
|
|
@@ -150,6 +152,7 @@ export const PARENTS: Readonly<Record<string, readonly string[]>> = {
|
|
|
150
152
|
"run.description_turn": ["request", "ship.round"],
|
|
151
153
|
"run.observe_workspace": ["request", "ship.round"],
|
|
152
154
|
"run.pr_post_step": ["request", "ship.round"],
|
|
155
|
+
"run.review_post_step": ["request", "ship.round"],
|
|
153
156
|
"run.reading_diff_join": ["request", "ship.round"],
|
|
154
157
|
"run.pr_description_join": ["request", "ship.round"],
|
|
155
158
|
"model.turn": ["run.agent"],
|
|
@@ -52,6 +52,9 @@ export function shimRoute(pathname: string): string | undefined {
|
|
|
52
52
|
if (pathname === "/costs" || pathname.startsWith("/costs/")) return "costs";
|
|
53
53
|
if (pathname.startsWith("/api/")) return "api";
|
|
54
54
|
if (pathname.startsWith("/admin/")) return "admin";
|
|
55
|
+
// The model proxy's two routes (docs/reference/specs/model-proxy.md): a bounded
|
|
56
|
+
// request per model call, forwarded to the container like everything else.
|
|
57
|
+
if (pathname === "/v1/messages" || pathname === "/v1/chat/completions") return "model-proxy";
|
|
55
58
|
if (pathname === "/docs" || pathname.startsWith("/docs/")) return "docs";
|
|
56
59
|
if (pathname === "/" || pathname === "/index.html") return "page";
|
|
57
60
|
return "other";
|