@coreplane/switchboard 1.207.1 → 1.209.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (27) hide show
  1. package/dist/assets/config/config.example.yaml +11 -11
  2. package/dist/assets/package-lock.json +3 -3
  3. package/dist/assets/package.json +1 -1
  4. package/dist/assets/source.json +3 -3
  5. package/dist/assets/src/agents/registry.ts +34 -16
  6. package/dist/assets/src/core/coordinator/contract.ts +11 -0
  7. package/dist/assets/src/core/coordinator/driver.ts +22 -1
  8. package/dist/assets/src/core/reviewVerdict.ts +40 -0
  9. package/dist/assets/src/core/runEvents.ts +29 -1
  10. package/dist/assets/src/core/runFriction.ts +5 -4
  11. package/dist/assets/src/core/runRecord.ts +10 -1
  12. package/dist/assets/src/core/ship/contract.ts +9 -0
  13. package/dist/assets/src/core/ship/coordinator.ts +17 -12
  14. package/dist/assets/src/core/trace/streamSpans.ts +3 -0
  15. package/dist/assets/src/core/trace/workerTrace.ts +3 -0
  16. package/dist/assets/web/dist/.vite/manifest.json +18 -18
  17. package/dist/assets/web/dist/assets/{ResidentDetailPage-D3P21yeI.js → ResidentDetailPage-CV3wfAIP.js} +1 -1
  18. package/dist/assets/web/dist/assets/{ResidentsIndexPage-DOYqnZ1q.js → ResidentsIndexPage-CLkWc50b.js} +1 -1
  19. package/dist/assets/web/dist/assets/{RunRoutePage-OmvrvPXY.js → RunRoutePage-CQYRfQ_B.js} +4 -4
  20. package/dist/assets/web/dist/assets/{RunsIndexPage-DWbSQtL4.js → RunsIndexPage-BLRPp_gk.js} +1 -1
  21. package/dist/assets/web/dist/assets/{ScheduledPage-CPKfJ4mR.js → ScheduledPage-C-VO4Ddl.js} +1 -1
  22. package/dist/assets/web/dist/assets/{StatusDot-COr8jTyM.js → StatusDot-CIAoBB5Y.js} +1 -1
  23. package/dist/assets/web/dist/assets/{Tooltip-fOqTZkNT.js → Tooltip-DEL1ic4g.js} +1 -1
  24. package/dist/assets/web/dist/assets/{dist-BVjAWgkb.js → dist-CawBR4t8.js} +1 -1
  25. package/dist/assets/web/dist/assets/{main-DZbJaqUb.js → main-CctUbVOl.js} +2 -2
  26. package/dist/cli.js +6570 -6602
  27. package/package.json +1 -1
@@ -278,20 +278,20 @@ workspaceDir: ./workspaces
278
278
  # tokenEnv: MEMORY_TOKEN # env var holding the Worker's bearer (default)
279
279
 
280
280
  # agent:ship pipeline caps (docs/reference/specs/agent-ship.md). `agent:ship in owner/repo:
281
- # <task>` runs the coding → review → fix loop to LGTM as one pipeline: at most
281
+ # <task>` hands the coding → review → fix loop to LGTM to the plan runner — the
282
+ # ShipCoordinator Workflow in the bot Worker, whose rounds are child runs — which
283
+ # needs the `coordinator` ingress entry, its grants, run history on the state
284
+ # Worker and PUBLIC_BASE_URL (docs/how-to/turn-features-on-and-off.md); without
285
+ # them the request is refused naming what is missing. Per unit: at most
282
286
  # `maxRounds` review rounds and `maxMinutes` minutes of wall clock — whichever
283
- # hits first ends the loop, and each child round's own budget is clipped to the
284
- # remaining pipeline time. `maxMinutes` is the ship preset's declared budget:
285
- # a channel or user `boundary.maxMinutes` and a per-message `budget:` directive
286
- # clip it like any preset's, and the card says so. Defaults shown; both must be
287
- # integers >= 1.
287
+ # hits first ends the unit, and each child round's own budget is clipped to the
288
+ # remaining time. `maxMinutes` is the ship preset's declared budget: a channel or
289
+ # user `boundary.maxMinutes` and a per-message `budget:` directive clip it like
290
+ # any preset's, and the card says so. Defaults shown; both must be integers >= 1;
291
+ # any other key under `ship` fails the load by name.
288
292
  # ship:
289
- # maxRounds: 3 # review rounds per pipeline
293
+ # maxRounds: 3 # review rounds per unit
290
294
  # maxMinutes: 120 # the ship preset's wall-clock budget in minutes (the registry's default)
291
- # coordinator: false # true hands every agent:ship request to the plan runner — the ShipCoordinator
292
- # # Workflow in the bot Worker — instead of the in-process round loop; needs the
293
- # # `coordinator` ingress entry, its grant, run history on the state Worker and
294
- # # PUBLIC_BASE_URL (see docs/how-to/turn-features-on-and-off.md)
295
295
 
296
296
  # The fan-out cap a spawning run meets (docs/reference/specs/agent-conductor.md).
297
297
  # `agent:conductor` starts child runs as the person who asked — each an
@@ -1,12 +1,12 @@
1
1
  {
2
2
  "name": "switchboard",
3
- "version": "1.207.1",
3
+ "version": "1.209.0",
4
4
  "lockfileVersion": 3,
5
5
  "requires": true,
6
6
  "packages": {
7
7
  "": {
8
8
  "name": "switchboard",
9
- "version": "1.207.1",
9
+ "version": "1.209.0",
10
10
  "license": "Apache-2.0",
11
11
  "workspaces": [
12
12
  "web",
@@ -18999,7 +18999,7 @@
18999
18999
  },
19000
19000
  "packages/switchboard": {
19001
19001
  "name": "@coreplane/switchboard",
19002
- "version": "1.207.1",
19002
+ "version": "1.209.0",
19003
19003
  "license": "Apache-2.0",
19004
19004
  "dependencies": {
19005
19005
  "@anthropic-ai/sdk": "^0.124.0",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "switchboard",
3
- "version": "1.207.1",
3
+ "version": "1.209.0",
4
4
  "private": true,
5
5
  "description": "Mention it in Slack and an agent reviews the PR, ships the fix, or answers the question — on the model you choose, with its tools running where you decide.",
6
6
  "license": "Apache-2.0",
@@ -1,5 +1,5 @@
1
1
  {
2
- "version": "1.207.1",
3
- "commit": "a5e85d6a364f558fdf079ec32e85eb6deadf5033",
4
- "builtAt": "2026-09-14T01:09:34.713Z"
2
+ "version": "1.209.0",
3
+ "commit": "921bd6b7bd8c7cd979c68435a47bc7ad169a3893",
4
+ "builtAt": "2026-09-14T05:12:19.226Z"
5
5
  }
@@ -1,8 +1,8 @@
1
- // Agent definitions. An agent is a system prompt + toolset + machine class + turn budget.
1
+ // Agent definitions. An agent is a system prompt + toolset + machine class + wall-clock budget.
2
2
  import type { Effort } from "../effort.js";
3
3
  import type { CacheTtl } from "../providers/types.js";
4
4
  import { BASH_TIMEOUT_MAX_MS } from "../execution/bashTimeout.js";
5
- import { CONTRACT_HEADING, CONTRACT_SECTION_HEADINGS } from "../core/ship/contract.js";
5
+ import { CONTRACT_HEADING, CONTRACT_SECTION_HEADINGS, PR_TITLE_GUARD } from "../core/ship/contract.js";
6
6
  // Which model runs it is resolved separately by the config layers, so any
7
7
  // agent can run on any configured provider/model.
8
8
 
@@ -42,13 +42,37 @@ export function machineNeedsRepo(machine: MachineClass): boolean {
42
42
  export const IDENTITIES = ["none", "read", "write"] as const;
43
43
  export type Identity = (typeof IDENTITIES)[number];
44
44
 
45
+ /** The pace that marks a run as looping rather than working: a model turn
46
+ * every ten seconds, sustained for the whole wall clock. A busy run takes
47
+ * 20–40 s a turn (a model think plus a tool call), so a run that averages six
48
+ * a minute from start to end is re-issuing calls, not making progress — and
49
+ * its turn cap ends it before the wall clock would, with a write-up that
50
+ * says so (docs/reference/specs/run-loop.md item 1). */
51
+ export const RUNAWAY_TURNS_PER_MINUTE = 6;
52
+
53
+ /** The turn cap a wall clock implies: `maxMinutes × RUNAWAY_TURNS_PER_MINUTE`.
54
+ * Every preset that runs the loop derives its `maxTurns` from this, so the
55
+ * cap is never a number a good run reaches — the minutes are the budget. */
56
+ export function runawayTurnCap(maxMinutes: number): number {
57
+ return maxMinutes * RUNAWAY_TURNS_PER_MINUTE;
58
+ }
59
+
60
+ /** A loop-running preset's budget as one fact: the wall clock, and the runaway
61
+ * guard derived from it. */
62
+ function loopBudget(maxMinutes: number): Pick<AgentDef, "maxMinutes" | "maxTurns"> {
63
+ return { maxMinutes, maxTurns: runawayTurnCap(maxMinutes) };
64
+ }
65
+
45
66
  export interface AgentDef {
46
67
  name: string;
47
68
  description: string;
48
69
  system: string;
49
70
  /** key into TOOLSETS: "full" | "readonly" | "web" | "assistant" | "explore" | "conductor" | "none" */
50
71
  toolset: "full" | "readonly" | "web" | "assistant" | "explore" | "conductor" | "none";
51
- /** backstop only — the wall clock below is the real budget */
72
+ /** The runaway guard, not a budget: `runawayTurnCap(maxMinutes)` for every
73
+ * preset that runs the loop (`loopBudget`). The wall clock below is the
74
+ * budget; a run that reaches this cap first was pacing like a loop, and its
75
+ * write-up says so. The proxy refuses model calls past it too. */
52
76
  maxTurns: number;
53
77
  maxTokens: number;
54
78
  /** hard wall-clock budget for the tool loop; at the deadline the agent is
@@ -97,7 +121,7 @@ export interface AgentDef {
97
121
  // and rendered against the head sha at render time, so a repush is a
98
122
  // re-render by Switchboard — the agent only resubmits when the CONTENT (line
99
123
  // numbers included) changed.
100
- const PR_DESCRIPTION_TEMPLATE = `PR description — submit it with the submit_pr_description tool for EVERY PR (this is the default, not something to wait to be asked for). Switchboard renders the GitHub body from the object you submit, so never author PR-body markdown yourself. Content contract per field (each renders as its own section): prose is unwrapped — no hard line breaks inside a paragraph. Always hyperlink the triggering issue/request. Never fabricate validation — state exactly what you ran and the real result. Keep each field concise, not padded.
124
+ const PR_DESCRIPTION_TEMPLATE = `PR description — submit it with the submit_pr_description tool for EVERY PR (this is the default, not something to wait to be asked for). Switchboard renders the GitHub body from the object you submit, so never author PR-body markdown yourself. Before submitting, judge your title with the ${PR_TITLE_GUARD} gate — \`npm run check:pr-title -- "<title>"\` — and submit only a title it accepts; the same gate refuses the PR in CI. Content contract per field (each renders as its own section): prose is unwrapped — no hard line breaks inside a paragraph. Always hyperlink the triggering issue/request. Never fabricate validation — state exactly what you ran and the real result. Keep each field concise, not padded.
101
125
  EVERY PR includes one that already exists when you push — opened by a person, by dependabot, or by an earlier run. After EVERY push to such a PR: read its current title and body (\`github_issue_get\` with the PR number works for pull requests; \`gh pr view\` where gh exists), judge them against the change as it now stands at the pushed head, and submit the object that describes the PR as it is NOW — carry forward what the existing body says that is still true (a dependency bump's release notes belong in whatWhy), add what you changed, and anchor the Tour at the new head. Switchboard replaces the PR's title and body with your rendering. A description that describes an earlier state of its branch is a bug; "it is someone else's PR" is never a reason to leave it.
102
126
  - **title**: the PR title — one line naming the change, specific enough to pick out of a PR list.
103
127
  - **TL;DR** (\`tldr\`, rendered first): two sentences for a naive reader with zero context — what this PR does and why it matters.
@@ -401,9 +425,8 @@ export const AGENTS: Record<string, AgentDef> = {
401
425
  // and mints no credential of its own.
402
426
  machine: "none",
403
427
  identity: "none",
404
- maxTurns: 8, // a repo read is 2-3 calls (repos → tree → file); an issue action 1-2; still fast
405
428
  maxTokens: 16000,
406
- maxMinutes: 5,
429
+ ...loopBudget(5),
407
430
  },
408
431
  coding: {
409
432
  name: "coding",
@@ -411,9 +434,8 @@ export const AGENTS: Record<string, AgentDef> = {
411
434
  system: CODING_SYSTEM,
412
435
  residentSystem: CODING_SYSTEM_RESIDENT,
413
436
  toolset: "full",
414
- maxTurns: 60, // scoping is capped at ~5 calls by the prompt; this is implementation room
415
437
  maxTokens: 64000,
416
- maxMinutes: 45,
438
+ ...loopBudget(45),
417
439
  // Coding steps run long: a single model turn can take 5-6 minutes and
418
440
  // installs/tests add more — a 5m cache entry would expire between
419
441
  // requests, so the 2× write buys reads for the whole run.
@@ -431,9 +453,8 @@ export const AGENTS: Record<string, AgentDef> = {
431
453
  toolset: "readonly",
432
454
  machine: "repo-resident",
433
455
  identity: "read", // a read-scoped token and a read-only worktree: it cannot post or push from inside
434
- maxTurns: 30, // backstop only; wall clock is the real budget (12 bound at ~4 min in practice)
435
456
  maxTokens: 64000,
436
- maxMinutes: 25, // safety net, not the mechanism — typical reviews land in ~5
457
+ ...loopBudget(25), // a safety net — typical reviews land in ~5 minutes
437
458
  effort: "medium", // fast turns; one big-context pass does the deep work
438
459
  },
439
460
  ship: {
@@ -468,9 +489,8 @@ export const AGENTS: Record<string, AgentDef> = {
468
489
  toolset: "web",
469
490
  machine: "none", // web I/O only; no workspace is provisioned
470
491
  identity: "none",
471
- maxTurns: 12,
472
492
  maxTokens: 24000,
473
- maxMinutes: 8,
493
+ ...loopBudget(8),
474
494
  effort: "medium",
475
495
  },
476
496
  explore: {
@@ -483,9 +503,8 @@ export const AGENTS: Record<string, AgentDef> = {
483
503
  // review depends on: a two-hour job shares no container with anyone.
484
504
  machine: "repo-cold",
485
505
  identity: "read", // a read-scoped token: it can clone and read, never push — whatever the caller holds
486
- maxTurns: 150, // a backstop for a two-hour loop of batched checks; the wall clock is the budget
487
506
  maxTokens: 64000,
488
- maxMinutes: 120,
507
+ ...loopBudget(120),
489
508
  // A detached job polled across calls makes long steps: a 5m cache entry
490
509
  // would expire between them, so the 2× write buys reads for the whole run.
491
510
  cacheTtl: "1h",
@@ -501,9 +520,8 @@ export const AGENTS: Record<string, AgentDef> = {
501
520
  // dispatcher, the GitHub reads are REST in the bot process.
502
521
  machine: "none",
503
522
  identity: "none",
504
- maxTurns: 40, // a spawn, then a poll per child every few minutes; the wall clock is the budget
505
523
  maxTokens: 32000,
506
- maxMinutes: 120, // long enough to outlast a coding child; every child is capped by what remains of it
524
+ ...loopBudget(120), // long enough to outlast a coding child; every child is capped by what remains of it
507
525
  // No built-in effort: the deployment decides, as for coding.
508
526
  },
509
527
  };
@@ -145,6 +145,11 @@ export interface CoordinatorUnit {
145
145
  /** The unit's board issue in the repository, when one titled by the unit id exists — the handoff's destination. */
146
146
  issue?: number;
147
147
  pr?: { number: number; url: string };
148
+ /** Resume at review (agent-ship item 10): the open pull request of ship's own
149
+ * the requester named, so the unit's pipeline opens at its first review round
150
+ * — no pre-check, no branch, no round 0. A task string's row only; written by
151
+ * the hand-off, read by the driver into the machine's input. */
152
+ resume?: { pr: number; headSha?: string; url?: string };
148
153
  /** The round boundaries the coordinator reported, oldest first (the `ship_round` vocabulary). */
149
154
  rounds: Array<{ index: number; agent: string; outcome: string; at: number }>;
150
155
  /** How the unit ended: the ending's kind and the thread's report, when it has. */
@@ -163,6 +168,11 @@ const isOptionalText = (v: unknown): boolean => v === undefined || isText(v);
163
168
  const isFinite = (v: unknown): v is number => typeof v === "number" && Number.isFinite(v);
164
169
  const isObject = (v: unknown): v is Record<string, unknown> => typeof v === "object" && v !== null;
165
170
  const isPr = (v: unknown): boolean => isObject(v) && isFinite(v.number) && isText(v.url, 2048);
171
+ const isResume = (v: unknown): boolean =>
172
+ isObject(v) &&
173
+ isFinite(v.pr) &&
174
+ (v.headSha === undefined || isText(v.headSha)) &&
175
+ (v.url === undefined || isText(v.url, 2048));
166
176
 
167
177
  /** Structural check on a record from outside the process (a Worker response, an HTTP body). */
168
178
  export function isCoordinatorInstance(v: unknown): v is CoordinatorInstance {
@@ -194,6 +204,7 @@ export function isCoordinatorUnit(v: unknown): v is CoordinatorUnit {
194
204
  if (!isOptionalText(r.threadKey) || !isOptionalText(r.sourceUrl)) return false;
195
205
  if (r.issue !== undefined && !isFinite(r.issue)) return false;
196
206
  if (r.pr !== undefined && !isPr(r.pr)) return false;
207
+ if (r.resume !== undefined && !isResume(r.resume)) return false;
197
208
  if (
198
209
  !Array.isArray(r.rounds) ||
199
210
  r.rounds.length > MAX_ROUNDS ||
@@ -233,7 +233,18 @@ function readRecordReturn(step: string, a: BotAnswer): StepReturn {
233
233
  // The typed artifacts as the bot's record carries them — shape-checked where
234
234
  // they were written (the run record's validator), read here as they are.
235
235
  const facts = run as unknown as Omit<Extract<ChildFacts, { finished: true }>, "finished" | "status">;
236
- const { finalReply, pr, headSha, description, verdict, reviewPosted, reviewHead, dispositions, handoff } = facts;
236
+ const {
237
+ finalReply,
238
+ pr,
239
+ headSha,
240
+ description,
241
+ verdict,
242
+ reviewPosted,
243
+ reviewPostReason,
244
+ reviewHead,
245
+ dispositions,
246
+ handoff,
247
+ } = facts;
237
248
  return {
238
249
  type: "read-record",
239
250
  step,
@@ -246,6 +257,7 @@ function readRecordReturn(step: string, a: BotAnswer): StepReturn {
246
257
  ...(description !== undefined ? { description } : {}),
247
258
  ...(verdict !== undefined ? { verdict } : {}),
248
259
  ...(reviewPosted !== undefined ? { reviewPosted } : {}),
260
+ ...(reviewPostReason !== undefined ? { reviewPostReason } : {}),
249
261
  ...(reviewHead !== undefined ? { reviewHead } : {}),
250
262
  ...(dispositions !== undefined ? { dispositions } : {}),
251
263
  ...(handoff !== undefined ? { handoff } : {}),
@@ -399,6 +411,10 @@ async function runUnit(
399
411
  const start = readUnitStart(
400
412
  answerOf("unit-start", await step.do(`${unit}/start`, STEP_CONFIG, () => call(bot, "unit-start", tag))),
401
413
  );
414
+ // A resume at review (agent-ship item 10) rides the unit's row: the pull
415
+ // request of ship's own the requester named opens the pipeline at its first
416
+ // review round, with no pre-check, no branch and no round 0.
417
+ const resume = plan.units.find((u) => u.unit === unit)?.resume;
402
418
  let state: UnitPipelineState = openUnitPipeline(
403
419
  {
404
420
  unit: { id: unit, branch: node.branch },
@@ -411,6 +427,7 @@ async function runUnit(
411
427
  // approved at its head and the checks are green; any other branch — a
412
428
  // task string's ship branch — waits for a person.
413
429
  merge: parsePlanBranch(node.branch) !== undefined ? "runner" : "person",
430
+ ...(resume !== undefined ? { resume } : {}),
414
431
  },
415
432
  start.at,
416
433
  );
@@ -425,10 +442,14 @@ async function runUnit(
425
442
  const body = { ...tag, index: note.index, agent: note.agent, outcome: note.outcome };
426
443
  await step.do(`${unit}/note/${++notes}`, STEP_CONFIG, () => call(bot, "round", body));
427
444
  } else {
445
+ // The last coding child's run is named so the bot can put its handoff
446
+ // — the deviations it recorded — on the unit's board issue beside the
447
+ // ending (agent-ship item 14).
428
448
  const body = {
429
449
  ...tag,
430
450
  ending: { kind: note.ending.kind, report: renderUnitReport(state) },
431
451
  ...(state.pr !== undefined ? { pr: state.pr } : {}),
452
+ ...(state.lastCodingRunId !== undefined ? { codingRunId: state.lastCodingRunId } : {}),
432
453
  };
433
454
  await step.do(`${unit}/end`, STEP_CONFIG, () => call(bot, "unit-end", body));
434
455
  }
@@ -259,6 +259,46 @@ export function isReviewVerdictShape(v: unknown): v is ReviewVerdict {
259
259
  return true;
260
260
  }
261
261
 
262
+ /** How a review run's post-step ended, as the run's record carries it
263
+ * (docs/reference/specs/agent-review.md item 18; run-history item 2): the verdict
264
+ * landed on a named pull request pinned to `head` (the verdict kind rides when
265
+ * one was submitted — a review that posted without a verdict posts the
266
+ * no-verdict line), or nothing landed and `reason` says why — a guard's
267
+ * refusal, an opt-out, no pull request, GitHub's own error. A coordinator's
268
+ * `read-record` answers `reviewPosted` from this before it asks GitHub, whose
269
+ * review list can lag a post it accepted a second ago. */
270
+ export type ReviewPost =
271
+ | { posted: true; target: { repo: string; number: number }; head: string; verdict?: ReviewVerdictKind }
272
+ | { posted: false; reason: string };
273
+
274
+ const REVIEW_POST_HEAD = /^[0-9a-f]{7,40}$/;
275
+
276
+ /** Structural check on a review post read back from a stored record: a posted
277
+ * outcome names its pull request and a 7-to-40-hex head, its verdict (when
278
+ * present) a known kind; a skipped one carries a string reason. */
279
+ export function isReviewPostShape(v: unknown): v is ReviewPost {
280
+ if (!isRecordLike(v)) return false;
281
+ if (v.posted === false) return typeof v.reason === "string";
282
+ if (v.posted !== true) return false;
283
+ const target = v.target;
284
+ if (
285
+ !isRecordLike(target) ||
286
+ typeof target.repo !== "string" ||
287
+ typeof target.number !== "number" ||
288
+ !Number.isInteger(target.number) ||
289
+ target.number <= 0
290
+ )
291
+ return false;
292
+ if (typeof v.head !== "string" || !REVIEW_POST_HEAD.test(v.head)) return false;
293
+ return v.verdict === undefined || VERDICT_KINDS.includes(v.verdict as string);
294
+ }
295
+
296
+ /** The skip's reason through the redaction seam (it may carry GitHub's own
297
+ * words); a posted outcome has no free text and is returned as it is. */
298
+ export function redactReviewPost(post: ReviewPost, redact: (s: string) => string = redactSecrets): ReviewPost {
299
+ return post.posted ? post : { posted: false, reason: redact(post.reason) };
300
+ }
301
+
262
302
  /** Structural check on a disposition set read back from a stored record. */
263
303
  export function isFindingDispositionsShape(v: unknown): v is FindingDisposition[] {
264
304
  return (
@@ -110,7 +110,15 @@ export type RunNoteKind =
110
110
  * request would target (docs/reference/specs/pr-description.md item 5) —
111
111
  * the summary names the branch. Published by the post-step, so a unit
112
112
  * that ends without a pull request says why on the record and the card. */
113
- | "pr_not_opened";
113
+ | "pr_not_opened"
114
+ /** A review run's post-step posted nothing to the pull request — a guard's
115
+ * refusal, an opt-out, no pull request resolved, GitHub's own error — and
116
+ * the summary names the pull request (when one was resolved) and the
117
+ * reason (docs/reference/specs/agent-review.md item 18). Published by the
118
+ * post-step beside the thread's Slack-only note, so the record says the
119
+ * verdict is Slack-only and a coordinator reading it never asks GitHub
120
+ * for a review that was never sent. */
121
+ | "review_not_posted";
114
122
 
115
123
  /** Every `RunNoteKind`, as a value (a reader that filters notes by kind uses
116
124
  * this; adding a kind to the union without adding it here is a type error). */
@@ -130,6 +138,7 @@ export const RUN_NOTE_KINDS = [
130
138
  "description_turn",
131
139
  "cold_sandbox",
132
140
  "pr_not_opened",
141
+ "review_not_posted",
133
142
  ] as const satisfies readonly RunNoteKind[];
134
143
  type _EveryKindListed = [RunNoteKind] extends [(typeof RUN_NOTE_KINDS)[number]] ? true : never;
135
144
  const _everyKindListed: _EveryKindListed = true;
@@ -438,6 +447,25 @@ export type RunEvent =
438
447
  * run record carries the PR URL as a fact of the run rather than only the
439
448
  * channel reply's projection of it. Additive: unknown → ignored. */
440
449
  | { type: "pr_opened"; url: string; number: number; created: boolean; seq?: number; at?: number }
450
+ /** The review post-step's outcome when the verdict landed
451
+ * (docs/reference/specs/agent-review.md item 18): the pull request it was
452
+ * posted to, the head it was pinned to (the carried head after a rebase,
453
+ * item 12) and the verdict kind when one was submitted. Published by the
454
+ * post-step straight to the registry BEFORE the stream finishes — the
455
+ * post-step runs inside the run loop, like the coding one — so the record
456
+ * carries the post as a fact of the run and a coordinator woken by the
457
+ * finish reads it there instead of asking GitHub, whose review list can
458
+ * lag the post it just accepted. A post that did not land is a
459
+ * `review_not_posted` note. Additive: unknown → ignored. */
460
+ | {
461
+ type: "review_posted";
462
+ repo: string;
463
+ number: number;
464
+ head: string;
465
+ verdict?: "approve" | "request_changes";
466
+ seq?: number;
467
+ at?: number;
468
+ }
441
469
  /** One `agent:ship` round boundary (docs/reference/specs/agent-ship.md item 12): the
442
470
  * pipeline publishes a `started` event when a round's child is dispatched
443
471
  * and one settle event when its outcome is known (`ShipRoundOutcome`), so
@@ -370,7 +370,7 @@ export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOp
370
370
  // final answer (`answer`) — are the run's story, not its steps: none counts
371
371
  // toward `eventCount`.
372
372
  let narrativeEvents = 0;
373
- let sideFactEvents = 0; // skill_use / review_artifact / pr_description / pr_opened / ship_round: facts about the run, not steps
373
+ let sideFactEvents = 0; // skill_use / review_artifact / pr_description / pr_opened / review_posted / ship_round: facts about the run, not steps
374
374
  let spanEvents = 0; // span_start / span_end (docs/reference/specs/tracing.md): timing records, not steps
375
375
  let wrapUp: { index: number; at?: number } | undefined;
376
376
  events.forEach((ev, index) => {
@@ -385,14 +385,15 @@ export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOp
385
385
  }
386
386
  // Side facts about the run, not steps: skill_use rides beside a use_skill
387
387
  // call that already produced its own tool pair; review_artifact,
388
- // pr_description, pr_opened and the ship_round boundaries are published
389
- // by the dispatcher/pipeline outside the model loop entirely. Counting
390
- // any of them would distort the story.
388
+ // pr_description, pr_opened, review_posted and the ship_round boundaries
389
+ // are published by the dispatcher/pipeline outside the model loop
390
+ // entirely. Counting any of them would distort the story.
391
391
  if (
392
392
  ev.type === "skill_use" ||
393
393
  ev.type === "review_artifact" ||
394
394
  ev.type === "pr_description" ||
395
395
  ev.type === "pr_opened" ||
396
+ ev.type === "review_posted" ||
396
397
  ev.type === "ship_round"
397
398
  ) {
398
399
  sideFactEvents++;
@@ -4,9 +4,11 @@ import type { RunEvent } from "./runEvents.js";
4
4
  import { isHeadMaterial, isSpanRecord } from "./runEvents.js";
5
5
  import { isHandoffShape, type Handoff } from "./ship/handoff.js";
6
6
  import {
7
+ type FindingDisposition,
7
8
  isFindingDispositionsShape,
9
+ isReviewPostShape,
8
10
  isReviewVerdictShape,
9
- type FindingDisposition,
11
+ type ReviewPost,
10
12
  type ReviewVerdict,
11
13
  } from "./reviewVerdict.js";
12
14
  import { IDEMPOTENCY_KEY_PATTERN, INSTANCE_ID_PATTERN } from "./coordinator/contract.js";
@@ -114,6 +116,12 @@ export interface RunRecord {
114
116
  /** The head a review run reviewed and posted against (7 to 40 lowercase hex),
115
117
  * after the settle; present only on a review run that pinned one. */
116
118
  reviewHead?: string;
119
+ /** How the review run's post-step ended (docs/reference/specs/agent-review.md
120
+ * item 18): the verdict posted to a named pull request at a pinned head, or
121
+ * not posted with the reason — what a coordinator's `read-record` answers
122
+ * `reviewPosted` from before it asks GitHub. Present only on a review run
123
+ * that reached its post-step; absent on records written before it existed. */
124
+ reviewPost?: ReviewPost;
117
125
  /** The dispositions a fix round submitted through `submit_dispositions`
118
126
  * (docs/reference/specs/agent-ship.md item 6), the last call's set, redacted;
119
127
  * present only on a coding run dispatched as a fix round that submitted one. */
@@ -502,6 +510,7 @@ export function isRunRecord(v: unknown): v is RunRecord {
502
510
  if (r.reviewHead !== undefined && (typeof r.reviewHead !== "string" || !REVIEW_HEAD_PATTERN.test(r.reviewHead)))
503
511
  return false;
504
512
  if (r.dispositions !== undefined && !isFindingDispositionsShape(r.dispositions)) return false;
513
+ if (r.reviewPost !== undefined && !isReviewPostShape(r.reviewPost)) return false;
505
514
  if (r.profile !== undefined && !isRunProfileRecord(r.profile)) return false;
506
515
  // A parent is named by a run id (item 46): the same shape as the record's own.
507
516
  if (r.parentRunId !== undefined && (typeof r.parentRunId !== "string" || !RUN_ID_PATTERN.test(r.parentRunId)))
@@ -79,6 +79,10 @@ export interface ChildContract {
79
79
  issue: { repo: string; number: number } | undefined;
80
80
  }
81
81
 
82
+ /** The PR-title gate's name, exported so another module can name the same
83
+ * gate without a copied string (`npm run check:pr-title`; CI's `title` check). */
84
+ export const PR_TITLE_GUARD = "check:pr-title";
85
+
82
86
  /** The guards a child may not weaken (AGENTS.md's Commands table), one line each on what they refuse. */
83
87
  export const GUARDS: readonly Guard[] = [
84
88
  {
@@ -96,6 +100,11 @@ export const GUARDS: readonly Guard[] = [
96
100
  refuses:
97
101
  "a new imprint in the public tree (a company, a person, a tracker reference, a plan id, a platform id, a date); the recorded list only shrinks",
98
102
  },
103
+ {
104
+ name: PR_TITLE_GUARD,
105
+ refuses:
106
+ "a title whose type, scope or grammar is not the changelog line, the scope being one of the code map's Areas",
107
+ },
99
108
  {
100
109
  name: "decisions:check",
101
110
  refuses:
@@ -4,11 +4,10 @@
4
4
  // Workflow instance in the bot's shim Worker with no model turn and no
5
5
  // credential — decides between the steps it asks the bot for. The Workflow
6
6
  // asks `nextAction`, performs it (a bot route, a `waitForEvent`, a sleep) and
7
- // feeds the answer to `applyReturn`; everything the in-process round loop
8
- // decides today (`runShipPipeline`: which round is next, what a child's end
9
- // means, when a cap ends the pipeline, when the pull request is merge-ready)
10
- // is decided here over the step returns instead, so the same endings hold
11
- // with the loop's process gone.
7
+ // feeds the answer to `applyReturn`; everything the pipeline decides — which
8
+ // round is next, what a child's end means, when a cap ends the pipeline, when
9
+ // the pull request is merge-ready — is decided here over the step returns, so
10
+ // the endings hold with no process of the pipeline's own to die.
12
11
  //
13
12
  // Two machines, both pure. The plan cursor walks a plan record's unit graph:
14
13
  // which units are ready (their dependencies merged), which one a failure
@@ -345,8 +344,12 @@ export type ChildFacts =
345
344
  description?: boolean;
346
345
  /** A review child's verdict. */
347
346
  verdict?: { verdict: ReviewVerdictKind; summary?: string; findings: Finding[] };
348
- /** Whether the review child's verdict landed on the pull request. */
347
+ /** Whether the review child's verdict landed on the pull request — the
348
+ * child's own record of its post, or GitHub's review list when the
349
+ * record is silent (http-ingress.md item 9). */
349
350
  reviewPosted?: boolean;
351
+ /** Why the child recorded no post, when it recorded one it chose or failed. */
352
+ reviewPostReason?: string;
350
353
  reviewHead?: string;
351
354
  /** A fix child's dispositions. */
352
355
  dispositions?: FindingDisposition[];
@@ -375,7 +378,7 @@ export type StepReturn =
375
378
  | { type: "merge"; step: string; outcome: "pending" | "refused"; reason: string; at: number }
376
379
  | { type: "sleep"; step: string };
377
380
 
378
- /** How one unit's pipeline ended — the truthful vocabulary the in-process loop
381
+ /** How one unit's pipeline ended — the truthful vocabulary the ship pipeline
379
382
  * has, plus the merge's own: `merged` by the runner, or found merged (`by:
380
383
  * other` — a person's merge, or an earlier attempt's that died after it, so
381
384
  * the runner merged nothing), `merge_ready` for a person, `merge_refused` by
@@ -512,7 +515,7 @@ function waitSliceMs(clock: number, until: number): number {
512
515
  }
513
516
 
514
517
  /** The child's budget: its preset's own, clipped to the pipeline's remaining
515
- * wall clock (the in-process loop's `clip`), never under the two minutes a
518
+ * wall clock (agent-ship item 8's clip), never under the two minutes a
516
519
  * spawn accepts. */
517
520
  function budgetMinutesFor(s: UnitPipelineState, preset: ChildPreset): number {
518
521
  return Math.max(2, Math.min(s.input.childMinutes[preset], Math.floor(remainingMs(s) / MIN)));
@@ -606,7 +609,7 @@ const roundNote = (round: RoundRef, outcome: ShipRoundOutcome): CoordinatorNote
606
609
  outcome,
607
610
  });
608
611
 
609
- /** Start a round if the reservation holds (the in-process loop's check: a
612
+ /** Start a round if the reservation holds (agent-ship item 8's check: a
610
613
  * child clipped under the reserve cannot do useful work). */
611
614
  function enterRound(s: UnitPipelineState, round: RoundRef, notes: CoordinatorNote[] = []): Transition {
612
615
  const remaining = remainingMs(s);
@@ -720,13 +723,15 @@ function settleReview(
720
723
  const notes = [roundNote(round, verdict.verdict)];
721
724
  if (verdict.verdict === "approve") {
722
725
  // Merge-ready stands on the POSTED approval: an approve whose post did
723
- // not land left no approving review on the pull request.
726
+ // not land left no approving review on the pull request. The reason is
727
+ // the child's own when it recorded one; how to continue is the report's
728
+ // re-issue line, in the runner's words (`renderUnitReport`).
724
729
  if (facts.reviewPosted === false)
725
730
  return end(
726
731
  next,
727
732
  {
728
733
  kind: "aborted",
729
- reason: `⚠️ The review approved, but the approval could not be posted — the pull request carries no approving review. Re-run ship with the pull request URL to retry the approval.`,
734
+ reason: `⚠️ The review approved, but the approval could not be posted${facts.reviewPostReason !== undefined ? ` (${facts.reviewPostReason})` : ""} — the pull request carries no approving review.`,
730
735
  round,
731
736
  reviewRounds: next.reviewRounds,
732
737
  },
@@ -1030,7 +1035,7 @@ function splitReport(s: UnitPipelineState): string {
1030
1035
  return lines.join("\n");
1031
1036
  }
1032
1037
 
1033
- /** The thread's report for a unit's ending — the in-process loop's own words
1038
+ /** The thread's report for a unit's ending — the ship pipeline's own words
1034
1039
  * for the endings it has, and the merge's for the ones it gains. */
1035
1040
  export function renderUnitReport(s: UnitPipelineState): string {
1036
1041
  const e = s.ending;
@@ -45,6 +45,7 @@ export const STREAMED_SPANS = [
45
45
  "run.description_turn",
46
46
  "run.observe_workspace",
47
47
  "run.pr_post_step",
48
+ "run.review_post_step",
48
49
  "run.reading_diff_join",
49
50
  "run.pr_description_join",
50
51
  "model.turn",
@@ -92,6 +93,7 @@ const GETTING_READY: ReadonlySet<string> = new Set([
92
93
  const FINISHING_UP: ReadonlySet<string> = new Set([
93
94
  "run.observe_workspace",
94
95
  "run.pr_post_step",
96
+ "run.review_post_step",
95
97
  "run.reading_diff_join",
96
98
  "run.pr_description_join",
97
99
  ]);
@@ -150,6 +152,7 @@ export const PARENTS: Readonly<Record<string, readonly string[]>> = {
150
152
  "run.description_turn": ["request", "ship.round"],
151
153
  "run.observe_workspace": ["request", "ship.round"],
152
154
  "run.pr_post_step": ["request", "ship.round"],
155
+ "run.review_post_step": ["request", "ship.round"],
153
156
  "run.reading_diff_join": ["request", "ship.round"],
154
157
  "run.pr_description_join": ["request", "ship.round"],
155
158
  "model.turn": ["run.agent"],
@@ -52,6 +52,9 @@ export function shimRoute(pathname: string): string | undefined {
52
52
  if (pathname === "/costs" || pathname.startsWith("/costs/")) return "costs";
53
53
  if (pathname.startsWith("/api/")) return "api";
54
54
  if (pathname.startsWith("/admin/")) return "admin";
55
+ // The model proxy's two routes (docs/reference/specs/model-proxy.md): a bounded
56
+ // request per model call, forwarded to the container like everything else.
57
+ if (pathname === "/v1/messages" || pathname === "/v1/chat/completions") return "model-proxy";
55
58
  if (pathname === "/docs" || pathname.startsWith("/docs/")) return "docs";
56
59
  if (pathname === "/" || pathname === "/index.html") return "page";
57
60
  return "other";