@coreplane/switchboard 1.251.0 → 1.252.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/dist/assets/config/config.example.yaml +4 -2
  2. package/dist/assets/deploy/cloudflare/preflight.mjs +19 -21
  3. package/dist/assets/deploy/cloudflare/worker.ts +6 -3
  4. package/dist/assets/deploy/cloudflare-memory/worker.ts +77 -12
  5. package/dist/assets/deploy/cloudflare-resident/memoryGuard.ts +212 -0
  6. package/dist/assets/deploy/cloudflare-resident/refresh.ts +1 -1
  7. package/dist/assets/deploy/cloudflare-resident/worker.ts +317 -56
  8. package/dist/assets/deploy/cloudflare-sandbox/Dockerfile +4 -2
  9. package/dist/assets/deploy/cloudflare-sandbox/runtime-supervisor.d.mts +31 -0
  10. package/dist/assets/deploy/cloudflare-sandbox/runtime-supervisor.mjs +119 -0
  11. package/dist/assets/package-lock.json +3 -3
  12. package/dist/assets/package.json +3 -2
  13. package/dist/assets/project.json +13 -9
  14. package/dist/assets/source.json +3 -3
  15. package/dist/assets/src/agents/registry.ts +5 -5
  16. package/dist/assets/src/core/budgets.ts +22 -0
  17. package/dist/assets/src/core/coordinator/contract.ts +42 -0
  18. package/dist/assets/src/core/coordinator/driver.ts +134 -10
  19. package/dist/assets/src/core/drain.ts +50 -0
  20. package/dist/assets/src/core/memory/engine.ts +98 -0
  21. package/dist/assets/src/core/memory/scorer.ts +12 -4
  22. package/dist/assets/src/core/memory/types.ts +69 -12
  23. package/dist/assets/src/core/modelCard.ts +32 -4
  24. package/dist/assets/src/core/modelPricing.ts +111 -1
  25. package/dist/assets/src/core/modelProxy/usage.ts +88 -0
  26. package/dist/assets/src/core/modelRegistry.ts +15 -1
  27. package/dist/assets/src/core/refusal.ts +4 -7
  28. package/dist/assets/src/core/reviewVerdict.ts +4 -0
  29. package/dist/assets/src/core/runEvents.ts +51 -2
  30. package/dist/assets/src/core/runFriction.ts +7 -2
  31. package/dist/assets/src/core/runLedger/types.ts +11 -0
  32. package/dist/assets/src/core/runUsage.ts +67 -13
  33. package/dist/assets/src/core/schedules.ts +3 -0
  34. package/dist/assets/src/core/ship/contract.ts +41 -14
  35. package/dist/assets/src/core/ship/coordinator.ts +380 -53
  36. package/dist/assets/src/core/ship/renewal.ts +10 -5
  37. package/dist/assets/src/core/trace/attrs.ts +24 -0
  38. package/dist/assets/src/core/types.ts +5 -5
  39. package/dist/assets/src/core/verbosity.ts +48 -0
  40. package/dist/assets/src/deploy/liveGate.ts +40 -13
  41. package/dist/assets/src/deploy/restart.ts +11 -12
  42. package/dist/assets/src/execution/residentDepCache.ts +50 -1
  43. package/dist/assets/src/execution/residentDepsStore.ts +40 -2
  44. package/dist/assets/src/execution/residentRefresh.ts +55 -3
  45. package/dist/assets/src/execution/residentSteps.ts +4 -0
  46. package/dist/assets/src/execution/sandboxErrors.ts +8 -0
  47. package/dist/assets/web/dist/.vite/manifest.json +55 -55
  48. package/dist/assets/web/dist/assets/{DeliveryPage-DUXd-Sl-.js → DeliveryPage-3ELQWM0r.js} +1 -1
  49. package/dist/assets/web/dist/assets/HomePage-BG_ok-K2.js +2 -0
  50. package/dist/assets/web/dist/assets/{PendingTurnRow-DDhMhrI7.js → PendingTurnRow-ChCQOLgZ.js} +1 -1
  51. package/dist/assets/web/dist/assets/{ResidentDetailPage-BnEoOnGQ.js → ResidentDetailPage-C9y3nbo8.js} +1 -1
  52. package/dist/assets/web/dist/assets/{ResidentsIndexPage-Dxpgf-l-.js → ResidentsIndexPage-i1RG9e7g.js} +1 -1
  53. package/dist/assets/web/dist/assets/RunFoldRow-D3wVpzBa.js +1 -0
  54. package/dist/assets/web/dist/assets/{RunRoutePage-9klVWhSF.js → RunRoutePage-B3IirUVi.js} +4 -4
  55. package/dist/assets/web/dist/assets/RunsIndexPage-DiFmtGaJ.js +1 -0
  56. package/dist/assets/web/dist/assets/{ScheduledPage-B_GgeJrb.js → ScheduledPage-DvYwM2TE.js} +1 -1
  57. package/dist/assets/web/dist/assets/{SettingsPage-BXX4R113.js → SettingsPage-Bo6yCyXZ.js} +1 -1
  58. package/dist/assets/web/dist/assets/{StatusDot-BOaw8le9.js → StatusDot-CAfS1AUi.js} +1 -1
  59. package/dist/assets/web/dist/assets/{Tooltip-DYZZ4l4V.js → Tooltip-tZoum_T-.js} +1 -1
  60. package/dist/assets/web/dist/assets/{UnitRoutePage-BaSW5Odq.js → UnitRoutePage-BmdOHwNn.js} +1 -1
  61. package/dist/assets/web/dist/assets/budgets-CbIyPAER.js +1 -0
  62. package/dist/assets/web/dist/assets/{dist-BCVXeBJ9.js → dist-DfbEpHXR.js} +1 -1
  63. package/dist/assets/web/dist/assets/indexRow-BT0cPVRw.js +1 -0
  64. package/dist/assets/web/dist/assets/{main-Dkcbtu3u.js → main-5Gm_1Gv8.js} +2 -2
  65. package/dist/assets/web/dist/assets/sseReplay-DmyMXfRC.js +11 -0
  66. package/dist/cli.js +2470 -902
  67. package/package.json +1 -1
  68. package/dist/assets/deploy/cloudflare-sandbox/runtime-supervisor.sh +0 -68
  69. package/dist/assets/web/dist/assets/HomePage-mSiqEEcN.js +0 -2
  70. package/dist/assets/web/dist/assets/RunFoldRow-CSo4-vld.js +0 -1
  71. package/dist/assets/web/dist/assets/RunsIndexPage-BplMIgaw.js +0 -1
  72. package/dist/assets/web/dist/assets/budgets-BvWYKPsY.js +0 -1
  73. package/dist/assets/web/dist/assets/indexRow-Bde9OZxG.js +0 -1
  74. package/dist/assets/web/dist/assets/sseReplay-DPwdsaok.js +0 -9
@@ -34,6 +34,7 @@
34
34
  // the shim Worker imports this by relative path.
35
35
 
36
36
  import { DEFAULT_GRANT, GRANT_RENEWALS_MAX, type Grant, type GrantSource } from "../budgets.js";
37
+ import { DEFAULT_VERBOSITY, isVerbosity, type Verbosity } from "../verbosity.js";
37
38
  import {
38
39
  applyReturn,
39
40
  cursorFinished,
@@ -52,6 +53,7 @@ import {
52
53
  type AddressSeveritySource,
53
54
  type PlanGraph,
54
55
  type PlanUnitNode,
56
+ type RoundChecks,
55
57
  type ShipCaps,
56
58
  type StepReturn,
57
59
  type UnitEnding,
@@ -59,7 +61,14 @@ import {
59
61
  stepPrefixOf,
60
62
  type UnitSession,
61
63
  } from "../ship/coordinator.js";
62
- import { checksSettledEventType, isCoordinatorUnit, runFinishedEventType, type CoordinatorUnit } from "./contract.js";
64
+ import {
65
+ checksSettledEventType,
66
+ childInterruptedEventType,
67
+ childResumedEventType,
68
+ isCoordinatorUnit,
69
+ runFinishedEventType,
70
+ type CoordinatorUnit,
71
+ } from "./contract.js";
63
72
 
64
73
  const MIN = 60_000;
65
74
 
@@ -83,7 +92,17 @@ export interface StepRunner {
83
92
  }
84
93
 
85
94
  export type CoordinatorStepRoute =
86
- "plan" | "unit-start" | "branch" | "spawn" | "read-record" | "pr-check" | "round" | "unit-end" | "merge" | "finish";
95
+ | "plan"
96
+ | "unit-start"
97
+ | "branch"
98
+ | "spawn"
99
+ | "read-record"
100
+ | "pr-check"
101
+ | "round"
102
+ | "unit-end"
103
+ | "checks"
104
+ | "merge"
105
+ | "finish";
87
106
 
88
107
  /** What a step stores: the bot's reply as the wire carried it — its status and
89
108
  * its text, read the same way on replay. Two numbers and a string, so the
@@ -170,6 +189,8 @@ interface PlanFacts {
170
189
  /** The grant beside them (decision 0046): what a renewal could spend, and which layer granted it. */
171
190
  grant: Grant;
172
191
  grantSource: GrantSource;
192
+ /** The request's verbosity as the plan route answers it (routing-and-config item 28): what the unit threads hear. */
193
+ verbosity: Verbosity;
173
194
  /** The instance's mark as the plan route answers it: a generated one-unit plan (a `plan` with no `path`). */
174
195
  generated: boolean;
175
196
  /** The runs page base the bot answered: the report links a child's write-up to its run page with it. */
@@ -218,6 +239,7 @@ function readPlan(a: BotAnswer): PlanFacts {
218
239
  grant: readGrant(b.grant),
219
240
  grantSource:
220
241
  b.grantSource === "run" || b.grantSource === "user" || b.grantSource === "channel" ? b.grantSource : "org",
242
+ verbosity: isVerbosity(b.verbosity) ? b.verbosity : DEFAULT_VERBOSITY,
221
243
  generated: b.generated === true,
222
244
  ...(typeof b.runPageBase === "string" && b.runPageBase.length > 0 ? { runPageBase: b.runPageBase } : {}),
223
245
  repo: b.repo,
@@ -349,6 +371,12 @@ function prCheckReturn(step: string, a: BotAnswer): StepReturn {
349
371
  ...(typeof headSha === "string" ? { headSha } : {}),
350
372
  ...(typeof a.body.autoMergeEnabled === "boolean" ? { autoMergeEnabled: a.body.autoMergeEnabled } : {}),
351
373
  ...(isCommitChecks(a.body.checks) ? { checks: a.body.checks } : {}),
374
+ // The ready-state facts beside the checks (agent-ship item 9): the
375
+ // pull request's mergeable state and its self-declared fix-up commits.
376
+ ...(typeof a.body.mergeableState === "string" ? { mergeableState: a.body.mergeableState } : {}),
377
+ ...(Array.isArray(a.body.fixupCommits) && a.body.fixupCommits.every((s: unknown) => typeof s === "string")
378
+ ? { fixupCommits: a.body.fixupCommits as string[] }
379
+ : {}),
352
380
  },
353
381
  at,
354
382
  };
@@ -366,6 +394,31 @@ function prCheckReturn(step: string, a: BotAnswer): StepReturn {
366
394
  throw new UnreadableAnswer("pr-check", a, "state");
367
395
  }
368
396
 
397
+ /** The round's checks read as the bot answered it (record 0055): the runs at
398
+ * the reviewed head, or none when GitHub could not be read — the machine
399
+ * treats an absent read as pending and asks again at the chunk's end. A retry
400
+ * ask's answer carries `retried` instead: whether the re-run was dispatched,
401
+ * so the machine never waits on a head an undispatched re-run left unchanged. */
402
+ function checksReturn(step: string, a: BotAnswer): StepReturn {
403
+ const { ok, checks, retried, at } = a.body;
404
+ if (ok !== true) throw new UnreadableAnswer("checks", a, "ok");
405
+ return {
406
+ type: "checks",
407
+ step,
408
+ ...(isRoundChecks(checks) ? { checks } : {}),
409
+ ...(typeof retried === "boolean" ? { retried } : {}),
410
+ at,
411
+ };
412
+ }
413
+
414
+ const isRoundChecks = (v: unknown): v is RoundChecks =>
415
+ isRecord(v) &&
416
+ typeof v.total === "number" &&
417
+ Array.isArray(v.pending) &&
418
+ v.pending.every((n: unknown) => typeof n === "string") &&
419
+ Array.isArray(v.failed) &&
420
+ v.failed.every((f: unknown) => isRecord(f) && typeof f.name === "string" && typeof f.conclusion === "string");
421
+
369
422
  function mergeReturn(step: string, a: BotAnswer): StepReturn {
370
423
  const { ok, outcome, by, sha, mergedAt, reason, at } = a.body;
371
424
  // The door found the pull request already merged after the approval: the
@@ -403,17 +456,58 @@ function answerOf(route: CoordinatorStepRoute, reply: BotReply): BotAnswer {
403
456
  return read.answer;
404
457
  }
405
458
 
406
- /** A wait's outcome: the event, or anything else — the machine confirms either by `read-record`. */
459
+ /** A wait's outcome: the event, or anything else — the machine confirms either by `read-record`.
460
+ *
461
+ * Three waits under one chunk (run-history item 47a): the child's finish
462
+ * (`run-finished-<runId>`), its deploy-roll interruption
463
+ * (`child-interrupted-<runId>`, the reattach path's word that the child
464
+ * closed `interrupted` — settled as the finish is, so the round ends at once
465
+ * with the child's own reason once `read-record` confirms it) and its resume
466
+ * (`child-resumed-<runId>`, the same run carrying on after a roll — consumed
467
+ * and re-armed, never a settlement: a resumed child keeps the wait). The
468
+ * chunk times out only once the finish AND the interruption waits both have;
469
+ * a resumed wait's own timeout decides nothing. */
407
470
  async function waitForRun(
408
471
  step: StepRunner,
409
472
  action: Extract<CoordinatorAction, { type: "wait" }>,
410
473
  ): Promise<"event" | "timeout"> {
411
- try {
412
- await step.waitForEvent(action.step, { type: runFinishedEventType(action.runId), timeout: action.timeoutMs });
413
- return "event";
414
- } catch {
415
- return "timeout";
416
- }
474
+ return await new Promise((resolve) => {
475
+ let settled = false;
476
+ let timeouts = 0;
477
+ const settle = (outcome: "event" | "timeout") => {
478
+ settled = true;
479
+ resolve(outcome);
480
+ };
481
+ const settling = (name: string, type: string) =>
482
+ step.waitForEvent(name, { type, timeout: action.timeoutMs }).then(
483
+ () => settle("event"),
484
+ () => {
485
+ timeouts += 1;
486
+ // Deferred a microtask so a resume that lands with the chunk's own
487
+ // end is still consumed (re-armed) before the timeout settles.
488
+ if (timeouts === 2) queueMicrotask(() => settle("timeout"));
489
+ },
490
+ );
491
+ void settling(action.step, runFinishedEventType(action.runId));
492
+ void settling(`${action.step}/interrupted`, childInterruptedEventType(action.runId));
493
+ // Each resumed event re-arms under the next durable name, so a second roll
494
+ // in the same chunk is still heard; a timeout here ends nothing, and a
495
+ // settled wait arms no further step.
496
+ const armResumed = (n: number): void => {
497
+ void step
498
+ .waitForEvent(n === 1 ? `${action.step}/resumed` : `${action.step}/resumed/${n}`, {
499
+ type: childResumedEventType(action.runId),
500
+ timeout: action.timeoutMs,
501
+ })
502
+ .then(
503
+ () => {
504
+ if (!settled) armResumed(n + 1);
505
+ },
506
+ () => {},
507
+ );
508
+ };
509
+ armResumed(1);
510
+ });
417
511
  }
418
512
 
419
513
  async function perform(
@@ -494,6 +588,24 @@ async function perform(
494
588
  }
495
589
  return { type: "wait-checks", step: action.step, outcome };
496
590
  }
591
+ case "checks":
592
+ // The round's checks step (record 0055): the bot reads the check runs at
593
+ // the reviewed head with the merge door's own reading — or, on a retry
594
+ // ask, re-runs the named failed checks' jobs first.
595
+ return checksReturn(
596
+ action.step,
597
+ answerOf(
598
+ "checks",
599
+ await step.do(action.step, STEP_CONFIG, () =>
600
+ call(bot, "checks", {
601
+ ...tag,
602
+ prNumber: action.prNumber,
603
+ headSha: action.headSha,
604
+ ...(action.retry !== undefined ? { retry: action.retry } : {}),
605
+ }),
606
+ ),
607
+ ),
608
+ );
497
609
  case "merge":
498
610
  return mergeReturn(
499
611
  action.step,
@@ -549,6 +661,7 @@ async function runUnit(
549
661
  addressSeveritySource: plan.addressSeveritySource,
550
662
  grant: plan.grant,
551
663
  grantSource: plan.grantSource,
664
+ verbosity: plan.verbosity,
552
665
  generated: plan.generated,
553
666
  ...(plan.runPageBase !== undefined ? { runPageBase: plan.runPageBase } : {}),
554
667
  ...(resume !== undefined ? { resume } : {}),
@@ -600,6 +713,11 @@ async function runUnit(
600
713
  // The checks at the approved head (record 0055): the report's
601
714
  // headline is a claim about them, never "merge-ready" over a red one.
602
715
  ...(check.pr.checks !== undefined ? { checks: check.pr.checks } : {}),
716
+ // The ready state beside them (agent-ship item 9): a conflicting
717
+ // head, or one carrying an unsquashed fix-up commit, is reported
718
+ // approved-but-not-merge-ready, never "merge-ready".
719
+ ...(check.pr.mergeableState !== undefined ? { mergeableState: check.pr.mergeableState } : {}),
720
+ ...(check.pr.fixupCommits !== undefined ? { fixupCommits: check.pr.fixupCommits } : {}),
603
721
  };
604
722
  } catch {
605
723
  // the report simply omits the fact
@@ -610,7 +728,13 @@ async function runUnit(
610
728
  // ending (agent-ship item 14).
611
729
  const body = {
612
730
  ...tag,
613
- ending: { kind: note.ending.kind, report: renderUnitReport(state, endFacts) },
731
+ // Two copies (routing-and-config item 28): the full report for the
732
+ // row and the board, and the thread's at the request's verbosity.
733
+ ending: {
734
+ kind: note.ending.kind,
735
+ report: renderUnitReport(state, endFacts),
736
+ threadReport: renderUnitReport(state, endFacts, state.input.verbosity ?? DEFAULT_VERBOSITY),
737
+ },
614
738
  ...(state.pr !== undefined ? { pr: state.pr } : {}),
615
739
  // A review_pending ending names the child's own last push so the next
616
740
  // attempt's pre-check can start at the review round (the row's lastPush).
@@ -19,11 +19,61 @@
19
19
  * (docs/reference/specs/run-history.md item 39); this is the wait for the rest. */
20
20
  export const DRAIN_DEADLINE_MS = 15 * 60_000;
21
21
 
22
+ /** One run holding a drain: the registry-active run's id and why it holds.
23
+ * The drain is held by the RUN REGISTRY's live rows, not by the dispatcher's
24
+ * in-flight count — the two can disagree (a registry row whose dispatcher-side
25
+ * run is gone still holds the drain for its full deadline) — so the lines an
26
+ * operator reads name these rows, never the count from the other ledger. */
27
+ export interface HeldRun {
28
+ id: string;
29
+ why: string;
30
+ }
31
+
32
+ /** Why a registry-active run holds the drain: the handoff (run-history item 39)
33
+ * did not mark it for the next generation, so this process must wait for it. */
34
+ export const HELD_NOT_HANDED_OFF = "not handed off";
35
+
36
+ /** One `id (why)` per held run, comma-separated — shared by the drain's hold
37
+ * line here and the deploy CLI's still-draining line (src/deploy/liveGate.ts). */
38
+ export function heldRunsText(held: readonly HeldRun[]): string {
39
+ return held.map((r) => `${r.id} (${r.why})`).join(", ");
40
+ }
41
+
42
+ /** The drain's hold line (slack-channel.md item 8): what actually holds the
43
+ * exit, by run id and reason — printed once the handoff has settled who stays. */
44
+ export function drainHoldLine(held: readonly HeldRun[]): string {
45
+ return `[drain] holding for ${held.length} registry-active run(s): ${heldRunsText(held)}`;
46
+ }
47
+
22
48
  /** The handoff's own budget (plan D8): after every resumable run is marked
23
49
  * `handoff`, the drain waits this long for pending history writes and
24
50
  * reflections, then exits — the next generation takes the runs. */
25
51
  export const HANDOFF_BUDGET_MS = 6_000;
26
52
 
53
+ /**
54
+ * The drain's wait bound, re-read on every poll of the drain loop
55
+ * (`src/index.ts`). While a run still holds the drain (registry-active, not
56
+ * handed off) the bound is the full DRAIN_DEADLINE_MS from the signal — the
57
+ * deadline is the bound for a run that will not end, never the schedule. The
58
+ * moment the held count reaches zero the bound collapses to HANDOFF_BUDGET_MS
59
+ * from that instant (never past the full deadline) — the same grace a drain
60
+ * that started with nothing held gets — so pending reflections and history
61
+ * writes, including the steady stream a handed-off run still executing here
62
+ * produces, get seconds to settle, not the deploy's remaining minutes. Once
63
+ * collapsed the bound never grows back: the socket is closed, so no new run
64
+ * can arrive to hold the drain again.
65
+ */
66
+ export function createDrainDeadline(drainStartedAt: number): (now: number, runsHeld: number) => number {
67
+ const full = drainStartedAt + DRAIN_DEADLINE_MS;
68
+ let collapsed: number | undefined;
69
+ return (now, runsHeld) => {
70
+ if (collapsed !== undefined) return collapsed;
71
+ if (runsHeld > 0) return full;
72
+ collapsed = Math.min(full, now + HANDOFF_BUDGET_MS);
73
+ return collapsed;
74
+ };
75
+ }
76
+
27
77
  /** Time budgeted for the replacement container to boot and reach Socket Mode
28
78
  * `connected` (image pull + Node start + Bolt handshake), when the catch-up
29
79
  * scan runs. */
@@ -94,6 +94,104 @@ export function planWrite(
94
94
  return { action: "insert", record: mint(cand), ...(target ? { supersede: target } : {}) };
95
95
  }
96
96
 
97
+ // ---- The write gate (docs/reference/specs/memory.md item 13) -----------------
98
+ // Status — the state of one pull request at one moment — and change
99
+ // descriptions — what one change did, which the spec and the diff already say —
100
+ // must never become facts: the prompt has asked for that since the write path
101
+ // shipped, and a rule a model is asked to follow is a rule it follows on
102
+ // average. The gate is therefore code, pure and total, and lives HERE in the
103
+ // shared engine so the bot's write path and the Worker's sweep run the exact
104
+ // same rule. A summary is never gated (episodic by definition).
105
+
106
+ /** A delivery predicate: words that say a change LANDED — one moment's news,
107
+ * never a lesson. Shared by the `delivery` marker and the reference marker's
108
+ * same-clause test. The auxiliary may sit up to two words from the participle
109
+ * ("is fixed and pushed"). */
110
+ const DELIVERY_PREDICATE =
111
+ /\b(?:(?:was|were|is|are)\s+(?:\w+\s+){0,2}?(?:pushed|merged|approved)|all\s+green|lgtm|ready\s+for\s+review|awaits?\s+ci|is\s+complete)\b/i;
112
+
113
+ /** A pull-request / issue / unit reference in subject position: the fact opens
114
+ * with the noun and a number, so the reference is what the fact is ABOUT. */
115
+ const REFERENCE_SUBJECT = /^\s*(?:pr|pull\s+request|issue|unit)s?\s*#?\d+\b/i;
116
+
117
+ /** A reference elsewhere (`#n`, `pull/n`, `issues/n`) rejects only beside a
118
+ * delivery predicate in the same clause — a citation inside a lesson ("the
119
+ * staged rebuild (issue 170) must budget the swap") is not status. */
120
+ const REFERENCE_IN_CLAUSE = /#\d+|\b(?:pull|issues)\/\d+/i;
121
+
122
+ /** A commit sha: 7–40 hex chars with at least one digit AND one letter, bounded
123
+ * by non-alphanumerics — the letter keeps a timestamp or a plain count out,
124
+ * the digit keeps "defaced" and "accede" out. An all-digit or all-letter sha
125
+ * is missed and accepted as the cost. */
126
+ const COMMIT_SHA = /(?<![a-z0-9])(?=[0-9a-f]*\d)(?=[0-9a-f]*[a-f])[0-9a-f]{7,40}(?![a-z0-9])/i;
127
+
128
+ /** A run id (`run` + 8 hex chars) or a branch by its path-like name. */
129
+ const RUN_OR_BRANCH = /\brun\s+[0-9a-f]{8}\b|\bbranch\s+\S*\/\S+/i;
130
+
131
+ /** `now` + a present-tense verb (one intervening word allowed): what a change
132
+ * "now does" is a change description, not a lesson. */
133
+ const NOW_VERB =
134
+ /\bnow\s+(?:\w+\s+)?(?:documents|preserves|displays|includes|carries|has|is|supports|shows|maps|controls|applies|uses)\b/i;
135
+
136
+ /** A passive change participle: "was implemented", "has been fixed", … */
137
+ const CHANGE_PARTICIPLE =
138
+ /\b(?:was|were|has\s+been|have\s+been)\s+(?:implemented|added|updated|fixed|documented|removed|renamed|introduced|extended)\b/i;
139
+
140
+ /** A plan-unit reference. */
141
+ const PLAN_UNIT = /\bunit\s+u?\d+\b/i;
142
+
143
+ /** "spec row" / "spec rows": what a spec row documents is the spec's to say. */
144
+ const SPEC_ROW = /\bspec\s+rows?\b/i;
145
+
146
+ /** The words that make a nearby number a test/check count… */
147
+ const COUNT_NOUNS = new Set(["test", "tests", "checks", "rows"]);
148
+ /** …and the outcome words that make that count status. */
149
+ const COUNT_OUTCOME = /\b(?:pass(?:es|ing|ed)?|green|fail(?:s|ing|ed)?)\b/i;
150
+
151
+ /** A clause: the unit within which the reference and count markers look for
152
+ * their second half. */
153
+ function clausesOf(text: string): string[] {
154
+ return text.split(/[;.!?\n—]+/);
155
+ }
156
+
157
+ /** A number within three words of a count noun, with an outcome word in the
158
+ * same clause — in either order ("all 12 tests passing", "green across 12
159
+ * checks"). */
160
+ function hasCountMarker(clause: string): boolean {
161
+ if (!COUNT_OUTCOME.test(clause)) return false;
162
+ const words = clause
163
+ .toLowerCase()
164
+ .split(/[^a-z0-9]+/)
165
+ .filter(Boolean);
166
+ return words.some(
167
+ (w, i) => /^\d+$/.test(w) && words.slice(Math.max(0, i - 3), i + 4).some((neighbour) => COUNT_NOUNS.has(neighbour)),
168
+ );
169
+ }
170
+
171
+ /** The status and change-description markers a fact text carries, by name —
172
+ * a table of named patterns, each firing independently. A non-empty answer
173
+ * rejects the fact (`parseReflection`) and, later, sweeps the stored row; the
174
+ * names are safe to log (never the text). Pure and total: never throws, and
175
+ * empty text carries no markers (it is rejected upstream as empty). */
176
+ export function rejectionMarkers(text: string): string[] {
177
+ const clauses = clausesOf(text);
178
+ const table: Array<[name: string, hit: boolean]> = [
179
+ [
180
+ "reference",
181
+ REFERENCE_SUBJECT.test(text) || clauses.some((c) => REFERENCE_IN_CLAUSE.test(c) && DELIVERY_PREDICATE.test(c)),
182
+ ],
183
+ ["sha", COMMIT_SHA.test(text)],
184
+ ["count", clauses.some(hasCountMarker)],
185
+ ["delivery", DELIVERY_PREDICATE.test(text)],
186
+ ["identifier", RUN_OR_BRANCH.test(text)],
187
+ ["now", NOW_VERB.test(text)],
188
+ ["changed", CHANGE_PARTICIPLE.test(text)],
189
+ ["unit", PLAN_UNIT.test(text)],
190
+ ["spec-row", SPEC_ROW.test(text)],
191
+ ];
192
+ return table.filter(([, hit]) => hit).map(([name]) => name);
193
+ }
194
+
97
195
  /** Build the record a store persists for a candidate. Ids are namespaced per
98
196
  * AGENTS.md invariant 4 (`mem:<scopeKey>:<seq>`); keywords default to the
99
197
  * text's tokens so keyword retrieval always has something to hit. */
@@ -20,10 +20,18 @@ export const DEFAULT_WEIGHTS: ScoreWeights = { keyword: 0.7, recency: 0.3 };
20
20
  * sweeper job (decay lives in the score). */
21
21
  export const RECENCY_TAU_MS = 7 * 24 * 60 * 60 * 1000;
22
22
 
23
- /** Default read budget: at most 8 records / ~800 tokens injected, regardless of
24
- * store size — context never bloats. */
25
- export const DEFAULT_MEMORY_LIMIT = 8;
26
- export const DEFAULT_MEMORY_TOKENS = 800;
23
+ /** Default read budget: at most 32 records / ~3000 tokens injected, regardless
24
+ * of store size — context never bloats. Raised from 8/800 with the repository
25
+ * window (docs/decisions/0061-…): the window's 24 facts plus the keyword hits
26
+ * must fit in one pool under one budget. */
27
+ export const DEFAULT_MEMORY_LIMIT = 32;
28
+ export const DEFAULT_MEMORY_TOKENS = 3000;
29
+
30
+ /** Default size of the repository window (`memory.repoWindow`): the newest
31
+ * facts of the run's bound repository, rendered ahead of the keyword hits.
32
+ * `0` disables the window. At ~320 chars per repository fact, 24 render in
33
+ * about 2000 tokens — the rest of the budget is the hits'. */
34
+ export const DEFAULT_REPO_WINDOW = 24;
27
35
 
28
36
  export interface MemoryBudget {
29
37
  maxRecords: number;
@@ -36,9 +36,11 @@ export interface MemoryRecord {
36
36
  supersedes?: string;
37
37
  /** `evicted`: dropped by the per-scope cap (least recently used);
38
38
  * `superseded`: replaced by a newer record; `forgotten`: removed by a human
39
- * via `memory forget`. Both are soft deletes — the row stays for
40
- * provenance but is invisible to retrieval, list, and dedup. */
41
- status: "active" | "superseded" | "forgotten" | "evicted";
39
+ * via `memory forget`; `swept`: retired by `memory sweep` because the text
40
+ * carries a status/change-description marker (`rejectionMarkers`). All are
41
+ * soft deletes — the row stays for provenance but is invisible to
42
+ * retrieval, list, and dedup. */
43
+ status: "active" | "superseded" | "forgotten" | "evicted" | "swept";
42
44
  }
43
45
 
44
46
  /** What the reflection extractor emits. The store assigns id/timestamps/useCount/
@@ -54,6 +56,45 @@ export interface MemoryCandidate {
54
56
  supersedes?: string;
55
57
  }
56
58
 
59
+ /** What one `write` batch actually did, per candidate action — the seam's
60
+ * receipt (the counters on the `[memory]` outcome line are these plus the
61
+ * parse gate's own). `restated` stays 0 until the restate action lands on the
62
+ * write plan; it is on the shape now so every store answers the same fields. */
63
+ export interface WriteCounts {
64
+ /** Candidates minted as new active records. */
65
+ inserted: number;
66
+ /** Candidates whose normalized text bumped an existing active record. */
67
+ deduped: number;
68
+ /** Candidates that bumped the shown record they restate (no insert). */
69
+ restated: number;
70
+ /** Records flipped to `superseded` by a candidate's pointer. */
71
+ superseded: number;
72
+ /** Records the per-scope cap evicted inside the same batch. */
73
+ evicted: number;
74
+ }
75
+
76
+ /** Narrowing filters for `MemoryStore.list` — each narrows, never ranks. */
77
+ export interface MemoryListOptions {
78
+ /** Whole-token text/keyword filter (the human command's `<words>`). */
79
+ query?: string;
80
+ /** Keep only records of this kind (the repository window lists facts). */
81
+ kind?: MemoryRecord["kind"];
82
+ }
83
+
84
+ /** What one sweep did — or, on the durable path, why it could not. The
85
+ * failure is a VALUE, never a throw: the one expected failure is an older
86
+ * Memory Worker generation without the `/sweep` route (a 404), which the
87
+ * command must report as a ⚠️ line, not crash on. */
88
+ export type SweepOutcome =
89
+ | {
90
+ ok: true;
91
+ /** Active facts the gate marked (flipped to `swept`, or merely counted under `dryRun`). */
92
+ swept: number;
93
+ /** The marked record ids — always safe to show (ids carry no record text). */
94
+ ids: string[];
95
+ }
96
+ | { ok: false; error: string };
97
+
57
98
  /** A retrieval request: which resource, what to match, how many at most. */
58
99
  export interface MemoryQuery {
59
100
  scopeKey: string;
@@ -71,19 +112,29 @@ export interface MemoryStore {
71
112
  retrieve(q: MemoryQuery, trace?: TraceOptions): Promise<MemoryRecord[]>;
72
113
  /** Persist distilled candidates. Dedup (identical normalized text → bump
73
114
  * `useCount`) and supersede (`supersedes` id → old record soft-deleted) live
74
- * inside the store. Driven by the post-run reflection pass (reflection.ts). */
75
- write(scopeKey: string, records: MemoryCandidate[]): Promise<void>;
76
- /** Human view: a scope's ACTIVE records, newest first, at most
77
- * `limit`. Unlike `retrieve` this never bumps usage. */
78
- /** `query`: when given, only records that a query token hits
115
+ * inside the store. Driven by the post-run reflection pass (reflection.ts).
116
+ * Answers what the batch did (`WriteCounts`), so the caller's outcome line
117
+ * reports the store's actions, never a guess. */
118
+ write(scopeKey: string, records: MemoryCandidate[]): Promise<WriteCounts>;
119
+ /** Human view and the repository window's read: a scope's ACTIVE records,
120
+ * newest first, at most `limit`. Unlike `retrieve` this never bumps usage.
121
+ * `opts.query`: when given, only records that a query token hits
79
122
  * (whole-token, text or keywords) are listed — the filter narrows, it
80
- * never ranks or bumps usage. */
81
- list(scopeKey: string, limit: number, query?: string): Promise<MemoryRecord[]>;
123
+ * never ranks or bumps usage. `opts.kind`: when given, only records of
124
+ * that kind (the repository window lists facts, never summaries). */
125
+ list(scopeKey: string, limit: number, opts?: MemoryListOptions): Promise<MemoryRecord[]>;
82
126
  /** Human control: soft-delete one ACTIVE record of this scope
83
127
  * (`status: "forgotten"`, row kept for provenance). Resolves true when a
84
128
  * record was forgotten, false when the id names nothing active in this
85
129
  * scope — a foreign-scope id can never be forgotten through another scope. */
86
130
  forget(scopeKey: string, id: string): Promise<boolean>;
131
+ /** Human control: retire this scope's ACTIVE facts whose text the write
132
+ * gate would reject today (`rejectionMarkers`, engine.ts) — `status:
133
+ * "swept"`, rows kept for provenance, summaries never touched. Under
134
+ * `dryRun` nothing flips and the marked ids are answered. Idempotent: a
135
+ * second sweep answers 0. Failures come back as `{ok: false}`, never a
136
+ * throw (an older Worker without the route is a reported failure). */
137
+ sweep(scopeKey: string, opts?: { dryRun?: boolean }): Promise<SweepOutcome>;
87
138
  }
88
139
 
89
140
  /** The resources memory is scoped to. `org` is the shared resource every
@@ -98,10 +149,16 @@ export type MemoryScope = "org" | "user" | "repo" | "channel";
98
149
  export interface MemoryConfig {
99
150
  /** Master switch. Default false → `NullMemoryStore` → zero behavior change. */
100
151
  enabled?: boolean;
101
- /** Max records retrieved/injected per request. Default 8. */
152
+ /** Max records retrieved/injected per request. Default 32. */
102
153
  limit?: number;
103
- /** Hard token budget for the injected block. Default ~800. */
154
+ /** Hard token budget for the injected block. Default ~3000. */
104
155
  maxTokens?: number;
156
+ /** The repository window: a run bound to a repository leads its block with
157
+ * that repository's newest facts — at most this many, read with
158
+ * `list(repoScope, repoWindow, { kind: "fact" })` — ahead of the keyword
159
+ * hits, under the same budget. Default 24; `0` disables the window (the
160
+ * repository scope is retrieved by keyword like the others). */
161
+ repoWindow?: number;
105
162
  /** Per-scope cap on ACTIVE records. A write that would leave a scope
106
163
  * over the cap evicts the least recently used records (soft delete, status
107
164
  * `evicted`) down to it, inside the same write. Default 500. */
@@ -32,11 +32,23 @@ export type LevelMap = Record<Effort, LevelWord | "refused"> | "unknown";
32
32
  export type InputSupport = boolean | "unknown";
33
33
  export type CacheRule = "automatic" | "markers" | "none" | "unknown";
34
34
 
35
+ /** One long-context tier of a card's rate (pi's rule, `calculateCost`): when
36
+ * the request's input side (input + cache reads + cache writes) exceeds
37
+ * `inputTokensAbove`, the WHOLE request re-rates at the tier. */
38
+ export interface CardPriceTier {
39
+ inputTokensAbove: number;
40
+ input: number;
41
+ output: number;
42
+ cacheRead: number;
43
+ cacheWrite: number;
44
+ }
45
+
35
46
  export interface CardPrice {
36
47
  input: number;
37
48
  output: number;
38
49
  cacheRead: number;
39
50
  cacheWrite: number;
51
+ tiers?: readonly CardPriceTier[];
40
52
  }
41
53
 
42
54
  export interface ModelCard {
@@ -123,14 +135,30 @@ function levelMapOf(map: Record<string, string | null> | undefined, reasoning: b
123
135
  return out;
124
136
  }
125
137
 
126
- function priceOf(
127
- raw: { input?: number; output?: number; cacheRead?: number; cacheWrite?: number } | undefined,
128
- ): CardPrice | undefined {
138
+ type RawPrice = {
139
+ input?: number;
140
+ output?: number;
141
+ cacheRead?: number;
142
+ cacheWrite?: number;
143
+ tiers?: Array<{ inputTokensAbove?: number } & Omit<RawPrice, "tiers">>;
144
+ };
145
+
146
+ function priceOf(raw: RawPrice | undefined): CardPrice | undefined {
129
147
  if (!raw) return undefined;
130
148
  const { input, output, cacheRead, cacheWrite } = raw;
131
149
  if (input === undefined || output === undefined || cacheRead === undefined || cacheWrite === undefined)
132
150
  return undefined;
133
- return { input, output, cacheRead, cacheWrite };
151
+ // A tier missing a field cannot re-rate the whole request; it is dropped,
152
+ // never guessed at the base rate.
153
+ const tiers = (raw.tiers ?? []).filter(
154
+ (t): t is CardPriceTier =>
155
+ t.inputTokensAbove !== undefined &&
156
+ t.input !== undefined &&
157
+ t.output !== undefined &&
158
+ t.cacheRead !== undefined &&
159
+ t.cacheWrite !== undefined,
160
+ );
161
+ return { input, output, cacheRead, cacheWrite, ...(tiers.length > 0 ? { tiers } : {}) };
134
162
  }
135
163
 
136
164
  /**