@coreplane/switchboard 1.213.0 → 1.215.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/config/config.example.yaml +7 -1
- package/dist/assets/deploy/cloudflare/artifactsCopy.ts +180 -0
- package/dist/assets/deploy/cloudflare/ensure-bucket.mjs +71 -0
- package/dist/assets/deploy/cloudflare/package.json +1 -1
- package/dist/assets/deploy/cloudflare/worker.ts +19 -1
- package/dist/assets/deploy/cloudflare/wrangler.template.jsonc +15 -0
- package/dist/assets/deploy/cloudflare-memory/worker.ts +568 -6
- package/dist/assets/deploy/cloudflare-memory/wrangler.template.jsonc +11 -2
- package/dist/assets/deploy/profile.example.json +1 -1
- package/dist/assets/package-lock.json +3 -3
- package/dist/assets/package.json +1 -1
- package/dist/assets/source.json +3 -3
- package/dist/assets/src/agents/registry.ts +39 -14
- package/dist/assets/src/config/profile.ts +8 -5
- package/dist/assets/src/core/coordinator/contract.ts +6 -0
- package/dist/assets/src/core/runEvents.ts +18 -9
- package/dist/assets/src/core/runFriction.ts +9 -2
- package/dist/assets/src/core/runLedger/sessionLog.ts +207 -0
- package/dist/assets/src/core/runLedger/transcript.ts +187 -0
- package/dist/assets/src/core/runLedger/types.ts +25 -4
- package/dist/assets/src/core/runRecord.ts +62 -1
- package/dist/assets/src/core/ship/coordinator.ts +70 -33
- package/dist/assets/src/core/trace/workerTrace.ts +2 -0
- package/dist/assets/src/deploy/profile.ts +11 -0
- package/dist/assets/src/execution/residentInstanceId.ts +11 -3
- package/dist/assets/src/execution/residentText.ts +17 -2
- package/dist/assets/src/providers/types.ts +6 -0
- package/dist/cli.js +1689 -627
- package/package.json +1 -1
|
@@ -35,7 +35,15 @@
|
|
|
35
35
|
// ONE DeliveryDO (named "delivery"): the delivery page's snapshot per
|
|
36
36
|
// repository — the merged pull requests' facts over the window, one row
|
|
37
37
|
// each, and when they were read (docs/reference/specs/delivery.md item 10).
|
|
38
|
-
{ "name": "DELIVERY", "class_name": "DeliveryDO" }
|
|
38
|
+
{ "name": "DELIVERY", "class_name": "DeliveryDO" },
|
|
39
|
+
// One SessionLogDO per session — a thread and an agent (the DO name is
|
|
40
|
+
// `<threadKey>:<agent>`): the transcript rows of every run of the
|
|
41
|
+
// session, a full-text index over them, and the notepad
|
|
42
|
+
// (docs/reference/specs/session-log.md). Kept after a run finishes; the
|
|
43
|
+
// retention sweep drops it once its last kept run is gone and no run is
|
|
44
|
+
// live on the thread. Runs claimed before it existed still finish on
|
|
45
|
+
// RunTranscriptDO, which stays bound until no such row is live.
|
|
46
|
+
{ "name": "SESSION_LOGS", "class_name": "SessionLogDO" }
|
|
39
47
|
]
|
|
40
48
|
},
|
|
41
49
|
// The bot shim's ship coordinator Workflow, bound across scripts by the bot's
|
|
@@ -65,6 +73,7 @@
|
|
|
65
73
|
// run's diagnosis); the FrictionDO that once mirrored those diagnoses is
|
|
66
74
|
// retired with its rows — docs/reference/specs/self-improvement.md item 1.
|
|
67
75
|
{ "tag": "v7", "deleted_classes": ["FrictionDO"] },
|
|
68
|
-
{ "tag": "v8", "new_sqlite_classes": ["DeliveryDO"] }
|
|
76
|
+
{ "tag": "v8", "new_sqlite_classes": ["DeliveryDO"] },
|
|
77
|
+
{ "tag": "v9", "new_sqlite_classes": ["SessionLogDO"] }
|
|
69
78
|
]
|
|
70
79
|
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
{
|
|
2
|
-
"$comment": "The deployment profile: where THIS installation runs. Copy to deploy/profile.json and fill it in (or run `switchboard deploy init`). The account is your Cloudflare account id; every hostname must be under `zone`, a zone in that account, unless the Worker names its own `zone` (also in the account); `workers.bot` is the one required Worker — leave `memory`, `resident` or `sandbox` out and `deploy plan` has no step for them (a bot-only profile is a one-step plan; without `memory` the config is not pushed anywhere). The project's docs site is not a Worker of an installation: it is the project's website, deployed by the project's own CI from project.json, so there is no `docs` entry. `configSource` is where `deploy all` reads the bot's runtime config from before building the image — a path, `github://owner/repo/path@ref` (needs CONFIG_REPO_TOKEN), or `op://Vault/Item/field` (needs OP_SERVICE_ACCOUNT_TOKEN); `secretsSource` is where `secrets put` reads values from — a directory of <NAME> files (the default when absent) or `op://Vault/Item`. `images` is where the bot, resident and sandbox container images come from: `registry` — the release's published images, copied once per version into your account registry by `deploy all` itself (or `deploy images` ahead of it) over HTTPS and referenced from there — no Docker anywhere, a Cloudflare API token with Containers Edit in CLOUDFLARE_API_TOKEN for the copy (an installation deploying published images; what `init` writes from the published package); or `build` (the default when absent) — each Worker's Dockerfile, built by wrangler where `deploy all` runs (a checkout; the project's own production). `deploy plan` reads this example when no profile exists; `deploy all` refuses it.",
|
|
2
|
+
"$comment": "The deployment profile: where THIS installation runs. Copy to deploy/profile.json and fill it in (or run `switchboard deploy init`). The account is your Cloudflare account id; every hostname must be under `zone`, a zone in that account, unless the Worker names its own `zone` (also in the account); `workers.bot` is the one required Worker — leave `memory`, `resident` or `sandbox` out and `deploy plan` has no step for them (a bot-only profile is a one-step plan; without `memory` the config is not pushed anywhere). The project's docs site is not a Worker of an installation: it is the project's website, deployed by the project's own CI from project.json, so there is no `docs` entry. `configSource` is where `deploy all` reads the bot's runtime config from before building the image — a path, `github://owner/repo/path@ref` (needs CONFIG_REPO_TOKEN), or `op://Vault/Item/field` (needs OP_SERVICE_ACCOUNT_TOKEN); `secretsSource` is where `secrets put` reads values from — a directory of <NAME> files (the default when absent) or `op://Vault/Item`. `images` is where the bot, resident and sandbox container images come from: `registry` — the release's published images, copied once per version into your account registry by `deploy all` itself (or `deploy images` ahead of it) over HTTPS and referenced from there — no Docker anywhere, a Cloudflare API token with Containers Edit in CLOUDFLARE_API_TOKEN for the copy (an installation deploying published images; what `init` writes from the published package); or `build` (the default when absent) — each Worker's Dockerfile, built by wrangler where `deploy all` runs (a checkout; the project's own production). `artifacts` (optional) names the R2 bucket a run's files move through (docs/reference/specs/execution.md item 20): with it the bot Worker binds the bucket and `deploy` creates it before the upload (the credential then needs Workers R2 Storage: Edit); the bot's runtime config `artifacts.r2.bucket` must say the same name. `deploy plan` reads this example when no profile exists; `deploy all` refuses it.",
|
|
3
3
|
"account": "00000000000000000000000000000000",
|
|
4
4
|
"zone": "example.com",
|
|
5
5
|
"workers": {
|
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "switchboard",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.215.0",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "switchboard",
|
|
9
|
-
"version": "1.
|
|
9
|
+
"version": "1.215.0",
|
|
10
10
|
"license": "Apache-2.0",
|
|
11
11
|
"workspaces": [
|
|
12
12
|
"web",
|
|
@@ -19000,7 +19000,7 @@
|
|
|
19000
19000
|
},
|
|
19001
19001
|
"packages/switchboard": {
|
|
19002
19002
|
"name": "@coreplane/switchboard",
|
|
19003
|
-
"version": "1.
|
|
19003
|
+
"version": "1.215.0",
|
|
19004
19004
|
"license": "Apache-2.0",
|
|
19005
19005
|
"dependencies": {
|
|
19006
19006
|
"@anthropic-ai/sdk": "^0.124.0",
|
package/dist/assets/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "switchboard",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.215.0",
|
|
4
4
|
"private": true,
|
|
5
5
|
"description": "Mention it in Slack and an agent reviews the PR, ships the fix, or answers the question — on the model you choose, with its tools running where you decide.",
|
|
6
6
|
"license": "Apache-2.0",
|
package/dist/assets/source.json
CHANGED
|
@@ -198,6 +198,8 @@ const RESIDENT_TOOLCHAIN = `The resident image carries ${IMAGE_TOOLCHAIN}; plus
|
|
|
198
198
|
// the sandbox and resident variants cannot drift on it.
|
|
199
199
|
const SHOW_FILES = `Files the person should SEE go through the attach_file tool: a screenshot from \`playwright screenshot\`, a rendered PDF, a recording — it posts the workspace file into this conversation, where an image renders inline. Use it whenever you produce an image worth showing (a visual change, a rendered page, a before/after); a link to a file on GitHub is not a picture.
|
|
200
200
|
SCREENSHOTS GO TO BOTH PLACES, ALL OF THEM: when the request asks for screenshots, or the change is visual, every capture is attached here with attach_file AND published on the pull request — commit the images to an assets branch (never the PR's own diff) and reference them from the description's validation section or a PR comment so they render inline there too — unless the request names one destination. Never attach a subset and link the rest.
|
|
201
|
+
Whole files: up to 1 GiB where the artifact store is configured (a recording, a large PDF), 10 MiB otherwise — the tool's result says which applies; over the limit, link to the file.
|
|
202
|
+
Files the person dropped on the thread that were too large to show you inline are already in ./attachments/ in your workspace when the turn's text names them (a video for ffmpeg, a large PDF, a zip); read them from there — never ask for a re-upload.
|
|
201
203
|
Text stays in your message; do not attach what you can say.`;
|
|
202
204
|
|
|
203
205
|
const CODING_SYSTEM = `You are Switchboard's coding agent, operating from a Slack request.
|
|
@@ -426,28 +428,43 @@ Report outcomes faithfully: a check you could not run is "could not check", neve
|
|
|
426
428
|
// preset. A child is a `dispatch()` run as the requesting user, in a thread of
|
|
427
429
|
// its own, under their permissions (docs/decisions/0002-dispatcher-is-the-only-orchestrator.md,
|
|
428
430
|
// docs/decisions/0007-authorization-policy-table.md): the prompt says exactly
|
|
429
|
-
// that, so the model never expects a child to
|
|
430
|
-
//
|
|
431
|
-
//
|
|
432
|
-
//
|
|
433
|
-
//
|
|
434
|
-
//
|
|
435
|
-
//
|
|
436
|
-
|
|
431
|
+
// that, so the model never expects a child to hold more than its requester
|
|
432
|
+
// does. A child is a reader of this conversation: it starts from the text so
|
|
433
|
+
// far plus the prompt, and the prompt says so. Machine `none`, identity
|
|
434
|
+
// `none`: it holds no workspace, no shell and no credential of its own; its
|
|
435
|
+
// reach is the run tools, the GitHub reads and URL reading. The prompt names
|
|
436
|
+
// the limits the spawn stage enforces — one level of depth, the fan-out cap,
|
|
437
|
+
// the parent's remaining clock — so a refusal is never a surprise, and renders
|
|
438
|
+
// the presets a child can run from the registry the way `help` renders its
|
|
439
|
+
// rows: every sibling whose identity is not `write`, with its own description,
|
|
440
|
+
// so the model picks from the real list and a preset that crosses the identity
|
|
441
|
+
// line moves the day its def does; the write presets are named as what a child
|
|
442
|
+
// never is, with the refusal's name.
|
|
443
|
+
function conductorSystem(siblings: readonly AgentDef[]): string {
|
|
444
|
+
const readers = siblings.filter((a) => a.identity !== "write");
|
|
445
|
+
const writers = siblings.filter((a) => a.identity === "write");
|
|
446
|
+
const rows = readers.map(
|
|
447
|
+
(a) => `- \`${a.name}\`${machineNeedsRepo(a.machine) ? " (needs the repository)" : ""}: ${a.description}`,
|
|
448
|
+
);
|
|
449
|
+
return `You are Switchboard's conductor: you coordinate other runs instead of doing the work yourself, answering a request from Slack.
|
|
437
450
|
|
|
438
451
|
You have no workspace and no shell. Your tools: \`spawn_run\` (start a child run), \`send_to_run\` (steer a live child: your text reaches it as a follow-up at its next step), \`await_runs\` (wait for your children to end and get each end — its status and final reply — back as data), \`list_runs\` (the runs you may see — your own children by default), \`get_run_status\` (one run: whether it is running, what it is doing, and its final reply once it finished), the GitHub reads — \`github_repos\`, \`github_tree\` / \`github_file\` (browse and read our repositories), \`github_search_code\`, \`github_issue_list\` / \`github_issue_get\` — \`web_fetch\` (read a public URL), and \`update_status\`.
|
|
439
452
|
|
|
440
453
|
WHAT A CHILD IS. A child is an ordinary Switchboard run started as the person who asked you — exactly the run they could start by hand with \`agent:<preset>\` — in a thread of its own in this channel, visible to everyone there, with its own status card and run page, and under their permissions: a preset they may not run, a repository they may not use, or a profile a boundary caps is refused in the child's thread, and the refusal comes back to you as the tool result naming the gate. Children cannot spawn children. You may have a few live at once (the deployment's \`spawn.maxChildren\`, three by default); a spawn past the cap is refused until one finishes. A child's wall clock is capped by what is left of yours.
|
|
441
454
|
|
|
442
|
-
THE PRESETS a child can run
|
|
455
|
+
THE PRESETS a child can run — a child reads, so only a preset whose identity is \`none\` or \`read\`:
|
|
456
|
+
${rows.join("\n")}
|
|
457
|
+
|
|
458
|
+
A preset that writes — ${writers.map((a) => `\`${a.name}\``).join(", ")} — is refused by name (\`spawn_identity\`): a spawned child never holds a write credential, so pushing a branch or opening a pull request is the requester's to start by hand with \`agent:<preset>\`; say so in your answer instead of spawning it.
|
|
443
459
|
|
|
444
460
|
ROUTED COMPOUNDS. A request may arrive already split: the router found independent parts, and the message ends with the line "Routed as a compound request: N independent parts" followed by a numbered list, one part per line as \`<preset>\`: <text>. Spawn exactly those children — one \`spawn_run\` per line, the preset as listed, the line's text as the child's prompt (it already stands alone; add the repository where the preset needs one) — then \`await_runs\` them all and compile. Never merge, drop or add a part; a part whose spawn is refused is reported as refused, by the gate's name.
|
|
445
461
|
|
|
446
|
-
HOW TO WORK. Fan out, await, compile. Read the request and split it into children only where the parts are independent; a request one preset answers is one child. Spawn each child with a
|
|
462
|
+
HOW TO WORK. Fan out, await, compile. Read the request and split it into children only where the parts are independent; a request one preset answers is one child. Spawn each child with a prompt that says what it should do, and the repository where the preset needs one: a child starts from this conversation's text so far — every user and assistant turn before your call, never your tool calls, their results or your thinking — and your prompt is its one new turn, so tell it what to do rather than repeat what was said. Then call \`await_runs\` once with every child's id: it returns when all of them have ended, or earlier — at the edge of your own budget, at a stop, or when a follow-up lands in this thread — and \`ended\` says which; a child still running at the cut keeps running (name it in your answer, or await again after a follow-up). Steer a child with \`send_to_run\` when the request changes or a child is heading the wrong way. A child that ended — finished, failed, interrupted by a restart — is reported as it ended and never restarted; spawn a new child if the work still matters. Then compile: one answer from the write-ups \`await_runs\` returned. Never do a child's job yourself, and never claim a child finished or found something you did not read from \`await_runs\` or \`get_run_status\`.
|
|
447
463
|
|
|
448
464
|
Maintain the user-facing status card with the update_status tool: one item per child (○ pending, ✱ running, ✓ finished — only once await_runs or get_run_status said so).
|
|
449
465
|
|
|
450
466
|
Use Slack-friendly formatting (no markdown headers; *bold*, bullets, code blocks). Your final message is posted to Slack: lead with the outcome, then one line per child — its preset, its thread, its status and its result in a sentence — and what is still running, if anything.`;
|
|
467
|
+
}
|
|
451
468
|
|
|
452
469
|
/** The compound answer's preset — the one preset absent from the router's
|
|
453
470
|
* table that a plain message still reaches: a message with two or more
|
|
@@ -469,11 +486,15 @@ export function presetDoor(def: AgentDef): PresetDoor {
|
|
|
469
486
|
return def.name === COMPOUND_PRESET ? "compound" : "directive";
|
|
470
487
|
}
|
|
471
488
|
|
|
472
|
-
|
|
489
|
+
/** Every preset that does the work: the ones a conductor's children are drawn
|
|
490
|
+
* from (those that read) and the ones it names as refused (those that write).
|
|
491
|
+
* The conductor is built from this list below, so its prompt renders its
|
|
492
|
+
* siblings and never itself. */
|
|
493
|
+
const WORK_PRESETS = {
|
|
473
494
|
general: {
|
|
474
495
|
name: "general",
|
|
475
496
|
description:
|
|
476
|
-
"Default assistant
|
|
497
|
+
"Default assistant: answers directly, reads URLs, manages issues; the preset for any question the org's GitHub answers. No workspace or shell.",
|
|
477
498
|
system: GENERAL_SYSTEM,
|
|
478
499
|
toolset: "assistant",
|
|
479
500
|
// The GitHub tools are REST in the bot process, so a general ask never
|
|
@@ -547,7 +568,7 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
547
568
|
research: {
|
|
548
569
|
name: "research",
|
|
549
570
|
description:
|
|
550
|
-
"Answers questions
|
|
571
|
+
"Answers questions that need the web (search, URL reading); GitHub for context, not for a question GitHub alone answers. No workspace.",
|
|
551
572
|
system: RESEARCH_SYSTEM,
|
|
552
573
|
toolset: "web",
|
|
553
574
|
machine: "none", // web I/O only; no workspace is provisioned
|
|
@@ -573,11 +594,15 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
573
594
|
cacheTtl: "1h",
|
|
574
595
|
// No built-in effort: the deployment decides, as for coding.
|
|
575
596
|
},
|
|
597
|
+
} satisfies Record<string, AgentDef>;
|
|
598
|
+
|
|
599
|
+
export const AGENTS: Record<string, AgentDef> = {
|
|
600
|
+
...WORK_PRESETS,
|
|
576
601
|
conductor: {
|
|
577
602
|
name: "conductor",
|
|
578
603
|
description:
|
|
579
604
|
"Coordinates other runs: spawns child runs as the requester — each in a thread of its own, under their permissions — follows them, and reports. No workspace or shell.",
|
|
580
|
-
system:
|
|
605
|
+
system: conductorSystem(Object.values(WORK_PRESETS)),
|
|
581
606
|
toolset: "conductor",
|
|
582
607
|
// Nothing is provisioned and no credential minted: the run tools call the
|
|
583
608
|
// dispatcher, the GitHub reads are REST in the bot process.
|
|
@@ -2,9 +2,9 @@
|
|
|
2
2
|
// the machine class its tools execute on, the identity it acts as, and the
|
|
3
3
|
// minutes it may run. A preset declares one; the run's EFFECTIVE profile is
|
|
4
4
|
// what the pipeline hands the factory, the ledger and the runner — never the
|
|
5
|
-
// preset's fields read again downstream. Pure and leaf:
|
|
6
|
-
//
|
|
7
|
-
import type
|
|
5
|
+
// preset's fields read again downstream. Pure and near-leaf: the only value
|
|
6
|
+
// import is the registry's `runawayTurnCap`, itself pure.
|
|
7
|
+
import { runawayTurnCap, type AgentDef, type Identity, type MachineClass } from "../agents/registry.js";
|
|
8
8
|
|
|
9
9
|
export type { Identity, MachineClass };
|
|
10
10
|
|
|
@@ -63,9 +63,12 @@ export function declaredProfile(preset: Pick<AgentDef, "machine" | "identity" |
|
|
|
63
63
|
/** The preset with its budget replaced by the effective profile's — the def
|
|
64
64
|
* the runner is handed, since its deadline, wrap-up warning and budget label
|
|
65
65
|
* read `maxMinutes` (the same clipped copy the ship pipeline hands its child
|
|
66
|
-
* rounds).
|
|
66
|
+
* rounds). `maxTurns` is re-derived from the clipped minutes with the
|
|
67
|
+
* registry's own `runawayTurnCap`, so the six-per-minute runaway guard holds
|
|
68
|
+
* for the budget the run actually has, not the preset's declared one.
|
|
69
|
+
* Always a copy: the shared `AgentDef` is never mutated. */
|
|
67
70
|
export function budgetedAgent(agent: AgentDef, profile: RunProfile): AgentDef {
|
|
68
|
-
return { ...agent, maxMinutes: profile.minutes };
|
|
71
|
+
return { ...agent, maxMinutes: profile.minutes, maxTurns: runawayTurnCap(profile.minutes) };
|
|
69
72
|
}
|
|
70
73
|
|
|
71
74
|
// ---- boundaries: a scope caps, never grants -----------------------------------
|
|
@@ -142,6 +142,10 @@ export interface CoordinatorUnit {
|
|
|
142
142
|
/** The unit's thread, once opened; a task's is the requesting thread from the start. */
|
|
143
143
|
threadKey?: string;
|
|
144
144
|
sourceUrl?: string;
|
|
145
|
+
/** The unit's review thread, opened once beside the unit's thread: every
|
|
146
|
+
* review round runs there (record 0034), so the review child's worktree is
|
|
147
|
+
* readonly and its own and no round wipes the coding thread's. */
|
|
148
|
+
reviewThread?: { threadKey: string; sourceUrl?: string };
|
|
145
149
|
/** The unit's board issue in the repository, when one titled by the unit id exists — the handoff's destination. */
|
|
146
150
|
issue?: number;
|
|
147
151
|
pr?: { number: number; url: string };
|
|
@@ -173,6 +177,7 @@ const isResume = (v: unknown): boolean =>
|
|
|
173
177
|
isFinite(v.pr) &&
|
|
174
178
|
(v.headSha === undefined || isText(v.headSha)) &&
|
|
175
179
|
(v.url === undefined || isText(v.url, 2048));
|
|
180
|
+
const isThread = (v: unknown): boolean => isObject(v) && isText(v.threadKey) && isOptionalText(v.sourceUrl);
|
|
176
181
|
|
|
177
182
|
/** Structural check on a record from outside the process (a Worker response, an HTTP body). */
|
|
178
183
|
export function isCoordinatorInstance(v: unknown): v is CoordinatorInstance {
|
|
@@ -202,6 +207,7 @@ export function isCoordinatorUnit(v: unknown): v is CoordinatorUnit {
|
|
|
202
207
|
if (!isText(r.unit, 32) || !isText(r.slug) || !isText(r.branch) || !isOptionalText(r.title)) return false;
|
|
203
208
|
if (!Array.isArray(r.dependsOn) || !r.dependsOn.every((d) => isText(d, 32))) return false;
|
|
204
209
|
if (!isOptionalText(r.threadKey) || !isOptionalText(r.sourceUrl)) return false;
|
|
210
|
+
if (r.reviewThread !== undefined && !isThread(r.reviewThread)) return false;
|
|
205
211
|
if (r.issue !== undefined && !isFinite(r.issue)) return false;
|
|
206
212
|
if (r.pr !== undefined && !isPr(r.pr)) return false;
|
|
207
213
|
if (r.resume !== undefined && !isResume(r.resume)) return false;
|
|
@@ -134,7 +134,11 @@ export type RunNoteKind =
|
|
|
134
134
|
/** The harness's gate refused a tool call the model asked for (harness-pi.md
|
|
135
135
|
* item 7): the summary names the tool and the rule; the model read the same
|
|
136
136
|
* reason as the tool's result. Published by the bot's authorize route. */
|
|
137
|
-
| "tool_refused"
|
|
137
|
+
| "tool_refused"
|
|
138
|
+
/** The stuck-loop guard fired (docs/reference/specs/run-loop.md item 18):
|
|
139
|
+
* the same tool call failed identically six times in a row, so the run is
|
|
140
|
+
* forced into its write-up instead of looping to the wall clock. */
|
|
141
|
+
| "stuck_loop";
|
|
138
142
|
|
|
139
143
|
/** Every `RunNoteKind`, as a value (a reader that filters notes by kind uses
|
|
140
144
|
* this; adding a kind to the union without adding it here is a type error). */
|
|
@@ -158,6 +162,7 @@ export const RUN_NOTE_KINDS = [
|
|
|
158
162
|
"compacted",
|
|
159
163
|
"harness_error",
|
|
160
164
|
"tool_refused",
|
|
165
|
+
"stuck_loop",
|
|
161
166
|
] as const satisfies readonly RunNoteKind[];
|
|
162
167
|
type _EveryKindListed = [RunNoteKind] extends [(typeof RUN_NOTE_KINDS)[number]] ? true : never;
|
|
163
168
|
const _everyKindListed: _EveryKindListed = true;
|
|
@@ -531,20 +536,24 @@ export type RunEvent =
|
|
|
531
536
|
* the router gave (redacted, capped — the same text the card's `routed:`
|
|
532
537
|
* line carries) and the model that decided. A compound route is `preset:
|
|
533
538
|
* "conductor"` with `parts` — one per child the conductor was told to
|
|
534
|
-
* spawn: its preset and its text, the child's whole prompt. A compound
|
|
535
|
-
*
|
|
536
|
-
*
|
|
537
|
-
*
|
|
538
|
-
*
|
|
539
|
-
*
|
|
540
|
-
*
|
|
541
|
-
*
|
|
539
|
+
* spawn: its preset and its text, the child's whole prompt. A compound
|
|
540
|
+
* answer that carried a write-identity part collapsed onto that preset
|
|
541
|
+
* (record 0034: a write ask is never a part): `preset` is the write preset
|
|
542
|
+
* the run is, `collapsed.presets` the preset each part named in answer
|
|
543
|
+
* order, and no `parts`. A compound the parse refused is recorded too, on
|
|
544
|
+
* the run that fell to the default: `preset` is `defaults.agent` and
|
|
545
|
+
* `reason` reads `compound_rejected: <why>` (the run's
|
|
546
|
+
* `run_meta.agentSource` stays `default`). Published by the dispatcher
|
|
547
|
+
* straight to the registry right after `run_meta`, once per run the router
|
|
548
|
+
* answered; absent on every run a directive, a sticky preset or a scope
|
|
549
|
+
* chose. Head material, like `run_meta`. Additive: unknown → ignored. */
|
|
542
550
|
| {
|
|
543
551
|
type: "route";
|
|
544
552
|
preset: string;
|
|
545
553
|
reason: string;
|
|
546
554
|
model: string;
|
|
547
555
|
parts?: ReadonlyArray<{ preset: string; text: string }>;
|
|
556
|
+
collapsed?: { presets: ReadonlyArray<string> };
|
|
548
557
|
seq?: number;
|
|
549
558
|
at?: number;
|
|
550
559
|
}
|
|
@@ -254,6 +254,13 @@ function unionMs(intervals: ReadonlyArray<{ start: number; end: number }>): numb
|
|
|
254
254
|
return total;
|
|
255
255
|
}
|
|
256
256
|
|
|
257
|
+
/** One tool call's identity for retry/streak accounting: the tool name plus
|
|
258
|
+
* the call's one-line summary (which carries the arguments — a bash command,
|
|
259
|
+
* a path, a url). Shared with the runner's stuck-loop guard
|
|
260
|
+
* (docs/reference/specs/run-loop.md item 18), so both count "the same call"
|
|
261
|
+
* identically. */
|
|
262
|
+
export const callSignature = (tool: string, summary: string): string => `${tool} ${summary}`;
|
|
263
|
+
|
|
257
264
|
/** Analyze a run's event stream. Pure and deterministic; never mutates `events`. */
|
|
258
265
|
export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOptions = {}): FrictionDiagnosis {
|
|
259
266
|
const slowToolMs = opts.slowToolMs ?? DEFAULT_SLOW_TOOL_MS;
|
|
@@ -405,7 +412,7 @@ export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOp
|
|
|
405
412
|
|
|
406
413
|
if (ev.type === "tool_call") {
|
|
407
414
|
toolCalls++;
|
|
408
|
-
if (failedCalls.has(
|
|
415
|
+
if (failedCalls.has(callSignature(ev.tool, ev.summary))) {
|
|
409
416
|
findings.push({
|
|
410
417
|
category: "retry",
|
|
411
418
|
severity: "low",
|
|
@@ -432,7 +439,7 @@ export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOp
|
|
|
432
439
|
? { interval: { start: span.startedAt, end: span.startedAt + durationMs } }
|
|
433
440
|
: {};
|
|
434
441
|
|
|
435
|
-
if (!ev.ok) failedCalls.add(
|
|
442
|
+
if (!ev.ok) failedCalls.add(callSignature(ev.tool, callSummary));
|
|
436
443
|
|
|
437
444
|
if (ev.infra) {
|
|
438
445
|
// An infra-level failure is the sandbox, not the command: classify once,
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
// The session log's pure rules (docs/decisions/0035-a-session-log-outlives-its-runs-compaction-is-a-pointer.md;
|
|
2
|
+
// docs/reference/specs/session-log.md): one log per thread and agent holds the
|
|
3
|
+
// transcript rows of every run of the session, each run a range of it. This
|
|
4
|
+
// file is node-free — the bot and `deploy/cloudflare-memory/worker.ts` import
|
|
5
|
+
// it alike — and holds what both sides must agree on: the object's name, the
|
|
6
|
+
// text the full-text index sees for a row, the byte policy's choice of what to
|
|
7
|
+
// drop, the tail cut a follow-up seeds from, and the sweep's drop decision.
|
|
8
|
+
|
|
9
|
+
import type { ChatMessage, ContentPart } from "../../providers/types.js";
|
|
10
|
+
import { DEFAULT_RETENTION_POLICY, utf8ByteLength } from "../runRecord.js";
|
|
11
|
+
import type { StoredRow } from "./transcript.js";
|
|
12
|
+
|
|
13
|
+
export { isRunSession, SESSION_KEY_PATTERN, type RunSession } from "../runRecord.js";
|
|
14
|
+
|
|
15
|
+
/** The byte policy's default: `RetentionPolicy.sessionLogMaxBytes`. */
|
|
16
|
+
export const DEFAULT_SESSION_LOG_MAX_BYTES = DEFAULT_RETENTION_POLICY.sessionLogMaxBytes;
|
|
17
|
+
|
|
18
|
+
/** The object's name: the thread and the agent, the pair record 0034 calls a
|
|
19
|
+
* session. A run without a resolved agent keys on a dash so the name still
|
|
20
|
+
* has both halves. */
|
|
21
|
+
export function sessionKey(threadKey: string, agent: string | undefined): string {
|
|
22
|
+
return `${threadKey}:${agent ?? "-"}`;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/** The request is the seed's last user turn (`splitSeed` reads the seed the
|
|
26
|
+
* same way); a seed with no user turn — or no turns — puts its first row there. */
|
|
27
|
+
export function requestIndex(seed: readonly ChatMessage[]): number {
|
|
28
|
+
for (let i = seed.length - 1; i >= 0; i--) if (seed[i].role === "user") return i;
|
|
29
|
+
return 0;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
export type RowKind = "text" | "tool_result" | "tool_use" | "attachment" | "compaction" | "other";
|
|
33
|
+
|
|
34
|
+
function parseStored(json: string): StoredRow | undefined {
|
|
35
|
+
try {
|
|
36
|
+
const v = JSON.parse(json) as unknown;
|
|
37
|
+
return typeof v === "object" && v !== null ? (v as StoredRow) : undefined;
|
|
38
|
+
} catch {
|
|
39
|
+
return undefined;
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/** What a stored row holds, for the byte policy (only a `tool_result` is
|
|
44
|
+
* replaceable) and the index (only text kinds carry text). */
|
|
45
|
+
export function rowKind(json: string): RowKind {
|
|
46
|
+
const stored = parseStored(json);
|
|
47
|
+
if (!stored) return "other";
|
|
48
|
+
if ("compaction" in stored) return "compaction";
|
|
49
|
+
const type = (stored.part as { type?: unknown }).type;
|
|
50
|
+
switch (type) {
|
|
51
|
+
case "text":
|
|
52
|
+
return "text";
|
|
53
|
+
case "tool_result":
|
|
54
|
+
return "tool_result";
|
|
55
|
+
case "tool_use":
|
|
56
|
+
return "tool_use";
|
|
57
|
+
case "image":
|
|
58
|
+
case "document":
|
|
59
|
+
return "attachment";
|
|
60
|
+
default:
|
|
61
|
+
return "other";
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
const textOfResultContent = (content: unknown): string => {
|
|
66
|
+
if (typeof content === "string") return content;
|
|
67
|
+
if (!Array.isArray(content)) return "";
|
|
68
|
+
return content
|
|
69
|
+
.filter((c): c is { type: "text"; text: string } => typeof c === "object" && c !== null && c.type === "text")
|
|
70
|
+
.map((c) => c.text)
|
|
71
|
+
.join("\n");
|
|
72
|
+
};
|
|
73
|
+
|
|
74
|
+
/** The text the full-text index holds for a row: a text part's text, a tool
|
|
75
|
+
* result's text (the failing test's name in a vitest report is findable), a
|
|
76
|
+
* tool call's name and arguments (a command is findable), a compaction row's
|
|
77
|
+
* summary; an attachment, a thinking block or an unreadable row index nothing. */
|
|
78
|
+
export function textOfStoredRow(json: string): string {
|
|
79
|
+
const stored = parseStored(json);
|
|
80
|
+
if (!stored) return "";
|
|
81
|
+
if ("compaction" in stored) return typeof stored.compaction?.summary === "string" ? stored.compaction.summary : "";
|
|
82
|
+
const part = stored.part as ContentPart | undefined;
|
|
83
|
+
if (!part) return "";
|
|
84
|
+
switch (part.type) {
|
|
85
|
+
case "text":
|
|
86
|
+
return part.text;
|
|
87
|
+
case "tool_result":
|
|
88
|
+
return textOfResultContent(part.content);
|
|
89
|
+
case "tool_use":
|
|
90
|
+
return `${part.name} ${JSON.stringify(part.input ?? {})}`;
|
|
91
|
+
default:
|
|
92
|
+
return "";
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/** The run record keeps this much of a tool's output; the marker says so. */
|
|
97
|
+
const RECORD_TOOL_OUTPUT_CHARS = "8,000 characters";
|
|
98
|
+
|
|
99
|
+
/** The row that replaces a tool result the byte policy drops: the same role
|
|
100
|
+
* and call id (the model's `tool_use` keeps its result, so the conversation
|
|
101
|
+
* stays valid), the error flag, and a text naming what went and where the
|
|
102
|
+
* rest still is. Undefined for any row that is not a tool result — user and
|
|
103
|
+
* assistant text are never dropped. */
|
|
104
|
+
export function droppedToolResultRow(json: string): string | undefined {
|
|
105
|
+
const stored = parseStored(json);
|
|
106
|
+
if (!stored || "compaction" in stored) return undefined;
|
|
107
|
+
const part = stored.part as ContentPart | undefined;
|
|
108
|
+
if (!part || part.type !== "tool_result") return undefined;
|
|
109
|
+
const marker: ContentPart = {
|
|
110
|
+
type: "tool_result",
|
|
111
|
+
toolUseId: part.toolUseId,
|
|
112
|
+
content: `[session log: this tool result (${utf8ByteLength(json)} bytes) was dropped to keep the session under its byte policy; the run record keeps its first ${RECORD_TOOL_OUTPUT_CHARS}]`,
|
|
113
|
+
...(part.isError === true ? { isError: true } : {}),
|
|
114
|
+
};
|
|
115
|
+
return JSON.stringify({ role: stored.role, part: marker });
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
export interface TrimCandidate {
|
|
119
|
+
id: number;
|
|
120
|
+
bytes: number;
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
/** Which tool-result rows the byte policy replaces once a log is `excess`
|
|
124
|
+
* bytes over its budget: the oldest first (the caller passes them oldest
|
|
125
|
+
* first), each freeing its bytes less the marker's, until the excess is
|
|
126
|
+
* covered or the candidates run out — never a user or assistant text row,
|
|
127
|
+
* which are not candidates. */
|
|
128
|
+
export function planSessionTrim(candidates: readonly TrimCandidate[], excess: number, markerBytes: number): number[] {
|
|
129
|
+
const ids: number[] = [];
|
|
130
|
+
let freed = 0;
|
|
131
|
+
for (const c of candidates) {
|
|
132
|
+
if (freed >= excess) break;
|
|
133
|
+
const gain = c.bytes - markerBytes;
|
|
134
|
+
if (gain <= 0) continue;
|
|
135
|
+
ids.push(c.id);
|
|
136
|
+
freed += gain;
|
|
137
|
+
}
|
|
138
|
+
return ids;
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
export interface TailRow {
|
|
142
|
+
idx: number;
|
|
143
|
+
bytes: number;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/** The tail a follow-up seeds from: walking the rows newest first, the first
|
|
147
|
+
* index of the oldest turn that still fits `maxBytes` with every newer turn
|
|
148
|
+
* — whole turns only, so no tool call is parted from its result. Undefined
|
|
149
|
+
* when even the newest turn is over the budget, or the log is empty. */
|
|
150
|
+
export function tailCut(rowsNewestFirst: readonly TailRow[], maxBytes: number): number | undefined {
|
|
151
|
+
let total = 0;
|
|
152
|
+
let from: number | undefined;
|
|
153
|
+
let i = 0;
|
|
154
|
+
while (i < rowsNewestFirst.length) {
|
|
155
|
+
const idx = rowsNewestFirst[i].idx;
|
|
156
|
+
let turnBytes = 0;
|
|
157
|
+
let j = i;
|
|
158
|
+
while (j < rowsNewestFirst.length && rowsNewestFirst[j].idx === idx) turnBytes += rowsNewestFirst[j++].bytes;
|
|
159
|
+
if (total + turnBytes > maxBytes) break;
|
|
160
|
+
total += turnBytes;
|
|
161
|
+
from = idx;
|
|
162
|
+
i = j;
|
|
163
|
+
}
|
|
164
|
+
return from;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/** What the sweep re-reads about one candidate right before dropping it
|
|
168
|
+
* (session-log item 7): whether a kept run record names the session, and
|
|
169
|
+
* whether any run is live on its thread. */
|
|
170
|
+
export interface DropCandidate {
|
|
171
|
+
key: string;
|
|
172
|
+
hasKeptRun: boolean;
|
|
173
|
+
threadLive: boolean;
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
export type DropDecision = "drop" | "kept-run" | "thread-live";
|
|
177
|
+
|
|
178
|
+
/** The sweep's decision (session-log item 7), one per candidate on the facts
|
|
179
|
+
* read at that moment: a session object is dropped only when no kept run
|
|
180
|
+
* record names it and no live run holds its thread — any live run on the
|
|
181
|
+
* thread blocks every session of that thread, since the live row is the
|
|
182
|
+
* conservative signal the sweep has. A kept run outranks a live thread in
|
|
183
|
+
* the answer, since it is the longer-lived reason. */
|
|
184
|
+
export function sessionsToDrop(candidates: readonly DropCandidate[]): Array<{ key: string; decision: DropDecision }> {
|
|
185
|
+
return candidates.map((c) => ({
|
|
186
|
+
key: c.key,
|
|
187
|
+
decision: c.hasKeptRun ? "kept-run" : c.threadLive ? "thread-live" : "drop",
|
|
188
|
+
}));
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
/** The attachments a stored row references: a part's own `dataRef`, and the
|
|
192
|
+
* `dataRef` of each of a tool result's nested parts — so a trimmed result
|
|
193
|
+
* can take with it what nothing else references (session-log item 5). */
|
|
194
|
+
export function attachmentRefsOf(json: string): string[] {
|
|
195
|
+
const stored = parseStored(json);
|
|
196
|
+
if (!stored || "compaction" in stored) return [];
|
|
197
|
+
const part = stored.part as (ContentPart & { dataRef?: unknown }) | undefined;
|
|
198
|
+
if (!part) return [];
|
|
199
|
+
const refs: string[] = [];
|
|
200
|
+
if (typeof part.dataRef === "string") refs.push(part.dataRef);
|
|
201
|
+
if (part.type === "tool_result" && Array.isArray(part.content)) {
|
|
202
|
+
for (const c of part.content as Array<{ dataRef?: unknown }>) {
|
|
203
|
+
if (typeof c?.dataRef === "string") refs.push(c.dataRef);
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
return refs;
|
|
207
|
+
}
|