@coreplane/switchboard 1.212.0 → 1.214.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -3
- package/dist/assets/config/config.example.yaml +48 -0
- package/dist/assets/deploy/cloudflare/artifactsCopy.ts +180 -0
- package/dist/assets/deploy/cloudflare/ensure-bucket.mjs +71 -0
- package/dist/assets/deploy/cloudflare/package.json +1 -1
- package/dist/assets/deploy/cloudflare/worker.ts +25 -1
- package/dist/assets/deploy/cloudflare/wrangler.template.jsonc +15 -0
- package/dist/assets/deploy/cloudflare-resident/Dockerfile +24 -0
- package/dist/assets/deploy/cloudflare-resident/worker.ts +353 -18
- package/dist/assets/deploy/cloudflare-resident/wrangler.template.jsonc +9 -1
- package/dist/assets/deploy/cloudflare-sandbox/Dockerfile +18 -0
- package/dist/assets/deploy/profile.example.json +1 -1
- package/dist/assets/deploy/secrets.manifest.json +18 -0
- package/dist/assets/package-lock.json +5 -3
- package/dist/assets/package.json +2 -1
- package/dist/assets/source.json +3 -3
- package/dist/assets/src/agents/registry.ts +76 -14
- package/dist/assets/src/config/profile.ts +8 -5
- package/dist/assets/src/core/redact.ts +11 -1
- package/dist/assets/src/core/runEvents.ts +46 -1
- package/dist/assets/src/core/runFriction.ts +15 -6
- package/dist/assets/src/core/runLedger/types.ts +4 -0
- package/dist/assets/src/core/runRecord.ts +12 -0
- package/dist/assets/src/core/trace/workerTrace.ts +6 -0
- package/dist/assets/src/deploy/profile.ts +11 -0
- package/dist/assets/src/execution/residentRefresh.ts +151 -5
- package/dist/assets/src/execution/residentText.ts +17 -2
- package/dist/assets/src/providers/types.ts +6 -0
- package/dist/assets/web/dist/.vite/manifest.json +19 -19
- package/dist/assets/web/dist/assets/{ResidentDetailPage-CM5nWw-Z.js → ResidentDetailPage-CvwjhjlG.js} +1 -1
- package/dist/assets/web/dist/assets/{ResidentsIndexPage-BKBnuvp3.js → ResidentsIndexPage-DoN6XuNx.js} +1 -1
- package/dist/assets/web/dist/assets/RunRoutePage-FzLNexjF.js +12 -0
- package/dist/assets/web/dist/assets/{RunsIndexPage-w2jJP5xu.js → RunsIndexPage-DhSUbF9k.js} +1 -1
- package/dist/assets/web/dist/assets/{ScheduledPage-D3-k9DPz.js → ScheduledPage-Bq0Yg4nT.js} +1 -1
- package/dist/assets/web/dist/assets/{StatusDot-BmFHnV8m.js → StatusDot-D8Wwt1KC.js} +1 -1
- package/dist/assets/web/dist/assets/{Tooltip-DoThP2fW.js → Tooltip-DpDK7jWZ.js} +1 -1
- package/dist/assets/web/dist/assets/{dist-YrRKtxsS.js → dist-B7BVkB7x.js} +1 -1
- package/dist/assets/web/dist/assets/{main-D6nzMf0k.js → main-BFkEOy3K.js} +2 -2
- package/dist/assets/web/dist/assets/main-DCH3Mezs.css +1 -0
- package/dist/cli.js +3227 -506
- package/package.json +2 -1
- package/dist/assets/web/dist/assets/RunRoutePage-Bu4CwEzt.js +0 -12
- package/dist/assets/web/dist/assets/main-CuENKPdD.css +0 -1
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "switchboard",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.214.0",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "switchboard",
|
|
9
|
-
"version": "1.
|
|
9
|
+
"version": "1.214.0",
|
|
10
10
|
"license": "Apache-2.0",
|
|
11
11
|
"workspaces": [
|
|
12
12
|
"web",
|
|
@@ -21,6 +21,7 @@
|
|
|
21
21
|
"dependencies": {
|
|
22
22
|
"@anthropic-ai/sdk": "^0.124.0",
|
|
23
23
|
"@slack/bolt": "^5.1.0",
|
|
24
|
+
"aws4fetch": "^1.0.20",
|
|
24
25
|
"e2b": "^2.46.1",
|
|
25
26
|
"mdast-util-from-markdown": "^2.0.3",
|
|
26
27
|
"undici": "^8.10.2",
|
|
@@ -18999,11 +19000,12 @@
|
|
|
18999
19000
|
},
|
|
19000
19001
|
"packages/switchboard": {
|
|
19001
19002
|
"name": "@coreplane/switchboard",
|
|
19002
|
-
"version": "1.
|
|
19003
|
+
"version": "1.214.0",
|
|
19003
19004
|
"license": "Apache-2.0",
|
|
19004
19005
|
"dependencies": {
|
|
19005
19006
|
"@anthropic-ai/sdk": "^0.124.0",
|
|
19006
19007
|
"@slack/bolt": "^5.1.0",
|
|
19008
|
+
"aws4fetch": "^1.0.20",
|
|
19007
19009
|
"e2b": "^2.46.1",
|
|
19008
19010
|
"mdast-util-from-markdown": "^2.0.3",
|
|
19009
19011
|
"undici": "^8.10.2",
|
package/dist/assets/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "switchboard",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.214.0",
|
|
4
4
|
"private": true,
|
|
5
5
|
"description": "Mention it in Slack and an agent reviews the PR, ships the fix, or answers the question — on the model you choose, with its tools running where you decide.",
|
|
6
6
|
"license": "Apache-2.0",
|
|
@@ -81,6 +81,7 @@
|
|
|
81
81
|
"dependencies": {
|
|
82
82
|
"@anthropic-ai/sdk": "^0.124.0",
|
|
83
83
|
"@slack/bolt": "^5.1.0",
|
|
84
|
+
"aws4fetch": "^1.0.20",
|
|
84
85
|
"e2b": "^2.46.1",
|
|
85
86
|
"mdast-util-from-markdown": "^2.0.3",
|
|
86
87
|
"undici": "^8.10.2",
|
package/dist/assets/source.json
CHANGED
|
@@ -42,6 +42,15 @@ export function machineNeedsRepo(machine: MachineClass): boolean {
|
|
|
42
42
|
export const IDENTITIES = ["none", "read", "write"] as const;
|
|
43
43
|
export type Identity = (typeof IDENTITIES)[number];
|
|
44
44
|
|
|
45
|
+
/** The loops a preset's runs can be driven by (docs/reference/specs/harness-pi.md
|
|
46
|
+
* item 1): `native`, the in-process turn loop (`src/runner.ts`), or `pi`, the
|
|
47
|
+
* pi coding agent in the run's own execution container, driven over its RPC
|
|
48
|
+
* protocol and bridged onto the run's events. A deployment's `harness:` block
|
|
49
|
+
* overrides a preset's own declaration (`effectiveHarness`,
|
|
50
|
+
* src/core/harness/select.ts). */
|
|
51
|
+
export const HARNESSES = ["native", "pi"] as const;
|
|
52
|
+
export type Harness = (typeof HARNESSES)[number];
|
|
53
|
+
|
|
45
54
|
/** The pace that marks a run as looping rather than working: a model turn
|
|
46
55
|
* every ten seconds, sustained for the whole wall clock. A busy run takes
|
|
47
56
|
* 20–40 s a turn (a model think plus a tool call), so a run that averages six
|
|
@@ -110,6 +119,11 @@ export interface AgentDef {
|
|
|
110
119
|
* discovery, no gh CLI. Selected by the dispatcher AFTER executor
|
|
111
120
|
* resolution via RunOptions.system; the shared AgentDef is never mutated. */
|
|
112
121
|
residentSystem?: string;
|
|
122
|
+
/** Which loop drives the preset's runs (`HARNESSES`): the native loop
|
|
123
|
+
* unless declared, and whatever a deployment's `harness.<preset>` says
|
|
124
|
+
* over that. Only a preset with a workspace can run on pi — pi is a process
|
|
125
|
+
* in the run's execution container. */
|
|
126
|
+
harness?: Harness;
|
|
113
127
|
}
|
|
114
128
|
|
|
115
129
|
// Every PR the coding agent ships carries a rich description by default —
|
|
@@ -184,6 +198,8 @@ const RESIDENT_TOOLCHAIN = `The resident image carries ${IMAGE_TOOLCHAIN}; plus
|
|
|
184
198
|
// the sandbox and resident variants cannot drift on it.
|
|
185
199
|
const SHOW_FILES = `Files the person should SEE go through the attach_file tool: a screenshot from \`playwright screenshot\`, a rendered PDF, a recording — it posts the workspace file into this conversation, where an image renders inline. Use it whenever you produce an image worth showing (a visual change, a rendered page, a before/after); a link to a file on GitHub is not a picture.
|
|
186
200
|
SCREENSHOTS GO TO BOTH PLACES, ALL OF THEM: when the request asks for screenshots, or the change is visual, every capture is attached here with attach_file AND published on the pull request — commit the images to an assets branch (never the PR's own diff) and reference them from the description's validation section or a PR comment so they render inline there too — unless the request names one destination. Never attach a subset and link the rest.
|
|
201
|
+
Whole files: up to 1 GiB where the artifact store is configured (a recording, a large PDF), 10 MiB otherwise — the tool's result says which applies; over the limit, link to the file.
|
|
202
|
+
Files the person dropped on the thread that were too large to show you inline are already in ./attachments/ in your workspace when the turn's text names them (a video for ffmpeg, a large PDF, a zip); read them from there — never ask for a re-upload.
|
|
187
203
|
Text stays in your message; do not attach what you can say.`;
|
|
188
204
|
|
|
189
205
|
const CODING_SYSTEM = `You are Switchboard's coding agent, operating from a Slack request.
|
|
@@ -412,34 +428,73 @@ Report outcomes faithfully: a check you could not run is "could not check", neve
|
|
|
412
428
|
// preset. A child is a `dispatch()` run as the requesting user, in a thread of
|
|
413
429
|
// its own, under their permissions (docs/decisions/0002-dispatcher-is-the-only-orchestrator.md,
|
|
414
430
|
// docs/decisions/0007-authorization-policy-table.md): the prompt says exactly
|
|
415
|
-
// that, so the model never expects a child to
|
|
416
|
-
//
|
|
417
|
-
//
|
|
418
|
-
//
|
|
419
|
-
//
|
|
420
|
-
//
|
|
421
|
-
//
|
|
422
|
-
|
|
431
|
+
// that, so the model never expects a child to hold more than its requester
|
|
432
|
+
// does. A child is a reader of this conversation: it starts from the text so
|
|
433
|
+
// far plus the prompt, and the prompt says so. Machine `none`, identity
|
|
434
|
+
// `none`: it holds no workspace, no shell and no credential of its own; its
|
|
435
|
+
// reach is the run tools, the GitHub reads and URL reading. The prompt names
|
|
436
|
+
// the limits the spawn stage enforces — one level of depth, the fan-out cap,
|
|
437
|
+
// the parent's remaining clock — so a refusal is never a surprise, and renders
|
|
438
|
+
// the presets a child can run from the registry the way `help` renders its
|
|
439
|
+
// rows: every sibling whose identity is not `write`, with its own description,
|
|
440
|
+
// so the model picks from the real list and a preset that crosses the identity
|
|
441
|
+
// line moves the day its def does; the write presets are named as what a child
|
|
442
|
+
// never is, with the refusal's name.
|
|
443
|
+
function conductorSystem(siblings: readonly AgentDef[]): string {
|
|
444
|
+
const readers = siblings.filter((a) => a.identity !== "write");
|
|
445
|
+
const writers = siblings.filter((a) => a.identity === "write");
|
|
446
|
+
const rows = readers.map(
|
|
447
|
+
(a) => `- \`${a.name}\`${machineNeedsRepo(a.machine) ? " (needs the repository)" : ""}: ${a.description}`,
|
|
448
|
+
);
|
|
449
|
+
return `You are Switchboard's conductor: you coordinate other runs instead of doing the work yourself, answering a request from Slack.
|
|
423
450
|
|
|
424
451
|
You have no workspace and no shell. Your tools: \`spawn_run\` (start a child run), \`send_to_run\` (steer a live child: your text reaches it as a follow-up at its next step), \`await_runs\` (wait for your children to end and get each end — its status and final reply — back as data), \`list_runs\` (the runs you may see — your own children by default), \`get_run_status\` (one run: whether it is running, what it is doing, and its final reply once it finished), the GitHub reads — \`github_repos\`, \`github_tree\` / \`github_file\` (browse and read our repositories), \`github_search_code\`, \`github_issue_list\` / \`github_issue_get\` — \`web_fetch\` (read a public URL), and \`update_status\`.
|
|
425
452
|
|
|
426
453
|
WHAT A CHILD IS. A child is an ordinary Switchboard run started as the person who asked you — exactly the run they could start by hand with \`agent:<preset>\` — in a thread of its own in this channel, visible to everyone there, with its own status card and run page, and under their permissions: a preset they may not run, a repository they may not use, or a profile a boundary caps is refused in the child's thread, and the refusal comes back to you as the tool result naming the gate. Children cannot spawn children. You may have a few live at once (the deployment's \`spawn.maxChildren\`, three by default); a spawn past the cap is refused until one finishes. A child's wall clock is capped by what is left of yours.
|
|
427
454
|
|
|
428
|
-
THE PRESETS a child can run
|
|
455
|
+
THE PRESETS a child can run — a child reads, so only a preset whose identity is \`none\` or \`read\`:
|
|
456
|
+
${rows.join("\n")}
|
|
457
|
+
|
|
458
|
+
A preset that writes — ${writers.map((a) => `\`${a.name}\``).join(", ")} — is refused by name (\`spawn_identity\`): a spawned child never holds a write credential, so pushing a branch or opening a pull request is the requester's to start by hand with \`agent:<preset>\`; say so in your answer instead of spawning it.
|
|
429
459
|
|
|
430
460
|
ROUTED COMPOUNDS. A request may arrive already split: the router found independent parts, and the message ends with the line "Routed as a compound request: N independent parts" followed by a numbered list, one part per line as \`<preset>\`: <text>. Spawn exactly those children — one \`spawn_run\` per line, the preset as listed, the line's text as the child's prompt (it already stands alone; add the repository where the preset needs one) — then \`await_runs\` them all and compile. Never merge, drop or add a part; a part whose spawn is refused is reported as refused, by the gate's name.
|
|
431
461
|
|
|
432
|
-
HOW TO WORK. Fan out, await, compile. Read the request and split it into children only where the parts are independent; a request one preset answers is one child. Spawn each child with a
|
|
462
|
+
HOW TO WORK. Fan out, await, compile. Read the request and split it into children only where the parts are independent; a request one preset answers is one child. Spawn each child with a prompt that says what it should do, and the repository where the preset needs one: a child starts from this conversation's text so far — every user and assistant turn before your call, never your tool calls, their results or your thinking — and your prompt is its one new turn, so tell it what to do rather than repeat what was said. Then call \`await_runs\` once with every child's id: it returns when all of them have ended, or earlier — at the edge of your own budget, at a stop, or when a follow-up lands in this thread — and \`ended\` says which; a child still running at the cut keeps running (name it in your answer, or await again after a follow-up). Steer a child with \`send_to_run\` when the request changes or a child is heading the wrong way. A child that ended — finished, failed, interrupted by a restart — is reported as it ended and never restarted; spawn a new child if the work still matters. Then compile: one answer from the write-ups \`await_runs\` returned. Never do a child's job yourself, and never claim a child finished or found something you did not read from \`await_runs\` or \`get_run_status\`.
|
|
433
463
|
|
|
434
464
|
Maintain the user-facing status card with the update_status tool: one item per child (○ pending, ✱ running, ✓ finished — only once await_runs or get_run_status said so).
|
|
435
465
|
|
|
436
466
|
Use Slack-friendly formatting (no markdown headers; *bold*, bullets, code blocks). Your final message is posted to Slack: lead with the outcome, then one line per child — its preset, its thread, its status and its result in a sentence — and what is still running, if anything.`;
|
|
467
|
+
}
|
|
437
468
|
|
|
438
|
-
|
|
469
|
+
/** The compound answer's preset — the one preset absent from the router's
|
|
470
|
+
* table that a plain message still reaches: a message with two or more
|
|
471
|
+
* independent asks routes to it with the parts named
|
|
472
|
+
* (docs/reference/specs/routing-and-config.md item 21). Its def below opts
|
|
473
|
+
* out of the table (`routable: false`); the router names it only in the
|
|
474
|
+
* compound form. */
|
|
475
|
+
export const COMPOUND_PRESET = "conductor";
|
|
476
|
+
|
|
477
|
+
/** How a plain message reaches a preset (docs/reference/specs/routing-and-config.md
|
|
478
|
+
* item 21), read off its def: `routed` — a row of the router's table, picked
|
|
479
|
+
* for a single ask; `compound` — the compound form alone, a message with
|
|
480
|
+
* several independent asks; `directive` — never picked, only `agent:<name>`.
|
|
481
|
+
* `help` renders its lines from this, so its words follow the registry. */
|
|
482
|
+
export type PresetDoor = "routed" | "compound" | "directive";
|
|
483
|
+
|
|
484
|
+
export function presetDoor(def: AgentDef): PresetDoor {
|
|
485
|
+
if (def.routable !== false) return "routed";
|
|
486
|
+
return def.name === COMPOUND_PRESET ? "compound" : "directive";
|
|
487
|
+
}
|
|
488
|
+
|
|
489
|
+
/** Every preset that does the work: the ones a conductor's children are drawn
|
|
490
|
+
* from (those that read) and the ones it names as refused (those that write).
|
|
491
|
+
* The conductor is built from this list below, so its prompt renders its
|
|
492
|
+
* siblings and never itself. */
|
|
493
|
+
const WORK_PRESETS = {
|
|
439
494
|
general: {
|
|
440
495
|
name: "general",
|
|
441
496
|
description:
|
|
442
|
-
"Default assistant
|
|
497
|
+
"Default assistant: answers directly, reads URLs, manages issues; the preset for any question the org's GitHub answers. No workspace or shell.",
|
|
443
498
|
system: GENERAL_SYSTEM,
|
|
444
499
|
toolset: "assistant",
|
|
445
500
|
// The GitHub tools are REST in the bot process, so a general ask never
|
|
@@ -466,6 +521,9 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
466
521
|
// `config set channel efforts.coding=…`, or `effort:` per request).
|
|
467
522
|
machine: "repo-resident",
|
|
468
523
|
identity: "write", // pushes branches and opens pull requests
|
|
524
|
+
// The native loop until the pi series moves this preset; a deployment
|
|
525
|
+
// flips it early with `harness: { coding: pi }` (docs/reference/specs/harness-pi.md).
|
|
526
|
+
harness: "native",
|
|
469
527
|
},
|
|
470
528
|
review: {
|
|
471
529
|
name: "review",
|
|
@@ -510,7 +568,7 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
510
568
|
research: {
|
|
511
569
|
name: "research",
|
|
512
570
|
description:
|
|
513
|
-
"Answers questions
|
|
571
|
+
"Answers questions that need the web (search, URL reading); GitHub for context, not for a question GitHub alone answers. No workspace.",
|
|
514
572
|
system: RESEARCH_SYSTEM,
|
|
515
573
|
toolset: "web",
|
|
516
574
|
machine: "none", // web I/O only; no workspace is provisioned
|
|
@@ -536,11 +594,15 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
536
594
|
cacheTtl: "1h",
|
|
537
595
|
// No built-in effort: the deployment decides, as for coding.
|
|
538
596
|
},
|
|
597
|
+
} satisfies Record<string, AgentDef>;
|
|
598
|
+
|
|
599
|
+
export const AGENTS: Record<string, AgentDef> = {
|
|
600
|
+
...WORK_PRESETS,
|
|
539
601
|
conductor: {
|
|
540
602
|
name: "conductor",
|
|
541
603
|
description:
|
|
542
604
|
"Coordinates other runs: spawns child runs as the requester — each in a thread of its own, under their permissions — follows them, and reports. No workspace or shell.",
|
|
543
|
-
system:
|
|
605
|
+
system: conductorSystem(Object.values(WORK_PRESETS)),
|
|
544
606
|
toolset: "conductor",
|
|
545
607
|
// Nothing is provisioned and no credential minted: the run tools call the
|
|
546
608
|
// dispatcher, the GitHub reads are REST in the bot process.
|
|
@@ -2,9 +2,9 @@
|
|
|
2
2
|
// the machine class its tools execute on, the identity it acts as, and the
|
|
3
3
|
// minutes it may run. A preset declares one; the run's EFFECTIVE profile is
|
|
4
4
|
// what the pipeline hands the factory, the ledger and the runner — never the
|
|
5
|
-
// preset's fields read again downstream. Pure and leaf:
|
|
6
|
-
//
|
|
7
|
-
import type
|
|
5
|
+
// preset's fields read again downstream. Pure and near-leaf: the only value
|
|
6
|
+
// import is the registry's `runawayTurnCap`, itself pure.
|
|
7
|
+
import { runawayTurnCap, type AgentDef, type Identity, type MachineClass } from "../agents/registry.js";
|
|
8
8
|
|
|
9
9
|
export type { Identity, MachineClass };
|
|
10
10
|
|
|
@@ -63,9 +63,12 @@ export function declaredProfile(preset: Pick<AgentDef, "machine" | "identity" |
|
|
|
63
63
|
/** The preset with its budget replaced by the effective profile's — the def
|
|
64
64
|
* the runner is handed, since its deadline, wrap-up warning and budget label
|
|
65
65
|
* read `maxMinutes` (the same clipped copy the ship pipeline hands its child
|
|
66
|
-
* rounds).
|
|
66
|
+
* rounds). `maxTurns` is re-derived from the clipped minutes with the
|
|
67
|
+
* registry's own `runawayTurnCap`, so the six-per-minute runaway guard holds
|
|
68
|
+
* for the budget the run actually has, not the preset's declared one.
|
|
69
|
+
* Always a copy: the shared `AgentDef` is never mutated. */
|
|
67
70
|
export function budgetedAgent(agent: AgentDef, profile: RunProfile): AgentDef {
|
|
68
|
-
return { ...agent, maxMinutes: profile.minutes };
|
|
71
|
+
return { ...agent, maxMinutes: profile.minutes, maxTurns: runawayTurnCap(profile.minutes) };
|
|
69
72
|
}
|
|
70
73
|
|
|
71
74
|
// ---- boundaries: a scope caps, never grants -----------------------------------
|
|
@@ -46,6 +46,14 @@ const REDACT: Array<{ re: RegExp; replace: string }> = [
|
|
|
46
46
|
const SECRET_COMPONENT =
|
|
47
47
|
/^(secret|token|password|passwd|pwd|credential|credentials|key|apikey|auth|session|sessionid|cookie)$/i;
|
|
48
48
|
|
|
49
|
+
// An identifier component that marks the whole identifier as an ERROR CODE, not
|
|
50
|
+
// a secret name: `github-token-mint-failed: HTTP 422 …` names a failure about a
|
|
51
|
+
// token, and the word after the colon is diagnosis, never a credential. Without
|
|
52
|
+
// this guard the assignment pass redacted `HTTP` out of that message
|
|
53
|
+
// (`github-token-mint-failed: «redacted» 422 …`), garbling the one line that
|
|
54
|
+
// tells an operator what went wrong.
|
|
55
|
+
const ERROR_CODE_COMPONENT = /^(failed|failure|error|err|refused|denied|invalid|missing|expired|unset|mismatch)$/i;
|
|
56
|
+
|
|
49
57
|
/** Redact the VALUE of any `<name> = value` / `<name>: value` where the name has
|
|
50
58
|
* a secret-marking component. Handles quoted values (with spaces) and unquoted,
|
|
51
59
|
* and a quoted NAME (`"password": "…"` in pasted JSON — the closing quote sits
|
|
@@ -64,7 +72,9 @@ function redactNamedAssignments(text: string): string {
|
|
|
64
72
|
// decides whether it names a secret (`_SECRET` → ["", "SECRET"]).
|
|
65
73
|
/(?<![A-Za-z0-9_-])([A-Za-z0-9_-]+)("?)(\s*[=:]\s*)("(?:[^"\\]|\\.)*"|'(?:[^'\\]|\\.)*'|[^\s]{4,})/g,
|
|
66
74
|
(whole, id: string, close: string, sep: string, val: string) => {
|
|
67
|
-
|
|
75
|
+
const parts = id.split(/[_-]/);
|
|
76
|
+
if (!parts.some((p) => SECRET_COMPONENT.test(p))) return whole;
|
|
77
|
+
if (parts.some((p) => ERROR_CODE_COMPONENT.test(p))) return whole;
|
|
68
78
|
const quote = val[0] === '"' || val[0] === "'" ? val[0] : "";
|
|
69
79
|
return `${id}${close}${sep}${quote}«redacted»${quote}`;
|
|
70
80
|
},
|
|
@@ -118,7 +118,27 @@ export type RunNoteKind =
|
|
|
118
118
|
* post-step beside the thread's Slack-only note, so the record says the
|
|
119
119
|
* verdict is Slack-only and a coordinator reading it never asks GitHub
|
|
120
120
|
* for a review that was never sent. */
|
|
121
|
-
| "review_not_posted"
|
|
121
|
+
| "review_not_posted"
|
|
122
|
+
/** A pi run's context was compacted (docs/reference/specs/harness-pi.md item
|
|
123
|
+
* 6): pi summarized its older turns into one entry and the model reads the
|
|
124
|
+
* summary from here on; the transcript keeps the originals, so the record
|
|
125
|
+
* is a superset of the model's context. The summary names the token counts
|
|
126
|
+
* before and after. Published by the pi bridge. */
|
|
127
|
+
| "compacted"
|
|
128
|
+
/** The pi harness itself failed in a way the run must show (harness-pi.md
|
|
129
|
+
* item 6): the extension threw, pi asked a dialog no one answers (answered
|
|
130
|
+
* cancelled), or pi emitted an event kind this build's bridge does not
|
|
131
|
+
* know — named, so a pi bump is visible in the first run's record. Published
|
|
132
|
+
* by the pi bridge. */
|
|
133
|
+
| "harness_error"
|
|
134
|
+
/** The harness's gate refused a tool call the model asked for (harness-pi.md
|
|
135
|
+
* item 7): the summary names the tool and the rule; the model read the same
|
|
136
|
+
* reason as the tool's result. Published by the bot's authorize route. */
|
|
137
|
+
| "tool_refused"
|
|
138
|
+
/** The stuck-loop guard fired (docs/reference/specs/run-loop.md item 18):
|
|
139
|
+
* the same tool call failed identically six times in a row, so the run is
|
|
140
|
+
* forced into its write-up instead of looping to the wall clock. */
|
|
141
|
+
| "stuck_loop";
|
|
122
142
|
|
|
123
143
|
/** Every `RunNoteKind`, as a value (a reader that filters notes by kind uses
|
|
124
144
|
* this; adding a kind to the union without adding it here is a type error). */
|
|
@@ -139,6 +159,10 @@ export const RUN_NOTE_KINDS = [
|
|
|
139
159
|
"cold_sandbox",
|
|
140
160
|
"pr_not_opened",
|
|
141
161
|
"review_not_posted",
|
|
162
|
+
"compacted",
|
|
163
|
+
"harness_error",
|
|
164
|
+
"tool_refused",
|
|
165
|
+
"stuck_loop",
|
|
142
166
|
] as const satisfies readonly RunNoteKind[];
|
|
143
167
|
type _EveryKindListed = [RunNoteKind] extends [(typeof RUN_NOTE_KINDS)[number]] ? true : never;
|
|
144
168
|
const _everyKindListed: _EveryKindListed = true;
|
|
@@ -383,6 +407,27 @@ export type RunEvent =
|
|
|
383
407
|
seq?: number;
|
|
384
408
|
at?: number;
|
|
385
409
|
}
|
|
410
|
+
/** A file moved through the artifact store (docs/reference/specs/execution.md item 20,
|
|
411
|
+
* record 0033): one the run received from its thread (`in`, staged before
|
|
412
|
+
* the turn) or one it sent (`out`, `attach_file`). The record keeps the
|
|
413
|
+
* store KEY and the facts a page needs to list the file — never a URL: the
|
|
414
|
+
* run page's proxy route (`/runs/:id/artifacts/<key>`, live-view.md item 26)
|
|
415
|
+
* mints a signed GET per request, so a stored record never carries a
|
|
416
|
+
* credential that expires or leaks, and the parser refuses a payload that
|
|
417
|
+
* tries. A side fact beside the tool pair that moved the file (like
|
|
418
|
+
* `skill_use`), never a step; the friction analyzer ignores it. */
|
|
419
|
+
| {
|
|
420
|
+
type: "artifact";
|
|
421
|
+
direction: "in" | "out";
|
|
422
|
+
/** The store key (`src/artifacts/keys.ts`): `runs/<runId>/out/<seq>-<basename>` or `threads/<thread>/in/<ts>/<i>-<basename>`. */
|
|
423
|
+
key: string;
|
|
424
|
+
/** The file's name as the person sees it (the Slack filename, the tool's `name`). */
|
|
425
|
+
name: string;
|
|
426
|
+
size: number;
|
|
427
|
+
contentType: string;
|
|
428
|
+
seq?: number;
|
|
429
|
+
at?: number;
|
|
430
|
+
}
|
|
386
431
|
/** A skill was loaded into the model's context (docs/reference/specs/skills.md). Emitted
|
|
387
432
|
* by the `use_skill` tool on a successful load — alongside, not instead of,
|
|
388
433
|
* its `tool_call`/`tool_result` pair — so skill use is a first-class fact in
|
|
@@ -254,6 +254,13 @@ function unionMs(intervals: ReadonlyArray<{ start: number; end: number }>): numb
|
|
|
254
254
|
return total;
|
|
255
255
|
}
|
|
256
256
|
|
|
257
|
+
/** One tool call's identity for retry/streak accounting: the tool name plus
|
|
258
|
+
* the call's one-line summary (which carries the arguments — a bash command,
|
|
259
|
+
* a path, a url). Shared with the runner's stuck-loop guard
|
|
260
|
+
* (docs/reference/specs/run-loop.md item 18), so both count "the same call"
|
|
261
|
+
* identically. */
|
|
262
|
+
export const callSignature = (tool: string, summary: string): string => `${tool} ${summary}`;
|
|
263
|
+
|
|
257
264
|
/** Analyze a run's event stream. Pure and deterministic; never mutates `events`. */
|
|
258
265
|
export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOptions = {}): FrictionDiagnosis {
|
|
259
266
|
const slowToolMs = opts.slowToolMs ?? DEFAULT_SLOW_TOOL_MS;
|
|
@@ -384,12 +391,14 @@ export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOp
|
|
|
384
391
|
return;
|
|
385
392
|
}
|
|
386
393
|
// Side facts about the run, not steps: skill_use rides beside a use_skill
|
|
387
|
-
// call that already produced its own tool pair
|
|
388
|
-
//
|
|
389
|
-
//
|
|
390
|
-
//
|
|
394
|
+
// call that already produced its own tool pair, and artifact beside the
|
|
395
|
+
// attach_file call (or the dispatcher's staging) that moved the file;
|
|
396
|
+
// review_artifact, pr_description, pr_opened, review_posted, the ship_round
|
|
397
|
+
// boundaries and the router's route are published by the dispatcher/pipeline
|
|
398
|
+
// outside the model loop entirely. Counting any of them would distort the story.
|
|
391
399
|
if (
|
|
392
400
|
ev.type === "skill_use" ||
|
|
401
|
+
ev.type === "artifact" ||
|
|
393
402
|
ev.type === "review_artifact" ||
|
|
394
403
|
ev.type === "pr_description" ||
|
|
395
404
|
ev.type === "pr_opened" ||
|
|
@@ -403,7 +412,7 @@ export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOp
|
|
|
403
412
|
|
|
404
413
|
if (ev.type === "tool_call") {
|
|
405
414
|
toolCalls++;
|
|
406
|
-
if (failedCalls.has(
|
|
415
|
+
if (failedCalls.has(callSignature(ev.tool, ev.summary))) {
|
|
407
416
|
findings.push({
|
|
408
417
|
category: "retry",
|
|
409
418
|
severity: "low",
|
|
@@ -430,7 +439,7 @@ export function analyzeRunFriction(events: readonly RunEvent[], opts: FrictionOp
|
|
|
430
439
|
? { interval: { start: span.startedAt, end: span.startedAt + durationMs } }
|
|
431
440
|
: {};
|
|
432
441
|
|
|
433
|
-
if (!ev.ok) failedCalls.add(
|
|
442
|
+
if (!ev.ok) failedCalls.add(callSignature(ev.tool, callSummary));
|
|
434
443
|
|
|
435
444
|
if (ev.infra) {
|
|
436
445
|
// An infra-level failure is the sandbox, not the command: classify once,
|
|
@@ -7,6 +7,7 @@ import type { ChatMessage, ToolDef } from "../../providers/types.js";
|
|
|
7
7
|
import type { ChannelVisibility } from "../authz/types.js";
|
|
8
8
|
import type { RunProfile } from "../../config/profile.js";
|
|
9
9
|
import type { RunEvent } from "../runEvents.js";
|
|
10
|
+
import type { RunSeed } from "../runRecord.js";
|
|
10
11
|
|
|
11
12
|
/** How long a generation's claim on a run lasts without a heartbeat. */
|
|
12
13
|
export const LEASE_MS = 30_000;
|
|
@@ -69,6 +70,9 @@ export interface LiveRunMeta {
|
|
|
69
70
|
* still sends the parent its event and a retried spawn finds its run. */
|
|
70
71
|
parentInstanceId?: string;
|
|
71
72
|
idempotencyKey?: string;
|
|
73
|
+
/** Where the run's conversation started (item 52), so a reclaimed run's
|
|
74
|
+
* record still says so: `parent` for a spawned child, `channel` otherwise. */
|
|
75
|
+
seed?: RunSeed;
|
|
72
76
|
/** Which executor the run attached: what `makeExecutor` chose. */
|
|
73
77
|
selection?: "resident" | "sandbox" | "local" | "none";
|
|
74
78
|
/** The worktree path the system prompt names. */
|
|
@@ -145,8 +145,18 @@ export interface RunRecord {
|
|
|
145
145
|
* item 48), stored at the claim so a retried spawn finds its run. Present
|
|
146
146
|
* exactly when `parentInstanceId` is — both or neither, never one alone. */
|
|
147
147
|
idempotencyKey?: string;
|
|
148
|
+
/** Where the run's conversation started (item 52): `channel` — its own
|
|
149
|
+
* thread's history, as for every run a person, a schedule or a coordinator
|
|
150
|
+
* started — or `parent` — a spawned child seeded from its parent's text
|
|
151
|
+
* turns at the spawn (`DispatchOptions.seed`). Absent on records written
|
|
152
|
+
* before it existed. */
|
|
153
|
+
seed?: RunSeed;
|
|
148
154
|
}
|
|
149
155
|
|
|
156
|
+
/** The two places a run's conversation can start (item 52). */
|
|
157
|
+
export const RUN_SEEDS = ["channel", "parent"] as const;
|
|
158
|
+
export type RunSeed = (typeof RUN_SEEDS)[number];
|
|
159
|
+
|
|
150
160
|
/** The profile as the record stores it: the run's effective profile plus the preset it came from. */
|
|
151
161
|
export type RunProfileRecord = RunProfile & { preset: string };
|
|
152
162
|
|
|
@@ -515,6 +525,8 @@ export function isRunRecord(v: unknown): v is RunRecord {
|
|
|
515
525
|
// A parent is named by a run id (item 46): the same shape as the record's own.
|
|
516
526
|
if (r.parentRunId !== undefined && (typeof r.parentRunId !== "string" || !RUN_ID_PATTERN.test(r.parentRunId)))
|
|
517
527
|
return false;
|
|
528
|
+
// Where the conversation started (item 52): one of the two words, or absent.
|
|
529
|
+
if (r.seed !== undefined && !RUN_SEEDS.includes(r.seed as RunSeed)) return false;
|
|
518
530
|
// A coordinator's child (item 48): the instance id in the platform's alphabet
|
|
519
531
|
// and the key `<instance>:<step>` — both or neither; one alone is no tag.
|
|
520
532
|
if ((r.parentInstanceId === undefined) !== (r.idempotencyKey === undefined)) return false;
|
|
@@ -52,9 +52,15 @@ export function shimRoute(pathname: string): string | undefined {
|
|
|
52
52
|
if (pathname === "/costs" || pathname.startsWith("/costs/")) return "costs";
|
|
53
53
|
if (pathname.startsWith("/api/")) return "api";
|
|
54
54
|
if (pathname.startsWith("/admin/")) return "admin";
|
|
55
|
+
// The artifact copy (deploy/cloudflare/artifactsCopy.ts): the shim's own route, a 1 GB stream.
|
|
56
|
+
if (pathname === "/artifacts/copy") return "artifacts";
|
|
55
57
|
// The model proxy's two routes (docs/reference/specs/model-proxy.md): a bounded
|
|
56
58
|
// request per model call, forwarded to the container like everything else.
|
|
57
59
|
if (pathname === "/v1/messages" || pathname === "/v1/chat/completions") return "model-proxy";
|
|
60
|
+
// The pi harness's three routes (docs/reference/specs/harness-pi.md item 7): a
|
|
61
|
+
// run's extension asking for its tools, a verdict, a relayed tool's result.
|
|
62
|
+
if (pathname === "/harness/tools" || pathname === "/harness/authorize" || pathname === "/harness/tool")
|
|
63
|
+
return "harness";
|
|
58
64
|
if (pathname === "/docs" || pathname.startsWith("/docs/")) return "docs";
|
|
59
65
|
if (pathname === "/" || pathname === "/index.html") return "page";
|
|
60
66
|
return "other";
|
|
@@ -89,6 +89,17 @@ export const profileSchema = z.object({
|
|
|
89
89
|
/** The Cloudflare Access application in front of the bot's dashboards, when
|
|
90
90
|
* there is one: the team domain the JWT is issued by and the app's AUD. */
|
|
91
91
|
access: z.object({ teamDomain: hostname, aud: z.string().regex(/^[0-9a-f]{64}$/) }).optional(),
|
|
92
|
+
/** The artifact store's bucket (docs/reference/specs/execution.md item 20), when the installation
|
|
93
|
+
* has one: the bot Worker's template binds it as `ARTIFACTS` and `deploy` creates it before the
|
|
94
|
+
* upload. The bot's runtime config (`artifacts.r2.bucket`) must name the same bucket — the two
|
|
95
|
+
* are held equal by `artifacts check`, not by the profile. R2's own bucket-name rules. */
|
|
96
|
+
artifacts: z
|
|
97
|
+
.object({
|
|
98
|
+
bucket: z
|
|
99
|
+
.string()
|
|
100
|
+
.regex(/^[a-z0-9][a-z0-9-]{1,61}[a-z0-9]$/, "an R2 bucket name: 3–63 lowercase letters, digits and hyphens"),
|
|
101
|
+
})
|
|
102
|
+
.optional(),
|
|
92
103
|
});
|
|
93
104
|
|
|
94
105
|
export type DeploymentProfile = z.infer<typeof profileSchema>;
|