@coreplane/switchboard 1.208.0 → 1.209.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "switchboard",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.209.0",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "switchboard",
|
|
9
|
-
"version": "1.
|
|
9
|
+
"version": "1.209.0",
|
|
10
10
|
"license": "Apache-2.0",
|
|
11
11
|
"workspaces": [
|
|
12
12
|
"web",
|
|
@@ -18999,7 +18999,7 @@
|
|
|
18999
18999
|
},
|
|
19000
19000
|
"packages/switchboard": {
|
|
19001
19001
|
"name": "@coreplane/switchboard",
|
|
19002
|
-
"version": "1.
|
|
19002
|
+
"version": "1.209.0",
|
|
19003
19003
|
"license": "Apache-2.0",
|
|
19004
19004
|
"dependencies": {
|
|
19005
19005
|
"@anthropic-ai/sdk": "^0.124.0",
|
package/dist/assets/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "switchboard",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.209.0",
|
|
4
4
|
"private": true,
|
|
5
5
|
"description": "Mention it in Slack and an agent reviews the PR, ships the fix, or answers the question — on the model you choose, with its tools running where you decide.",
|
|
6
6
|
"license": "Apache-2.0",
|
package/dist/assets/source.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
// Agent definitions. An agent is a system prompt + toolset + machine class +
|
|
1
|
+
// Agent definitions. An agent is a system prompt + toolset + machine class + wall-clock budget.
|
|
2
2
|
import type { Effort } from "../effort.js";
|
|
3
3
|
import type { CacheTtl } from "../providers/types.js";
|
|
4
4
|
import { BASH_TIMEOUT_MAX_MS } from "../execution/bashTimeout.js";
|
|
5
|
-
import { CONTRACT_HEADING, CONTRACT_SECTION_HEADINGS } from "../core/ship/contract.js";
|
|
5
|
+
import { CONTRACT_HEADING, CONTRACT_SECTION_HEADINGS, PR_TITLE_GUARD } from "../core/ship/contract.js";
|
|
6
6
|
// Which model runs it is resolved separately by the config layers, so any
|
|
7
7
|
// agent can run on any configured provider/model.
|
|
8
8
|
|
|
@@ -42,13 +42,37 @@ export function machineNeedsRepo(machine: MachineClass): boolean {
|
|
|
42
42
|
export const IDENTITIES = ["none", "read", "write"] as const;
|
|
43
43
|
export type Identity = (typeof IDENTITIES)[number];
|
|
44
44
|
|
|
45
|
+
/** The pace that marks a run as looping rather than working: a model turn
|
|
46
|
+
* every ten seconds, sustained for the whole wall clock. A busy run takes
|
|
47
|
+
* 20–40 s a turn (a model think plus a tool call), so a run that averages six
|
|
48
|
+
* a minute from start to end is re-issuing calls, not making progress — and
|
|
49
|
+
* its turn cap ends it before the wall clock would, with a write-up that
|
|
50
|
+
* says so (docs/reference/specs/run-loop.md item 1). */
|
|
51
|
+
export const RUNAWAY_TURNS_PER_MINUTE = 6;
|
|
52
|
+
|
|
53
|
+
/** The turn cap a wall clock implies: `maxMinutes × RUNAWAY_TURNS_PER_MINUTE`.
|
|
54
|
+
* Every preset that runs the loop derives its `maxTurns` from this, so the
|
|
55
|
+
* cap is never a number a good run reaches — the minutes are the budget. */
|
|
56
|
+
export function runawayTurnCap(maxMinutes: number): number {
|
|
57
|
+
return maxMinutes * RUNAWAY_TURNS_PER_MINUTE;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** A loop-running preset's budget as one fact: the wall clock, and the runaway
|
|
61
|
+
* guard derived from it. */
|
|
62
|
+
function loopBudget(maxMinutes: number): Pick<AgentDef, "maxMinutes" | "maxTurns"> {
|
|
63
|
+
return { maxMinutes, maxTurns: runawayTurnCap(maxMinutes) };
|
|
64
|
+
}
|
|
65
|
+
|
|
45
66
|
export interface AgentDef {
|
|
46
67
|
name: string;
|
|
47
68
|
description: string;
|
|
48
69
|
system: string;
|
|
49
70
|
/** key into TOOLSETS: "full" | "readonly" | "web" | "assistant" | "explore" | "conductor" | "none" */
|
|
50
71
|
toolset: "full" | "readonly" | "web" | "assistant" | "explore" | "conductor" | "none";
|
|
51
|
-
/**
|
|
72
|
+
/** The runaway guard, not a budget: `runawayTurnCap(maxMinutes)` for every
|
|
73
|
+
* preset that runs the loop (`loopBudget`). The wall clock below is the
|
|
74
|
+
* budget; a run that reaches this cap first was pacing like a loop, and its
|
|
75
|
+
* write-up says so. The proxy refuses model calls past it too. */
|
|
52
76
|
maxTurns: number;
|
|
53
77
|
maxTokens: number;
|
|
54
78
|
/** hard wall-clock budget for the tool loop; at the deadline the agent is
|
|
@@ -97,7 +121,7 @@ export interface AgentDef {
|
|
|
97
121
|
// and rendered against the head sha at render time, so a repush is a
|
|
98
122
|
// re-render by Switchboard — the agent only resubmits when the CONTENT (line
|
|
99
123
|
// numbers included) changed.
|
|
100
|
-
const PR_DESCRIPTION_TEMPLATE = `PR description — submit it with the submit_pr_description tool for EVERY PR (this is the default, not something to wait to be asked for). Switchboard renders the GitHub body from the object you submit, so never author PR-body markdown yourself. Content contract per field (each renders as its own section): prose is unwrapped — no hard line breaks inside a paragraph. Always hyperlink the triggering issue/request. Never fabricate validation — state exactly what you ran and the real result. Keep each field concise, not padded.
|
|
124
|
+
const PR_DESCRIPTION_TEMPLATE = `PR description — submit it with the submit_pr_description tool for EVERY PR (this is the default, not something to wait to be asked for). Switchboard renders the GitHub body from the object you submit, so never author PR-body markdown yourself. Before submitting, judge your title with the ${PR_TITLE_GUARD} gate — \`npm run check:pr-title -- "<title>"\` — and submit only a title it accepts; the same gate refuses the PR in CI. Content contract per field (each renders as its own section): prose is unwrapped — no hard line breaks inside a paragraph. Always hyperlink the triggering issue/request. Never fabricate validation — state exactly what you ran and the real result. Keep each field concise, not padded.
|
|
101
125
|
EVERY PR includes one that already exists when you push — opened by a person, by dependabot, or by an earlier run. After EVERY push to such a PR: read its current title and body (\`github_issue_get\` with the PR number works for pull requests; \`gh pr view\` where gh exists), judge them against the change as it now stands at the pushed head, and submit the object that describes the PR as it is NOW — carry forward what the existing body says that is still true (a dependency bump's release notes belong in whatWhy), add what you changed, and anchor the Tour at the new head. Switchboard replaces the PR's title and body with your rendering. A description that describes an earlier state of its branch is a bug; "it is someone else's PR" is never a reason to leave it.
|
|
102
126
|
- **title**: the PR title — one line naming the change, specific enough to pick out of a PR list.
|
|
103
127
|
- **TL;DR** (\`tldr\`, rendered first): two sentences for a naive reader with zero context — what this PR does and why it matters.
|
|
@@ -401,9 +425,8 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
401
425
|
// and mints no credential of its own.
|
|
402
426
|
machine: "none",
|
|
403
427
|
identity: "none",
|
|
404
|
-
maxTurns: 8, // a repo read is 2-3 calls (repos → tree → file); an issue action 1-2; still fast
|
|
405
428
|
maxTokens: 16000,
|
|
406
|
-
|
|
429
|
+
...loopBudget(5),
|
|
407
430
|
},
|
|
408
431
|
coding: {
|
|
409
432
|
name: "coding",
|
|
@@ -411,9 +434,8 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
411
434
|
system: CODING_SYSTEM,
|
|
412
435
|
residentSystem: CODING_SYSTEM_RESIDENT,
|
|
413
436
|
toolset: "full",
|
|
414
|
-
maxTurns: 60, // scoping is capped at ~5 calls by the prompt; this is implementation room
|
|
415
437
|
maxTokens: 64000,
|
|
416
|
-
|
|
438
|
+
...loopBudget(45),
|
|
417
439
|
// Coding steps run long: a single model turn can take 5-6 minutes and
|
|
418
440
|
// installs/tests add more — a 5m cache entry would expire between
|
|
419
441
|
// requests, so the 2× write buys reads for the whole run.
|
|
@@ -431,9 +453,8 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
431
453
|
toolset: "readonly",
|
|
432
454
|
machine: "repo-resident",
|
|
433
455
|
identity: "read", // a read-scoped token and a read-only worktree: it cannot post or push from inside
|
|
434
|
-
maxTurns: 30, // backstop only; wall clock is the real budget (12 bound at ~4 min in practice)
|
|
435
456
|
maxTokens: 64000,
|
|
436
|
-
|
|
457
|
+
...loopBudget(25), // a safety net — typical reviews land in ~5 minutes
|
|
437
458
|
effort: "medium", // fast turns; one big-context pass does the deep work
|
|
438
459
|
},
|
|
439
460
|
ship: {
|
|
@@ -468,9 +489,8 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
468
489
|
toolset: "web",
|
|
469
490
|
machine: "none", // web I/O only; no workspace is provisioned
|
|
470
491
|
identity: "none",
|
|
471
|
-
maxTurns: 12,
|
|
472
492
|
maxTokens: 24000,
|
|
473
|
-
|
|
493
|
+
...loopBudget(8),
|
|
474
494
|
effort: "medium",
|
|
475
495
|
},
|
|
476
496
|
explore: {
|
|
@@ -483,9 +503,8 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
483
503
|
// review depends on: a two-hour job shares no container with anyone.
|
|
484
504
|
machine: "repo-cold",
|
|
485
505
|
identity: "read", // a read-scoped token: it can clone and read, never push — whatever the caller holds
|
|
486
|
-
maxTurns: 150, // a backstop for a two-hour loop of batched checks; the wall clock is the budget
|
|
487
506
|
maxTokens: 64000,
|
|
488
|
-
|
|
507
|
+
...loopBudget(120),
|
|
489
508
|
// A detached job polled across calls makes long steps: a 5m cache entry
|
|
490
509
|
// would expire between them, so the 2× write buys reads for the whole run.
|
|
491
510
|
cacheTtl: "1h",
|
|
@@ -501,9 +520,8 @@ export const AGENTS: Record<string, AgentDef> = {
|
|
|
501
520
|
// dispatcher, the GitHub reads are REST in the bot process.
|
|
502
521
|
machine: "none",
|
|
503
522
|
identity: "none",
|
|
504
|
-
maxTurns: 40, // a spawn, then a poll per child every few minutes; the wall clock is the budget
|
|
505
523
|
maxTokens: 32000,
|
|
506
|
-
|
|
524
|
+
...loopBudget(120), // long enough to outlast a coding child; every child is capped by what remains of it
|
|
507
525
|
// No built-in effort: the deployment decides, as for coding.
|
|
508
526
|
},
|
|
509
527
|
};
|
package/dist/cli.js
CHANGED
|
@@ -1560,6 +1560,12 @@ var init_contract = __esm({
|
|
|
1560
1560
|
function machineNeedsRepo(machine) {
|
|
1561
1561
|
return machine === "repo-cold" || machine === "repo-resident";
|
|
1562
1562
|
}
|
|
1563
|
+
function runawayTurnCap(maxMinutes) {
|
|
1564
|
+
return maxMinutes * RUNAWAY_TURNS_PER_MINUTE;
|
|
1565
|
+
}
|
|
1566
|
+
function loopBudget(maxMinutes) {
|
|
1567
|
+
return { maxMinutes, maxTurns: runawayTurnCap(maxMinutes) };
|
|
1568
|
+
}
|
|
1563
1569
|
function getAgent(name) {
|
|
1564
1570
|
const a = AGENTS[name];
|
|
1565
1571
|
if (!a) {
|
|
@@ -1567,7 +1573,7 @@ function getAgent(name) {
|
|
|
1567
1573
|
}
|
|
1568
1574
|
return a;
|
|
1569
1575
|
}
|
|
1570
|
-
var MACHINE_CLASSES, IDENTITIES, PR_DESCRIPTION_TEMPLATE, NEVER_MERGE, CONTRACT_HEADINGS_LIST, UNIT_CONTRACT, UNIT_HANDOFF, IMAGE_TOOLCHAIN, SANDBOX_TOOLCHAIN, RESIDENT_TOOLCHAIN, CODING_SYSTEM, CODING_SYSTEM_RESIDENT, REVIEW_VERDICT_INSTRUCTION, REVIEW_SPEC_CHECK, REVIEW_UNIT_CONTRACT, REVIEW_WHOLE_CHANGE, REVIEW_SYSTEM, REVIEW_SYSTEM_RESIDENT, RESEARCH_SYSTEM, GENERAL_SYSTEM, EXPLORE_SYSTEM, CONDUCTOR_SYSTEM, AGENTS;
|
|
1576
|
+
var MACHINE_CLASSES, IDENTITIES, RUNAWAY_TURNS_PER_MINUTE, PR_DESCRIPTION_TEMPLATE, NEVER_MERGE, CONTRACT_HEADINGS_LIST, UNIT_CONTRACT, UNIT_HANDOFF, IMAGE_TOOLCHAIN, SANDBOX_TOOLCHAIN, RESIDENT_TOOLCHAIN, CODING_SYSTEM, CODING_SYSTEM_RESIDENT, REVIEW_VERDICT_INSTRUCTION, REVIEW_SPEC_CHECK, REVIEW_UNIT_CONTRACT, REVIEW_WHOLE_CHANGE, REVIEW_SYSTEM, REVIEW_SYSTEM_RESIDENT, RESEARCH_SYSTEM, GENERAL_SYSTEM, EXPLORE_SYSTEM, CONDUCTOR_SYSTEM, AGENTS;
|
|
1571
1577
|
var init_registry = __esm({
|
|
1572
1578
|
"../../src/agents/registry.ts"() {
|
|
1573
1579
|
"use strict";
|
|
@@ -1575,7 +1581,8 @@ var init_registry = __esm({
|
|
|
1575
1581
|
init_contract();
|
|
1576
1582
|
MACHINE_CLASSES = ["none", "blank", "repo-cold", "repo-resident"];
|
|
1577
1583
|
IDENTITIES = ["none", "read", "write"];
|
|
1578
|
-
|
|
1584
|
+
RUNAWAY_TURNS_PER_MINUTE = 6;
|
|
1585
|
+
PR_DESCRIPTION_TEMPLATE = `PR description \u2014 submit it with the submit_pr_description tool for EVERY PR (this is the default, not something to wait to be asked for). Switchboard renders the GitHub body from the object you submit, so never author PR-body markdown yourself. Before submitting, judge your title with the ${PR_TITLE_GUARD} gate \u2014 \`npm run check:pr-title -- "<title>"\` \u2014 and submit only a title it accepts; the same gate refuses the PR in CI. Content contract per field (each renders as its own section): prose is unwrapped \u2014 no hard line breaks inside a paragraph. Always hyperlink the triggering issue/request. Never fabricate validation \u2014 state exactly what you ran and the real result. Keep each field concise, not padded.
|
|
1579
1586
|
EVERY PR includes one that already exists when you push \u2014 opened by a person, by dependabot, or by an earlier run. After EVERY push to such a PR: read its current title and body (\`github_issue_get\` with the PR number works for pull requests; \`gh pr view\` where gh exists), judge them against the change as it now stands at the pushed head, and submit the object that describes the PR as it is NOW \u2014 carry forward what the existing body says that is still true (a dependency bump's release notes belong in whatWhy), add what you changed, and anchor the Tour at the new head. Switchboard replaces the PR's title and body with your rendering. A description that describes an earlier state of its branch is a bug; "it is someone else's PR" is never a reason to leave it.
|
|
1580
1587
|
- **title**: the PR title \u2014 one line naming the change, specific enough to pick out of a PR list.
|
|
1581
1588
|
- **TL;DR** (\`tldr\`, rendered first): two sentences for a naive reader with zero context \u2014 what this PR does and why it matters.
|
|
@@ -1766,10 +1773,8 @@ Use Slack-friendly formatting (no markdown headers; *bold*, bullets, code blocks
|
|
|
1766
1773
|
// and mints no credential of its own.
|
|
1767
1774
|
machine: "none",
|
|
1768
1775
|
identity: "none",
|
|
1769
|
-
maxTurns: 8,
|
|
1770
|
-
// a repo read is 2-3 calls (repos → tree → file); an issue action 1-2; still fast
|
|
1771
1776
|
maxTokens: 16e3,
|
|
1772
|
-
|
|
1777
|
+
...loopBudget(5)
|
|
1773
1778
|
},
|
|
1774
1779
|
coding: {
|
|
1775
1780
|
name: "coding",
|
|
@@ -1777,10 +1782,8 @@ Use Slack-friendly formatting (no markdown headers; *bold*, bullets, code blocks
|
|
|
1777
1782
|
system: CODING_SYSTEM,
|
|
1778
1783
|
residentSystem: CODING_SYSTEM_RESIDENT,
|
|
1779
1784
|
toolset: "full",
|
|
1780
|
-
maxTurns: 60,
|
|
1781
|
-
// scoping is capped at ~5 calls by the prompt; this is implementation room
|
|
1782
1785
|
maxTokens: 64e3,
|
|
1783
|
-
|
|
1786
|
+
...loopBudget(45),
|
|
1784
1787
|
// Coding steps run long: a single model turn can take 5-6 minutes and
|
|
1785
1788
|
// installs/tests add more — a 5m cache entry would expire between
|
|
1786
1789
|
// requests, so the 2× write buys reads for the whole run.
|
|
@@ -1800,11 +1803,9 @@ Use Slack-friendly formatting (no markdown headers; *bold*, bullets, code blocks
|
|
|
1800
1803
|
machine: "repo-resident",
|
|
1801
1804
|
identity: "read",
|
|
1802
1805
|
// a read-scoped token and a read-only worktree: it cannot post or push from inside
|
|
1803
|
-
maxTurns: 30,
|
|
1804
|
-
// backstop only; wall clock is the real budget (12 bound at ~4 min in practice)
|
|
1805
1806
|
maxTokens: 64e3,
|
|
1806
|
-
|
|
1807
|
-
// safety net
|
|
1807
|
+
...loopBudget(25),
|
|
1808
|
+
// a safety net — typical reviews land in ~5 minutes
|
|
1808
1809
|
effort: "medium"
|
|
1809
1810
|
// fast turns; one big-context pass does the deep work
|
|
1810
1811
|
},
|
|
@@ -1838,9 +1839,8 @@ Use Slack-friendly formatting (no markdown headers; *bold*, bullets, code blocks
|
|
|
1838
1839
|
machine: "none",
|
|
1839
1840
|
// web I/O only; no workspace is provisioned
|
|
1840
1841
|
identity: "none",
|
|
1841
|
-
maxTurns: 12,
|
|
1842
1842
|
maxTokens: 24e3,
|
|
1843
|
-
|
|
1843
|
+
...loopBudget(8),
|
|
1844
1844
|
effort: "medium"
|
|
1845
1845
|
},
|
|
1846
1846
|
explore: {
|
|
@@ -1853,10 +1853,8 @@ Use Slack-friendly formatting (no markdown headers; *bold*, bullets, code blocks
|
|
|
1853
1853
|
machine: "repo-cold",
|
|
1854
1854
|
identity: "read",
|
|
1855
1855
|
// a read-scoped token: it can clone and read, never push — whatever the caller holds
|
|
1856
|
-
maxTurns: 150,
|
|
1857
|
-
// a backstop for a two-hour loop of batched checks; the wall clock is the budget
|
|
1858
1856
|
maxTokens: 64e3,
|
|
1859
|
-
|
|
1857
|
+
...loopBudget(120),
|
|
1860
1858
|
// A detached job polled across calls makes long steps: a 5m cache entry
|
|
1861
1859
|
// would expire between them, so the 2× write buys reads for the whole run.
|
|
1862
1860
|
cacheTtl: "1h"
|
|
@@ -1871,10 +1869,8 @@ Use Slack-friendly formatting (no markdown headers; *bold*, bullets, code blocks
|
|
|
1871
1869
|
// dispatcher, the GitHub reads are REST in the bot process.
|
|
1872
1870
|
machine: "none",
|
|
1873
1871
|
identity: "none",
|
|
1874
|
-
maxTurns: 40,
|
|
1875
|
-
// a spawn, then a poll per child every few minutes; the wall clock is the budget
|
|
1876
1872
|
maxTokens: 32e3,
|
|
1877
|
-
|
|
1873
|
+
...loopBudget(120)
|
|
1878
1874
|
// long enough to outlast a coding child; every child is capped by what remains of it
|
|
1879
1875
|
// No built-in effort: the deployment decides, as for coding.
|
|
1880
1876
|
}
|
|
@@ -14526,7 +14522,7 @@ ${evidence}`);
|
|
|
14526
14522
|
case "wrap_up":
|
|
14527
14523
|
return `Runs keep reaching the wrap-up warning (${p.runIds.length} runs; ${formatDuration(p.durationMs, "report")} spent winding down). Either the affected agent's \`maxMinutes\` (\`src/agents/registry.ts\`) is too tight for this shape of work, or the prompt should push batching (fewer, larger tool calls) \u2014 the evidence rows say which agent and how close to the deadline each run got.`;
|
|
14528
14524
|
case "budget_hit":
|
|
14529
|
-
return p.signature === "turns" ? `Runs
|
|
14525
|
+
return p.signature === "turns" ? `Runs outpace the runaway guard: the turn cap is \`RUNAWAY_TURNS_PER_MINUTE\` (${RUNAWAY_TURNS_PER_MINUTE} turns a minute) over the affected agent's wall clock, a pace a working run does not sustain \u2014 so a run that reaches it is looping, not working. Read the evidence rows for a retry loop (the same call re-issued turn after turn) and fix its cause where the agent reads before acting (the target repo's AGENTS.md, the resident command table), or have the prompt batch tool calls (several commands per \`bash\` call). The cap is derived from \`maxMinutes\` (\`src/agents/registry.ts\`), not a knob to turn.` : `Runs exhaust the TIME budget and are cut off mid-work. Raise \`maxMinutes\` for the affected agent (\`src/agents/registry.ts\`), or split the task shape that triggers it \u2014 a run that is forced to write up findings is a run whose work was wasted.`;
|
|
14530
14526
|
case "infra_failure":
|
|
14531
14527
|
if (p.signature === "sandbox_dead") {
|
|
14532
14528
|
return `The sandbox died mid-run in ${p.runIds.length} runs. Check container sizing first (the \`deploy/cloudflare-*/wrangler.template.jsonc\` comments: the 1 GiB \`basic\` tier died running vitest; thread sandboxes are \`standard-4\` \u2014 4 vCPU / 12 GiB / 20 GB, the platform's largest \u2014 and residents a custom type of the same size), then the memory footprint of the failing command, then the sandbox/resident Worker logs (\`deploy/bin/cf-logs\`) around the affected runs.`;
|
|
@@ -14559,6 +14555,7 @@ var init_frictionProposals = __esm({
|
|
|
14559
14555
|
"../../src/core/frictionProposals.ts"() {
|
|
14560
14556
|
"use strict";
|
|
14561
14557
|
init_runFriction();
|
|
14558
|
+
init_registry();
|
|
14562
14559
|
init_formatDuration();
|
|
14563
14560
|
DEFAULT_MIN_RUNS = 2;
|
|
14564
14561
|
MAX_EXAMPLES = 5;
|
|
@@ -26248,21 +26245,33 @@ async function runLoop(opts, now, note, emit, agentSpan) {
|
|
|
26248
26245
|
return await finishSandboxDead(complete2, opts, messages, system, diagnosis);
|
|
26249
26246
|
}
|
|
26250
26247
|
const wasTimeout = now() >= deadline;
|
|
26248
|
+
const elapsedMs = opts.agent.maxMinutes * 6e4 - (deadline - now());
|
|
26249
|
+
const pace = `${turn} model turn${turn === 1 ? "" : "s"} in ${elapsedMinutes(elapsedMs)}`;
|
|
26251
26250
|
note(
|
|
26252
26251
|
wasTimeout ? "time_budget_exhausted" : "turn_budget_exhausted",
|
|
26253
|
-
|
|
26252
|
+
wasTimeout ? "time budget exhausted \u2014 writing up findings so far" : `turn guard fired: ${pace}, a pace that looks like a loop \u2014 writing up findings so far`
|
|
26254
26253
|
);
|
|
26254
|
+
const writeUp = "Write your final answer now from what you have learned so far: report your findings/results to date, then state plainly which parts of the task you did not get to and what a follow-up (in this thread, to reuse this workspace) should focus on.";
|
|
26255
26255
|
const text = await runFinale(
|
|
26256
26256
|
complete2,
|
|
26257
26257
|
opts,
|
|
26258
26258
|
messages,
|
|
26259
26259
|
system,
|
|
26260
|
-
|
|
26260
|
+
wasTimeout ? `You have reached the time budget and can make no more tool calls. ${writeUp}` : `You have hit the run's turn guard \u2014 ${pace}, a pace that looks like a loop \u2014 and can make no more tool calls. ${writeUp}`
|
|
26261
26261
|
);
|
|
26262
|
-
|
|
26263
|
-
|
|
26262
|
+
if (wasTimeout) {
|
|
26263
|
+
return text ? `\u26A0\uFE0F _Hit the ${opts.agent.maxMinutes}-minute budget before finishing \u2014 findings so far:_
|
|
26264
26264
|
|
|
26265
|
-
${text}` : `Stopped at the ${
|
|
26265
|
+
${text}` : `Stopped at the ${opts.agent.maxMinutes}-minute budget without finishing. Partial work may exist in the workspace \u2014 narrow the task and try again.`;
|
|
26266
|
+
}
|
|
26267
|
+
return text ? `\u26A0\uFE0F _Stopped after ${pace} \u2014 that pace looks like a loop; findings so far:_
|
|
26268
|
+
|
|
26269
|
+
${text}` : `Stopped after ${pace} \u2014 that pace looks like a loop \u2014 without finishing. Partial work may exist in the workspace \u2014 look for a retry loop in the run's events before trying again.`;
|
|
26270
|
+
}
|
|
26271
|
+
function elapsedMinutes(ms) {
|
|
26272
|
+
const minutes = Math.round(ms / 6e4);
|
|
26273
|
+
if (minutes < 1) return "under a minute";
|
|
26274
|
+
return `${minutes} minute${minutes === 1 ? "" : "s"}`;
|
|
26266
26275
|
}
|
|
26267
26276
|
async function runFinale(complete2, opts, messages, system, instruction) {
|
|
26268
26277
|
messages.push({ role: "user", content: [{ type: "text", text: instruction }] });
|
|
@@ -37004,7 +37013,7 @@ async function handleAdmitted(door, req, deps) {
|
|
|
37004
37013
|
grant2.publish({
|
|
37005
37014
|
type: "run_note",
|
|
37006
37015
|
kind: "turn_budget_exhausted",
|
|
37007
|
-
summary: `model proxy refused a call past the ${turn.maxTurns}-turn
|
|
37016
|
+
summary: `model proxy refused a call past the run's ${turn.maxTurns}-turn guard (${used})`,
|
|
37008
37017
|
at: deps.clock()
|
|
37009
37018
|
});
|
|
37010
37019
|
log(`[model-proxy] 403 turn_budget_exhausted run=${grant2.runId} turns=${turn.turns}/${turn.maxTurns}`);
|
|
@@ -37012,7 +37021,7 @@ async function handleAdmitted(door, req, deps) {
|
|
|
37012
37021
|
shape,
|
|
37013
37022
|
403,
|
|
37014
37023
|
"turn_budget_exhausted",
|
|
37015
|
-
`the run
|
|
37024
|
+
`the run is past its ${turn.maxTurns}-turn guard (${used})`
|
|
37016
37025
|
);
|
|
37017
37026
|
}
|
|
37018
37027
|
const payload = JSON.stringify(pinRequest(shape, body, grant2));
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@coreplane/switchboard",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.209.0",
|
|
4
4
|
"description": "Mention it in Slack and an agent reviews the PR, ships the fix, or answers the question — on the model you choose, with its tools running where you decide.",
|
|
5
5
|
"license": "Apache-2.0",
|
|
6
6
|
"homepage": "https://openswitchboard.dev",
|