switchroom 0.19.16 → 0.19.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/run-hook.sh +148 -0
- package/bin/workspace-dynamic-hook.sh +147 -38
- package/dist/agent-scheduler/index.js +11 -3
- package/dist/auth-broker/index.js +29 -4
- package/dist/cli/notion-write-pretool.mjs +11 -3
- package/dist/cli/switchroom.js +8307 -7620
- package/dist/host-control/main.js +626 -36
- package/dist/vault/approvals/kernel-server.js +30 -5
- package/dist/vault/broker/server.js +71 -18
- package/package.json +3 -2
- package/profiles/_base/start.sh.hbs +8 -4
- package/profiles/coding/CLAUDE.md.hbs +1 -1
- package/profiles/default/CLAUDE.md.hbs +3 -3
- package/profiles/executive-assistant/CLAUDE.md.hbs +1 -1
- package/profiles/health-coach/CLAUDE.md.hbs +1 -1
- package/skills/mental-model-curator/SKILL.md +8 -6
- package/telegram-plugin/bridge/bridge.ts +11 -19
- package/telegram-plugin/bridge/mcp-instructions.ts +87 -0
- package/telegram-plugin/dist/bridge/bridge.js +15 -20
- package/telegram-plugin/dist/gateway/gateway.js +763 -373
- package/telegram-plugin/dist/server.js +19 -20
- package/telegram-plugin/gateway/boot-card.ts +5 -1
- package/telegram-plugin/gateway/boot-probes.ts +113 -0
- package/telegram-plugin/gateway/config-approval-handler.test.ts +54 -0
- package/telegram-plugin/gateway/config-approval-handler.ts +16 -1
- package/telegram-plugin/gateway/disconnect-flush.ts +17 -0
- package/telegram-plugin/gateway/gateway.ts +43 -1
- package/telegram-plugin/gateway/handback-preturn-signal.ts +61 -7
- package/telegram-plugin/gateway/ipc-protocol.ts +5 -0
- package/telegram-plugin/gateway/ipc-server.ts +13 -0
- package/telegram-plugin/gateway/liveness-wiring.ts +125 -5
- package/telegram-plugin/gateway/obligation-ledger.ts +84 -4
- package/telegram-plugin/gateway/resume-inbound-builder.ts +13 -4
- package/telegram-plugin/gateway/stream-render.ts +24 -5
- package/telegram-plugin/hooks/secret-guard-pretool.mjs +249 -76
- package/telegram-plugin/registry/turns-schema.test.ts +8 -3
- package/telegram-plugin/registry/turns-schema.ts +40 -12
- package/telegram-plugin/runtime-metrics.ts +14 -0
- package/telegram-plugin/silence-poke.ts +138 -0
- package/telegram-plugin/tests/boot-probe-drift.test.ts +152 -0
- package/telegram-plugin/tests/gateway-disconnect-flush.test.ts +32 -0
- package/telegram-plugin/tests/handback-preturn-signal.test.ts +62 -0
- package/telegram-plugin/tests/helpers/liveness-wiring-fixture.ts +178 -0
- package/telegram-plugin/tests/ipc-server-validate-config-approval.test.ts +95 -0
- package/telegram-plugin/tests/mcp-instructions-budget.test.ts +184 -0
- package/telegram-plugin/tests/multitopic-routing-wiring.test.ts +22 -2
- package/telegram-plugin/tests/obligation-determinism.test.ts +114 -3
- package/telegram-plugin/tests/obligation-ledger.test.ts +310 -0
- package/telegram-plugin/tests/registry-turns.test.ts +13 -0
- package/telegram-plugin/tests/resume-inbound-builder.test.ts +15 -0
- package/telegram-plugin/tests/secret-guard-pretool.test.ts +347 -16
- package/telegram-plugin/tests/silence-poke-orphan-reap.test.ts +392 -0
- package/telegram-plugin/tests/silence-poke-teardown-notice.test.ts +301 -0
- package/telegram-plugin/tests/stream-render-golden.test.ts +103 -1
- package/telegram-plugin/tests/tts-normalize.test.ts +43 -0
- package/telegram-plugin/tests/voice-normalize-text.test.ts +212 -3
- package/telegram-plugin/tts-normalize.ts +6 -4
- package/telegram-plugin/voice-normalize-text.ts +168 -11
- package/vendor/hindsight-memory/CHANGELOG.md +73 -0
- package/vendor/hindsight-memory/scripts/lib/config.py +8 -3
- package/vendor/hindsight-memory/scripts/lib/directives.py +62 -4
- package/vendor/hindsight-memory/scripts/recall.py +257 -12
- package/vendor/hindsight-memory/scripts/retain.py +12 -6
- package/vendor/hindsight-memory/scripts/tests/test_directives.py +80 -9
- package/vendor/hindsight-memory/scripts/tests/test_recall_integration.py +362 -18
- package/vendor/hindsight-memory/settings.json +1 -1
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Validation contract for the `request_config_approval` IPC message —
|
|
3
|
+
* hostd's approval-card request — with focus on the KEN-129 optional
|
|
4
|
+
* `title` header override.
|
|
5
|
+
*
|
|
6
|
+
* Mixed-version safety pins:
|
|
7
|
+
* - `title` absent must validate (old hostd → new gateway).
|
|
8
|
+
* - a valid `title` must validate (new hostd → new gateway).
|
|
9
|
+
* - malformed titles (empty / oversize / non-string) are rejected —
|
|
10
|
+
* the validator is the security boundary on the client→gateway
|
|
11
|
+
* direction.
|
|
12
|
+
*
|
|
13
|
+
* (The reverse direction — new hostd → OLD gateway — is safe because
|
|
14
|
+
* the old validator checks only known fields' types and ignores extra
|
|
15
|
+
* keys; pinned by the "extra unknown fields" case below matching that
|
|
16
|
+
* permissive behaviour on the current validator too.)
|
|
17
|
+
*
|
|
18
|
+
* Companion to ipc-server-validate-{operator,inject-inbound}.test.ts.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import { describe, it, expect } from 'vitest'
|
|
22
|
+
import { validateClientMessage } from '../gateway/ipc-server.js'
|
|
23
|
+
|
|
24
|
+
function base() {
|
|
25
|
+
return {
|
|
26
|
+
type: 'request_config_approval' as const,
|
|
27
|
+
requestId: 'relnotify-abc123',
|
|
28
|
+
agentName: 'klanker',
|
|
29
|
+
reason: 'fleet is behind',
|
|
30
|
+
unifiedDiff: 'plan text',
|
|
31
|
+
timeoutMs: 3_600_000,
|
|
32
|
+
}
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
describe('validateClientMessage — request_config_approval', () => {
|
|
36
|
+
it('accepts the base message without a title (old-hostd compat)', () => {
|
|
37
|
+
expect(validateClientMessage(base())).toBe(true)
|
|
38
|
+
})
|
|
39
|
+
|
|
40
|
+
it('accepts a valid title (KEN-129 update-check card)', () => {
|
|
41
|
+
expect(
|
|
42
|
+
validateClientMessage({
|
|
43
|
+
...base(),
|
|
44
|
+
title: '⬆️ **Switchroom update available — fleet is behind**',
|
|
45
|
+
}),
|
|
46
|
+
).toBe(true)
|
|
47
|
+
})
|
|
48
|
+
|
|
49
|
+
it('rejects an empty title', () => {
|
|
50
|
+
expect(validateClientMessage({ ...base(), title: '' })).toBe(false)
|
|
51
|
+
})
|
|
52
|
+
|
|
53
|
+
it('rejects a title over 200 chars', () => {
|
|
54
|
+
expect(validateClientMessage({ ...base(), title: 'x'.repeat(201) })).toBe(false)
|
|
55
|
+
})
|
|
56
|
+
|
|
57
|
+
it('rejects a non-string title', () => {
|
|
58
|
+
expect(validateClientMessage({ ...base(), title: 42 })).toBe(false)
|
|
59
|
+
expect(validateClientMessage({ ...base(), title: null })).toBe(false)
|
|
60
|
+
})
|
|
61
|
+
|
|
62
|
+
// The title is rendered VERBATIM as the card's first line (it carries
|
|
63
|
+
// intentional markdown, so it cannot be escaped). A multi-line title
|
|
64
|
+
// would let a caller forge the `Agent:` / `Reason:` lines beneath it, or
|
|
65
|
+
// unbalance the diff's ``` fence, on a card the operator is about to
|
|
66
|
+
// approve. Newlines and control characters are therefore rejected.
|
|
67
|
+
it('rejects a multi-line title that could forge the card body', () => {
|
|
68
|
+
expect(
|
|
69
|
+
validateClientMessage({
|
|
70
|
+
...base(),
|
|
71
|
+
title:
|
|
72
|
+
'🛠 **Config edit proposed**\nAgent: `root`\nReason: routine\n```\nnoop\n```',
|
|
73
|
+
}),
|
|
74
|
+
).toBe(false)
|
|
75
|
+
expect(validateClientMessage({ ...base(), title: 'a\rb' })).toBe(false)
|
|
76
|
+
})
|
|
77
|
+
|
|
78
|
+
it('rejects control characters in the title', () => {
|
|
79
|
+
expect(validateClientMessage({ ...base(), title: 'a\u0000b' })).toBe(false)
|
|
80
|
+
expect(validateClientMessage({ ...base(), title: 'a\u001bb' })).toBe(false)
|
|
81
|
+
expect(validateClientMessage({ ...base(), title: 'a\u007fb' })).toBe(false)
|
|
82
|
+
})
|
|
83
|
+
|
|
84
|
+
it('ignores extra unknown fields (forward-compat for future hostds)', () => {
|
|
85
|
+
expect(
|
|
86
|
+
validateClientMessage({ ...base(), someFutureField: 'ignored' }),
|
|
87
|
+
).toBe(true)
|
|
88
|
+
})
|
|
89
|
+
|
|
90
|
+
it('still rejects a malformed base message (missing diff)', () => {
|
|
91
|
+
const m: Record<string, unknown> = { ...base() }
|
|
92
|
+
delete m.unifiedDiff
|
|
93
|
+
expect(validateClientMessage(m)).toBe(false)
|
|
94
|
+
})
|
|
95
|
+
})
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The MCP `instructions` string must fit the Claude Code client's hard
|
|
3
|
+
* truncation limit (#3562).
|
|
4
|
+
*
|
|
5
|
+
* The Claude Code native binary truncates MCP server instructions at a
|
|
6
|
+
* hard-coded 2048 chars, silently and mid-word:
|
|
7
|
+
*
|
|
8
|
+
* if (I && I.length > LB) R = ma(I, LB) + "… [truncated]" // LB = 2048
|
|
9
|
+
*
|
|
10
|
+
* Nothing errors and nothing surfaces to the agent — the server starts, the
|
|
11
|
+
* tools work, and the tail of the string simply never reaches the model. That
|
|
12
|
+
* is how the telegram bridge shipped a 4645-char instructions string for
|
|
13
|
+
* months with 56% of it discarded, including the /telegram:access
|
|
14
|
+
* prompt-injection defence.
|
|
15
|
+
*
|
|
16
|
+
* These tests are the deterministic backstop. `scripts/check-mcp-instructions-budget.mjs`
|
|
17
|
+
* (in `npm run lint`) is the twin: it resolves the module the bridge actually
|
|
18
|
+
* imports and measures the same runtime value that the SDK will hand to the
|
|
19
|
+
* client.
|
|
20
|
+
*/
|
|
21
|
+
import { describe, it, expect } from "vitest";
|
|
22
|
+
import { readFileSync } from "node:fs";
|
|
23
|
+
import { resolve } from "node:path";
|
|
24
|
+
import {
|
|
25
|
+
MCP_INSTRUCTIONS,
|
|
26
|
+
MCP_INSTRUCTIONS_BUDGET,
|
|
27
|
+
MCP_INSTRUCTIONS_LIMIT,
|
|
28
|
+
} from "../bridge/mcp-instructions.js";
|
|
29
|
+
|
|
30
|
+
describe("MCP instructions budget", () => {
|
|
31
|
+
it("pins the client's hard limit at 2048 (not ours to change)", () => {
|
|
32
|
+
// Sourced from the minified constant `LB` in @anthropic-ai/claude-code
|
|
33
|
+
// v2.1.219. If a future client version changes this, update it here
|
|
34
|
+
// deliberately — do not raise it to make a long string pass.
|
|
35
|
+
expect(MCP_INSTRUCTIONS_LIMIT).toBe(2048);
|
|
36
|
+
});
|
|
37
|
+
|
|
38
|
+
it("keeps the authored budget strictly below the client's hard limit", () => {
|
|
39
|
+
expect(MCP_INSTRUCTIONS_BUDGET).toBeLessThan(MCP_INSTRUCTIONS_LIMIT);
|
|
40
|
+
});
|
|
41
|
+
|
|
42
|
+
it("leaves a real safety margin below the hard limit", () => {
|
|
43
|
+
// The budget exists to fail an edit BEFORE the client silently cuts it.
|
|
44
|
+
// Raising it to within a handful of chars of 2048 would leave no room to
|
|
45
|
+
// notice, so the margin is asserted rather than trusted to review.
|
|
46
|
+
const margin = MCP_INSTRUCTIONS_LIMIT - MCP_INSTRUCTIONS_BUDGET;
|
|
47
|
+
expect(
|
|
48
|
+
margin,
|
|
49
|
+
`budget ${MCP_INSTRUCTIONS_BUDGET} is only ${margin} chars below the ` +
|
|
50
|
+
`${MCP_INSTRUCTIONS_LIMIT} hard limit — too close to act as an early ` +
|
|
51
|
+
`warning. Lower the budget instead of raising it.`,
|
|
52
|
+
).toBeGreaterThanOrEqual(50);
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
it("leaves usable authoring headroom under the budget", () => {
|
|
56
|
+
// The counterpart to the margin above: a budget pinned two chars above the
|
|
57
|
+
// current string is a nuisance rail, and nuisance rails get "fixed" by
|
|
58
|
+
// deleting content — which is precisely what caused #3562. If this fails,
|
|
59
|
+
// MOVE mechanical detail into a tool `description` (those are not capped)
|
|
60
|
+
// rather than raising the budget toward the limit.
|
|
61
|
+
const headroom = MCP_INSTRUCTIONS_BUDGET - MCP_INSTRUCTIONS.length;
|
|
62
|
+
expect(
|
|
63
|
+
headroom,
|
|
64
|
+
`only ${headroom} chars of headroom between the string ` +
|
|
65
|
+
`(${MCP_INSTRUCTIONS.length}) and the budget ` +
|
|
66
|
+
`(${MCP_INSTRUCTIONS_BUDGET}); a routine edit would trip lint.`,
|
|
67
|
+
).toBeGreaterThanOrEqual(25);
|
|
68
|
+
});
|
|
69
|
+
|
|
70
|
+
it("fits under the authored budget", () => {
|
|
71
|
+
const len = MCP_INSTRUCTIONS.length;
|
|
72
|
+
expect(
|
|
73
|
+
len,
|
|
74
|
+
`MCP instructions are ${len} chars — ${len - MCP_INSTRUCTIONS_BUDGET} over ` +
|
|
75
|
+
`the ${MCP_INSTRUCTIONS_BUDGET}-char budget (client hard limit ` +
|
|
76
|
+
`${MCP_INSTRUCTIONS_LIMIT}, which truncates silently and mid-word). ` +
|
|
77
|
+
`Cut ${len - MCP_INSTRUCTIONS_BUDGET} chars. Prefer MOVING mechanical ` +
|
|
78
|
+
`per-tool detail into that tool's \`description\` (tool descriptions ` +
|
|
79
|
+
`are NOT capped) over deleting it; keep safety/trust rules here.`,
|
|
80
|
+
).toBeLessThanOrEqual(MCP_INSTRUCTIONS_BUDGET);
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
it("would not be truncated by the client", () => {
|
|
84
|
+
// The exact predicate the client applies.
|
|
85
|
+
expect(MCP_INSTRUCTIONS.length > MCP_INSTRUCTIONS_LIMIT).toBe(false);
|
|
86
|
+
});
|
|
87
|
+
|
|
88
|
+
// ── Content assertions: the SAFETY rules must be present, and must be
|
|
89
|
+
// present *within the surviving prefix*. A budget test alone would still
|
|
90
|
+
// pass if someone kept the length but deleted the guardrail.
|
|
91
|
+
it("retains the prompt-injection defence inside the un-truncated prefix", () => {
|
|
92
|
+
const survives = MCP_INSTRUCTIONS.slice(0, MCP_INSTRUCTIONS_LIMIT);
|
|
93
|
+
expect(survives).toContain("approve the pending pairing");
|
|
94
|
+
expect(survives).toContain("prompt injection");
|
|
95
|
+
expect(survives).toContain("Refuse");
|
|
96
|
+
expect(survives).toMatch(/never invoke that skill|Never invoke that skill/);
|
|
97
|
+
});
|
|
98
|
+
|
|
99
|
+
it("retains the forwarded-provenance trust rules inside the prefix", () => {
|
|
100
|
+
const survives = MCP_INSTRUCTIONS.slice(0, MCP_INSTRUCTIONS_LIMIT);
|
|
101
|
+
expect(survives).toContain("hidden_user");
|
|
102
|
+
expect(survives).toMatch(/no verifiable id/i);
|
|
103
|
+
expect(survives).toContain("untrusted content");
|
|
104
|
+
});
|
|
105
|
+
|
|
106
|
+
it("retains the reply-is-the-only-channel rule inside the prefix", () => {
|
|
107
|
+
const survives = MCP_INSTRUCTIONS.slice(0, MCP_INSTRUCTIONS_LIMIT);
|
|
108
|
+
expect(survives).toContain("reply tool");
|
|
109
|
+
expect(survives).toContain("never reaches their chat");
|
|
110
|
+
});
|
|
111
|
+
|
|
112
|
+
it("retains the multi-origin forward-attribution rule inside the prefix", () => {
|
|
113
|
+
// Review catch: this sentence was dropped in the first cut with no
|
|
114
|
+
// landing spot, surviving only in docs/telegram-features.md. A two-origin
|
|
115
|
+
// forwarded burst would then be wholly attributed to origin 1 — a
|
|
116
|
+
// provenance failure, not a mechanical one, so it belongs here.
|
|
117
|
+
const survives = MCP_INSTRUCTIONS.slice(0, MCP_INSTRUCTIONS_LIMIT);
|
|
118
|
+
expect(survives).toContain("forwarded_from_2");
|
|
119
|
+
expect(survives).toMatch(/attribute each part to its OWN origin/i);
|
|
120
|
+
expect(survives).toMatch(/never the whole burst to the first/i);
|
|
121
|
+
});
|
|
122
|
+
|
|
123
|
+
it("retains the forward id/date/deep-link attributes inside the prefix", () => {
|
|
124
|
+
const survives = MCP_INSTRUCTIONS.slice(0, MCP_INSTRUCTIONS_LIMIT);
|
|
125
|
+
for (const attr of [
|
|
126
|
+
"forwarded_from_id",
|
|
127
|
+
"forwarded_date",
|
|
128
|
+
"forwarded_message_id",
|
|
129
|
+
"t.me/<channel>/<id>",
|
|
130
|
+
"attachment_count",
|
|
131
|
+
]) {
|
|
132
|
+
expect(survives).toContain(attr);
|
|
133
|
+
}
|
|
134
|
+
});
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
/**
|
|
138
|
+
* The PR's thesis is "MOVED, not deleted" — mechanical detail was cut from the
|
|
139
|
+
* instructions only because it landed in a tool `description`, which is not
|
|
140
|
+
* subject to the 2048-char cap. Nothing tested that the landing actually
|
|
141
|
+
* happened, so deleting the destination text passed the whole suite.
|
|
142
|
+
*
|
|
143
|
+
* These assertions pin the destinations. Comments are stripped first: a
|
|
144
|
+
* matching comment in bridge.ts must NOT be able to satisfy an assertion about
|
|
145
|
+
* text the AGENT reads.
|
|
146
|
+
*/
|
|
147
|
+
describe("moved instruction content landed in tool descriptions", () => {
|
|
148
|
+
const raw = readFileSync(
|
|
149
|
+
resolve(__dirname, "..", "bridge", "bridge.ts"),
|
|
150
|
+
"utf-8",
|
|
151
|
+
);
|
|
152
|
+
// Strip block and line comments so only real agent-facing strings match.
|
|
153
|
+
const bridgeCode = raw
|
|
154
|
+
.replace(/\/\*[\s\S]*?\*\//g, "")
|
|
155
|
+
.replace(/(^|[^:])\/\/.*$/gm, "$1");
|
|
156
|
+
|
|
157
|
+
it("forum-topic routing landed in the reply description", () => {
|
|
158
|
+
expect(bridgeCode).toContain("FORUM TOPICS:");
|
|
159
|
+
expect(bridgeCode).toMatch(/do NOT pass message_thread_id on a normal reply/);
|
|
160
|
+
expect(bridgeCode).toContain("origin_turn_id");
|
|
161
|
+
});
|
|
162
|
+
|
|
163
|
+
it("format modes landed in the reply description", () => {
|
|
164
|
+
expect(bridgeCode).toContain("FORMAT:");
|
|
165
|
+
expect(bridgeCode).toMatch(/default format is "html"/);
|
|
166
|
+
expect(bridgeCode).toContain("markdownv2");
|
|
167
|
+
});
|
|
168
|
+
|
|
169
|
+
it("history-buffer rationale landed in get_recent_messages", () => {
|
|
170
|
+
expect(bridgeCode).toMatch(/no history endpoint/);
|
|
171
|
+
expect(bridgeCode).toMatch(/buffer survives restarts/);
|
|
172
|
+
});
|
|
173
|
+
|
|
174
|
+
it("the comment-stripping itself works (guards the assertions above)", () => {
|
|
175
|
+
// If stripping silently no-ops, every assertion in this block becomes
|
|
176
|
+
// satisfiable by a comment. Prove it removes both comment forms.
|
|
177
|
+
const sample = 'const a = 1 // FORUM TOPICS:\n/* FORMAT: */ const b = 2';
|
|
178
|
+
const stripped = sample
|
|
179
|
+
.replace(/\/\*[\s\S]*?\*\//g, "")
|
|
180
|
+
.replace(/(^|[^:])\/\/.*$/gm, "$1");
|
|
181
|
+
expect(stripped).not.toContain("FORUM TOPICS:");
|
|
182
|
+
expect(stripped).not.toContain("FORMAT:");
|
|
183
|
+
});
|
|
184
|
+
});
|
|
@@ -11,6 +11,7 @@
|
|
|
11
11
|
import { describe, it, expect } from 'vitest'
|
|
12
12
|
import { readFileSync } from 'node:fs'
|
|
13
13
|
import { resolve } from 'node:path'
|
|
14
|
+
import { MCP_INSTRUCTIONS } from '../bridge/mcp-instructions.js'
|
|
14
15
|
|
|
15
16
|
// #2996 P2: executeReply's body moved VERBATIM to outbound-send-path.ts
|
|
16
17
|
// (`sendReply`); the reply-path routing assertions read the window there.
|
|
@@ -39,6 +40,17 @@ const streamSrc = readFileSync(
|
|
|
39
40
|
'utf-8',
|
|
40
41
|
)
|
|
41
42
|
const gatewayAndStreamSrc = gatewaySrc + '\n' + streamSrc
|
|
43
|
+
// #3562 — the MCP `instructions` string was extracted out of bridge.ts into
|
|
44
|
+
// its own module (it must fit the Claude Code client's 2048-char truncation
|
|
45
|
+
// limit, so it is now length-guarded independently). bridge.ts still owns the
|
|
46
|
+
// per-tool `description` strings.
|
|
47
|
+
//
|
|
48
|
+
// Deliberately NOT concatenated with the instructions module: an earlier
|
|
49
|
+
// revision merged both files into one blob, which meant an assertion could no
|
|
50
|
+
// longer tell WHICH file carried a phrase, and a matching COMMENT satisfied it
|
|
51
|
+
// just as well as real agent-facing text. Instructions text is asserted
|
|
52
|
+
// against the imported runtime constant instead (see below) — a comment cannot
|
|
53
|
+
// satisfy that, by construction.
|
|
42
54
|
const bridgeSrc = readFileSync(
|
|
43
55
|
resolve(__dirname, '..', 'bridge', 'bridge.ts'),
|
|
44
56
|
'utf-8',
|
|
@@ -151,8 +163,16 @@ describe('component 4 — per-turn topic framing', () => {
|
|
|
151
163
|
})
|
|
152
164
|
|
|
153
165
|
it('the bridge instructions frame each channel message as the current topic', () => {
|
|
154
|
-
|
|
155
|
-
|
|
166
|
+
// Asserted against the IMPORTED runtime constant, not file text: this is
|
|
167
|
+
// the exact string handed to the MCP client, so a comment (or the phrase
|
|
168
|
+
// living in some other file) cannot satisfy it. Wording was tightened in
|
|
169
|
+
// #3562 to fit the 2048-char budget; the invariant is unchanged — the
|
|
170
|
+
// instructions must scope the agent to the current message's topic and
|
|
171
|
+
// forbid answering a queued other-topic message in the same turn.
|
|
172
|
+
expect(MCP_INSTRUCTIONS).toMatch(/answer only the current message/i)
|
|
173
|
+
expect(MCP_INSTRUCTIONS).toMatch(
|
|
174
|
+
/do not also answer a pending message from another topic/i,
|
|
175
|
+
)
|
|
156
176
|
})
|
|
157
177
|
})
|
|
158
178
|
|
|
@@ -95,6 +95,16 @@ function runSchedule(
|
|
|
95
95
|
graceMs = 0,
|
|
96
96
|
bgGraceMs = 0,
|
|
97
97
|
bgAlwaysActive = false,
|
|
98
|
+
// #3550 — per-represent grace. Exercised so the determinism proof actually
|
|
99
|
+
// reaches `representGraceStillProtecting`; without this the fuzz never passes
|
|
100
|
+
// representGraceMs and the branch is unvisited.
|
|
101
|
+
representGraceMs = 0,
|
|
102
|
+
// #3550 discriminator knob. When FALSE, a re-present turn ends WITHOUT ever
|
|
103
|
+
// stamping noteTurnEnded — the "re-present still in flight" shape, in which
|
|
104
|
+
// the early-out provably cannot fire and the full represent window is held.
|
|
105
|
+
// Running the same seed both ways is what turns the represent-grace case from
|
|
106
|
+
// a code-path exercise into a test that FAILS if the early-out is removed.
|
|
107
|
+
stampTurnEndAfterRepresent = true,
|
|
98
108
|
): Sim {
|
|
99
109
|
const PATH = "/state/agent/telegram/obligations.json";
|
|
100
110
|
const store = memStore();
|
|
@@ -107,6 +117,7 @@ function runSchedule(
|
|
|
107
117
|
// terminal (no livelock).
|
|
108
118
|
let clock = 1_000_000;
|
|
109
119
|
const SWEEP_TICK = 5_000;
|
|
120
|
+
const TURN_DURATION = 1_000; // virtual ms a turn occupies before it ends
|
|
110
121
|
const r = rng(seed);
|
|
111
122
|
|
|
112
123
|
const pending = [...msgs]; // not yet received
|
|
@@ -128,9 +139,20 @@ function runSchedule(
|
|
|
128
139
|
const had = (turnsHad.get(id) ?? 0);
|
|
129
140
|
const attemptIndex = had; // 0-based
|
|
130
141
|
turnsHad.set(id, had + 1);
|
|
142
|
+
const isRepresentTurn = attemptIndex > 0;
|
|
131
143
|
if (byId.get(id)!.answerOnAttempt === attemptIndex) {
|
|
132
144
|
close(id, "answered");
|
|
133
|
-
} else if (
|
|
145
|
+
} else if (isRepresentTurn && !stampTurnEndAfterRepresent) {
|
|
146
|
+
// In-flight shape: the turn produced nothing observable, so nothing is
|
|
147
|
+
// stamped and the per-represent window is held for its full duration.
|
|
148
|
+
// (Deliberately NOT advancing the clock either — the two runs must differ
|
|
149
|
+
// only in whether the early-out can fire.)
|
|
150
|
+
} else if ((graceMs > 0 || representGraceMs > 0) && ledger.isOpen(id)) {
|
|
151
|
+
// A turn takes real time: its end is strictly LATER than the represent
|
|
152
|
+
// that triggered it. Without this the sim stamps lastTurnEndedAt equal to
|
|
153
|
+
// lastRepresentedAt and the #3550 early-out (which requires strictly
|
|
154
|
+
// greater) is never reached — the branch would go unproven.
|
|
155
|
+
clock += TURN_DURATION;
|
|
134
156
|
ledger.noteTurnEnded(id, clock);
|
|
135
157
|
}
|
|
136
158
|
};
|
|
@@ -163,12 +185,13 @@ function runSchedule(
|
|
|
163
185
|
deliverTurn(m.id); // original turn (attempt 0)
|
|
164
186
|
} else if (open) {
|
|
165
187
|
const decision =
|
|
166
|
-
graceMs > 0 || bgGraceMs > 0
|
|
188
|
+
graceMs > 0 || bgGraceMs > 0 || representGraceMs > 0
|
|
167
189
|
? ledger.decideAtIdle({
|
|
168
190
|
now: clock,
|
|
169
191
|
graceMs,
|
|
170
192
|
backgroundWorkActive: bgGraceMs > 0 && bgAlwaysActive,
|
|
171
193
|
backgroundGraceMs: bgGraceMs,
|
|
194
|
+
representGraceMs,
|
|
172
195
|
})
|
|
173
196
|
: ledger.decideAtIdle();
|
|
174
197
|
if (decision.action === "none") {
|
|
@@ -181,7 +204,10 @@ function runSchedule(
|
|
|
181
204
|
// INVARIANT (no double-ask): a terminated obligation must never resurface.
|
|
182
205
|
expect(terminals.has(o.originTurnId)).toBe(false);
|
|
183
206
|
if (decision.action === "represent") {
|
|
184
|
-
|
|
207
|
+
// Stamp on the SAME virtual clock as noteTurnEnded, otherwise the
|
|
208
|
+
// per-represent window is measured against a wall-clock `Date.now()`
|
|
209
|
+
// default and never interacts with the grace under test.
|
|
210
|
+
ledger.markRepresented(o.originTurnId, clock);
|
|
185
211
|
deliverTurn(o.originTurnId); // the re-present turn
|
|
186
212
|
} else if (decision.action === "escalate") {
|
|
187
213
|
if (ESC_IN_FLIGHT.has(o.originTurnId)) continue;
|
|
@@ -340,6 +366,91 @@ describe("obligation determinism — every inbound reaches a terminal, no silent
|
|
|
340
366
|
}
|
|
341
367
|
});
|
|
342
368
|
|
|
369
|
+
it("holds across 3000 schedules WITH the per-represent grace on, and the #3550 early-out DISCRIMINATES (same terminals, strictly fewer sweeps)", () => {
|
|
370
|
+
// Honest framing of what this case is and is not.
|
|
371
|
+
//
|
|
372
|
+
// The terminal assertions below are NOT a discriminator for the #3550 diff:
|
|
373
|
+
// a terminal is a function of `answerOnAttempt` / `escalateFailsFor` alone,
|
|
374
|
+
// and the worst-case schedule settles in ~72 sweep steps against a CAP of
|
|
375
|
+
// 10_000. Revert the early-out and every terminal — and the step budget —
|
|
376
|
+
// still holds. Those assertions are a NO-LIVELOCK / NO-LOSS guard on the
|
|
377
|
+
// represent-grace path, nothing more, and are labelled as such.
|
|
378
|
+
//
|
|
379
|
+
// The discriminator is the paired run. The SAME seed is run twice, differing
|
|
380
|
+
// in exactly one thing: whether a re-present turn ends (stamping
|
|
381
|
+
// noteTurnEnded) or stays in flight. The early-out fires only in the first,
|
|
382
|
+
// so:
|
|
383
|
+
// - terminals must be IDENTICAL — the early-out changes WHEN the ladder
|
|
384
|
+
// advances, never WHERE it lands; and
|
|
385
|
+
// - the ended-turn run must take STRICTLY FEWER sweep steps — which is
|
|
386
|
+
// the early-out actually firing and shortening rungs. Delete the
|
|
387
|
+
// early-out and both runs hold the full 120s window, the step counts
|
|
388
|
+
// become equal, and this assertion fails.
|
|
389
|
+
const ANSWER = [0, 1, 2, 3, 99];
|
|
390
|
+
const ESCFAIL = [0, 1, 2, 3, 5];
|
|
391
|
+
const GRACE_MS = 45_000; // trailing-answer grace, still armed
|
|
392
|
+
const REPR_GRACE_MS = 120_000; // mirrors OBLIGATION_REPRESENT_GRACE_MS default
|
|
393
|
+
let discriminated = 0; // schedules the early-out demonstrably shortened
|
|
394
|
+
let totalRetired = 0;
|
|
395
|
+
let totalInFlight = 0;
|
|
396
|
+
for (let seed = 1; seed <= 3000; seed++) {
|
|
397
|
+
const r = rng(seed * 7919);
|
|
398
|
+
const n = 1 + Math.floor(r() * 5);
|
|
399
|
+
const msgs: Msg[] = [];
|
|
400
|
+
for (let i = 0; i < n; i++) {
|
|
401
|
+
const msgId = seed * 100 + i;
|
|
402
|
+
msgs.push({
|
|
403
|
+
id: `c:3#${msgId}`,
|
|
404
|
+
msgId,
|
|
405
|
+
answerOnAttempt: pick(ANSWER, r),
|
|
406
|
+
escalateFailsFor: pick(ESCFAIL, r),
|
|
407
|
+
});
|
|
408
|
+
}
|
|
409
|
+
const retired = runSchedule(msgs, seed * 104729, GRACE_MS, 0, false, REPR_GRACE_MS, true);
|
|
410
|
+
const inFlight = runSchedule(msgs, seed * 104729, GRACE_MS, 0, false, REPR_GRACE_MS, false);
|
|
411
|
+
|
|
412
|
+
// No-livelock / no-loss guard (NOT the #3550 discriminator).
|
|
413
|
+
expect(retired.steps).toBeLessThan(10_000);
|
|
414
|
+
expect(inFlight.steps).toBeLessThan(10_000);
|
|
415
|
+
for (const m of msgs) {
|
|
416
|
+
const t = retired.terminals.get(m.id);
|
|
417
|
+
expect(t, `repr seed=${seed} msg=${m.id} answer=${m.answerOnAttempt} escFail=${m.escalateFailsFor}`).toBeDefined();
|
|
418
|
+
if (m.answerOnAttempt <= MAX_REPRESENTS) {
|
|
419
|
+
expect(t).toBe("answered");
|
|
420
|
+
} else if (m.escalateFailsFor < ESCALATE_MAX) {
|
|
421
|
+
expect(t).toBe("escalation-delivered");
|
|
422
|
+
} else {
|
|
423
|
+
expect(t).toBe("escalation-give-up");
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
// DISCRIMINATOR 1 — the early-out is terminal-neutral.
|
|
428
|
+
expect(
|
|
429
|
+
[...retired.terminals].sort(),
|
|
430
|
+
`terminals diverged at seed=${seed}`,
|
|
431
|
+
).toEqual([...inFlight.terminals].sort());
|
|
432
|
+
|
|
433
|
+
// DISCRIMINATOR 2 — the early-out actually fires and shortens rungs.
|
|
434
|
+
// Directional per seed (retiring a grace can never ADD sweeps)…
|
|
435
|
+
expect(
|
|
436
|
+
retired.steps,
|
|
437
|
+
`early-out LENGTHENED seed=${seed} (retired=${retired.steps} inFlight=${inFlight.steps})`,
|
|
438
|
+
).toBeLessThanOrEqual(inFlight.steps);
|
|
439
|
+
// …and strictly shorter on the schedules that actually sit out a rung.
|
|
440
|
+
// Not every schedule does: with several messages interleaved, other work
|
|
441
|
+
// can advance the virtual clock past the window with no waiting sweep, so
|
|
442
|
+
// the strict inequality is asserted in aggregate rather than per seed.
|
|
443
|
+
if (retired.steps < inFlight.steps) discriminated++;
|
|
444
|
+
totalRetired += retired.steps;
|
|
445
|
+
totalInFlight += inFlight.steps;
|
|
446
|
+
}
|
|
447
|
+
// Delete the early-out and BOTH runs hold the full 120s window: every seed
|
|
448
|
+
// becomes equal, `discriminated` drops to 0 and the totals converge. These
|
|
449
|
+
// two assertions are what make this case a real test of the #3550 diff.
|
|
450
|
+
expect(discriminated, "the #3550 early-out never shortened a single schedule").toBeGreaterThan(1_000);
|
|
451
|
+
expect(totalRetired).toBeLessThan(totalInFlight);
|
|
452
|
+
});
|
|
453
|
+
|
|
343
454
|
it("a delivered-but-unanswered obligation survives a restart and is escalated, not lost", () => {
|
|
344
455
|
// Deterministic single case: model NEVER answers, escalation succeeds first try,
|
|
345
456
|
// with a restart forced mid-life via a seed that triggers the 0.15 branch.
|