switchroom 0.19.34 → 0.19.35
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/auth-broker/index.js +7 -1
- package/dist/cli/switchroom.js +334 -42
- package/dist/host-control/main.js +8 -2
- package/dist/vault/approvals/kernel-server.js +7 -1
- package/dist/vault/broker/server.js +7 -1
- package/package.json +1 -1
- package/telegram-plugin/dist/gateway/gateway.js +169 -27
- package/telegram-plugin/gateway/gateway.ts +5 -2
- package/telegram-plugin/gateway/model-command.ts +245 -23
- package/telegram-plugin/tests/model-command.test.ts +415 -4
- package/vendor/hindsight-memory/scripts/subagent_retain.py +103 -7
- package/vendor/hindsight-memory/scripts/tests/test_subagent_retain_learnings.py +377 -0
|
@@ -50,6 +50,7 @@ function makeDeps(overrides: Partial<ModelCommandDeps> = {}) {
|
|
|
50
50
|
const deps: ModelCommandDeps = {
|
|
51
51
|
getAgentName: () => "klanker",
|
|
52
52
|
getConfiguredModel: () => "claude-sonnet-5",
|
|
53
|
+
getConfiguredEffort: () => "low",
|
|
53
54
|
escapeHtml: (s) =>
|
|
54
55
|
s.replace(/&/g, "&").replace(/</g, "<").replace(/>/g, ">"),
|
|
55
56
|
preBlock: (s) => `<pre>${s}</pre>`,
|
|
@@ -241,7 +242,7 @@ import {
|
|
|
241
242
|
MODEL_CALLBACK_PAGE_EXTERNAL,
|
|
242
243
|
SR_MODEL_LABELS,
|
|
243
244
|
SR_MODEL_ALIASES,
|
|
244
|
-
|
|
245
|
+
EXTRA_CLAUDE_MENU_TOKENS,
|
|
245
246
|
expandSrAlias,
|
|
246
247
|
externalModelNames,
|
|
247
248
|
type ModelMenuDeps,
|
|
@@ -496,14 +497,14 @@ describe("classifyDiscoveredOptions", () => {
|
|
|
496
497
|
});
|
|
497
498
|
});
|
|
498
499
|
|
|
499
|
-
describe("externalModelNames / SR_MODEL_LABELS /
|
|
500
|
+
describe("externalModelNames / SR_MODEL_LABELS / EXTRA_CLAUDE_MENU_TOKENS", () => {
|
|
500
501
|
it("seeds from the curated alias targets and merges discovered sr-*", () => {
|
|
501
502
|
const names = externalModelNames(["sr-gpt-oss-20b"]);
|
|
502
503
|
expect(names).toContain("sr-gemini-2.5-flash");
|
|
503
504
|
expect(names).toContain("sr-gpt-oss-20b");
|
|
504
505
|
});
|
|
505
|
-
it("Fable is offered as an extra Claude
|
|
506
|
-
expect(
|
|
506
|
+
it("Fable is offered as an extra Claude menu token", () => {
|
|
507
|
+
expect(EXTRA_CLAUDE_MENU_TOKENS.some((t) => t.token === "fable")).toBe(true);
|
|
507
508
|
});
|
|
508
509
|
it("labels are display-only strings", () => {
|
|
509
510
|
expect(SR_MODEL_LABELS["sr-glm-5"]).toBeTypeOf("string");
|
|
@@ -911,3 +912,413 @@ describe("unvalidatedIdCaveat — immediate fail-fast warn on free-text claude-*
|
|
|
911
912
|
expect(alias.text).not.toContain("can't be validated before launch");
|
|
912
913
|
});
|
|
913
914
|
});
|
|
915
|
+
|
|
916
|
+
// ---------------------------------------------------------------------------
|
|
917
|
+
// Short `/model` spellings for pinned Claude ids (CLAUDE_MODEL_ALIASES).
|
|
918
|
+
//
|
|
919
|
+
// Every assertion here FAILS on the pre-change code: `opus48` / `opus-4-8` were
|
|
920
|
+
// passed through verbatim to the carrier (claude would 4xx or silently serve the
|
|
921
|
+
// fallback), were refused by the offline-trust gate, and were not reachable from
|
|
922
|
+
// the picker. The negative block pins the design constraint that made this a
|
|
923
|
+
// separate map from SR_MODEL_ALIASES: these are Anthropic OAuth-passthrough
|
|
924
|
+
// models and must NEVER be classified as external/OpenRouter routes.
|
|
925
|
+
// ---------------------------------------------------------------------------
|
|
926
|
+
import { readFileSync } from "node:fs";
|
|
927
|
+
import {
|
|
928
|
+
CLAUDE_MODEL_ALIASES,
|
|
929
|
+
canonicalModelToken,
|
|
930
|
+
expandClaudeAlias,
|
|
931
|
+
expandModelAlias,
|
|
932
|
+
thinkingEffortCaveat,
|
|
933
|
+
buildModelMenu as buildModelMenuForAliases,
|
|
934
|
+
} from "../gateway/model-command.js";
|
|
935
|
+
|
|
936
|
+
const OPUS_48 = "claude-opus-4-8";
|
|
937
|
+
|
|
938
|
+
describe("CLAUDE_MODEL_ALIASES — short spellings for a pinned Claude id", () => {
|
|
939
|
+
it("both requested spellings expand to the canonical id", () => {
|
|
940
|
+
expect(expandModelAlias("opus48")).toBe(OPUS_48);
|
|
941
|
+
expect(expandModelAlias("opus-4-8")).toBe(OPUS_48);
|
|
942
|
+
expect(expandClaudeAlias("opus48")).toBe(OPUS_48);
|
|
943
|
+
expect(expandClaudeAlias("opus-4-8")).toBe(OPUS_48);
|
|
944
|
+
});
|
|
945
|
+
|
|
946
|
+
it("expansion is case-insensitive", () => {
|
|
947
|
+
for (const spelling of ["OPUS48", "Opus48", "OPUS-4-8", "Opus-4-8"]) {
|
|
948
|
+
expect(expandModelAlias(spelling)).toBe(OPUS_48);
|
|
949
|
+
}
|
|
950
|
+
});
|
|
951
|
+
|
|
952
|
+
it("leaves every other token alone (sr aliases still expand, ids pass through)", () => {
|
|
953
|
+
expect(expandModelAlias("opus")).toBe("opus");
|
|
954
|
+
expect(expandModelAlias("default")).toBe("default");
|
|
955
|
+
expect(expandModelAlias("flash")).toBe("sr-gemini-2.5-flash");
|
|
956
|
+
expect(expandModelAlias("sr-glm-5")).toBe("sr-glm-5");
|
|
957
|
+
expect(expandModelAlias(OPUS_48)).toBe(OPUS_48);
|
|
958
|
+
// The Claude map must not swallow an sr-* shortcut, nor vice versa.
|
|
959
|
+
expect(expandClaudeAlias("flash")).toBe("flash");
|
|
960
|
+
expect(expandSrAlias("opus48")).toBe("opus48");
|
|
961
|
+
});
|
|
962
|
+
|
|
963
|
+
it("every target is a full claude-* id and a legal model arg", () => {
|
|
964
|
+
for (const target of Object.values(CLAUDE_MODEL_ALIASES)) {
|
|
965
|
+
expect(target.startsWith("claude-")).toBe(true);
|
|
966
|
+
expect(isValidModelArg(target)).toBe(true);
|
|
967
|
+
}
|
|
968
|
+
});
|
|
969
|
+
|
|
970
|
+
it("typed `/model opus48` relaunches on the canonical id, not the shortcut", async () => {
|
|
971
|
+
const { deps, relaunchCalls } = makeDeps();
|
|
972
|
+
const reply = await handleModelCommand({ kind: "set", model: "opus48" }, deps);
|
|
973
|
+
expect(relaunchCalls).toEqual([
|
|
974
|
+
{ model: OPUS_48, reason: `user: /model ${OPUS_48} (session-only relaunch)` },
|
|
975
|
+
]);
|
|
976
|
+
expect(reply.text).toContain(OPUS_48);
|
|
977
|
+
// The raw shortcut must never reach the carrier / `claude --model`.
|
|
978
|
+
expect(relaunchCalls[0].model).not.toBe("opus48");
|
|
979
|
+
});
|
|
980
|
+
|
|
981
|
+
it("typed `/model opus-4-8` relaunches on the canonical id", async () => {
|
|
982
|
+
const { deps, relaunchCalls } = makeDeps();
|
|
983
|
+
await handleModelCommand({ kind: "set", model: "opus-4-8" }, deps);
|
|
984
|
+
expect(relaunchCalls).toEqual([
|
|
985
|
+
{ model: OPUS_48, reason: `user: /model ${OPUS_48} (session-only relaunch)` },
|
|
986
|
+
]);
|
|
987
|
+
});
|
|
988
|
+
|
|
989
|
+
it("typed `/model OPUS48` (mixed case) relaunches on the canonical id", async () => {
|
|
990
|
+
const { deps, relaunchCalls } = makeDeps();
|
|
991
|
+
await handleModelCommand({ kind: "set", model: "OPUS48" }, deps);
|
|
992
|
+
expect(relaunchCalls).toEqual([expect.objectContaining({ model: OPUS_48 })]);
|
|
993
|
+
});
|
|
994
|
+
|
|
995
|
+
it("a mid-turn `/model opus48` QUEUES the expanded id (start.sh never sees the shortcut)", () => {
|
|
996
|
+
const busy = { currentTurnActive: true, turnInFlight: false, menuEnabled: true };
|
|
997
|
+
expect(planModelCommand({ kind: "set", model: "opus48" }, busy)).toEqual({
|
|
998
|
+
kind: "queue",
|
|
999
|
+
target: OPUS_48,
|
|
1000
|
+
});
|
|
1001
|
+
expect(planModelCommand({ kind: "set", model: "opus-4-8" }, busy)).toEqual({
|
|
1002
|
+
kind: "queue",
|
|
1003
|
+
target: OPUS_48,
|
|
1004
|
+
});
|
|
1005
|
+
});
|
|
1006
|
+
|
|
1007
|
+
it("a queued shortcut survives a boot: both the KEY and the TARGET are offline-trusted", () => {
|
|
1008
|
+
expect(isOfflineTrustedModelToken("opus48")).toBe(true);
|
|
1009
|
+
expect(isOfflineTrustedModelToken("opus-4-8")).toBe(true);
|
|
1010
|
+
expect(isOfflineTrustedModelToken("OPUS48")).toBe(true);
|
|
1011
|
+
expect(isOfflineTrustedModelToken(OPUS_48)).toBe(true);
|
|
1012
|
+
// Still refuses an arbitrary hand-typed pinned id.
|
|
1013
|
+
expect(isOfflineTrustedModelToken("claude-opus-4-9")).toBe(false);
|
|
1014
|
+
});
|
|
1015
|
+
|
|
1016
|
+
it("parser → expansion round-trip: both spellings reach the canonical id", () => {
|
|
1017
|
+
// The parser alone is not the contract (it tokenizes anything shape-legal —
|
|
1018
|
+
// `opus_4_8` parses too and expands to nothing, see the near-miss issue).
|
|
1019
|
+
// What must hold is the PIPELINE: parse, then the one expansion every
|
|
1020
|
+
// `/model` path applies, lands on the canonical id.
|
|
1021
|
+
for (const text of ["/model opus48", "/model opus-4-8", "/model OPUS48"]) {
|
|
1022
|
+
const parsed = parseModelCommand(text);
|
|
1023
|
+
expect(parsed?.kind).toBe("set");
|
|
1024
|
+
expect(expandModelAlias((parsed as { model: string }).model)).toBe(OPUS_48);
|
|
1025
|
+
}
|
|
1026
|
+
});
|
|
1027
|
+
|
|
1028
|
+
it("NEGATIVE: the shortcuts are Claude, never external / sr-*", () => {
|
|
1029
|
+
for (const token of ["opus48", "opus-4-8", "OPUS48", OPUS_48]) {
|
|
1030
|
+
expect(isSrModel(token)).toBe(false);
|
|
1031
|
+
expect(isClaudeModel(token)).toBe(true);
|
|
1032
|
+
}
|
|
1033
|
+
// Not in the external map, in either direction.
|
|
1034
|
+
for (const key of Object.keys(CLAUDE_MODEL_ALIASES)) {
|
|
1035
|
+
expect(key in SR_MODEL_ALIASES).toBe(false);
|
|
1036
|
+
expect(Object.values(SR_MODEL_ALIASES)).not.toContain(key);
|
|
1037
|
+
}
|
|
1038
|
+
for (const target of Object.values(CLAUDE_MODEL_ALIASES)) {
|
|
1039
|
+
expect(Object.values(SR_MODEL_ALIASES)).not.toContain(target);
|
|
1040
|
+
expect(SR_MODEL_LABELS[target]).toBeUndefined();
|
|
1041
|
+
}
|
|
1042
|
+
// And never rendered on the 🌐 External keyboard page.
|
|
1043
|
+
const external = externalModelNames(["sr-gpt-oss-20b"]);
|
|
1044
|
+
for (const token of [...Object.keys(CLAUDE_MODEL_ALIASES), ...Object.values(CLAUDE_MODEL_ALIASES)]) {
|
|
1045
|
+
expect(external).not.toContain(token);
|
|
1046
|
+
}
|
|
1047
|
+
});
|
|
1048
|
+
|
|
1049
|
+
it("the ACK names the canonical id and never echoes the raw shortcut back", async () => {
|
|
1050
|
+
// Renamed from a "NEGATIVE: … no external ack copy" assertion that was
|
|
1051
|
+
// vacuous: `sr-` / `billed separately` are equally absent from the
|
|
1052
|
+
// pre-change ack, so it could not fail on the bug it claimed to guard.
|
|
1053
|
+
// (The real external-classification constraint is pinned by the
|
|
1054
|
+
// "NEGATIVE: the shortcuts are Claude, never external / sr-*" case above,
|
|
1055
|
+
// which DOES fail without the expansion.) What is genuinely discriminating
|
|
1056
|
+
// here is that the operator is told the id they actually got.
|
|
1057
|
+
const { deps } = makeDeps();
|
|
1058
|
+
const reply = await handleModelCommand({ kind: "set", model: "opus48" }, deps);
|
|
1059
|
+
expect(reply.text).toContain(OPUS_48);
|
|
1060
|
+
expect(reply.text).not.toMatch(/\bopus48\b/);
|
|
1061
|
+
expect(reply.text).not.toContain("billed separately");
|
|
1062
|
+
});
|
|
1063
|
+
|
|
1064
|
+
it("selectable: the pinned id is a main-page keyboard button", async () => {
|
|
1065
|
+
const { deps } = makeMenuDeps();
|
|
1066
|
+
const menu = await buildModelMenuForAliases(deps);
|
|
1067
|
+
const buttons = (menu.keyboard ?? []).flat();
|
|
1068
|
+
expect(buttons.some((b) => b.callback_data === `${MODEL_CALLBACK_ALIAS}${OPUS_48}`)).toBe(true);
|
|
1069
|
+
expect(buttons.some((b) => b.text === "Opus 4.8")).toBe(true);
|
|
1070
|
+
});
|
|
1071
|
+
|
|
1072
|
+
it("selectable mid-turn too: the static quick list carries the same button", async () => {
|
|
1073
|
+
const { deps } = makeMenuDeps({ isBusy: () => true });
|
|
1074
|
+
const menu = await buildModelMenuForAliases(deps);
|
|
1075
|
+
const buttons = (menu.keyboard ?? []).flat();
|
|
1076
|
+
expect(buttons.some((b) => b.callback_data === `${MODEL_CALLBACK_ALIAS}${OPUS_48}`)).toBe(true);
|
|
1077
|
+
});
|
|
1078
|
+
|
|
1079
|
+
it("end-to-end: the RENDERED 'Opus 4.8' button's own payload relaunches on the canonical id", async () => {
|
|
1080
|
+
// Drive the payload the keyboard actually mints rather than a hand-typed
|
|
1081
|
+
// `mdl:alias:claude-opus-4-8` — the latter is recognized on the pre-change
|
|
1082
|
+
// code too, so it could not fail on the bug it guards. Rendering first ties
|
|
1083
|
+
// the button's existence and its payload together in one assertion.
|
|
1084
|
+
const { deps, relaunchCalls } = makeMenuDeps();
|
|
1085
|
+
const menu = await buildModelMenuForAliases(deps);
|
|
1086
|
+
const button = (menu.keyboard ?? []).flat().find((b) => b.text === "Opus 4.8");
|
|
1087
|
+
expect(button).toBeDefined();
|
|
1088
|
+
await handleModelMenuCallback(button!.callback_data, deps);
|
|
1089
|
+
expect(relaunchCalls).toEqual([expect.objectContaining({ model: OPUS_48 })]);
|
|
1090
|
+
});
|
|
1091
|
+
|
|
1092
|
+
it("a shortcut arriving on a callback is expanded, not rejected as unrecognized", async () => {
|
|
1093
|
+
const { deps, relaunchCalls } = makeMenuDeps();
|
|
1094
|
+
await handleModelMenuCallback(`${MODEL_CALLBACK_ALIAS}opus48`, deps);
|
|
1095
|
+
expect(relaunchCalls).toEqual([expect.objectContaining({ model: OPUS_48 })]);
|
|
1096
|
+
});
|
|
1097
|
+
|
|
1098
|
+
it("discoverable: help and the dashboard both name every spelling", async () => {
|
|
1099
|
+
const { deps } = makeDeps();
|
|
1100
|
+
const help = await handleModelCommand({ kind: "help" }, deps);
|
|
1101
|
+
const show = await handleModelCommand({ kind: "show" }, deps);
|
|
1102
|
+
for (const spelling of Object.keys(CLAUDE_MODEL_ALIASES)) {
|
|
1103
|
+
expect(help.text).toContain(spelling);
|
|
1104
|
+
expect(show.text).toContain(`/model ${spelling}`);
|
|
1105
|
+
}
|
|
1106
|
+
expect(help.text).toContain(OPUS_48);
|
|
1107
|
+
expect(show.text).toContain(OPUS_48);
|
|
1108
|
+
});
|
|
1109
|
+
|
|
1110
|
+
it("the exemption itself: a curated TARGET carries no 'can't be validated' caveat", async () => {
|
|
1111
|
+
// The riskiest line in the change is the
|
|
1112
|
+
// `Object.values(CLAUDE_MODEL_ALIASES).includes(lower)` early return in
|
|
1113
|
+
// `unvalidatedIdCaveat`. Driving `/model opus48` does NOT exercise it: on
|
|
1114
|
+
// the pre-change code `opus48` doesn't start with `claude-`, so the caveat
|
|
1115
|
+
// is null for an unrelated reason. Test the full canonical id directly.
|
|
1116
|
+
const deps = { escapeHtml: (s: string) => s, getConfiguredEffort: () => "low" };
|
|
1117
|
+
for (const target of Object.values(CLAUDE_MODEL_ALIASES)) {
|
|
1118
|
+
expect(unvalidatedIdCaveat(deps, target)).toBeNull();
|
|
1119
|
+
}
|
|
1120
|
+
// …and via the ack, on the full-id spelling as well as the shortcut.
|
|
1121
|
+
const { deps: cmdDeps } = makeDeps();
|
|
1122
|
+
for (const spelling of ["opus48", "opus-4-8", OPUS_48]) {
|
|
1123
|
+
const reply = await handleModelCommand({ kind: "set", model: spelling }, cmdDeps);
|
|
1124
|
+
expect(reply.text).not.toContain("can't be validated before launch");
|
|
1125
|
+
}
|
|
1126
|
+
// An UNcurated pinned id still warns — the exemption is not a blanket
|
|
1127
|
+
// silencing of the caveat.
|
|
1128
|
+
const other = await handleModelCommand({ kind: "set", model: "claude-opus-4-9" }, cmdDeps);
|
|
1129
|
+
expect(other.text).toContain("can't be validated before launch");
|
|
1130
|
+
});
|
|
1131
|
+
});
|
|
1132
|
+
|
|
1133
|
+
// ---------------------------------------------------------------------------
|
|
1134
|
+
// MAJOR-1: expansion and the caveat exemption must agree on ONE canonical form.
|
|
1135
|
+
//
|
|
1136
|
+
// Pre-fix, `expandModelAlias` emitted its argument VERBATIM on a map miss (no
|
|
1137
|
+
// lowercase, no trim) while `unvalidatedIdCaveat` lowercased before comparing.
|
|
1138
|
+
// So `/model CLAUDE-OPUS-4-8` (phone autocapitalize) was launched as the
|
|
1139
|
+
// uncanonical token `CLAUDE-OPUS-4-8` — an id claude cannot resolve, hence a
|
|
1140
|
+
// silent `--fallback-model` serve — under a clean green ack, because the
|
|
1141
|
+
// exemption matched the lowercased form. Both halves are asserted here.
|
|
1142
|
+
// ---------------------------------------------------------------------------
|
|
1143
|
+
describe("canonicalization — one form reaches `claude --model` and the caveat gate", () => {
|
|
1144
|
+
it("canonicalModelToken lowercases claude-* ids, trims everything, touches nothing else", () => {
|
|
1145
|
+
expect(canonicalModelToken("CLAUDE-OPUS-4-8")).toBe(OPUS_48);
|
|
1146
|
+
expect(canonicalModelToken("Claude-Opus-4-8")).toBe(OPUS_48);
|
|
1147
|
+
expect(canonicalModelToken(` ${OPUS_48} `)).toBe(OPUS_48);
|
|
1148
|
+
expect(canonicalModelToken("Claude-Opus-4-9")).toBe("claude-opus-4-9");
|
|
1149
|
+
// Non-claude tokens keep their case — sr-* ids are not ours to rewrite.
|
|
1150
|
+
expect(canonicalModelToken("sr-GLM-5")).toBe("sr-GLM-5");
|
|
1151
|
+
expect(canonicalModelToken(" sr-glm-5 ")).toBe("sr-glm-5");
|
|
1152
|
+
expect(canonicalModelToken("opus")).toBe("opus");
|
|
1153
|
+
});
|
|
1154
|
+
|
|
1155
|
+
it("expandModelAlias emits the canonical form on a MISS, not the raw argument", () => {
|
|
1156
|
+
expect(expandModelAlias("CLAUDE-OPUS-4-8")).toBe(OPUS_48);
|
|
1157
|
+
expect(expandModelAlias("Claude-Opus-4-8")).toBe(OPUS_48);
|
|
1158
|
+
expect(expandModelAlias("Claude-Opus-4-9")).toBe("claude-opus-4-9");
|
|
1159
|
+
// …and trims, matching what `unvalidatedIdCaveat` has always done.
|
|
1160
|
+
expect(expandModelAlias(" opus48 ")).toBe(OPUS_48);
|
|
1161
|
+
expect(expandModelAlias(` ${OPUS_48} `)).toBe(OPUS_48);
|
|
1162
|
+
});
|
|
1163
|
+
|
|
1164
|
+
it("a mixed-case pinned id is LAUNCHED canonical, not verbatim", async () => {
|
|
1165
|
+
for (const spelling of ["CLAUDE-OPUS-4-8", "Claude-Opus-4-8"]) {
|
|
1166
|
+
const { deps, relaunchCalls } = makeDeps();
|
|
1167
|
+
const reply = await handleModelCommand({ kind: "set", model: spelling }, deps);
|
|
1168
|
+
expect(relaunchCalls).toEqual([
|
|
1169
|
+
{ model: OPUS_48, reason: `user: /model ${OPUS_48} (session-only relaunch)` },
|
|
1170
|
+
]);
|
|
1171
|
+
// The uncanonical token must not survive anywhere the operator can see
|
|
1172
|
+
// it either — the ack names what actually boots.
|
|
1173
|
+
expect(reply.text).not.toContain(spelling);
|
|
1174
|
+
expect(reply.text).toContain(OPUS_48);
|
|
1175
|
+
}
|
|
1176
|
+
});
|
|
1177
|
+
|
|
1178
|
+
it("an UNCURATED mixed-case pinned id is canonicalized AND still warned about", async () => {
|
|
1179
|
+
const { deps, relaunchCalls } = makeDeps();
|
|
1180
|
+
const reply = await handleModelCommand({ kind: "set", model: "CLAUDE-OPUS-4-9" }, deps);
|
|
1181
|
+
expect(relaunchCalls[0].model).toBe("claude-opus-4-9");
|
|
1182
|
+
expect(reply.text).toContain("can't be validated before launch");
|
|
1183
|
+
});
|
|
1184
|
+
|
|
1185
|
+
it("the caveat exemption agrees with expansion across the whole casing matrix", () => {
|
|
1186
|
+
const deps = { escapeHtml: (s: string) => s, getConfiguredEffort: () => "low" };
|
|
1187
|
+
for (const spelling of [
|
|
1188
|
+
OPUS_48,
|
|
1189
|
+
"CLAUDE-OPUS-4-8",
|
|
1190
|
+
"Claude-Opus-4-8",
|
|
1191
|
+
` ${OPUS_48} `,
|
|
1192
|
+
" CLAUDE-OPUS-4-8 ",
|
|
1193
|
+
]) {
|
|
1194
|
+
// Exempt (curated target) …
|
|
1195
|
+
expect(unvalidatedIdCaveat(deps, spelling)).toBeNull();
|
|
1196
|
+
// … and that is only sound because expansion emits the SAME form.
|
|
1197
|
+
expect(expandModelAlias(spelling)).toBe(OPUS_48);
|
|
1198
|
+
}
|
|
1199
|
+
for (const spelling of ["claude-opus-4-9", "CLAUDE-OPUS-4-9", "Claude-Sonnet-9"]) {
|
|
1200
|
+
expect(unvalidatedIdCaveat(deps, spelling)).not.toBeNull();
|
|
1201
|
+
}
|
|
1202
|
+
});
|
|
1203
|
+
|
|
1204
|
+
it("the callback shape gate still runs on the RAW payload (trim must not widen it)", async () => {
|
|
1205
|
+
const { deps, relaunchCalls } = makeMenuDeps();
|
|
1206
|
+
// `expandModelAlias` trims; the shape gate exists to guarantee a single
|
|
1207
|
+
// whitespace-free token reaches the tmux pane, so a padded payload must
|
|
1208
|
+
// still be refused rather than trimmed into legality.
|
|
1209
|
+
const out = await handleModelMenuCallback(`${MODEL_CALLBACK_ALIAS} ${OPUS_48} `, deps);
|
|
1210
|
+
expect(out.answer).toBe("Invalid model name");
|
|
1211
|
+
expect(relaunchCalls).toEqual([]);
|
|
1212
|
+
});
|
|
1213
|
+
});
|
|
1214
|
+
|
|
1215
|
+
// ---------------------------------------------------------------------------
|
|
1216
|
+
// MAJOR-2: the #1978 adaptive-thinking risk must surface at SWITCH time.
|
|
1217
|
+
//
|
|
1218
|
+
// `assessThinkingEffortRisk` had exactly one consumer — `switchroom doctor`,
|
|
1219
|
+
// which reads the CONFIGURED `model:` out of switchroom.yaml. A `/model` switch
|
|
1220
|
+
// is session-scoped, so every route onto a pinned Opus 4.x id (typed, shortcut,
|
|
1221
|
+
// or the one-tap "Opus 4.8" button this change adds) reached the exact
|
|
1222
|
+
// (pinned Opus 4.x, effort > low) combination doctor exists to warn about with
|
|
1223
|
+
// no warning anywhere. These fail on BOTH the pre-change code and the
|
|
1224
|
+
// pre-fix PR head.
|
|
1225
|
+
// ---------------------------------------------------------------------------
|
|
1226
|
+
const MERGE_400_MARKER = "issue #1978";
|
|
1227
|
+
|
|
1228
|
+
describe("thinking_effort × pinned Opus 4.x — surfaced when the switch is made", () => {
|
|
1229
|
+
it("typed switch onto a pinned Opus 4.x at medium effort warns in the ack", async () => {
|
|
1230
|
+
const { deps } = makeDeps({ getConfiguredEffort: () => "medium" });
|
|
1231
|
+
const reply = await handleModelCommand({ kind: "set", model: OPUS_48 }, deps);
|
|
1232
|
+
expect(reply.text).toContain(MERGE_400_MARKER);
|
|
1233
|
+
expect(reply.text).toContain("medium");
|
|
1234
|
+
});
|
|
1235
|
+
|
|
1236
|
+
it("the SHORTCUT spelling warns too — the risk follows the expanded target", async () => {
|
|
1237
|
+
const { deps } = makeDeps({ getConfiguredEffort: () => "high" });
|
|
1238
|
+
const reply = await handleModelCommand({ kind: "set", model: "opus48" }, deps);
|
|
1239
|
+
expect(reply.text).toContain(MERGE_400_MARKER);
|
|
1240
|
+
});
|
|
1241
|
+
|
|
1242
|
+
it("the one-tap 'Opus 4.8' BUTTON warns too — the easiest route into the combo", async () => {
|
|
1243
|
+
const { deps } = makeMenuDeps({ getConfiguredEffort: () => "medium" });
|
|
1244
|
+
const menu = await buildModelMenuForAliases(deps);
|
|
1245
|
+
const button = (menu.keyboard ?? []).flat().find((b) => b.text === "Opus 4.8");
|
|
1246
|
+
const out = await handleModelMenuCallback(button!.callback_data, deps);
|
|
1247
|
+
expect(out.reply.text).toContain(MERGE_400_MARKER);
|
|
1248
|
+
});
|
|
1249
|
+
|
|
1250
|
+
it("NEGATIVE: safe combos stay quiet — low effort, Opus 5, the bare alias, sonnet", async () => {
|
|
1251
|
+
const atLow = makeDeps({ getConfiguredEffort: () => "low" });
|
|
1252
|
+
expect(
|
|
1253
|
+
(await handleModelCommand({ kind: "set", model: OPUS_48 }, atLow.deps)).text,
|
|
1254
|
+
).not.toContain(MERGE_400_MARKER);
|
|
1255
|
+
|
|
1256
|
+
const atMedium = makeDeps({ getConfiguredEffort: () => "medium" });
|
|
1257
|
+
for (const safe of ["opus", "claude-opus-5-20260601", "sonnet", "sr-glm-5"]) {
|
|
1258
|
+
const reply = await handleModelCommand({ kind: "set", model: safe }, atMedium.deps);
|
|
1259
|
+
expect(reply.text).not.toContain(MERGE_400_MARKER);
|
|
1260
|
+
}
|
|
1261
|
+
});
|
|
1262
|
+
|
|
1263
|
+
it("NEGATIVE: an unreadable effort never blocks or fails the switch", async () => {
|
|
1264
|
+
const { deps, relaunchCalls } = makeDeps({
|
|
1265
|
+
getConfiguredEffort: () => { throw new Error("agent list unavailable"); },
|
|
1266
|
+
});
|
|
1267
|
+
const reply = await handleModelCommand({ kind: "set", model: OPUS_48 }, deps);
|
|
1268
|
+
expect(relaunchCalls).toEqual([expect.objectContaining({ model: OPUS_48 })]);
|
|
1269
|
+
expect(reply.text).not.toContain(MERGE_400_MARKER);
|
|
1270
|
+
|
|
1271
|
+
const unset = makeDeps({ getConfiguredEffort: () => null });
|
|
1272
|
+
expect(
|
|
1273
|
+
(await handleModelCommand({ kind: "set", model: OPUS_48 }, unset.deps)).text,
|
|
1274
|
+
).not.toContain(MERGE_400_MARKER);
|
|
1275
|
+
});
|
|
1276
|
+
|
|
1277
|
+
it("thinkingEffortCaveat is advisory only — it never rewrites the effort", () => {
|
|
1278
|
+
const escapeHtml = (s: string) => s;
|
|
1279
|
+
expect(thinkingEffortCaveat({ escapeHtml, getConfiguredEffort: () => "max" }, OPUS_48))
|
|
1280
|
+
.toContain(MERGE_400_MARKER);
|
|
1281
|
+
expect(thinkingEffortCaveat({ escapeHtml, getConfiguredEffort: () => "low" }, OPUS_48))
|
|
1282
|
+
.toBeNull();
|
|
1283
|
+
expect(thinkingEffortCaveat({ escapeHtml, getConfiguredEffort: () => "max" }, "opus"))
|
|
1284
|
+
.toBeNull();
|
|
1285
|
+
});
|
|
1286
|
+
});
|
|
1287
|
+
|
|
1288
|
+
// ---------------------------------------------------------------------------
|
|
1289
|
+
// The queued-across-restart persist path (gateway.ts). `persistQueuedCommandForRestart`
|
|
1290
|
+
// is module-local in a ~25k-line file that cannot be imported into a unit test,
|
|
1291
|
+
// so this is a source ratchet in the style of
|
|
1292
|
+
// `tests/scaffold.session-model.test.ts`: it pins the ONE line that decides
|
|
1293
|
+
// which token survives a restart. Pre-change that call used `expandSrAlias`, so
|
|
1294
|
+
// a queued `/model opus48` was persisted as the raw shortcut and start.sh
|
|
1295
|
+
// launched a token claude cannot resolve.
|
|
1296
|
+
// ---------------------------------------------------------------------------
|
|
1297
|
+
describe("gateway persist path uses the shared expansion, not the sr-only one", () => {
|
|
1298
|
+
const gatewaySrc = readFileSync(
|
|
1299
|
+
new URL("../gateway/gateway.ts", import.meta.url),
|
|
1300
|
+
"utf-8",
|
|
1301
|
+
);
|
|
1302
|
+
|
|
1303
|
+
it("the `.session-model` carrier write expands through expandModelAlias", () => {
|
|
1304
|
+
expect(gatewaySrc).toContain(
|
|
1305
|
+
"writeSessionModelFile(agentDir, expandModelAlias(action.arg), configured)",
|
|
1306
|
+
);
|
|
1307
|
+
expect(gatewaySrc).not.toContain("writeSessionModelFile(agentDir, expandSrAlias(");
|
|
1308
|
+
// …and the symbol is actually imported, so the above is live code.
|
|
1309
|
+
expect(gatewaySrc).toMatch(/^\s*expandModelAlias,$/m);
|
|
1310
|
+
});
|
|
1311
|
+
|
|
1312
|
+
it("the MODEL deps wire the configured effort for the switch-time guard", () => {
|
|
1313
|
+
// Anchored on the model deps' own `getConfiguredModel` reader — the effort
|
|
1314
|
+
// reader already existed on the EFFORT deps (`buildEffortDeps`), so a bare
|
|
1315
|
+
// substring match would pass without the model side being wired at all.
|
|
1316
|
+
const lines = gatewaySrc.split("\n");
|
|
1317
|
+
const anchor = lines.findIndex((l) =>
|
|
1318
|
+
l.includes("return data?.agents?.find(a => a.name === getMyAgentName())?.model ?? null"),
|
|
1319
|
+
);
|
|
1320
|
+
expect(anchor).toBeGreaterThan(-1);
|
|
1321
|
+
const window = lines.slice(anchor, anchor + 15).join("\n");
|
|
1322
|
+
expect(window).toContain("getConfiguredEffort: () => getConfiguredEffortForPersist(),");
|
|
1323
|
+
});
|
|
1324
|
+
});
|
|
@@ -3,11 +3,24 @@
|
|
|
3
3
|
|
|
4
4
|
Delegated (sub-agent / Task-tool) work is the biggest systematic memory hole:
|
|
5
5
|
the main-session Stop retain only ever reads the parent ``transcript_path``, so
|
|
6
|
-
a worker's hours of
|
|
7
|
-
|
|
6
|
+
a worker's hours of reasoning — the constraints it discovered, the structural
|
|
7
|
+
dead ends it ruled out — reach memory only as the terse final report the parent
|
|
8
8
|
transcript captures. This hook closes that hole by retaining a bounded window of
|
|
9
9
|
the *sidechain* transcript when a sub-agent terminates.
|
|
10
10
|
|
|
11
|
+
Scope (Ken, 2026-07-29): LEARNINGS, not raw transcripts and tools. The window is
|
|
12
|
+
retained on the TEXT-ONLY formatting path (``run_subagent_retain`` forces
|
|
13
|
+
``retainToolCalls = False`` on its config copy), so what reaches memory is the
|
|
14
|
+
sub-agent's own prose — reasoning, findings, final report — and NOT tool_use
|
|
15
|
+
inputs, tool_result bodies, file contents or diffs.
|
|
16
|
+
|
|
17
|
+
The text-only path cannot format to nothing here: the volume gate already
|
|
18
|
+
requires ``MIN_HUMAN_TURNS`` GENUINE human turns (tool_result-only user messages
|
|
19
|
+
are explicitly not counted, ``count_human_turns``), and every such turn is a
|
|
20
|
+
plain-string user message that ``_extract_text_content`` returns verbatim. So a
|
|
21
|
+
window that clears the gate always carries at least its instruction turns, and
|
|
22
|
+
``build_retain_payload`` never returns None on this path for volume reasons.
|
|
23
|
+
|
|
11
24
|
Probe result (Claude Code 2.1.215, PR5 Task 0 — recorded in the PR body):
|
|
12
25
|
the ``SubagentStop`` hook input carries BOTH the parent ``transcript_path`` AND
|
|
13
26
|
a first-class ``agent_transcript_path`` pointing straight at the sidechain
|
|
@@ -62,8 +75,11 @@ from lib.pacing import inflight_lock
|
|
|
62
75
|
# stays byte-identical to the main path where it matters (dedup ids, formatting).
|
|
63
76
|
from retain import build_retain_payload, read_transcript
|
|
64
77
|
|
|
65
|
-
# Retain the last N human turns of the sidechain.
|
|
66
|
-
#
|
|
78
|
+
# Retain the last N human turns of the sidechain. The window is formatted on the
|
|
79
|
+
# TEXT-ONLY path (``retainToolCalls`` is forced False for the sidechain — see
|
|
80
|
+
# ``run_subagent_retain``), so tool_use inputs and tool_result bodies are dropped
|
|
81
|
+
# entirely rather than passed through / truncated. ``_extract_message_blocks`` is
|
|
82
|
+
# still imported here, but only for the volume gate's char count.
|
|
67
83
|
SIDECHAIN_WINDOW_TURNS = 40
|
|
68
84
|
|
|
69
85
|
# Volume gate floors — SubagentStop fires for every Task, so skip trivial forks.
|
|
@@ -256,6 +272,14 @@ def non_tool_result_char_count(messages: list, stop_at: int | None = None) -> in
|
|
|
256
272
|
the floor even if it emitted a large tool_result, while a real worker's
|
|
257
273
|
commands and decisions count.
|
|
258
274
|
|
|
275
|
+
KNOWN MISMATCH (accepted, tracked as a follow-up): this counts ``tool_use``
|
|
276
|
+
inputs, but those are no longer RETAINED — the sidechain payload is built on
|
|
277
|
+
the text-only path. So the gate can clear on tool volume that contributes
|
|
278
|
+
nothing to the stored memory. It cannot produce an EMPTY retain (the
|
|
279
|
+
``MIN_HUMAN_TURNS`` floor guarantees prose-bearing user turns — see the
|
|
280
|
+
module docstring), so this is a precision issue in the gate, not a
|
|
281
|
+
correctness bug. Tightening it to count only text is a separate change.
|
|
282
|
+
|
|
259
283
|
``stop_at`` (review finding 4 — early short-circuit): return as soon as the
|
|
260
284
|
running total reaches this many chars. The gate only needs to know whether
|
|
261
285
|
the floor is CLEARED, not the exact size — so on a large worker transcript
|
|
@@ -396,9 +420,32 @@ def run_subagent_retain(hook_input: dict) -> dict:
|
|
|
396
420
|
# main-session retains: reuse retain.py's ``slice_document_id`` recipe via a
|
|
397
421
|
# composite session key so (a) re-fires of the SAME sub-agent window upsert
|
|
398
422
|
# server-side, and (b) it never collides with the parent's own
|
|
399
|
-
# ``{session_id}-r...`` documents.
|
|
400
|
-
#
|
|
401
|
-
#
|
|
423
|
+
# ``{session_id}-r...`` documents.
|
|
424
|
+
#
|
|
425
|
+
# Client-side diffing against the parent's final-report retain is still NOT
|
|
426
|
+
# attempted, but NOT because "hindsight's consolidation dedups" — the
|
|
427
|
+
# original claim here (design item 4) was FALSE and is the assumption that
|
|
428
|
+
# licensed the sidechain volume. Verified against the engine source
|
|
429
|
+
# (upstream image ``ghcr.io/vectorize-io/hindsight``):
|
|
430
|
+
# * The only SEMANTIC dedup lives in
|
|
431
|
+
# ``hindsight_api/engine/consolidation/consolidator.py`` and is a guard
|
|
432
|
+
# on the consolidator's OWN OUTPUT: ``_dedup_adjudicate`` probes a newly
|
|
433
|
+
# created/updated ``observation`` against existing ones, passing the
|
|
434
|
+
# literal fact-type list ``["observation"]`` to
|
|
435
|
+
# ``retrieve_semantic_bm25_combined``.
|
|
436
|
+
# * ``world`` and ``experience`` are the raw extracted facts that FEED the
|
|
437
|
+
# consolidator (it selects ``fact_type IN ('experience', 'world')`` for
|
|
438
|
+
# unconsolidated rows) — they never traverse the observation dedup path.
|
|
439
|
+
# * Grepping ``hindsight_api/engine/retain/*.py`` for dedup finds only
|
|
440
|
+
# content-hash CHUNK dedup (``chunk_storage.compute_chunk_hash``), i.e.
|
|
441
|
+
# byte-identical-chunk skipping. There is no semantic dedup on the
|
|
442
|
+
# retain path at all.
|
|
443
|
+
# So overlap between a sidechain retain and the parent's own retain persists
|
|
444
|
+
# as extra ``world``/``experience`` rows forever. Volume control has to come
|
|
445
|
+
# from retaining LESS (see ``retainToolCalls`` below), not from a downstream
|
|
446
|
+
# dedup that does not exist. Diffing is still skipped here because the
|
|
447
|
+
# deterministic document_id already makes re-fires of the SAME window upsert,
|
|
448
|
+
# which is the duplicate class this path can actually create.
|
|
402
449
|
sub_session_id = f"{session_id}-sub-{agent_id}"
|
|
403
450
|
|
|
404
451
|
# Sidechain tags + a topic-friendly parent link. Reuse the config-driven tag
|
|
@@ -407,6 +454,55 @@ def run_subagent_retain(hook_input: dict) -> dict:
|
|
|
407
454
|
# (recallTagWeights); ``parent_session:<id>`` lets a fresh session pull a
|
|
408
455
|
# worker's process facts by parent.
|
|
409
456
|
sub_config = dict(config)
|
|
457
|
+
|
|
458
|
+
# Learnings, not raw transcripts and tools (Ken, 2026-07-29): "I don't want
|
|
459
|
+
# sub-agents' transcripts but definitely their learnings should be captured
|
|
460
|
+
# but not raw transcripts and tools."
|
|
461
|
+
#
|
|
462
|
+
# With ``retainToolCalls`` on, ``build_retain_payload`` formats the window
|
|
463
|
+
# via ``_prepare_json_transcript`` → ``_extract_message_blocks``, which emits
|
|
464
|
+
# every ``tool_use.input`` verbatim (an entire ``Write.content``, a full
|
|
465
|
+
# ``Edit`` diff, a full ``Bash.command``) plus every ``tool_result`` at up to
|
|
466
|
+
# 2,000 chars per block.
|
|
467
|
+
#
|
|
468
|
+
# The cost mechanism is DOCUMENT FAN-OUT, not one oversized payload: a large
|
|
469
|
+
# payload is not truncated, it is CAPPED and SPLIT by
|
|
470
|
+
# ``lib.retain_split.split_retain_content`` at ``retain_content_limit()``
|
|
471
|
+
# (33,000 chars on the shipped inputs) into ``{base}-p{i}of{n}`` parts
|
|
472
|
+
# (``client.py`` ~:235). Real sidechain retains have been observed splitting
|
|
473
|
+
# into 38 parts / 1,022 messages, and every part is separately chunked and
|
|
474
|
+
# LLM-extracted into ``world``/``experience`` rows. Shrinking the content is
|
|
475
|
+
# therefore the lever that reduces rows.
|
|
476
|
+
#
|
|
477
|
+
# Forcing it OFF for the sidechain path routes the window through
|
|
478
|
+
# ``_prepare_text_transcript`` → ``_extract_text_content``, keeping assistant
|
|
479
|
+
# ``text`` blocks and channel-message tool_use text. Measured on real fleet
|
|
480
|
+
# sidechains: 5.5x smaller over 30 sampled transcripts (2,012,376 → 367,796
|
|
481
|
+
# chars), and independently 6.4x over the 84 most recent GATE-PASSING ones
|
|
482
|
+
# (26,278,261 → 4,085,678 chars) — of which 0 formatted to empty and 0 came
|
|
483
|
+
# out under 500 chars.
|
|
484
|
+
#
|
|
485
|
+
# ACCEPTED LOSS — this is not a clean "tool noise only" filter.
|
|
486
|
+
# ``_extract_text_content`` also drops image blocks, ALL ``tool_result``
|
|
487
|
+
# content, and sub-agent Task report bodies. So a fact that existed ONLY in
|
|
488
|
+
# tool output is unrecoverable: if a query returned ``43442`` and the agent
|
|
489
|
+
# merely replied "checked, it's the backlog", the number is gone. That trade
|
|
490
|
+
# is deliberate and is what Ken asked for; sub-agent prose was separately
|
|
491
|
+
# measured to carry durable learnings at essentially the main-session rate
|
|
492
|
+
# (5.24% vs 6.87%), so the learnings themselves survive.
|
|
493
|
+
#
|
|
494
|
+
# Mid-session safety: ``slice_document_id`` derives the BASE id from the
|
|
495
|
+
# slice's first/last uuids, not from the formatted text, so a re-fire of the
|
|
496
|
+
# same window still targets the same base id. Note the part SUFFIX embeds the
|
|
497
|
+
# part total, so a document that used to split into n parts and now splits
|
|
498
|
+
# into m < n leaves parts m+1..n in place — they are not overwritten and not
|
|
499
|
+
# duplicated. Historical sidechain rows are untouched by this change; this
|
|
500
|
+
# fixes INTAKE going forward only.
|
|
501
|
+
#
|
|
502
|
+
# Deliberately set on the COPY, not on ``config``: the parent session's own
|
|
503
|
+
# Stop retain keeps whatever the operator configured.
|
|
504
|
+
sub_config["retainToolCalls"] = False
|
|
505
|
+
|
|
410
506
|
base_tags = list(config.get("retainTags") or [])
|
|
411
507
|
extra_tags = ["sidechain", f"parent_session:{session_id}"]
|
|
412
508
|
if agent_type:
|