prism-mcp-server 20.21.13 → 20.21.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +64 -7
- package/dist/storage/inferMetricsLedger.js +34 -0
- package/dist/tools/ledgerHandlers.js +7 -4
- package/dist/tools/prismInferHandler.js +11 -0
- package/dist/tools/promptRouteHandler.js +57 -4
- package/dist/tools/savingsHandler.js +40 -0
- package/dist/tools/scopedSkillTriggers.js +6 -2
- package/dist/tools/skillRouting.js +69 -8
- package/dist/tools/taskRouterHandler.js +39 -2
- package/dist/utils/qualityGate.js +12 -4
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -158,6 +158,39 @@ or by re-enabling after each run.
|
|
|
158
158
|
<details>
|
|
159
159
|
<summary>Release history (optional)</summary>
|
|
160
160
|
|
|
161
|
+
## What's New in v20.21.15
|
|
162
|
+
|
|
163
|
+
### Follow-ups: see what runs locally, and a stricter screen when the classifier is down
|
|
164
|
+
|
|
165
|
+
- `local_savings` (and `prism savings`) shows a line for follow-ups, calls that
|
|
166
|
+
carried your conversation. It covers how many the local model answered (and
|
|
167
|
+
how many of those the 9b answered), how many the on-device screen refused and
|
|
168
|
+
at which stage, how many your plan's multi-turn limits refused, and how many
|
|
169
|
+
went to the cloud.
|
|
170
|
+
- A short chat answer ("16", "Yes") counts as an answer. Before, the quality
|
|
171
|
+
gate treated it as empty, and a paid plan re-asked the cloud.
|
|
172
|
+
- When the on-device classifier fails every read on a follow-up, the follow-up
|
|
173
|
+
is no longer answered on a keyword check alone. It goes to the cloud if your
|
|
174
|
+
plan allows it, and is refused otherwise.
|
|
175
|
+
|
|
176
|
+
## What's New in v20.21.14
|
|
177
|
+
|
|
178
|
+
### Skills load when they help, and you can see why they loaded
|
|
179
|
+
|
|
180
|
+
- Background task notifications, reports from other agents and continuation
|
|
181
|
+
summaries no longer load skills in the middle of a task, so a word inside an
|
|
182
|
+
agent's report cannot pull in unrelated rules.
|
|
183
|
+
- Pasting Prism's startup output into a prompt no longer loads skills that
|
|
184
|
+
happen to share a word with it.
|
|
185
|
+
- Tasks that say they need host tools or reserved judgment stay with the host
|
|
186
|
+
instead of failing on the local worker.
|
|
187
|
+
- Every routed-skills header ends with the routing table version (for example
|
|
188
|
+
"Routing table v41."), so a past skill load can be checked against the exact
|
|
189
|
+
rules that chose it.
|
|
190
|
+
- Skills with their own triggers now load even when their file uses Windows
|
|
191
|
+
line endings or mentions `prompt_triggers:` in its description; before, they
|
|
192
|
+
were delivered but silently never loaded.
|
|
193
|
+
|
|
161
194
|
## What's New in v20.21.13
|
|
162
195
|
|
|
163
196
|
### Paid plans and trials are clear in Account & Settings
|
|
@@ -1427,11 +1460,11 @@ It is paid because it cannot run without Synalux behind it:
|
|
|
1427
1460
|
```typescript
|
|
1428
1461
|
// Call 1
|
|
1429
1462
|
prism_infer({ prompt: "My project codename is Nightjar. Reply OK.", mode: "chat" })
|
|
1430
|
-
// → "OK" (local
|
|
1463
|
+
// → "OK" (local model, $0)
|
|
1431
1464
|
|
|
1432
1465
|
// Call 2 — the model never saw call 1
|
|
1433
1466
|
prism_infer({ prompt: "What is my codename? One word.", mode: "chat" })
|
|
1434
|
-
// → "I don't have that information." (local
|
|
1467
|
+
// → "I don't have that information." (local model, correct and useless)
|
|
1435
1468
|
|
|
1436
1469
|
// Call 3 — a coding follow-up with no thread
|
|
1437
1470
|
prism_infer({ prompt: "Now add a timeout parameter to it.", mode: "code" })
|
|
@@ -1456,7 +1489,7 @@ prism_infer({
|
|
|
1456
1489
|
prompt: "What is my codename? One word.",
|
|
1457
1490
|
mode: "chat",
|
|
1458
1491
|
})
|
|
1459
|
-
// → "Nightjar" (local
|
|
1492
|
+
// → "Nightjar" (local model, $0; history_turns: 2)
|
|
1460
1493
|
|
|
1461
1494
|
prism_infer({
|
|
1462
1495
|
messages: [
|
|
@@ -1466,7 +1499,7 @@ prism_infer({
|
|
|
1466
1499
|
prompt: "Write the one-line call that stores its result in n.",
|
|
1467
1500
|
mode: "code",
|
|
1468
1501
|
})
|
|
1469
|
-
// → "n = countActiveUsers(data)" (local
|
|
1502
|
+
// → "n = countActiveUsers(data)" (local model, $0)
|
|
1470
1503
|
|
|
1471
1504
|
// A turn the on-device screen finds uncertain, alone or in context, is not
|
|
1472
1505
|
// served locally: it goes to Synalux cloud on a paid plan, or is refused with
|
|
@@ -1483,9 +1516,14 @@ prism_infer({
|
|
|
1483
1516
|
// → "SYN-4471" (Gemini 3.6 Flash; used_cloud: true)
|
|
1484
1517
|
```
|
|
1485
1518
|
|
|
1486
|
-
Measured on the real
|
|
1487
|
-
benign follow-
|
|
1488
|
-
|
|
1519
|
+
Measured on 20.21.14 through the real handler, on the local 9b with cloud off:
|
|
1520
|
+
- Every benign follow-up that reached the local 9b with its conversation
|
|
1521
|
+
attached was answered correctly. Without the conversation, most answers were
|
|
1522
|
+
invented.
|
|
1523
|
+
- The on-device screen still refuses too many benign follow-ups: 11 of 26 in
|
|
1524
|
+
our tests. Each refusal names its reason. Cutting these false refusals is
|
|
1525
|
+
the current work.
|
|
1526
|
+
- Every reserved multi-turn probe was refused before any generation (11 of 11).
|
|
1489
1527
|
</details>
|
|
1490
1528
|
|
|
1491
1529
|
| Mode | Think | Model | Use case |
|
|
@@ -1538,6 +1576,10 @@ host, or run `prism savings` from a terminal — `--period all|month|week|sessio
|
|
|
1538
1576
|
prism-coder:9b: 41 call(s), ~505K tokens
|
|
1539
1577
|
prism-coder:4b: 12 call(s), ~4.8K tokens
|
|
1540
1578
|
|
|
1579
|
+
Follow-ups with your conversation:
|
|
1580
|
+
12 answered locally (12 by the 9b) · 2 refused by the on-device screen · 0 sent to cloud
|
|
1581
|
+
Refusals by stage: follow-up alone 1 · turns together 1
|
|
1582
|
+
|
|
1541
1583
|
Counts tokens a local model handled instead of your cloud model. On the token
|
|
1542
1584
|
axis, the token count is measured — a floor, with known undercounts listed
|
|
1543
1585
|
when present. On the displacement axis, prism cannot observe the call your
|
|
@@ -1559,6 +1601,21 @@ the same tokens; and most users are on flat plans where a currency figure means
|
|
|
1559
1601
|
nothing at all. Tokens are the one unit prism measured itself. If you know your
|
|
1560
1602
|
own effective rate, multiply — the split is printed for exactly that reason.
|
|
1561
1603
|
|
|
1604
|
+
**Follow-ups.** Calls that carried your conversation (`messages`) get their
|
|
1605
|
+
own line:
|
|
1606
|
+
- how many the local model answered, and how many of those the 9b answered;
|
|
1607
|
+
- how many the on-device screen refused, and at which stage (the follow-up read
|
|
1608
|
+
alone, an earlier turn read alone, or the turns read together);
|
|
1609
|
+
- how many went to the cloud.
|
|
1610
|
+
|
|
1611
|
+
A refused follow-up went back to your host instead of being answered locally,
|
|
1612
|
+
so this line shows how much of your follow-up work local serving actually
|
|
1613
|
+
took. Refusals because your plan does not include multi-turn history, or the
|
|
1614
|
+
history is over its cap, are counted on their own line. They are recorded only
|
|
1615
|
+
when the host asks for a report (`escalation: "report"`); otherwise they fail
|
|
1616
|
+
before anything is recorded. The line appears for the `week`, `month`, `all`
|
|
1617
|
+
and `--days` views, which read the durable ledger.
|
|
1618
|
+
|
|
1562
1619
|
Refused calls are excluded, the VS Code panel-playground share is disclosed
|
|
1563
1620
|
separately, and the known sources of undercount are listed inline rather than
|
|
1564
1621
|
left implicit — so the durable (`week`/`month`/`all`/`--days`) headline is a
|
|
@@ -284,6 +284,32 @@ export async function queryLocalSavings(sinceTs) {
|
|
|
284
284
|
GROUP BY COALESCE(model, backend)`,
|
|
285
285
|
args: whereArgs,
|
|
286
286
|
});
|
|
287
|
+
const followWhere = `WHERE history_turns > 0${sinceTs != null ? " AND ts >= ?" : ""}`;
|
|
288
|
+
const PLAN_REFUSAL = `COALESCE(refusal_reason, '') IN ('multi_turn_not_in_plan', 'history_over_plan_cap')`;
|
|
289
|
+
// A 9b model: '9b' right after a non-digit, so a future '19b' is not counted.
|
|
290
|
+
const NINE_B = `(LOWER(COALESCE(model, backend)) GLOB '*[^0-9]9b*' OR LOWER(COALESCE(model, backend)) GLOB '9b*')`;
|
|
291
|
+
const follow = await client.execute({
|
|
292
|
+
sql: `SELECT
|
|
293
|
+
SUM(CASE WHEN ${SERVED_LOCAL} THEN 1 ELSE 0 END) AS served,
|
|
294
|
+
SUM(CASE WHEN ${SERVED_LOCAL} AND ${NINE_B} THEN 1 ELSE 0 END) AS served_9b,
|
|
295
|
+
SUM(CASE WHEN used_cloud = 1 THEN 1 ELSE 0 END) AS cloud,
|
|
296
|
+
SUM(CASE WHEN used_cloud = 0 AND NOT (${SERVED_LOCAL}) AND NOT (${PLAN_REFUSAL}) THEN 1 ELSE 0 END) AS refused,
|
|
297
|
+
SUM(CASE WHEN used_cloud = 0 AND NOT (${SERVED_LOCAL}) AND ${PLAN_REFUSAL} THEN 1 ELSE 0 END) AS refused_by_plan
|
|
298
|
+
FROM infer_metrics ${followWhere}`,
|
|
299
|
+
args: whereArgs,
|
|
300
|
+
});
|
|
301
|
+
const followLayers = await client.execute({
|
|
302
|
+
sql: `SELECT COALESCE(refusal_layer, 'unrecorded') AS layer, COUNT(*) AS n
|
|
303
|
+
FROM infer_metrics ${followWhere} AND used_cloud = 0 AND NOT (${SERVED_LOCAL})
|
|
304
|
+
AND NOT (${PLAN_REFUSAL})
|
|
305
|
+
GROUP BY COALESCE(refusal_layer, 'unrecorded')`,
|
|
306
|
+
args: whereArgs,
|
|
307
|
+
});
|
|
308
|
+
const f = follow.rows[0];
|
|
309
|
+
const refused_by_layer = {};
|
|
310
|
+
for (const row of followLayers.rows) {
|
|
311
|
+
refused_by_layer[String(row.layer)] = Number(row.n ?? 0);
|
|
312
|
+
}
|
|
287
313
|
const r = agg.rows[0];
|
|
288
314
|
const by_model = {};
|
|
289
315
|
for (const row of byM.rows) {
|
|
@@ -310,6 +336,14 @@ export async function queryLocalSavings(sinceTs) {
|
|
|
310
336
|
first_ts: r.first_ts == null ? null : Number(r.first_ts),
|
|
311
337
|
last_ts: r.last_ts == null ? null : Number(r.last_ts),
|
|
312
338
|
by_model,
|
|
339
|
+
followups: {
|
|
340
|
+
served_local: Number(f.served ?? 0),
|
|
341
|
+
served_local_9b: Number(f.served_9b ?? 0),
|
|
342
|
+
refused: Number(f.refused ?? 0),
|
|
343
|
+
refused_by_plan: Number(f.refused_by_plan ?? 0),
|
|
344
|
+
cloud: Number(f.cloud ?? 0),
|
|
345
|
+
refused_by_layer,
|
|
346
|
+
},
|
|
313
347
|
};
|
|
314
348
|
}
|
|
315
349
|
catch (e) {
|
|
@@ -1693,7 +1693,7 @@ export async function sessionLoadContextHandler(args, options = {}) {
|
|
|
1693
1693
|
// skill budget.
|
|
1694
1694
|
if (typeof prompt === "string" && prompt.trim()) {
|
|
1695
1695
|
try {
|
|
1696
|
-
const {
|
|
1696
|
+
const { resolvePromptRouting } = await import("./skillRouting.js");
|
|
1697
1697
|
// The manifest's routing_version is the only version signal available
|
|
1698
1698
|
// here; without it a stale cached table would never be detected on
|
|
1699
1699
|
// this path, since there is no portal response to compare against.
|
|
@@ -1704,7 +1704,8 @@ export async function sessionLoadContextHandler(args, options = {}) {
|
|
|
1704
1704
|
// `prompt_triggers` in their own frontmatter and are matched here, on
|
|
1705
1705
|
// device, from bodies already cached for injection.
|
|
1706
1706
|
const scoped = await collectSkillTriggersOnThisMachine();
|
|
1707
|
-
const
|
|
1707
|
+
const routed = await resolvePromptRouting(prompt, Number.isFinite(manifestVersion) && manifestVersion > 0 ? manifestVersion : undefined, scoped?.triggers);
|
|
1708
|
+
const matched = routed.names.filter((name) => entitledSkillNames.has(name) || scoped?.localNames.has(name));
|
|
1708
1709
|
if (matched.length > 0) {
|
|
1709
1710
|
const shown = matched.slice(0, MAX_SYMPTOM_SKILLS);
|
|
1710
1711
|
const overflow = matched.length - shown.length;
|
|
@@ -1717,7 +1718,8 @@ export async function sessionLoadContextHandler(args, options = {}) {
|
|
|
1717
1718
|
symptomSkillSuffix = `\n\n**Symptom-triggered skills:** ${shown.join(", ")}` +
|
|
1718
1719
|
(overflow > 0 ? `, … ${overflow} more` : "") +
|
|
1719
1720
|
`\nThe first message matches these skills' trigger rules. Follow them before ` +
|
|
1720
|
-
`proposing any change.\n
|
|
1721
|
+
`proposing any change.\n` +
|
|
1722
|
+
(typeof routed.tableVersion === "number" ? `Routing table v${routed.tableVersion}.\n` : "");
|
|
1721
1723
|
// INLINE the top match's body rather than pointing at it. Naming a
|
|
1722
1724
|
// skill is not delivering it: bodies reach agents only as files under
|
|
1723
1725
|
// the canonical root, and hosts outside that mirror have no path to
|
|
@@ -2213,7 +2215,7 @@ export async function collectSkillTriggersOnThisMachine() {
|
|
|
2213
2215
|
*/
|
|
2214
2216
|
export async function runPromptRouteFromCache(prompt, loaded) {
|
|
2215
2217
|
const { routePrompt } = await import("./promptRouteHandler.js");
|
|
2216
|
-
const { resolvePromptSkillNames, _setStorage } = await import("./skillRouting.js");
|
|
2218
|
+
const { resolvePromptSkillNames, resolvePromptRouting, _setStorage } = await import("./skillRouting.js");
|
|
2217
2219
|
// The CLI is a fresh process per prompt: without storage wiring the keyword
|
|
2218
2220
|
// table can neither be read from disk (offline = dead routing) nor
|
|
2219
2221
|
// persisted after a fetch (every prompt = a network GET). The server paths
|
|
@@ -2221,6 +2223,7 @@ export async function runPromptRouteFromCache(prompt, loaded) {
|
|
|
2221
2223
|
_setStorage(async (key, value) => { await setSetting(key, value); }, async (key) => getSetting(key, ""));
|
|
2222
2224
|
return routePrompt(prompt, loaded, {
|
|
2223
2225
|
resolvePromptSkillNames,
|
|
2226
|
+
resolvePromptRouting,
|
|
2224
2227
|
collectTriggers: collectSkillTriggersOnThisMachine,
|
|
2225
2228
|
entitledNames: async () => {
|
|
2226
2229
|
try {
|
|
@@ -1744,6 +1744,17 @@ export async function runInfer(args, deps) {
|
|
|
1744
1744
|
break;
|
|
1745
1745
|
}
|
|
1746
1746
|
}
|
|
1747
|
+
// A classifier that answered none of this request's window reads
|
|
1748
|
+
// read none of them: the keyword net must not become their sole
|
|
1749
|
+
// guard (review round 16). The consecutive-ERROR breaker catches
|
|
1750
|
+
// this only from its third read, and a follow-up that re-sends the
|
|
1751
|
+
// same turns leaves one or two uncached windows (measured
|
|
1752
|
+
// 2026-09-25: a dead classifier, two failed reads, served locally
|
|
1753
|
+
// on the keyword net).
|
|
1754
|
+
if (budget.calls > 0 && budget.consecutiveErrors === budget.calls && !budget.tripped) {
|
|
1755
|
+
budget.tripped = true;
|
|
1756
|
+
attempts.push({ tier: "layer1", reason: "layer1_screen_all_reads_failed" });
|
|
1757
|
+
}
|
|
1747
1758
|
// A budget or breaker trip raises to UNCERTAIN whatever the cache
|
|
1748
1759
|
// held (text: cloud or refused; with an image: local only).
|
|
1749
1760
|
if (budget.tripped)
|
|
@@ -44,6 +44,44 @@ export const MAX_ROUTED_CHARS = 30_000;
|
|
|
44
44
|
* this bounds what may ride INLINE through a hook — anything larger must be
|
|
45
45
|
* our own offload file with an imperative pointer, not the host's silent one. */
|
|
46
46
|
export const HOOK_INLINE_SAFE_CHARS = 9_800;
|
|
47
|
+
/**
|
|
48
|
+
* Turns a host delivers as if a person typed them, although no person wrote
|
|
49
|
+
* them. Measured 2026-09-23 over 30 days of Claude Code sessions: most of the
|
|
50
|
+
* prompt hook's skill loads came from these turns, and few of those loads
|
|
51
|
+
* helped the task. A reviewer agent's report that mentions Supabase is not a
|
|
52
|
+
* request for the Supabase skill.
|
|
53
|
+
*
|
|
54
|
+
* Matched at the START only, never as a substring: a person who pastes a
|
|
55
|
+
* notification after their own words still routes. Tags are matched without
|
|
56
|
+
* their closing ">" so attributes do not hide them, and the relay line needs
|
|
57
|
+
* its colon so a sentence that merely begins with the same words still
|
|
58
|
+
* routes. The known cost is the reverse: a message that BEGINS with one of
|
|
59
|
+
* these markers, such as a raw notification pasted with nothing before it, is
|
|
60
|
+
* treated as machine-written.
|
|
61
|
+
*/
|
|
62
|
+
const MACHINE_TURN_PREFIXES = [
|
|
63
|
+
"<task-notification",
|
|
64
|
+
"<agent-message",
|
|
65
|
+
"Another Claude session sent a message:",
|
|
66
|
+
"<cross-session-message",
|
|
67
|
+
"This session is being continued from a previous conversation",
|
|
68
|
+
];
|
|
69
|
+
/**
|
|
70
|
+
* The part of a turn that routing reads. A person's text passes through
|
|
71
|
+
* unchanged, and a machine-written turn yields "". The exception is a finished
|
|
72
|
+
* BACKGROUND COMMAND: its one-line summary names the command the agent itself
|
|
73
|
+
* chose to run ("Build the app for the simulator"), so the summary routes.
|
|
74
|
+
* Its output never does.
|
|
75
|
+
*/
|
|
76
|
+
export function routableText(prompt) {
|
|
77
|
+
const text = (prompt || "").trimStart();
|
|
78
|
+
if (text.startsWith("<task-notification")) {
|
|
79
|
+
const head = text.split("<result>")[0];
|
|
80
|
+
const summary = /<summary>([\s\S]*?)<\/summary>/.exec(head)?.[1]?.trim() ?? "";
|
|
81
|
+
return summary.startsWith("Background command") ? summary : "";
|
|
82
|
+
}
|
|
83
|
+
return MACHINE_TURN_PREFIXES.some((marker) => text.startsWith(marker)) ? "" : prompt;
|
|
84
|
+
}
|
|
47
85
|
/**
|
|
48
86
|
* Match a prompt and return ONLY skills the caller does not already have.
|
|
49
87
|
*
|
|
@@ -51,15 +89,27 @@ export const HOOK_INLINE_SAFE_CHARS = 9_800;
|
|
|
51
89
|
* the thing under test.
|
|
52
90
|
*/
|
|
53
91
|
export async function routePrompt(prompt, loaded, deps) {
|
|
54
|
-
const
|
|
55
|
-
if (!
|
|
92
|
+
const supplied = (prompt || "").trim();
|
|
93
|
+
if (!supplied) {
|
|
56
94
|
return { names: [], alreadyLoaded: [], overflow: [], text: "No prompt supplied — nothing to route." };
|
|
57
95
|
}
|
|
96
|
+
const trimmed = routableText(supplied).trim();
|
|
97
|
+
if (!trimmed) {
|
|
98
|
+
return { names: [], alreadyLoaded: [], overflow: [], text: "Machine-written turn — not routed." };
|
|
99
|
+
}
|
|
58
100
|
const scoped = await deps.collectTriggers().catch(() => undefined);
|
|
59
101
|
const version = await deps.manifestVersion().catch(() => undefined);
|
|
60
102
|
let matched = [];
|
|
103
|
+
let tableVersion;
|
|
61
104
|
try {
|
|
62
|
-
|
|
105
|
+
if (deps.resolvePromptRouting) {
|
|
106
|
+
const routed = await deps.resolvePromptRouting(trimmed, version, scoped?.triggers);
|
|
107
|
+
matched = routed.names;
|
|
108
|
+
tableVersion = routed.tableVersion;
|
|
109
|
+
}
|
|
110
|
+
else {
|
|
111
|
+
matched = await deps.resolvePromptSkillNames(trimmed, version, scoped?.triggers);
|
|
112
|
+
}
|
|
63
113
|
}
|
|
64
114
|
catch (error) {
|
|
65
115
|
// Routing must never take down the turn that asked for it.
|
|
@@ -124,9 +174,12 @@ export async function routePrompt(prompt, loaded, deps) {
|
|
|
124
174
|
const overflowNote = overflow.length > 0
|
|
125
175
|
? `\n\nAlso matched, not injected: ${overflowShown.join(", ")}${overflow.length > overflowShown.length ? ` (+${overflow.length - overflowShown.length} more)` : ""}.`
|
|
126
176
|
: "";
|
|
177
|
+
// The table version makes a recorded load attributable after the table
|
|
178
|
+
// changes; without it, a transcript cannot say which rules produced a load.
|
|
179
|
+
const versionNote = typeof tableVersion === "number" ? `\n\nRouting table v${tableVersion}.` : "";
|
|
127
180
|
const header = `**Skills now active for this task:** ${delivered.join(", ")}\n\n` +
|
|
128
181
|
`These apply to the work you are about to do. Read and follow them before proceeding.` +
|
|
129
|
-
overflowNote;
|
|
182
|
+
overflowNote + versionNote;
|
|
130
183
|
return { names: delivered, alreadyLoaded, overflow, header, blocks, text: `${header}\n\n${blocks.join("\n\n")}` };
|
|
131
184
|
}
|
|
132
185
|
/**
|
|
@@ -129,6 +129,44 @@ function basisLine(s) {
|
|
|
129
129
|
"your host would have made, so whether all of it would have hit the cloud is an assumption. " +
|
|
130
130
|
`Read it as: at most this much displacement, of ${volumeWord} this token volume.`;
|
|
131
131
|
}
|
|
132
|
+
/** Plain names for the on-device screen's stages, as recorded in refusal_layer. */
|
|
133
|
+
const FOLLOWUP_STAGE_NAMES = {
|
|
134
|
+
isolated: "earlier turn alone",
|
|
135
|
+
prompt: "follow-up alone",
|
|
136
|
+
context: "turns together",
|
|
137
|
+
rules: "rules",
|
|
138
|
+
backstop: "keyword net",
|
|
139
|
+
budget: "screening limit",
|
|
140
|
+
};
|
|
141
|
+
/**
|
|
142
|
+
* Follow-ups: calls that carried the conversation. Shown even when none was
|
|
143
|
+
* served, because refusals are the part a user needs to see: a follow-up the
|
|
144
|
+
* screen refused went back to the host instead of being answered locally.
|
|
145
|
+
*/
|
|
146
|
+
export function followupLines(f) {
|
|
147
|
+
if (!f)
|
|
148
|
+
return [];
|
|
149
|
+
const total = f.served_local + f.refused + f.refused_by_plan + f.cloud;
|
|
150
|
+
if (total === 0)
|
|
151
|
+
return [];
|
|
152
|
+
const nineB = f.served_local > 0 ? ` (${fmt(f.served_local_9b)} by the 9b)` : "";
|
|
153
|
+
const lines = [
|
|
154
|
+
"",
|
|
155
|
+
" Follow-ups with your conversation:",
|
|
156
|
+
` ${fmt(f.served_local)} answered locally${nineB} · ${fmt(f.refused)} refused by the on-device screen · ${fmt(f.cloud)} sent to cloud`,
|
|
157
|
+
];
|
|
158
|
+
if (f.refused_by_plan > 0) {
|
|
159
|
+
lines.push(` ${fmt(f.refused_by_plan)} refused by your plan's multi-turn limits`);
|
|
160
|
+
}
|
|
161
|
+
// Most refusals first; ties in a fixed stage order, so the line never depends on SQL row order.
|
|
162
|
+
const order = (k) => { const i = Object.keys(FOLLOWUP_STAGE_NAMES).indexOf(k); return i < 0 ? 99 : i; };
|
|
163
|
+
const stages = Object.entries(f.refused_by_layer).filter(([, n]) => n > 0)
|
|
164
|
+
.sort((a, b) => b[1] - a[1] || order(a[0]) - order(b[0]) || a[0].localeCompare(b[0]));
|
|
165
|
+
if (stages.length > 0) {
|
|
166
|
+
lines.push(` Refusals by stage: ${stages.map(([k, n]) => `${FOLLOWUP_STAGE_NAMES[k] ?? k} ${fmt(n)}`).join(" · ")}`);
|
|
167
|
+
}
|
|
168
|
+
return lines;
|
|
169
|
+
}
|
|
132
170
|
export function renderSavings(s, period, customDays) {
|
|
133
171
|
const label = customDays !== undefined && Number.isFinite(customDays) && customDays > 0
|
|
134
172
|
? `LAST ${Math.floor(customDays)} DAYS`
|
|
@@ -144,6 +182,7 @@ export function renderSavings(s, period, customDays) {
|
|
|
144
182
|
lines.push(period === "session"
|
|
145
183
|
? " Delegate work with prism_infer, or use session_task_route to pick targets automatically."
|
|
146
184
|
: " Once prism starts serving locally, displaced token volume shows up here.");
|
|
185
|
+
lines.push(...followupLines(s.followups));
|
|
147
186
|
return { text: lines.join("\n"), data: { ...s, period } };
|
|
148
187
|
}
|
|
149
188
|
const totalRouted = s.local_calls + s.cloud_calls;
|
|
@@ -162,6 +201,7 @@ export function renderSavings(s, period, customDays) {
|
|
|
162
201
|
lines.push(` ${name}: ${fmt(m.calls)} call(s), ~${abbreviate(t)} tokens`);
|
|
163
202
|
}
|
|
164
203
|
}
|
|
204
|
+
lines.push(...followupLines(s.followups));
|
|
165
205
|
lines.push("");
|
|
166
206
|
lines.push(` ${basisLine(s)}`);
|
|
167
207
|
const caveats = caveatsFor(s);
|
|
@@ -112,7 +112,9 @@ export function extractSkillTriggers(skillName, content) {
|
|
|
112
112
|
// idiom is a plain data-property op for any key. Same treatment at every
|
|
113
113
|
// trigger accumulator in this file and in ledgerHandlers' merge.
|
|
114
114
|
const result = { triggers: Object.create(null), errors: [] };
|
|
115
|
-
|
|
115
|
+
// A SKILL.md saved with Windows line endings failed the `^---\n` match and
|
|
116
|
+
// loaded with no triggers and no error. skillDigest already accepts \r\n.
|
|
117
|
+
const frontmatter = content.replace(/\r\n?/g, "\n").match(/^---\n([\s\S]*?)\n---/);
|
|
116
118
|
if (!frontmatter)
|
|
117
119
|
return result;
|
|
118
120
|
const body = frontmatter[1];
|
|
@@ -128,7 +130,9 @@ export function extractSkillTriggers(skillName, content) {
|
|
|
128
130
|
else {
|
|
129
131
|
const blockStart = body.match(/^prompt_triggers:\s*$/m);
|
|
130
132
|
if (blockStart) {
|
|
131
|
-
|
|
133
|
+
// The match's own index, not indexOf: the key text can also end an
|
|
134
|
+
// earlier line (a description), and indexOf would start there.
|
|
135
|
+
const after = body.slice((blockStart.index ?? 0) + blockStart[0].length);
|
|
132
136
|
for (const line of after.split("\n")) {
|
|
133
137
|
// Stop at the next top-level key: an unterminated list must not swallow
|
|
134
138
|
// the rest of the frontmatter and turn `description:` into a trigger.
|
|
@@ -336,7 +336,8 @@ export function stripQuotedEvidenceForRouting(prompt, promptKeywords = {}) {
|
|
|
336
336
|
// text (adversarial review, confirmed with a repro). Line-anchoring means
|
|
337
337
|
// eating text now requires two line-start fences — which IS a fenced block.
|
|
338
338
|
//
|
|
339
|
-
//
|
|
339
|
+
// A removed FENCED BLOCK must sever BOTH proximity-window classes in the
|
|
340
|
+
// real table:
|
|
340
341
|
// - `.{0,N}` windows: `.` does not cross \n (no pattern uses the s-flag),
|
|
341
342
|
// so a newline severs them.
|
|
342
343
|
// - `\s*`/`\s+`-glued windows (34 of 58 live patterns, e.g.
|
|
@@ -346,7 +347,29 @@ export function stripQuotedEvidenceForRouting(prompt, promptKeywords = {}) {
|
|
|
346
347
|
// includes \x1F (unit separator): non-space (blocks \s runs), non-word
|
|
347
348
|
// (leaves \b semantics as a space would), and severed from dot-windows
|
|
348
349
|
// by the flanking newlines.
|
|
350
|
+
// A removed skill NAME is replaced differently — see neutralize below.
|
|
349
351
|
const SEVER = '\n\x1f\n';
|
|
352
|
+
// A stripped name keeps its length and each character's kind: ASCII
|
|
353
|
+
// lowercase letters become "q", ASCII uppercase "Q", ASCII digits "0", and
|
|
354
|
+
// every other character (separators, non-ASCII letters) stays as it was. \b, \w, \d, \s, ., [a-z] and the like read the same at every
|
|
355
|
+
// position as on the raw text, so a trigger stops matching only if it needs
|
|
356
|
+
// the identity of the name's letters — which is what stripping exists to remove.
|
|
357
|
+
// Earlier masks leaked: a line break cut "Draft an ABA <name> plan" apart
|
|
358
|
+
// (review round 1), a non-word \x1F run cut [-\w] windows (round 3), and
|
|
359
|
+
// "_" cut [ a-z-] windows (round 4).
|
|
360
|
+
// Known limitation, stated as a class: the mask keeps each ASCII letter's
|
|
361
|
+
// and digit's kind but not its identity. A trigger that can tell one letter (or one
|
|
362
|
+
// digit) from another — a word, a letter range such as [n-s], the mask
|
|
363
|
+
// letters themselves, or a backreference such as (\w)\1 — may match a
|
|
364
|
+
// stripped name differently, in either direction. A trigger that cannot
|
|
365
|
+
// tell them apart matches exactly as on the raw text. No mask smaller than
|
|
366
|
+
// the alphabet avoids this; triggers come from the routing table and
|
|
367
|
+
// account owners, so it is disclosed and pinned in tests, not defended.
|
|
368
|
+
// Also disclosed: a name glued to a following letter or digit other than
|
|
369
|
+
// a plural "s" ("…-protocol7") is not recognized as the name, so its
|
|
370
|
+
// trigger words still route; the segment anchor below is what keeps
|
|
371
|
+
// "fix-ci" from firing inside "prefix-ci".
|
|
372
|
+
const neutralize = (name) => name.replace(/[a-z]/g, 'q').replace(/[A-Z]/g, 'Q').replace(/[0-9]/g, '0');
|
|
350
373
|
let out = prompt
|
|
351
374
|
.replace(/^[ \t]*```[^\n]*\n[\s\S]*?\n[ \t]*```[ \t]*$/gm, SEVER)
|
|
352
375
|
.replace(/^[ \t]*~~~[^\n]*\n[\s\S]*?\n[ \t]*~~~[ \t]*$/gm, SEVER);
|
|
@@ -374,8 +397,20 @@ export function stripQuotedEvidenceForRouting(prompt, promptKeywords = {}) {
|
|
|
374
397
|
if (typeof n === "string")
|
|
375
398
|
names.add(n);
|
|
376
399
|
}
|
|
377
|
-
//
|
|
378
|
-
|
|
400
|
+
// Protected skills are never prompt-routed, so the table above does not name
|
|
401
|
+
// them, but every pasted startup log lists them and a protected name can
|
|
402
|
+
// carry another skill's trigger word: "aba-precision-protocol" satisfied the
|
|
403
|
+
// clinical \baba\b trigger. They are exact names, not English.
|
|
404
|
+
for (const n of REQUIRED_PROTECTED_SKILL_NAMES)
|
|
405
|
+
names.add(n);
|
|
406
|
+
// Every name is matched against the SAME unmasked text and the union of
|
|
407
|
+
// the spans is masked once below. Masking name by name let a longer name
|
|
408
|
+
// consume the head of an overlapping one, which then no longer matched and
|
|
409
|
+
// left its tail — and its trigger words — unmasked (new review, cycle 1).
|
|
410
|
+
// A contained name is covered by the union, so order does not matter.
|
|
411
|
+
const source = out;
|
|
412
|
+
const masked = new Uint8Array(source.length);
|
|
413
|
+
for (const name of names) {
|
|
379
414
|
// Bounds mirror the routing table's own name policy (≤128 chars). An
|
|
380
415
|
// overlong or hostile name from a poisoned table must degrade to
|
|
381
416
|
// "not stripped", never to a thrown SyntaxError that kills routing for
|
|
@@ -407,11 +442,26 @@ export function stripQuotedEvidenceForRouting(prompt, promptKeywords = {}) {
|
|
|
407
442
|
continue;
|
|
408
443
|
try {
|
|
409
444
|
const escaped = name.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
|
410
|
-
|
|
445
|
+
const re = new RegExp(`(?<![A-Za-z0-9])${escaped}s?(?![A-Za-z0-9])`, 'gi');
|
|
446
|
+
// exec, restarting one unit after each match START, so an occurrence
|
|
447
|
+
// that overlaps an earlier one of the same name ("aba-aba" twice in
|
|
448
|
+
// "aba-aba-aba") is found too; matchAll skips it. lastIndex strictly
|
|
449
|
+
// increases, so this stays linear for a bounded name length.
|
|
450
|
+
for (let m = re.exec(source); m; m = re.exec(source)) {
|
|
451
|
+
masked.fill(1, m.index, m.index + m[0].length);
|
|
452
|
+
re.lastIndex = m.index + 1;
|
|
453
|
+
}
|
|
411
454
|
}
|
|
412
455
|
catch { /* skip unbuildable names — same policy as the matcher */ }
|
|
413
456
|
}
|
|
414
|
-
|
|
457
|
+
if (!masked.includes(1))
|
|
458
|
+
return source;
|
|
459
|
+
// split('') indexes UTF-16 code units, the same units matchAll reports.
|
|
460
|
+
const units = source.split('');
|
|
461
|
+
for (let i = 0; i < units.length; i++)
|
|
462
|
+
if (masked[i])
|
|
463
|
+
units[i] = neutralize(units[i]);
|
|
464
|
+
return units.join('');
|
|
415
465
|
}
|
|
416
466
|
/**
|
|
417
467
|
* Verbatim port of portal resolve/route.ts prompt-matching block + the sort
|
|
@@ -470,8 +520,16 @@ export function _applyPromptRouting(base, prompt, promptKeywords) {
|
|
|
470
520
|
* indefinitely and never detect drift). The skill manifest carries one.
|
|
471
521
|
*/
|
|
472
522
|
export async function resolvePromptSkillNames(prompt, expectVersion, scopedTriggers) {
|
|
523
|
+
return (await resolvePromptRouting(prompt, expectVersion, scopedTriggers)).names;
|
|
524
|
+
}
|
|
525
|
+
/**
|
|
526
|
+
* resolvePromptSkillNames, plus which routing table produced the match. A load
|
|
527
|
+
* recorded without its table version cannot be attributed after the table
|
|
528
|
+
* changes, so callers that show routed skills also show the version.
|
|
529
|
+
*/
|
|
530
|
+
export async function resolvePromptRouting(prompt, expectVersion, scopedTriggers) {
|
|
473
531
|
if (!prompt)
|
|
474
|
-
return [];
|
|
532
|
+
return { names: [] };
|
|
475
533
|
const kw = await fetchKeywordTable(expectVersion);
|
|
476
534
|
// Scoped triggers must still route when the PUBLIC table is unavailable:
|
|
477
535
|
// they are declared in skill bodies already on this machine and owe nothing
|
|
@@ -479,7 +537,7 @@ export async function resolvePromptSkillNames(prompt, expectVersion, scopedTrigg
|
|
|
479
537
|
// depend on a public file it can never appear in.
|
|
480
538
|
const publicKeywords = kw?.prompt_keywords ?? {};
|
|
481
539
|
if (!kw && !scopedTriggers)
|
|
482
|
-
return [];
|
|
540
|
+
return { names: [] };
|
|
483
541
|
// NULL-PROTOTYPE, not a literal (round-4 review): with a plain object, a
|
|
484
542
|
// scoped pattern whose TEXT is an inherited property name made both sides
|
|
485
543
|
// of the merge below misbehave — `combined['constructor'] ?? []` read the
|
|
@@ -500,7 +558,10 @@ export async function resolvePromptSkillNames(prompt, expectVersion, scopedTrigg
|
|
|
500
558
|
continue;
|
|
501
559
|
combined[pattern] = [...(combined[pattern] ?? []), ...clean];
|
|
502
560
|
}
|
|
503
|
-
return
|
|
561
|
+
return {
|
|
562
|
+
names: _applyPromptRouting([], stripQuotedEvidenceForRouting(prompt, combined), combined).map((s) => s.name),
|
|
563
|
+
tableVersion: kw?.version,
|
|
564
|
+
};
|
|
504
565
|
}
|
|
505
566
|
/**
|
|
506
567
|
* Free tier resolves to an empty set portal-side, so adding prompt-matched
|
|
@@ -89,6 +89,35 @@ const HOST_TOOL_ACTION_GROUPS = [
|
|
|
89
89
|
],
|
|
90
90
|
},
|
|
91
91
|
];
|
|
92
|
+
/**
|
|
93
|
+
* Requirements the host states about its own task. Hosts write the task
|
|
94
|
+
* description, and many say outright that the work needs host tools or
|
|
95
|
+
* reserved judgment ("Needs host tools to inspect…", "Reserved security
|
|
96
|
+
* judgment: …"). The local worker cannot run tools, so these are hard host
|
|
97
|
+
* boundaries. The tool pattern must start at host/repository/repo/filesystem so
|
|
98
|
+
* "open-source tools" does not count.
|
|
99
|
+
*
|
|
100
|
+
* Any occurrence counts — negated, contrasted, or merely mentioned. Reading
|
|
101
|
+
* English negation with patterns kept misrouting ("do not skip host tools",
|
|
102
|
+
* "no host tools should be omitted"), and the costs are lopsided: a wrong
|
|
103
|
+
* host route costs one host turn, a wrong claw route a delegation that comes
|
|
104
|
+
* back refused or rejected. On two months of real routes, ignoring negation
|
|
105
|
+
* changed no decision.
|
|
106
|
+
*/
|
|
107
|
+
// Modifier chains are bounded ({0,3}): an unbounded chain restarted at every
|
|
108
|
+
// "repository" and made a near-miss input quadratic. Modifiers and "reserved
|
|
109
|
+
// … judgment" qualifiers are the closed sets seen in two months of real task
|
|
110
|
+
// descriptions, so prose like "she reserved her judgment" does not match, and
|
|
111
|
+
// "and" counts only before another modifier ("host git and shell tools", not
|
|
112
|
+
// "the host and tools").
|
|
113
|
+
const SELF_DECLARED_HOST_REQUIREMENTS = [
|
|
114
|
+
/\b(?:host|repository|repo|filesystem)(?:[- /](?:and[- ])?(?:side|repository|filesystem|file|source|shell|git|browser|test|web|docker|documentation|ci|process|external|deployment)){0,3}[- ]tools?\b/i,
|
|
115
|
+
/\breserved(?:[- /](?:adversarial|security|compliance|release|review|clinical|product|host|auth|phi|safety|tenant|lifecycle|isolation)){0,3}[- ]judge?ment\b/i,
|
|
116
|
+
/\b(?:security|compliance|tenant[- ]isolation)[- ]judge?ment\b/i,
|
|
117
|
+
];
|
|
118
|
+
function hasSelfDeclaredHostRequirement(description) {
|
|
119
|
+
return SELF_DECLARED_HOST_REQUIREMENTS.some((pattern) => pattern.test(description));
|
|
120
|
+
}
|
|
92
121
|
/** Complexity signals that should select 27B when the task is bounded. */
|
|
93
122
|
const HIGH_COMPLEXITY_KEYWORDS = [
|
|
94
123
|
"complex logic", "algorithm", "dynamic programming", "constraint solver",
|
|
@@ -170,6 +199,8 @@ function assessDelegability(args) {
|
|
|
170
199
|
reasons.push("reserved host judgment");
|
|
171
200
|
if (toolWorkflowHits > 0)
|
|
172
201
|
reasons.push("host tools or external state required");
|
|
202
|
+
if (hasSelfDeclaredHostRequirement(description))
|
|
203
|
+
reasons.push("host requirement stated in the task");
|
|
173
204
|
if (toolActionGroups.length >= HOST_TOOL_ACTION_GROUP_THRESHOLD) {
|
|
174
205
|
reasons.push(`host workflow actions: ${toolActionGroups.join(", ")}`);
|
|
175
206
|
}
|
|
@@ -362,13 +393,19 @@ export function computeRoute(args) {
|
|
|
362
393
|
const { task_description, files_involved, estimated_scope } = args;
|
|
363
394
|
// ── Cold-start / edge case: insufficient input ──
|
|
364
395
|
if (!task_description || task_description.trim().length < 10) {
|
|
396
|
+
// A stated host requirement is a hard boundary on this path too: the
|
|
397
|
+
// handler's experience bias and local tie-break both gate on the flag.
|
|
398
|
+
const hostRequirement = hasSelfDeclaredHostRequirement(task_description ?? "");
|
|
365
399
|
return {
|
|
366
400
|
target: "host",
|
|
367
|
-
confidence: 0.5,
|
|
401
|
+
confidence: hostRequirement ? HARD_HOST_BOUNDARY_CONFIDENCE : 0.5,
|
|
368
402
|
needs_history: false,
|
|
369
403
|
complexity_score: 5,
|
|
370
|
-
rationale:
|
|
404
|
+
rationale: hostRequirement
|
|
405
|
+
? "Host boundary: host requirement stated in the task."
|
|
406
|
+
: "Insufficient information for confident routing. Defaulting to host model.",
|
|
371
407
|
recommended_tool: null,
|
|
408
|
+
...(hostRequirement ? { _hardHostBoundary: true } : {}),
|
|
372
409
|
};
|
|
373
410
|
}
|
|
374
411
|
// ── Compute individual signals ──
|
|
@@ -10,7 +10,7 @@ export const TOOL_CALL_BLEED_RE = /<\|tool_call\|>|<\|tool_call_end\|>/;
|
|
|
10
10
|
* @param stripped Response AFTER think-stripping (use stripThink first)
|
|
11
11
|
* @param thinkOnly True if the response was only <think> blocks with no answer
|
|
12
12
|
* @param finishReason Ollama's finish_reason if available (e.g. "length" = truncated)
|
|
13
|
-
* @param mode Inference mode — "route"
|
|
13
|
+
* @param mode Inference mode — "route": empty only when blank; "chat": empty only with no letter or digit; "code"/unset: 4 chars or fewer
|
|
14
14
|
*/
|
|
15
15
|
export function passesQualityGate(stripped, thinkOnly, finishReason, mode) {
|
|
16
16
|
// Signal 1: Think-only — model reasoned but produced no answer (check before empty)
|
|
@@ -19,9 +19,17 @@ export function passesQualityGate(stripped, thinkOnly, finishReason, mode) {
|
|
|
19
19
|
}
|
|
20
20
|
// Signal 2: Mode-aware empty floor.
|
|
21
21
|
// Route legitimately returns 1–4 char labels ("P1", "YES", "CO4", "FIXED").
|
|
22
|
-
//
|
|
23
|
-
|
|
24
|
-
|
|
22
|
+
// Chat answers can be one short value: a follow-up asking "what is x times
|
|
23
|
+
// 6?" is correctly answered "42". Measured 2026-09-24: under the old <5
|
|
24
|
+
// floor, correct chat answers "16" and "36" failed here and, on a paid
|
|
25
|
+
// plan, were thrown away and re-asked of the cloud. So chat is empty only
|
|
26
|
+
// when it has no letter or digit at all.
|
|
27
|
+
// Code keeps <5: a 1–4 char code answer ("Hi", "DONE") is not an answer.
|
|
28
|
+
const trimmed = stripped.trim();
|
|
29
|
+
const empty = mode === "route" ? trimmed.length === 0
|
|
30
|
+
: mode === "chat" ? !/[\p{L}\p{N}]/u.test(trimmed)
|
|
31
|
+
: trimmed.length <= 4;
|
|
32
|
+
if (empty) {
|
|
25
33
|
return { pass: false, reason: "empty_response" };
|
|
26
34
|
}
|
|
27
35
|
// Signal 3: Hard truncation — Ollama reports finish_reason="length"
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "prism-mcp-server",
|
|
3
|
-
"version": "20.21.
|
|
3
|
+
"version": "20.21.15",
|
|
4
4
|
"mcpName": "io.github.dcostenco/prism-coder",
|
|
5
5
|
"description": "Persistent session memory for AI coding agents that never leaves your machine — including the on-device model that reasons over it. Restores your prior decisions, open TODOs, and changed files across sessions; adds associative recall of related past work, semantic drift detection, and local inference. Local-first by default. Works with Claude Code, Cursor, and Codex.",
|
|
6
6
|
"module": "index.ts",
|