prism-mcp-server 20.18.2 → 20.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +111 -2
- package/dist/cli.js +2 -1
- package/dist/config.js +9 -9
- package/dist/dashboard/server.js +1 -1
- package/dist/localFirstPolicy.js +2 -0
- package/dist/modelConvergeRunner.js +14 -0
- package/dist/scholar/webScholar.js +47 -14
- package/dist/tools/definitions.js +1 -1
- package/dist/tools/ledgerHandlers.js +20 -1
- package/dist/tools/prismInferHandler.js +578 -82
- package/dist/tools/skillRouting.js +6 -6
- package/dist/tools/taskRouterHandler.js +47 -0
- package/dist/utils/braveApi.js +53 -11
- package/dist/utils/entitlements.js +32 -0
- package/dist/utils/layer1.js +38 -16
- package/dist/utils/modelConverge.js +47 -2
- package/dist/utils/portalError.js +41 -0
- package/dist/utils/safetyGate.js +11 -1
- package/dist/utils/synaluxSearch.js +4 -1
- package/package.json +1 -1
- package/dist/utils/googleSearchApi.js +0 -42
|
@@ -19,19 +19,20 @@
|
|
|
19
19
|
* directly — all cloud traffic goes via the synalux portal so billing,
|
|
20
20
|
* tier gating, and HIPAA audit are enforced in one place.
|
|
21
21
|
*/
|
|
22
|
+
import { createHash } from "node:crypto";
|
|
22
23
|
import { pickLocalModel, fmtGb, MODEL_TIERS, resolveOllamaName } from "../utils/modelPicker.js";
|
|
23
24
|
import { getSynaluxJwt, invalidateSynaluxJwt } from "../utils/synaluxJwt.js";
|
|
24
25
|
import { getAvailableMemoryBytes } from "../utils/availableMemory.js";
|
|
25
26
|
import { downscaleImages, productionDownscaleDeps, resolveMaxImageEdge } from "../utils/imageDownscale.js";
|
|
26
27
|
import { PRISM_SYNALUX_BASE_URL, PRISM_LOCAL_LLM_URL, PRISM_USER_ID, SYNALUX_CONFIGURED, } from "../config.js";
|
|
27
28
|
import { debugLog } from "../utils/logger.js";
|
|
28
|
-
import { getEntitlements, clampCeiling } from "../utils/entitlements.js";
|
|
29
|
+
import { getEntitlements, clampCeiling, multiTurnPolicy, ABSOLUTE_MULTI_TURN } from "../utils/entitlements.js";
|
|
29
30
|
import { ddLog } from "../utils/ddLogger.js";
|
|
30
31
|
import { stripThink } from "../utils/thinkStrip.js";
|
|
31
32
|
import { passesQualityGate } from "../utils/qualityGate.js";
|
|
32
33
|
import { applyDeterministicCodingRepairs, buildCodingRepairPrompt, passesCodingQualityGate, } from "../utils/codingQualityPolicy.js";
|
|
33
34
|
import { checkInputSafety, checkOutputSafety } from "../utils/safetyGate.js";
|
|
34
|
-
import { callLayer1 as defaultCallLayer1, keywordBackstop, reservedCategory } from "../utils/layer1.js";
|
|
35
|
+
import { callLayer1 as defaultCallLayer1, classifyDeterministicLayer1, keywordBackstop, reservedCategory, MAX_CLASSIFIER_PROMPT_LENGTH } from "../utils/layer1.js";
|
|
35
36
|
import { recordInference, recordThinkOnlyRetry, formatInferenceMetrics, estimateTokens } from "../utils/inferenceMetrics.js";
|
|
36
37
|
import { appendInferMetric } from "../storage/inferMetricsLedger.js";
|
|
37
38
|
import { getStorage } from "../storage/index.js";
|
|
@@ -92,6 +93,215 @@ export const VISION_SYSTEM_PROMPT = "You read developer screenshots precisely. A
|
|
|
92
93
|
+ "Identify the most specific, innermost location the evidence points to rather "
|
|
93
94
|
+ "than the first thing you see.";
|
|
94
95
|
export const MAX_INFER_IMAGES = 8;
|
|
96
|
+
/** Tokens a history adds to the prompt body: content plus ~8 tokens of
|
|
97
|
+
* chat-template framing per message (role markers and separators). */
|
|
98
|
+
export function historyTokenEstimate(history) {
|
|
99
|
+
if (!history?.length)
|
|
100
|
+
return 0;
|
|
101
|
+
return history.reduce((n, t) => n + estimateTokens(t.content) + 8, 0);
|
|
102
|
+
}
|
|
103
|
+
/** History and current prompt as ONE text for the deterministic screens
|
|
104
|
+
* (reserved-category attribution, keyword backstop). The semantic classifier
|
|
105
|
+
* reads each turn alone, then each turn and the prompt in context — see the
|
|
106
|
+
* Layer 1 block and contextWindows. */
|
|
107
|
+
function screenedText(args) {
|
|
108
|
+
const history = args.messages ?? [];
|
|
109
|
+
return history.length ? [...history.map(t => t.content), args.prompt].join("\n") : args.prompt;
|
|
110
|
+
}
|
|
111
|
+
/** Role-labelled transcript, current prompt last. contextWindows() are its
|
|
112
|
+
* per-turn tails; the deterministic screens use screenedText. Exported for
|
|
113
|
+
* tests. */
|
|
114
|
+
export function screeningTranscript(args) {
|
|
115
|
+
return [...(args.messages ?? []), { role: "user", content: args.prompt }]
|
|
116
|
+
.map(t => `${t.role === "user" ? "User" : "Assistant"}: ${t.content}`)
|
|
117
|
+
.join("\n");
|
|
118
|
+
}
|
|
119
|
+
/** One context window per turn (and one for the current prompt): the last
|
|
120
|
+
* HISTORY_TURN_WINDOW_CHARS chars of the role-labelled transcript ENDING at
|
|
121
|
+
* that turn. A context read can only RAISE the verdict: intent spread across
|
|
122
|
+
* turns that each read clean alone (measured 2026-09-16: the two halves of a
|
|
123
|
+
* restraint request in separate user turns, clean apart, reserved together)
|
|
124
|
+
* is caught by the window ending at the later half — when both parts fall
|
|
125
|
+
* inside one window, i.e. the last HISTORY_TURN_WINDOW_CHARS chars of the
|
|
126
|
+
* transcript up to the END of the later part's turn; windows exist only at
|
|
127
|
+
* turn ends, so a later part at the start of a long turn, or parts further
|
|
128
|
+
* apart than that, are never in one read (the limit). No context read ever
|
|
129
|
+
* lowers or replaces an isolated verdict, so no window containing OTHER
|
|
130
|
+
* turns adjudicates a turn (review rounds 12–22: every "defer UNCERTAIN to
|
|
131
|
+
* context" variant was measured bypassable by a classifier-directed note in
|
|
132
|
+
* whichever window decided; round 23 restored these windows as raise-only
|
|
133
|
+
* reads after dropping them left a >3,600-char prefix unscreened for
|
|
134
|
+
* cross-turn intent). Anchored on the turn's end, so a later prompt is
|
|
135
|
+
* never in an earlier turn's window, and an eviction at the plan cap
|
|
136
|
+
* changes only the windows the evicted turn was in — one or two for long
|
|
137
|
+
* turns, every one while the whole transcript still fits in one window
|
|
138
|
+
* (a cost, not a safety property). */
|
|
139
|
+
export function contextWindows(args) {
|
|
140
|
+
const labelled = [...(args.messages ?? []), { role: "user", content: args.prompt }]
|
|
141
|
+
.map(t => `${t.role === "user" ? "User" : "Assistant"}: ${t.content}`);
|
|
142
|
+
const out = [];
|
|
143
|
+
let transcript = "";
|
|
144
|
+
for (const line of labelled) {
|
|
145
|
+
transcript = transcript ? `${transcript}\n${line}` : line;
|
|
146
|
+
out.push(transcript.slice(-HISTORY_TURN_WINDOW_CHARS));
|
|
147
|
+
}
|
|
148
|
+
return out;
|
|
149
|
+
}
|
|
150
|
+
/** Most severe of two Layer 1 verdicts. A reserved turn anywhere in the
|
|
151
|
+
* conversation is a reserved conversation. */
|
|
152
|
+
const LAYER1_SEVERITY = {
|
|
153
|
+
OBVIOUS_NOT_RESERVED: 0, UNCERTAIN_LENGTH: 1, ERROR: 2, UNCERTAIN: 3, OBVIOUS_RESERVED: 4,
|
|
154
|
+
};
|
|
155
|
+
function worseLayer1Verdict(a, b) {
|
|
156
|
+
return LAYER1_SEVERITY[b] > LAYER1_SEVERITY[a] ? b : a;
|
|
157
|
+
}
|
|
158
|
+
/** The whole conversation for the cloud client: history plus the current
|
|
159
|
+
* turn as its last entry. Empty object when there is no history, so a
|
|
160
|
+
* single-turn call still sends the bare `prompt` it always did. */
|
|
161
|
+
/** Trailing history argument for the local call — present ONLY when there is
|
|
162
|
+
* history, so a single-turn call keeps the exact arity it always had (mocks
|
|
163
|
+
* and harnesses that pin the argument list stay valid). */
|
|
164
|
+
/** Layer 1 classifies up to 4,000 chars in full and excerpts beyond that.
|
|
165
|
+
* History turns are cut into overlapping windows under that limit so every
|
|
166
|
+
* region of every turn is classified. Exported for tests. */
|
|
167
|
+
export const HISTORY_TURN_WINDOW_CHARS = 3_600;
|
|
168
|
+
export const HISTORY_TURN_WINDOW_OVERLAP = 200;
|
|
169
|
+
/** The deterministic co-occurrence rules (restraint+document, diagnos+determine…)
|
|
170
|
+
* are proximity rules: over a whole 20k-char pasted file, "diagnose" and
|
|
171
|
+
* "determine" 14k chars apart fired one (measured 2026-09-16, +20% of real
|
|
172
|
+
* source files refused). They run over 7,200-char windows advancing by
|
|
173
|
+
* 3,400 (the classifier stride), so ANY two terms up to 3,800 chars apart —
|
|
174
|
+
* more than one classifier window, about one prompt — share a window
|
|
175
|
+
* wherever they sit in the turn (a 7,000-char stride left a pair straddling
|
|
176
|
+
* the boundary in no window: round-5 review). Wider apart than that is not
|
|
177
|
+
* one intent. */
|
|
178
|
+
export const DETERMINISTIC_FLOOR_WINDOW_CHARS = 7_200;
|
|
179
|
+
export const DETERMINISTIC_FLOOR_WINDOW_OVERLAP = 3_800;
|
|
180
|
+
export function windowsOf(content, size, overlap) {
|
|
181
|
+
if (content.length <= size)
|
|
182
|
+
return [content];
|
|
183
|
+
const isHigh = (i) => { const c = content.charCodeAt(i); return c >= 0xd800 && c <= 0xdbff; };
|
|
184
|
+
const isLow = (i) => { const c = content.charCodeAt(i); return c >= 0xdc00 && c <= 0xdfff; };
|
|
185
|
+
const out = [];
|
|
186
|
+
const step = size - overlap;
|
|
187
|
+
for (let i = 0; i < content.length; i += step) {
|
|
188
|
+
// Never cut a surrogate pair: a window that starts on a low or ends
|
|
189
|
+
// on a high surrogate is malformed text for the classifier.
|
|
190
|
+
let start = i;
|
|
191
|
+
if (start > 0 && isLow(start))
|
|
192
|
+
start += 1;
|
|
193
|
+
let end = Math.min(content.length, start + size);
|
|
194
|
+
if (end < content.length && isHigh(end - 1))
|
|
195
|
+
end += 1;
|
|
196
|
+
out.push(content.slice(start, end));
|
|
197
|
+
if (end >= content.length)
|
|
198
|
+
break;
|
|
199
|
+
}
|
|
200
|
+
return out;
|
|
201
|
+
}
|
|
202
|
+
export function historyTurnWindows(content) {
|
|
203
|
+
return windowsOf(content, HISTORY_TURN_WINDOW_CHARS, HISTORY_TURN_WINDOW_OVERLAP);
|
|
204
|
+
}
|
|
205
|
+
/** Verdict cache for history windows, keyed by a hash of model + text — no
|
|
206
|
+
* turn text is retained. A follow-up re-sends the same accepted turns, so
|
|
207
|
+
* without this an n-turn conversation re-screens every prior turn on every
|
|
208
|
+
* call (quadratic classifier work; review 2026-09-16). ERROR verdicts are
|
|
209
|
+
* transient and never cached; a window classified WITH images never goes
|
|
210
|
+
* through here (the key has no image bytes in it). */
|
|
211
|
+
/** Aggregate classifier-call budget for one request's history screen — a
|
|
212
|
+
* safety net at the STRUCTURAL maximum (49 turns / 128k chars of history
|
|
213
|
+
* alone: 49 base windows + 37 extra for the long ones = 86; one context
|
|
214
|
+
* window per turn and one for the prompt = 50; 136 budgeted misses, 137
|
|
215
|
+
* calls with the prompt's own), not a plan-level limit: every shape the
|
|
216
|
+
* caps allow fits under it with 33 calls of headroom, so a paid call never
|
|
217
|
+
* trips it, and a runaway loop cannot exceed it. Beyond it the screen
|
|
218
|
+
* raises to UNCERTAIN. The real bounds are the plan caps (enterprise: 30
|
|
219
|
+
* turns alone + 31 context + 1 ≈ 62 calls on a cold cache; a follow-up that
|
|
220
|
+
* appends pays its new turns alone, the prompt alone and their context
|
|
221
|
+
* windows; one that evicts the oldest turn also pays every context window
|
|
222
|
+
* that turn was in) and the consecutive-ERROR breaker below (review rounds
|
|
223
|
+
* 13–23). */
|
|
224
|
+
export let LAYER1_SCREEN_CALL_BUDGET = 170;
|
|
225
|
+
export function _setScreenCallBudgetForTest(n) { LAYER1_SCREEN_CALL_BUDGET = n ?? 170; }
|
|
226
|
+
/** A dead or stalled classifier answers ERROR after its 1.5 s + 5 s retry
|
|
227
|
+
* budget; across a long history that is minutes of nothing. After this many
|
|
228
|
+
* consecutive uncached ERRORs the remaining windows are UNCERTAIN without a
|
|
229
|
+
* call — and the read that reaches the threshold trips it too (review
|
|
230
|
+
* round 23: checked only before a call, a third ERROR on the last read left
|
|
231
|
+
* the aggregate on the ERROR path) — UNCERTAIN for a text call (cloud when
|
|
232
|
+
* allowed, else refused; a call carrying an image keeps the image policy,
|
|
233
|
+
* local only), NOT the ERROR path:
|
|
234
|
+
* the regex-only keyword net must not become the sole guard for windows the
|
|
235
|
+
* classifier never read (review round 16). */
|
|
236
|
+
export const LAYER1_SCREEN_ERROR_BREAKER = 3;
|
|
237
|
+
const LAYER1_HISTORY_CACHE_MAX = 1_000;
|
|
238
|
+
/** Entries expire so a classifier alias updated in place (same name, new
|
|
239
|
+
* weights) cannot keep serving a clearance the old weights gave. */
|
|
240
|
+
export const LAYER1_HISTORY_CACHE_TTL_MS = 15 * 60_000;
|
|
241
|
+
const layer1HistoryCache = new Map();
|
|
242
|
+
export function _resetLayer1HistoryCacheForTest() { layer1HistoryCache.clear(); }
|
|
243
|
+
async function classifyHistoryWindow(l1fn, window, ollamaUrl, model, budget) {
|
|
244
|
+
const key = createHash("sha256").update(model).update("\0").update(window).digest("hex");
|
|
245
|
+
// performance.now() is monotonic: a wall-clock rollback must not extend
|
|
246
|
+
// a cached clearance (review round 3).
|
|
247
|
+
const hit = layer1HistoryCache.get(key);
|
|
248
|
+
if (hit && hit.expiresAt > performance.now())
|
|
249
|
+
return hit.verdict;
|
|
250
|
+
if (hit)
|
|
251
|
+
layer1HistoryCache.delete(key);
|
|
252
|
+
// Cache misses cost a model call; over budget the screen fails closed,
|
|
253
|
+
// and a classifier that keeps failing is not asked again this request.
|
|
254
|
+
if (budget && budget.consecutiveErrors >= LAYER1_SCREEN_ERROR_BREAKER) {
|
|
255
|
+
budget.tripped = true;
|
|
256
|
+
return "UNCERTAIN";
|
|
257
|
+
}
|
|
258
|
+
if (budget && ++budget.calls > LAYER1_SCREEN_CALL_BUDGET) {
|
|
259
|
+
budget.tripped = true;
|
|
260
|
+
return "UNCERTAIN";
|
|
261
|
+
}
|
|
262
|
+
const verdict = await l1fn(window, ollamaUrl, model, undefined, undefined, { deterministic: false });
|
|
263
|
+
if (budget) {
|
|
264
|
+
budget.consecutiveErrors = verdict === "ERROR" ? budget.consecutiveErrors + 1 : 0;
|
|
265
|
+
if (budget.consecutiveErrors >= LAYER1_SCREEN_ERROR_BREAKER) {
|
|
266
|
+
budget.tripped = true;
|
|
267
|
+
return "UNCERTAIN";
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
if (verdict !== "ERROR") {
|
|
271
|
+
if (layer1HistoryCache.size >= LAYER1_HISTORY_CACHE_MAX) {
|
|
272
|
+
const oldest = layer1HistoryCache.keys().next().value;
|
|
273
|
+
if (oldest !== undefined)
|
|
274
|
+
layer1HistoryCache.delete(oldest);
|
|
275
|
+
}
|
|
276
|
+
layer1HistoryCache.set(key, { verdict, expiresAt: performance.now() + LAYER1_HISTORY_CACHE_TTL_MS });
|
|
277
|
+
}
|
|
278
|
+
return verdict;
|
|
279
|
+
}
|
|
280
|
+
function historyArgs(args) {
|
|
281
|
+
return args.messages?.length ? [args.messages] : [];
|
|
282
|
+
}
|
|
283
|
+
/** Total history characters — what the plan cap and the truncation floor count. */
|
|
284
|
+
function historyChars(args) {
|
|
285
|
+
return (args.messages ?? []).reduce((n, m) => n + m.content.length, 0);
|
|
286
|
+
}
|
|
287
|
+
/** The portal's message-count cap on /api/v1/prism/inference. The current
|
|
288
|
+
* prompt is appended as the last message, so 50 prior turns make 51 and the
|
|
289
|
+
* portal answers 413 — the client must refuse first (review 2026-09-16). */
|
|
290
|
+
export const CLOUD_HISTORY_MAX_MESSAGES = 50;
|
|
291
|
+
/** Byte-exact mirror of how /api/v1/prism/inference flattens `messages`
|
|
292
|
+
* before its 32 KB check: role-labelled lines joined by newline plus the
|
|
293
|
+
* trailing "Assistant:" cue. Any drift here re-opens the 10-byte window in
|
|
294
|
+
* which the client accepts what the portal rejects. Exported for tests. */
|
|
295
|
+
export function portalFlattenedTranscript(messages) {
|
|
296
|
+
return messages
|
|
297
|
+
.map(m => `${m.role === "user" ? "User" : "Assistant"}: ${m.content}`)
|
|
298
|
+
.join("\n") + "\nAssistant:";
|
|
299
|
+
}
|
|
300
|
+
function cloudHistory(args) {
|
|
301
|
+
if (!args.messages?.length)
|
|
302
|
+
return {};
|
|
303
|
+
return { messages: [...args.messages, { role: "user", content: args.prompt }] };
|
|
304
|
+
}
|
|
95
305
|
/** Bytes per supplied image. Beyond this the base64 blows request memory and
|
|
96
306
|
* the tier timeout before the model ever sees it. */
|
|
97
307
|
export const MAX_IMAGE_BYTES = 12 * 1024 * 1024;
|
|
@@ -183,7 +393,14 @@ export const PRISM_INFER_TOOL = {
|
|
|
183
393
|
"When `project` is provided, loads the dashboard-configured quick/standard/deep handoff and bounded history " +
|
|
184
394
|
"as untrusted historical context for a memory-aware local worker. " +
|
|
185
395
|
"Use this for code generation, summarisation, classification, or any synth task you would " +
|
|
186
|
-
"otherwise hand to the cloud model — it costs $0 when the local hit succeeds."
|
|
396
|
+
"otherwise hand to the cloud model — it costs $0 when the local hit succeeds. " +
|
|
397
|
+
"For a FOLLOW-UP to an earlier prism_infer answer, pass the accepted prior turns as `messages` " +
|
|
398
|
+
"(paid plans): without them the worker answers the follow-up from nothing and fabricates. " +
|
|
399
|
+
"Every entitlement-resolved result reports `multi_turn` (your plan's caps) and `history_turns` (what was sent); " +
|
|
400
|
+
"the crisis intercept reports only `history_turns`. " +
|
|
401
|
+
"History over the plan's caps is refused (history_over_plan_cap), never trimmed; a free plan " +
|
|
402
|
+
"or a host with no portal is refused (multi_turn_not_in_plan). Hosts that compact large " +
|
|
403
|
+
"schemas may drop parameter text, so the contract lives here.",
|
|
187
404
|
inputSchema: {
|
|
188
405
|
type: "object",
|
|
189
406
|
properties: {
|
|
@@ -191,17 +408,33 @@ export const PRISM_INFER_TOOL = {
|
|
|
191
408
|
type: "array",
|
|
192
409
|
items: { type: "string" },
|
|
193
410
|
maxItems: MAX_INFER_IMAGES,
|
|
194
|
-
description: "Screenshots or frames
|
|
195
|
-
"
|
|
196
|
-
"skipped rather than shown the prompt without the image.",
|
|
411
|
+
description: "Screenshots or frames: absolute file paths or raw base64. Needs a vision-capable " +
|
|
412
|
+
"tier; tiers without vision are skipped, never shown the prompt without the image.",
|
|
197
413
|
},
|
|
198
414
|
prompt: {
|
|
199
415
|
type: "string",
|
|
200
|
-
description: "The user prompt.
|
|
416
|
+
description: "The user prompt.",
|
|
417
|
+
},
|
|
418
|
+
messages: {
|
|
419
|
+
type: "array",
|
|
420
|
+
description: "Prior turns of THIS conversation, oldest first; `prompt` stays the current turn. " +
|
|
421
|
+
"Send only accepted turns, as a brief, not a transcript; text only. A paid Synalux " +
|
|
422
|
+
"plan feature: the plan sets turn and character caps; free plan or no portal is " +
|
|
423
|
+
"refused (multi_turn_not_in_plan); over-cap or malformed history is refused with " +
|
|
424
|
+
"the caps named, never trimmed. Turns are safety-screened, counted against the " +
|
|
425
|
+
"tier's context, forwarded to the cloud on escalation (32 KB cap), never stored.",
|
|
426
|
+
items: {
|
|
427
|
+
type: "object",
|
|
428
|
+
properties: {
|
|
429
|
+
role: { type: "string", enum: ["user", "assistant"] },
|
|
430
|
+
content: { type: "string" },
|
|
431
|
+
},
|
|
432
|
+
required: ["role", "content"],
|
|
433
|
+
},
|
|
201
434
|
},
|
|
202
435
|
system: {
|
|
203
436
|
type: "string",
|
|
204
|
-
description: "
|
|
437
|
+
description: "System instruction prepended to the prompt.",
|
|
205
438
|
},
|
|
206
439
|
max_tokens: {
|
|
207
440
|
type: "number",
|
|
@@ -210,135 +443,167 @@ export const PRISM_INFER_TOOL = {
|
|
|
210
443
|
},
|
|
211
444
|
temperature: {
|
|
212
445
|
type: "number",
|
|
213
|
-
description: "Sampling temperature
|
|
446
|
+
description: "Sampling temperature; default 0 = deterministic.",
|
|
214
447
|
default: 0,
|
|
215
448
|
},
|
|
216
449
|
model_ceiling: {
|
|
217
450
|
type: "string",
|
|
218
451
|
enum: ["27b", "9b", "4b", "2b"],
|
|
219
|
-
description: "
|
|
452
|
+
description: "Largest tier the picker may select; '9b' forbids 27B even if RAM allows.",
|
|
220
453
|
},
|
|
221
454
|
task_complexity: {
|
|
222
455
|
type: "number",
|
|
223
456
|
minimum: 1,
|
|
224
457
|
maximum: 10,
|
|
225
|
-
description: "
|
|
226
|
-
"
|
|
458
|
+
description: "1-10 workload hint prism_infer (not the task router) uses to pick the initial " +
|
|
459
|
+
"local tier and thinking mode; explicit model_ceiling/think win.",
|
|
227
460
|
},
|
|
228
461
|
project: {
|
|
229
462
|
type: "string",
|
|
230
|
-
description: "
|
|
231
|
-
"
|
|
463
|
+
description: "Prism project whose dashboard-depth handoff and recent session memory go to the " +
|
|
464
|
+
"local worker as historical data.",
|
|
232
465
|
},
|
|
233
466
|
context_depth: {
|
|
234
467
|
type: "string",
|
|
235
468
|
enum: ["quick", "standard", "deep"],
|
|
236
|
-
description: "Project-memory depth
|
|
469
|
+
description: "Project-memory depth; defaults to the dashboard setting when `project` is given.",
|
|
237
470
|
},
|
|
238
471
|
conversation_id: {
|
|
239
472
|
type: "string",
|
|
240
|
-
description: "Conversation id
|
|
473
|
+
description: "Conversation id from session_bootstrap (telemetry, continuity).",
|
|
241
474
|
},
|
|
242
475
|
cloud_fallback: {
|
|
243
476
|
type: "boolean",
|
|
244
|
-
description: "
|
|
477
|
+
description: "Fall through to the Synalux portal cascade on local failure. Default false: saving tokens is the point.",
|
|
245
478
|
default: false,
|
|
246
479
|
},
|
|
247
480
|
timeout_ms: {
|
|
248
481
|
type: "number",
|
|
249
|
-
description: "
|
|
482
|
+
description: "Per-call timeout override. Default by tier: 27B 120s, 9B 60s, 4B 20s, 2B 15s.",
|
|
250
483
|
},
|
|
251
484
|
evidence: {
|
|
252
485
|
type: "array",
|
|
253
|
-
description: "
|
|
254
|
-
"
|
|
255
|
-
"
|
|
256
|
-
"these snippets or the draft is refused.",
|
|
486
|
+
description: "Snippets the output must be grounded in. With `verify: true`, every assertive " +
|
|
487
|
+
"claim (numbers, names, dates, codes, $ amounts) must be ENTAILED by a snippet " +
|
|
488
|
+
"or the draft is refused.",
|
|
257
489
|
items: {
|
|
258
490
|
type: "object",
|
|
259
491
|
properties: {
|
|
260
|
-
source: { type: "string", description: "
|
|
261
|
-
content: { type: "string", description: "The
|
|
492
|
+
source: { type: "string", description: "Snippet label, e.g. 'tool:knowledge_search#3'." },
|
|
493
|
+
content: { type: "string", description: "The snippet text." },
|
|
262
494
|
},
|
|
263
495
|
required: ["source", "content"],
|
|
264
496
|
},
|
|
265
497
|
},
|
|
266
498
|
verify: {
|
|
267
499
|
type: "boolean",
|
|
268
|
-
description: "
|
|
269
|
-
"
|
|
270
|
-
"
|
|
271
|
-
"NEUTRAL or CONTRADICTED claims are refused.",
|
|
500
|
+
description: "L3 grounding verifier; default true when `evidence` is given. A second model " +
|
|
501
|
+
"(qwen3.5:4b by default) checks the draft against `evidence`; NEUTRAL or " +
|
|
502
|
+
"CONTRADICTED claims are refused.",
|
|
272
503
|
},
|
|
273
504
|
verifier_model: {
|
|
274
505
|
type: "string",
|
|
275
|
-
description: "
|
|
506
|
+
description: "Verifier model override. Default qwen3.5:4b.",
|
|
276
507
|
},
|
|
277
508
|
verifier_timeout_ms: {
|
|
278
509
|
type: "number",
|
|
279
|
-
description: "
|
|
510
|
+
description: "Verifier hard timeout override. Default 2000 ms.",
|
|
280
511
|
default: 2000,
|
|
281
512
|
},
|
|
282
513
|
mode: {
|
|
283
514
|
type: "string",
|
|
284
515
|
enum: ["route", "chat", "code"],
|
|
285
|
-
description: "
|
|
286
|
-
"
|
|
287
|
-
"
|
|
288
|
-
"In chat/code modes, prefers the 27B tier and enables <think> reasoning.",
|
|
516
|
+
description: "'route' (default): MCP tool routing, fast, no thinking. 'chat': conversation, " +
|
|
517
|
+
"thinking on, cloud escalation on failure. 'code': code generation, thinking on, " +
|
|
518
|
+
"larger context. chat/code prefer the 27B tier.",
|
|
289
519
|
default: "route",
|
|
290
520
|
},
|
|
291
521
|
allowed_tools: {
|
|
292
522
|
type: "array",
|
|
293
523
|
maxItems: MAX_ROUTE_TOOLS,
|
|
294
524
|
items: { type: "string" },
|
|
295
|
-
description: "Tool names
|
|
296
|
-
"
|
|
297
|
-
"Defaults to Prism's seven trained routing tools.",
|
|
525
|
+
description: "Tool names advertised to the route model; well-formed calls outside this list " +
|
|
526
|
+
"are suppressed in route mode. Default: Prism's seven trained routing tools.",
|
|
298
527
|
},
|
|
299
528
|
route_guard: {
|
|
300
529
|
type: "string",
|
|
301
530
|
enum: ["auto", "local"],
|
|
302
|
-
description: "
|
|
303
|
-
"
|
|
304
|
-
"route correction. 'local' keeps the prompt and draft entirely on-device.",
|
|
531
|
+
description: "'auto' (default): local advertised-tool contract plus, on paid plans, the private " +
|
|
532
|
+
"Synalux deterministic route correction. 'local': prompt and draft stay on-device.",
|
|
305
533
|
default: "auto",
|
|
306
534
|
},
|
|
307
535
|
think: {
|
|
308
536
|
type: "boolean",
|
|
309
|
-
description: "
|
|
310
|
-
"
|
|
537
|
+
description: "<think> reasoning. Default true for chat/code, false for route; better on complex " +
|
|
538
|
+
"tasks, adds ~2-5s.",
|
|
311
539
|
},
|
|
312
540
|
strict_entitlements: {
|
|
313
541
|
type: "boolean",
|
|
314
|
-
description: "Fail loud instead of running with ASSUMED free-tier limits
|
|
315
|
-
"
|
|
316
|
-
"
|
|
317
|
-
"
|
|
318
|
-
"unconfigured machines are unaffected. Default: false.",
|
|
542
|
+
description: "Fail loud instead of running with ASSUMED free-tier limits: when entitlements " +
|
|
543
|
+
"fell back to free because the portal was unreachable (source='fallback_free'), " +
|
|
544
|
+
"throw instead of silently applying free clamps. Portal-confirmed free plans and " +
|
|
545
|
+
"unconfigured machines are unaffected.",
|
|
319
546
|
default: false,
|
|
320
547
|
},
|
|
321
548
|
escalation: {
|
|
322
549
|
type: "string",
|
|
323
550
|
enum: ["serve", "report"],
|
|
324
|
-
description: "
|
|
325
|
-
"
|
|
326
|
-
"
|
|
327
|
-
"
|
|
328
|
-
"served-anyway) output is explicitly flagged so callers can distinguish " +
|
|
329
|
-
"success / degraded / refused.",
|
|
551
|
+
description: "'serve' (default): safety refusals throw, gate-failed output may be served. " +
|
|
552
|
+
"'report': every terminal path returns a structured gate_outcome; refused results " +
|
|
553
|
+
"come back as {status:'refused', output:''} and degraded (served gate-failed) " +
|
|
554
|
+
"output is flagged.",
|
|
330
555
|
default: "serve",
|
|
331
556
|
},
|
|
332
557
|
},
|
|
333
558
|
required: ["prompt"],
|
|
334
559
|
},
|
|
335
560
|
};
|
|
561
|
+
/** Why a `messages` value fails the structural (absolute) contract, or null
|
|
562
|
+
* when it passes. The MCP handler surfaces this text so an over-ceiling or
|
|
563
|
+
* malformed history is refused with the ceiling named, not as a generic
|
|
564
|
+
* "invalid arguments" (review 2026-09-16). Plan caps are checked later. */
|
|
565
|
+
export function messagesProblem(messages) {
|
|
566
|
+
if (!Array.isArray(messages))
|
|
567
|
+
return "must be an array of {role, content} turns";
|
|
568
|
+
if (messages.length > ABSOLUTE_MULTI_TURN.max_turns) {
|
|
569
|
+
return `has ${messages.length} turns; the absolute ceiling is ${ABSOLUTE_MULTI_TURN.max_turns} (plans cap lower)`;
|
|
570
|
+
}
|
|
571
|
+
let chars = 0;
|
|
572
|
+
for (const [i, m] of messages.entries()) {
|
|
573
|
+
if (typeof m !== "object" || m === null)
|
|
574
|
+
return `turn ${i} must be an object {role, content}`;
|
|
575
|
+
const t = m;
|
|
576
|
+
// user/assistant only: a `system` turn here would be a second system
|
|
577
|
+
// prompt behind the safety-bearing one.
|
|
578
|
+
if (t.role !== "user" && t.role !== "assistant")
|
|
579
|
+
return `turn ${i} role must be 'user' or 'assistant'`;
|
|
580
|
+
if (typeof t.content !== "string" || !t.content.trim())
|
|
581
|
+
return `turn ${i} content must be a non-empty string`;
|
|
582
|
+
// text only: a turn carrying images (or anything else) would bypass
|
|
583
|
+
// the image screen, which sees the current call's images only.
|
|
584
|
+
if (Object.keys(t).some(k => k !== "role" && k !== "content"))
|
|
585
|
+
return `turn ${i} may carry only role and content (text only)`;
|
|
586
|
+
chars += t.content.length;
|
|
587
|
+
}
|
|
588
|
+
if (chars > ABSOLUTE_MULTI_TURN.max_chars) {
|
|
589
|
+
return `totals ${chars} chars; the absolute ceiling is ${ABSOLUTE_MULTI_TURN.max_chars} (plans cap lower)`;
|
|
590
|
+
}
|
|
591
|
+
return null;
|
|
592
|
+
}
|
|
593
|
+
/** With history, the current prompt is bounded like the history itself, so
|
|
594
|
+
* the transcript screen has a hard ceiling of classifier work (review
|
|
595
|
+
* round 12: an uncapped prompt made the window count unbounded). Anything
|
|
596
|
+
* this long is already over every local window; the cap changes no
|
|
597
|
+
* routing outcome. */
|
|
598
|
+
export const MULTI_TURN_PROMPT_MAX_CHARS = ABSOLUTE_MULTI_TURN.max_chars;
|
|
336
599
|
export function isPrismInferArgs(args) {
|
|
337
600
|
if (typeof args !== "object" || args === null)
|
|
338
601
|
return false;
|
|
339
602
|
const a = args;
|
|
340
603
|
if (typeof a.prompt !== "string" || !a.prompt.trim())
|
|
341
604
|
return false;
|
|
605
|
+
if (Array.isArray(a.messages) && a.messages.length > 0 && a.prompt.length > MULTI_TURN_PROMPT_MAX_CHARS)
|
|
606
|
+
return false;
|
|
342
607
|
if (a.system !== undefined && typeof a.system !== "string")
|
|
343
608
|
return false;
|
|
344
609
|
if (a.images !== undefined) {
|
|
@@ -347,6 +612,8 @@ export function isPrismInferArgs(args) {
|
|
|
347
612
|
if (a.images.some((i) => typeof i !== "string" || !i.trim()))
|
|
348
613
|
return false;
|
|
349
614
|
}
|
|
615
|
+
if (a.messages !== undefined && messagesProblem(a.messages) !== null)
|
|
616
|
+
return false;
|
|
350
617
|
if (a.max_tokens !== undefined && typeof a.max_tokens !== "number")
|
|
351
618
|
return false;
|
|
352
619
|
if (a.temperature !== undefined && typeof a.temperature !== "number")
|
|
@@ -715,11 +982,16 @@ export async function probeVision(url, model) {
|
|
|
715
982
|
* that copied one layer of production and assumed the rest; measuring against a
|
|
716
983
|
* hand-rolled copy of this function would repeat that.
|
|
717
984
|
*/
|
|
718
|
-
export async function callOllamaGenerate(url, model, prompt, system, maxTokens, temperature, timeoutMs, think, images) {
|
|
985
|
+
export async function callOllamaGenerate(url, model, prompt, system, maxTokens, temperature, timeoutMs, think, images, history) {
|
|
719
986
|
try {
|
|
720
987
|
const messages = [];
|
|
721
988
|
if (system)
|
|
722
989
|
messages.push({ role: "system", content: system });
|
|
990
|
+
// Prior turns sit between the system message and the current turn, in
|
|
991
|
+
// the model's own chat template — the form every installed tier read
|
|
992
|
+
// correctly in the 2026-09-15 probes, including turns another tier wrote.
|
|
993
|
+
for (const t of history ?? [])
|
|
994
|
+
messages.push({ role: t.role, content: t.content });
|
|
723
995
|
messages.push({ role: "user", content: prompt, ...(images?.length ? { images } : {}) });
|
|
724
996
|
const body = {
|
|
725
997
|
model,
|
|
@@ -799,7 +1071,28 @@ function makeReservedRefusal(verdict, attempts, category = null, cloudWasAllowed
|
|
|
799
1071
|
});
|
|
800
1072
|
return new ReservedRefusalError(verdict, attempts, category, cloudWasAllowed);
|
|
801
1073
|
}
|
|
802
|
-
|
|
1074
|
+
/** Portal cap on the flattened conversation (`ROLE: content` lines) — see
|
|
1075
|
+
* portal/src/app/api/v1/prism/inference/route.ts MAX_PROMPT_BYTES. */
|
|
1076
|
+
export const CLOUD_HISTORY_CAP_BYTES = 32 * 1024;
|
|
1077
|
+
/** Exported for tests: the cap check must be provable without a portal. */
|
|
1078
|
+
export async function callSynaluxInference(prompt, maxTokens, timeoutMs, opts) {
|
|
1079
|
+
// /api/v1/prism/inference accepts `messages` (≤ 50) OR `prompt`, flattens the
|
|
1080
|
+
// former to a role-labelled transcript, and rejects a flattened prompt over
|
|
1081
|
+
// 32 KB with 413. Pure validation, so it runs before the base-URL check and
|
|
1082
|
+
// the JWT exchange: an oversize conversation fails loud before any network
|
|
1083
|
+
// call and before anything is spent.
|
|
1084
|
+
if (opts?.messages?.length) {
|
|
1085
|
+
if (opts.messages.length > CLOUD_HISTORY_MAX_MESSAGES)
|
|
1086
|
+
return { ok: false, reason: "history_over_cloud_cap" };
|
|
1087
|
+
if (Buffer.byteLength(portalFlattenedTranscript(opts.messages), "utf8") > CLOUD_HISTORY_CAP_BYTES) {
|
|
1088
|
+
return { ok: false, reason: "history_over_cloud_cap" };
|
|
1089
|
+
}
|
|
1090
|
+
}
|
|
1091
|
+
else if (Buffer.byteLength(prompt, "utf8") > CLOUD_HISTORY_CAP_BYTES) {
|
|
1092
|
+
// Same portal cap on the single-prompt body; fail fast instead of a
|
|
1093
|
+
// doomed round trip that ends in 413 (review round 2, 2026-09-16).
|
|
1094
|
+
return { ok: false, reason: "prompt_over_cloud_cap" };
|
|
1095
|
+
}
|
|
803
1096
|
if (!PRISM_SYNALUX_BASE_URL)
|
|
804
1097
|
return { ok: false, reason: "no_synalux_base_url" };
|
|
805
1098
|
const jwt = await getSynaluxJwt();
|
|
@@ -809,7 +1102,14 @@ async function callSynaluxInference(prompt, maxTokens, timeoutMs, opts) {
|
|
|
809
1102
|
// reserved=true tells the portal this prompt was refused by local Layer-1
|
|
810
1103
|
// as reserved clinical content: it must be served by the portal's
|
|
811
1104
|
// reserved-capable cloud backend or refused — never by a local model.
|
|
812
|
-
const reqBody = JSON.stringify({
|
|
1105
|
+
const reqBody = JSON.stringify({
|
|
1106
|
+
// With history the conversation travels as `messages` (the current turn
|
|
1107
|
+
// is its last entry) and `prompt` is omitted, because the portal reads
|
|
1108
|
+
// `prompt` first and would ignore the turns.
|
|
1109
|
+
...(opts?.messages?.length ? { messages: opts.messages } : { prompt }),
|
|
1110
|
+
max_tokens: maxTokens,
|
|
1111
|
+
...(opts?.reserved ? { reserved: true } : {}),
|
|
1112
|
+
});
|
|
813
1113
|
try {
|
|
814
1114
|
let res = await fetch(url, {
|
|
815
1115
|
method: "POST",
|
|
@@ -1040,7 +1340,15 @@ export async function runInfer(args, deps) {
|
|
|
1040
1340
|
const t0 = Date.now();
|
|
1041
1341
|
const temperature = args.temperature ?? 0;
|
|
1042
1342
|
// ── L1 Safety — deterministic input interception ────────────
|
|
1043
|
-
|
|
1343
|
+
// Over the current turn AND every history turn: a first-person crisis
|
|
1344
|
+
// disclosure in a prior turn must meet the same intercept the portal
|
|
1345
|
+
// applies to the flattened conversation (adversarial review 2026-09-16).
|
|
1346
|
+
// Per turn, not over a join: two adjacent turns must not synthesise a
|
|
1347
|
+
// phrase neither contains. USER turns only: the intercept models a
|
|
1348
|
+
// first-person disclosure, and the worker's own prior answer ("here is a
|
|
1349
|
+
// jumping-off point for the refactor") is not one (review round 2).
|
|
1350
|
+
const safetyIntercept = [...(args.messages ?? []).filter(t => t.role === "user").map(t => t.content), args.prompt]
|
|
1351
|
+
.map(checkInputSafety).find(Boolean) ?? null;
|
|
1044
1352
|
if (safetyIntercept) {
|
|
1045
1353
|
return {
|
|
1046
1354
|
output: safetyIntercept,
|
|
@@ -1050,6 +1358,9 @@ export async function runInfer(args, deps) {
|
|
|
1050
1358
|
latency_ms: Date.now() - t0,
|
|
1051
1359
|
used_cloud: false,
|
|
1052
1360
|
attempts: [{ tier: "l1_safety", reason: "crisis_or_medical_intercept" }],
|
|
1361
|
+
// Entitlements are not resolved yet on this path (no network before
|
|
1362
|
+
// the intercept), so `multi_turn` is absent; what was sent is not.
|
|
1363
|
+
history_turns: args.messages?.length ?? 0,
|
|
1053
1364
|
};
|
|
1054
1365
|
}
|
|
1055
1366
|
// ── Entitlement enforcement ──────────────────────────────────
|
|
@@ -1086,8 +1397,8 @@ export async function runInfer(args, deps) {
|
|
|
1086
1397
|
// Per-tier adjustment happens in the tier loop — a tier that reasons before
|
|
1087
1398
|
// answering needs room for the reasoning as well as the answer.
|
|
1088
1399
|
const localMaxTokens = Math.min(args.max_tokens ?? 1024, 8192);
|
|
1089
|
-
// Retained for the log line
|
|
1090
|
-
//
|
|
1400
|
+
// Retained for the log line, which describes the request rather than a
|
|
1401
|
+
// specific backend.
|
|
1091
1402
|
const maxTokens = cloudMaxTokens;
|
|
1092
1403
|
// Cloud fallback only for paid plans
|
|
1093
1404
|
// let, not const: the reserved-image branch pins this off mid-call so no
|
|
@@ -1121,7 +1432,12 @@ export async function runInfer(args, deps) {
|
|
|
1121
1432
|
const wantReport = args.escalation === "report";
|
|
1122
1433
|
// Shared per-result entitlement metadata (§5.5) — spread into every
|
|
1123
1434
|
// terminal result so callers can audit which plan/provenance applied.
|
|
1124
|
-
const entMeta = {
|
|
1435
|
+
const entMeta = {
|
|
1436
|
+
plan: ent.plan,
|
|
1437
|
+
entitlements_source: entSource,
|
|
1438
|
+
multi_turn: multiTurnPolicy(ent),
|
|
1439
|
+
history_turns: args.messages?.length ?? 0,
|
|
1440
|
+
};
|
|
1125
1441
|
const refusedResult = (reason) => ({
|
|
1126
1442
|
output: "",
|
|
1127
1443
|
backend: "refused",
|
|
@@ -1135,6 +1451,33 @@ export async function runInfer(args, deps) {
|
|
|
1135
1451
|
});
|
|
1136
1452
|
debugLog(`[prism_infer] plan=${ent.plan} ceiling=${effectiveCeiling} max_tokens=${maxTokens} ` +
|
|
1137
1453
|
`cloud=${allowCloud} verify=${canVerify} route_guard=${canUsePrivateRouteGuard}`);
|
|
1454
|
+
// Multi-turn policy — the portal's, not ours. Enforced here (not in the
|
|
1455
|
+
// validator) because the caps are entitlements, resolved per call.
|
|
1456
|
+
if (args.messages?.length) {
|
|
1457
|
+
const policy = multiTurnPolicy(ent);
|
|
1458
|
+
const turns = args.messages.length;
|
|
1459
|
+
const chars = args.messages.reduce((n, t) => n + t.content.length, 0);
|
|
1460
|
+
if (!policy.enabled) {
|
|
1461
|
+
attempts.push({ tier: "entitlements", reason: "multi_turn_not_in_plan" });
|
|
1462
|
+
if (wantReport)
|
|
1463
|
+
return refusedResult("multi_turn_not_in_plan");
|
|
1464
|
+
// A portal outage assumes free-plan limits; say so instead of
|
|
1465
|
+
// telling a paying customer to upgrade (review 2026-09-16).
|
|
1466
|
+
const why = entSource === "fallback_free"
|
|
1467
|
+
? "the Synalux portal was unreachable, so free-plan limits are assumed " +
|
|
1468
|
+
"(entitlements_source=fallback_free); retry when it is back"
|
|
1469
|
+
: `multi-turn history is not included in the ${ent.plan} plan`;
|
|
1470
|
+
throw new Error(`prism_infer: ${why}. Send a single prompt, or upgrade: ${ent.upgrade_url}`);
|
|
1471
|
+
}
|
|
1472
|
+
if (turns > policy.max_turns || chars > policy.max_chars) {
|
|
1473
|
+
attempts.push({ tier: "entitlements", reason: "history_over_plan_cap" });
|
|
1474
|
+
if (wantReport)
|
|
1475
|
+
return refusedResult("history_over_plan_cap");
|
|
1476
|
+
throw new Error(`prism_infer: history of ${turns} turn(s) / ${chars} chars exceeds the ${ent.plan} plan's ` +
|
|
1477
|
+
`cap of ${policy.max_turns} turns / ${policy.max_chars} chars. Send fewer, shorter turns ` +
|
|
1478
|
+
`(a brief, not a transcript); nothing was trimmed for you.`);
|
|
1479
|
+
}
|
|
1480
|
+
}
|
|
1138
1481
|
// Log tier enforcement to Datadog for monetization visibility
|
|
1139
1482
|
const ceilingClamped = effectiveCeiling !== (requestedCeiling ?? ent.model_ceiling);
|
|
1140
1483
|
const tokensClamped = maxTokens < (args.max_tokens ?? 1024);
|
|
@@ -1168,13 +1511,16 @@ export async function runInfer(args, deps) {
|
|
|
1168
1511
|
attempts.push({ tier: "ollama_probe", reason: "unreachable" });
|
|
1169
1512
|
}
|
|
1170
1513
|
// ── §E Layer 1 semantic pre-classifier ──────────────────────────────────
|
|
1171
|
-
// Runs for ALL tiers when Ollama is reachable. RESERVED
|
|
1172
|
-
// to cloud if available; otherwise refuse (fail-closed)
|
|
1514
|
+
// Runs for ALL tiers when Ollama is reachable. RESERVED text escalates
|
|
1515
|
+
// to cloud if available; otherwise refuse (fail-closed); a request with
|
|
1516
|
+
// an image keeps the image policy below (local only). Free-tier users
|
|
1173
1517
|
// without cloud still get classified — a RESERVED verdict refuses the
|
|
1174
1518
|
// request rather than silently routing to local.
|
|
1175
|
-
//
|
|
1176
|
-
//
|
|
1177
|
-
|
|
1519
|
+
// No recursion guard: the classifier (layer1.ts) calls Ollama directly and
|
|
1520
|
+
// never re-enters runInfer, so the old "mode=route + max_tokens<=16 is the
|
|
1521
|
+
// classifier" skip only ever served as a caller-controlled bypass of the
|
|
1522
|
+
// safety screen (two independent reviews, 2026-09-16). Every call is
|
|
1523
|
+
// screened.
|
|
1178
1524
|
// Resolved BEFORE Layer 1: the classifier must see the same images the
|
|
1179
1525
|
// model will. Classifying only the text prompt let a screenshot of
|
|
1180
1526
|
// clinical material through a gate that never looked at it.
|
|
@@ -1210,7 +1556,7 @@ export async function runInfer(args, deps) {
|
|
|
1210
1556
|
attempts.push({ tier: "verifier", reason: "verifier_skipped_images_stay_local" });
|
|
1211
1557
|
gatedArgs = { ...gatedArgs, route_guard: "local", verify: false };
|
|
1212
1558
|
}
|
|
1213
|
-
if (installed
|
|
1559
|
+
if (installed) {
|
|
1214
1560
|
const l1fn = deps.callLayer1 ?? defaultCallLayer1;
|
|
1215
1561
|
const l1Model = resolveOllamaName("prism-coder:4b", installed);
|
|
1216
1562
|
// The classifier must be able to SEE what it is classifying. Ollama
|
|
@@ -1238,10 +1584,138 @@ export async function runInfer(args, deps) {
|
|
|
1238
1584
|
}
|
|
1239
1585
|
}
|
|
1240
1586
|
// 4th arg is fetchImpl (default), 5th is the images the classifier must see.
|
|
1241
|
-
|
|
1587
|
+
// Single turn: one call, unchanged. With history, three layers: the
|
|
1588
|
+
// deterministic floor per turn, every turn read alone (every verdict
|
|
1589
|
+
// kept: reserved and uncertain fail closed for text, error follows
|
|
1590
|
+
// the single-prompt error path), then each turn and the prompt in
|
|
1591
|
+
// context (raise only) — see below.
|
|
1592
|
+
let l1;
|
|
1593
|
+
if (!args.messages?.length) {
|
|
1594
|
+
// Single turn: the exact call it always was.
|
|
1595
|
+
l1 = await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages);
|
|
1596
|
+
}
|
|
1597
|
+
else {
|
|
1598
|
+
l1 = "OBVIOUS_NOT_RESERVED";
|
|
1599
|
+
// 1. Deterministic floor, per TURN and role-aware, regex only.
|
|
1600
|
+
for (const turn of args.messages) {
|
|
1601
|
+
// Role matters for the deterministic OPERATIONAL rules (write
|
|
1602
|
+
// auth code, auth bypass, ship/deploy, PHI exposure): they
|
|
1603
|
+
// classify a request, and by description they match ordinary
|
|
1604
|
+
// code — measured 2026-09-16, half of this repo's files and
|
|
1605
|
+
// the worker's own code answers refused the follow-up when
|
|
1606
|
+
// re-sent as an assistant turn. A USER turn is a request and
|
|
1607
|
+
// gets them; an ASSISTANT turn is the worker's prior output
|
|
1608
|
+
// and does not. Clinical rules run on every turn.
|
|
1609
|
+
// Roles come from the host's `messages`, not from the text:
|
|
1610
|
+
// the host is the trusted orchestrator and the alternative —
|
|
1611
|
+
// request rules over the worker's own answers — refused half
|
|
1612
|
+
// of this repo's files. The semantic classifier still reads
|
|
1613
|
+
// every window whatever the label says.
|
|
1614
|
+
const isUser = turn.role === "user";
|
|
1615
|
+
// Co-occurrence rules are proximity rules: 7,200-char windows
|
|
1616
|
+
// advancing by 3,400, so any two terms up to 3,800 chars apart
|
|
1617
|
+
// share a window wherever they sit (review rounds 2-5). The
|
|
1618
|
+
// artifact exemption ("add auth_bypass as a test fixture
|
|
1619
|
+
// label…") is decided per slice too: an exemption thousands
|
|
1620
|
+
// of chars away from a trigger is not the same clause.
|
|
1621
|
+
for (const slice of windowsOf(turn.content, DETERMINISTIC_FLOOR_WINDOW_CHARS, DETERMINISTIC_FLOOR_WINDOW_OVERLAP)) {
|
|
1622
|
+
const det = classifyDeterministicLayer1(slice, { operational: isUser });
|
|
1623
|
+
if (det)
|
|
1624
|
+
l1 = worseLayer1Verdict(l1, det);
|
|
1625
|
+
}
|
|
1626
|
+
}
|
|
1627
|
+
// 2. Semantic floor, per TURN in isolation; every verdict read
|
|
1628
|
+
// alone is kept: OBVIOUS_RESERVED is final (nothing
|
|
1629
|
+
// written later can lower it — measured 2026-09-16, a classifier-
|
|
1630
|
+
// directed note placed later cleared a reserved earlier turn when
|
|
1631
|
+
// the two shared one window); UNCERTAIN is kept (cloud when the
|
|
1632
|
+
// plan allows it, else refused — never local for a text-only call;
|
|
1633
|
+
// a call carrying an image keeps the image policy below, local
|
|
1634
|
+
// only); ERROR is kept and takes the path a single-prompt ERROR
|
|
1635
|
+
// always took (cloud when it is allowed and answers; otherwise the
|
|
1636
|
+
// keyword net over the whole conversation decides, and keyword-
|
|
1637
|
+
// clean text is served locally; three in a row trip to UNCERTAIN —
|
|
1638
|
+
// an availability policy the owner accepted for single turns, kept
|
|
1639
|
+
// identical here, so this one path is NOT fail-closed). Deferring
|
|
1640
|
+
// UNCERTAIN to "context" was
|
|
1641
|
+
// tried in four shapes and each was measured bypassable: a note in
|
|
1642
|
+
// whichever window decided flipped the classifier. A turn read
|
|
1643
|
+
// alone is the one read no later text can touch. The deterministic
|
|
1644
|
+
// rules stay role-aware (step 1); the semantic read is not.
|
|
1645
|
+
const budget = { calls: 0, consecutiveErrors: 0, tripped: false };
|
|
1646
|
+
// Skipped once the deterministic floor has already refused: the
|
|
1647
|
+
// verdict cannot move and every read would be spent for nothing.
|
|
1648
|
+
history: for (const turn of l1 === "OBVIOUS_RESERVED" ? [] : args.messages) {
|
|
1649
|
+
for (const window of historyTurnWindows(turn.content)) {
|
|
1650
|
+
if (!window.trim())
|
|
1651
|
+
continue;
|
|
1652
|
+
const alone = await classifyHistoryWindow(l1fn, window, deps.ollamaUrl, l1Model, budget);
|
|
1653
|
+
if (alone === "OBVIOUS_RESERVED") {
|
|
1654
|
+
l1 = "OBVIOUS_RESERVED";
|
|
1655
|
+
break history;
|
|
1656
|
+
}
|
|
1657
|
+
l1 = worseLayer1Verdict(l1, alone);
|
|
1658
|
+
}
|
|
1659
|
+
}
|
|
1660
|
+
// The current prompt is a request: its deterministic floor runs
|
|
1661
|
+
// here explicitly (not only inside the classifier entry point, so
|
|
1662
|
+
// an injected classifier cannot skip it), in the same proximity
|
|
1663
|
+
// slices as a turn; then, unless the routine fast path below
|
|
1664
|
+
// applies, it is read alone with its images and that verdict is
|
|
1665
|
+
// kept like a turn's.
|
|
1666
|
+
let promptRoutine = true;
|
|
1667
|
+
for (const slice of windowsOf(args.prompt, DETERMINISTIC_FLOOR_WINDOW_CHARS, DETERMINISTIC_FLOOR_WINDOW_OVERLAP)) {
|
|
1668
|
+
const promptDet = classifyDeterministicLayer1(slice);
|
|
1669
|
+
if (promptDet)
|
|
1670
|
+
l1 = worseLayer1Verdict(l1, promptDet);
|
|
1671
|
+
if (promptDet !== "OBVIOUS_NOT_RESERVED")
|
|
1672
|
+
promptRoutine = false;
|
|
1673
|
+
}
|
|
1674
|
+
// The classifier entry point's own whole-prompt deterministic pass
|
|
1675
|
+
// is switched off here — it would undo the slicing above (words
|
|
1676
|
+
// 14k chars apart firing one rule; review round 18). The routine
|
|
1677
|
+
// fast path it provided is kept explicitly and on ITS boundary: a
|
|
1678
|
+
// prompt of at most 4,000 chars whose rules verdict is routine,
|
|
1679
|
+
// with no images, skips the model. Longer prompts always reach
|
|
1680
|
+
// the entry point, whose full-text keyword floor must run
|
|
1681
|
+
// (review round 19: skipping it there bypassed that floor).
|
|
1682
|
+
const promptFastPath = promptRoutine && args.prompt.length <= MAX_CLASSIFIER_PROMPT_LENGTH && (resolvedImages?.length ?? 0) === 0;
|
|
1683
|
+
if (l1 !== "OBVIOUS_RESERVED" && !promptFastPath) {
|
|
1684
|
+
l1 = worseLayer1Verdict(l1, await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages, { deterministic: false }));
|
|
1685
|
+
}
|
|
1686
|
+
// 3. Context, raise only: one window per turn and one for the
|
|
1687
|
+
// prompt (see contextWindows), cached like any window. Skipped
|
|
1688
|
+
// once the verdict is UNCERTAIN or RESERVED: only a raise to
|
|
1689
|
+
// RESERVED is possible and the two take the same branch below;
|
|
1690
|
+
// the recorded label is then the isolated verdict, not the
|
|
1691
|
+
// strongest a context read might have returned.
|
|
1692
|
+
if (l1 === "UNCERTAIN" || l1 === "OBVIOUS_RESERVED") {
|
|
1693
|
+
// Audit: "context never read" is distinguishable from "context read clean".
|
|
1694
|
+
attempts.push({ tier: "layer1", reason: `layer1_context_skipped_${l1.toLowerCase()}` });
|
|
1695
|
+
}
|
|
1696
|
+
else {
|
|
1697
|
+
for (const window of contextWindows(args)) {
|
|
1698
|
+
if (!window.trim())
|
|
1699
|
+
continue;
|
|
1700
|
+
l1 = worseLayer1Verdict(l1, await classifyHistoryWindow(l1fn, window, deps.ollamaUrl, l1Model, budget));
|
|
1701
|
+
if (l1 === "UNCERTAIN" || l1 === "OBVIOUS_RESERVED")
|
|
1702
|
+
break;
|
|
1703
|
+
}
|
|
1704
|
+
}
|
|
1705
|
+
// A budget or breaker trip raises to UNCERTAIN whatever the cache
|
|
1706
|
+
// held (text: cloud or refused; with an image: local only).
|
|
1707
|
+
if (budget.tripped)
|
|
1708
|
+
l1 = worseLayer1Verdict(l1, "UNCERTAIN");
|
|
1709
|
+
if (budget.calls > LAYER1_SCREEN_CALL_BUDGET) {
|
|
1710
|
+
attempts.push({ tier: "layer1", reason: `layer1_screen_over_budget:${LAYER1_SCREEN_CALL_BUDGET}` });
|
|
1711
|
+
}
|
|
1712
|
+
if (budget.consecutiveErrors >= LAYER1_SCREEN_ERROR_BREAKER) {
|
|
1713
|
+
attempts.push({ tier: "layer1", reason: `layer1_screen_error_breaker:${LAYER1_SCREEN_ERROR_BREAKER}` });
|
|
1714
|
+
}
|
|
1715
|
+
}
|
|
1242
1716
|
// Null when the deterministic floor did not fire — the verdict then came
|
|
1243
1717
|
// from the semantic classifier, which has no per-rule attribution.
|
|
1244
|
-
const reservedCat = reservedCategory(args
|
|
1718
|
+
const reservedCat = reservedCategory(screenedText(args));
|
|
1245
1719
|
if ((l1 === "OBVIOUS_RESERVED" || l1 === "UNCERTAIN")
|
|
1246
1720
|
&& (resolvedImages?.length ?? 0) > 0) {
|
|
1247
1721
|
// Clinical images are PROCESSED, never refused (ruling 2026-08-18:
|
|
@@ -1283,7 +1757,7 @@ export async function runInfer(args, deps) {
|
|
|
1283
1757
|
}
|
|
1284
1758
|
if (allowCloud) {
|
|
1285
1759
|
const cloudTimeout = args.timeout_ms ?? 90_000;
|
|
1286
|
-
const cloud = await deps.callCloud(args.prompt, maxTokens, cloudTimeout, { reserved: true });
|
|
1760
|
+
const cloud = await deps.callCloud(args.prompt, maxTokens, cloudTimeout, { reserved: true, ...cloudHistory(args) });
|
|
1287
1761
|
if (cloud.ok && cloud.output) {
|
|
1288
1762
|
// Defense in depth (§5.1): the escalation target for reserved
|
|
1289
1763
|
// content must be STRONGER than the local model that refused
|
|
@@ -1348,7 +1822,7 @@ export async function runInfer(args, deps) {
|
|
|
1348
1822
|
}
|
|
1349
1823
|
if (allowCloud) {
|
|
1350
1824
|
const cloudTimeout = args.timeout_ms ?? 90_000;
|
|
1351
|
-
const cloud = await deps.callCloud(args.prompt, maxTokens, cloudTimeout);
|
|
1825
|
+
const cloud = await deps.callCloud(args.prompt, maxTokens, cloudTimeout, cloudHistory(args));
|
|
1352
1826
|
if (cloud.ok && cloud.output) {
|
|
1353
1827
|
return await applyVerification(cloud.output, gatedArgs, deps, {
|
|
1354
1828
|
backend: cloud.backend ?? "synalux",
|
|
@@ -1364,7 +1838,7 @@ export async function runInfer(args, deps) {
|
|
|
1364
1838
|
}
|
|
1365
1839
|
attempts.push({ tier: "synalux", reason: cloud.reason ?? "unknown" });
|
|
1366
1840
|
}
|
|
1367
|
-
const backstop = keywordBackstop(args
|
|
1841
|
+
const backstop = keywordBackstop(screenedText(args));
|
|
1368
1842
|
debugLog(`[prism_infer] keyword backstop verdict=${backstop}`);
|
|
1369
1843
|
attempts.push({ tier: "keyword_backstop", reason: `backstop_${backstop.toLowerCase()}` });
|
|
1370
1844
|
if (backstop === "OBVIOUS_RESERVED") {
|
|
@@ -1572,6 +2046,7 @@ export async function runInfer(args, deps) {
|
|
|
1572
2046
|
// effectiveSystem, not args.system: the default vision prompt is 46
|
|
1573
2047
|
// estimated tokens and the model is charged for them.
|
|
1574
2048
|
const promptBodyEst = estimateImageTokens(resolvedImages?.length ?? 0) + estimateTokens(args.prompt)
|
|
2049
|
+
+ historyTokenEstimate(args.messages)
|
|
1575
2050
|
+ (effectiveSystem ? estimateTokens(effectiveSystem) : 0);
|
|
1576
2051
|
// Prefer what the model reports over what the table remembers. A
|
|
1577
2052
|
// tag with a pinned num_ctx is authoritative; one without keeps the
|
|
@@ -1632,14 +2107,14 @@ export async function runInfer(args, deps) {
|
|
|
1632
2107
|
const tierTokens = enableThink && tier.minLocalTokens
|
|
1633
2108
|
? Math.min(Math.max(localMaxTokens, tier.minLocalTokens), 8192)
|
|
1634
2109
|
: localMaxTokens;
|
|
1635
|
-
let result = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, enableThink, resolvedImages);
|
|
2110
|
+
let result = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, enableThink, resolvedImages, ...historyArgs(args));
|
|
1636
2111
|
// Think-only retry: model burned all tokens on <think>, empty content.
|
|
1637
2112
|
// Retry same model with think=false rather than falling to a smaller tier.
|
|
1638
2113
|
// One-shot: think=false cannot re-trigger think_only (no thinking to burn).
|
|
1639
2114
|
if (!result.ok && result.reason === "think_only" && enableThink) {
|
|
1640
2115
|
debugLog(`[prism_infer] ${tier.tag} returned think-only — retrying with think=false`);
|
|
1641
2116
|
recordThinkOnlyRetry();
|
|
1642
|
-
result = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, false, resolvedImages);
|
|
2117
|
+
result = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, false, resolvedImages, ...historyArgs(args));
|
|
1643
2118
|
}
|
|
1644
2119
|
if (result.ok) {
|
|
1645
2120
|
// Did ollama silently drop part of the prompt?
|
|
@@ -1701,10 +2176,15 @@ export async function runInfer(args, deps) {
|
|
|
1701
2176
|
// and strictly more permissive — it keeps short prompts out
|
|
1702
2177
|
// without inheriting the estimate's blind spots.
|
|
1703
2178
|
const halfCtx = liveCtx != null ? Math.floor(liveCtx / 2) : null;
|
|
2179
|
+
// The floor is on the whole INPUT: with history, a short
|
|
2180
|
+
// "continue" behind 50k chars of turns is exactly the case that
|
|
2181
|
+
// collapses (adversarial review 2026-09-16), and the prompt
|
|
2182
|
+
// alone would never reach the floor.
|
|
2183
|
+
const inputChars = args.prompt.length + historyChars(args);
|
|
1704
2184
|
const looksTruncated = halfCtx != null
|
|
1705
2185
|
&& result.promptTokens != null
|
|
1706
2186
|
&& Math.abs(result.promptTokens - halfCtx) <= 8
|
|
1707
|
-
&&
|
|
2187
|
+
&& inputChars >= halfCtx;
|
|
1708
2188
|
if (looksTruncated) {
|
|
1709
2189
|
debugLog(`[prism_infer] ${tier.tag} evaluated ${result.promptTokens} tokens ≈ num_ctx/2 on a ${promptTokensEst}-token estimate — prompt was truncated`);
|
|
1710
2190
|
attempts.push({ tier: tier.tag, reason: `input_truncated:${result.promptTokens}_of_${liveCtx}` });
|
|
@@ -1734,7 +2214,7 @@ export async function runInfer(args, deps) {
|
|
|
1734
2214
|
if (!gate.pass && gate.reason === "hard_truncation" && enableThink) {
|
|
1735
2215
|
debugLog(`[prism_infer] ${tier.tag} truncated mid-answer — retrying with think=false`);
|
|
1736
2216
|
attempts.push({ tier: tier.tag, reason: "hard_truncation_retry" });
|
|
1737
|
-
const retried = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, false, resolvedImages);
|
|
2217
|
+
const retried = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, false, resolvedImages, ...historyArgs(args));
|
|
1738
2218
|
if (retried.ok) {
|
|
1739
2219
|
const retriedStrip = stripThink(retried.text);
|
|
1740
2220
|
const retriedGate = passesQualityGate(retriedStrip.stripped, retriedStrip.thinkOnly, retried.doneReason, mode);
|
|
@@ -1778,15 +2258,24 @@ export async function runInfer(args, deps) {
|
|
|
1778
2258
|
break;
|
|
1779
2259
|
}
|
|
1780
2260
|
const repair = buildCodingRepairPrompt(args.prompt, output, failedReason);
|
|
1781
|
-
|
|
1782
|
-
|
|
2261
|
+
// effectiveSystem, not args.system: the repair carries the
|
|
2262
|
+
// first call's images, so it keeps the default vision
|
|
2263
|
+
// instruction too. Sized like the first call — history and
|
|
2264
|
+
// images counted, against the live window (review 2026-09-16).
|
|
2265
|
+
const repairSystem = effectiveSystem
|
|
2266
|
+
? `${effectiveSystem}\n\n${repair.system}`
|
|
1783
2267
|
: repair.system;
|
|
1784
|
-
const repairPromptTokens =
|
|
2268
|
+
const repairPromptTokens = estimateImageTokens(resolvedImages?.length ?? 0) +
|
|
2269
|
+
estimateTokens(repair.prompt) +
|
|
2270
|
+
historyTokenEstimate(args.messages) +
|
|
1785
2271
|
estimateTokens(repairSystem) +
|
|
1786
2272
|
CTX_TEMPLATE_MARGIN;
|
|
1787
|
-
if (repairPromptTokens <=
|
|
2273
|
+
if (repairPromptTokens <= effectiveCtx) {
|
|
1788
2274
|
attempts.push({ tier: tier.tag, reason: `code_repair:${failedReason}` });
|
|
1789
|
-
|
|
2275
|
+
// Same images and history as the first call: a repair
|
|
2276
|
+
// of a follow-up without its context "repairs" against
|
|
2277
|
+
// nothing (adversarial review 2026-09-16).
|
|
2278
|
+
const repaired = await deps.callLocal(deps.ollamaUrl, ollamaName, repair.prompt, repairSystem, maxTokens, 0, timeout, false, resolvedImages, ...historyArgs(args));
|
|
1790
2279
|
if (repaired.ok) {
|
|
1791
2280
|
const repairedStripped = stripThink(repaired.text);
|
|
1792
2281
|
const repairedGenericGate = passesQualityGate(repairedStripped.stripped, repairedStripped.thinkOnly, repaired.doneReason, mode);
|
|
@@ -1872,7 +2361,7 @@ export async function runInfer(args, deps) {
|
|
|
1872
2361
|
throw new Error(`prism_infer: no local vision tier could serve this image request, and cloud fallback is refused ` +
|
|
1873
2362
|
`for image inputs (screenshots stay on this device). attempts=${JSON.stringify(attempts)}`);
|
|
1874
2363
|
}
|
|
1875
|
-
const cloud = await deps.callCloud(args.prompt, maxTokens, cloudTimeout);
|
|
2364
|
+
const cloud = await deps.callCloud(args.prompt, maxTokens, cloudTimeout, cloudHistory(args));
|
|
1876
2365
|
if (cloud.ok && cloud.output) {
|
|
1877
2366
|
return await applyVerification(cloud.output, gatedArgs, deps, {
|
|
1878
2367
|
backend: cloud.backend ?? "synalux",
|
|
@@ -2080,7 +2569,14 @@ export async function inferText(prompt, opts = {}) {
|
|
|
2080
2569
|
}
|
|
2081
2570
|
export async function prismInferHandler(args) {
|
|
2082
2571
|
if (!isPrismInferArgs(args)) {
|
|
2083
|
-
|
|
2572
|
+
const raw = typeof args === "object" && args !== null ? args : {};
|
|
2573
|
+
const mp = raw.messages !== undefined ? messagesProblem(raw.messages) : null;
|
|
2574
|
+
const longPrompt = Array.isArray(raw.messages) && raw.messages.length > 0 && typeof raw.prompt === "string" && raw.prompt.length > MULTI_TURN_PROMPT_MAX_CHARS;
|
|
2575
|
+
throw new Error(mp
|
|
2576
|
+
? `Invalid arguments for prism_infer: messages ${mp}`
|
|
2577
|
+
: longPrompt
|
|
2578
|
+
? `Invalid arguments for prism_infer: with messages, prompt is capped at ${MULTI_TURN_PROMPT_MAX_CHARS} chars (got ${raw.prompt.length})`
|
|
2579
|
+
: "Invalid arguments for prism_infer (need {prompt: string})");
|
|
2084
2580
|
}
|
|
2085
2581
|
try {
|
|
2086
2582
|
const prepared = await prepareMemoryAwareInferArgs(args);
|