prism-mcp-server 20.18.2 → 20.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -19,19 +19,20 @@
19
19
  * directly — all cloud traffic goes via the synalux portal so billing,
20
20
  * tier gating, and HIPAA audit are enforced in one place.
21
21
  */
22
+ import { createHash } from "node:crypto";
22
23
  import { pickLocalModel, fmtGb, MODEL_TIERS, resolveOllamaName } from "../utils/modelPicker.js";
23
24
  import { getSynaluxJwt, invalidateSynaluxJwt } from "../utils/synaluxJwt.js";
24
25
  import { getAvailableMemoryBytes } from "../utils/availableMemory.js";
25
26
  import { downscaleImages, productionDownscaleDeps, resolveMaxImageEdge } from "../utils/imageDownscale.js";
26
27
  import { PRISM_SYNALUX_BASE_URL, PRISM_LOCAL_LLM_URL, PRISM_USER_ID, SYNALUX_CONFIGURED, } from "../config.js";
27
28
  import { debugLog } from "../utils/logger.js";
28
- import { getEntitlements, clampCeiling } from "../utils/entitlements.js";
29
+ import { getEntitlements, clampCeiling, multiTurnPolicy, ABSOLUTE_MULTI_TURN } from "../utils/entitlements.js";
29
30
  import { ddLog } from "../utils/ddLogger.js";
30
31
  import { stripThink } from "../utils/thinkStrip.js";
31
32
  import { passesQualityGate } from "../utils/qualityGate.js";
32
33
  import { applyDeterministicCodingRepairs, buildCodingRepairPrompt, passesCodingQualityGate, } from "../utils/codingQualityPolicy.js";
33
34
  import { checkInputSafety, checkOutputSafety } from "../utils/safetyGate.js";
34
- import { callLayer1 as defaultCallLayer1, keywordBackstop, reservedCategory } from "../utils/layer1.js";
35
+ import { callLayer1 as defaultCallLayer1, classifyDeterministicLayer1, keywordBackstop, reservedCategory, MAX_CLASSIFIER_PROMPT_LENGTH } from "../utils/layer1.js";
35
36
  import { recordInference, recordThinkOnlyRetry, formatInferenceMetrics, estimateTokens } from "../utils/inferenceMetrics.js";
36
37
  import { appendInferMetric } from "../storage/inferMetricsLedger.js";
37
38
  import { getStorage } from "../storage/index.js";
@@ -92,6 +93,215 @@ export const VISION_SYSTEM_PROMPT = "You read developer screenshots precisely. A
92
93
  + "Identify the most specific, innermost location the evidence points to rather "
93
94
  + "than the first thing you see.";
94
95
  export const MAX_INFER_IMAGES = 8;
96
+ /** Tokens a history adds to the prompt body: content plus ~8 tokens of
97
+ * chat-template framing per message (role markers and separators). */
98
+ export function historyTokenEstimate(history) {
99
+ if (!history?.length)
100
+ return 0;
101
+ return history.reduce((n, t) => n + estimateTokens(t.content) + 8, 0);
102
+ }
103
+ /** History and current prompt as ONE text for the deterministic screens
104
+ * (reserved-category attribution, keyword backstop). The semantic classifier
105
+ * reads each turn alone, then each turn and the prompt in context — see the
106
+ * Layer 1 block and contextWindows. */
107
+ function screenedText(args) {
108
+ const history = args.messages ?? [];
109
+ return history.length ? [...history.map(t => t.content), args.prompt].join("\n") : args.prompt;
110
+ }
111
+ /** Role-labelled transcript, current prompt last. contextWindows() are its
112
+ * per-turn tails; the deterministic screens use screenedText. Exported for
113
+ * tests. */
114
+ export function screeningTranscript(args) {
115
+ return [...(args.messages ?? []), { role: "user", content: args.prompt }]
116
+ .map(t => `${t.role === "user" ? "User" : "Assistant"}: ${t.content}`)
117
+ .join("\n");
118
+ }
119
+ /** One context window per turn (and one for the current prompt): the last
120
+ * HISTORY_TURN_WINDOW_CHARS chars of the role-labelled transcript ENDING at
121
+ * that turn. A context read can only RAISE the verdict: intent spread across
122
+ * turns that each read clean alone (measured 2026-09-16: the two halves of a
123
+ * restraint request in separate user turns, clean apart, reserved together)
124
+ * is caught by the window ending at the later half — when both parts fall
125
+ * inside one window, i.e. the last HISTORY_TURN_WINDOW_CHARS chars of the
126
+ * transcript up to the END of the later part's turn; windows exist only at
127
+ * turn ends, so a later part at the start of a long turn, or parts further
128
+ * apart than that, are never in one read (the limit). No context read ever
129
+ * lowers or replaces an isolated verdict, so no window containing OTHER
130
+ * turns adjudicates a turn (review rounds 12–22: every "defer UNCERTAIN to
131
+ * context" variant was measured bypassable by a classifier-directed note in
132
+ * whichever window decided; round 23 restored these windows as raise-only
133
+ * reads after dropping them left a >3,600-char prefix unscreened for
134
+ * cross-turn intent). Anchored on the turn's end, so a later prompt is
135
+ * never in an earlier turn's window, and an eviction at the plan cap
136
+ * changes only the windows the evicted turn was in — one or two for long
137
+ * turns, every one while the whole transcript still fits in one window
138
+ * (a cost, not a safety property). */
139
+ export function contextWindows(args) {
140
+ const labelled = [...(args.messages ?? []), { role: "user", content: args.prompt }]
141
+ .map(t => `${t.role === "user" ? "User" : "Assistant"}: ${t.content}`);
142
+ const out = [];
143
+ let transcript = "";
144
+ for (const line of labelled) {
145
+ transcript = transcript ? `${transcript}\n${line}` : line;
146
+ out.push(transcript.slice(-HISTORY_TURN_WINDOW_CHARS));
147
+ }
148
+ return out;
149
+ }
150
+ /** Most severe of two Layer 1 verdicts. A reserved turn anywhere in the
151
+ * conversation is a reserved conversation. */
152
+ const LAYER1_SEVERITY = {
153
+ OBVIOUS_NOT_RESERVED: 0, UNCERTAIN_LENGTH: 1, ERROR: 2, UNCERTAIN: 3, OBVIOUS_RESERVED: 4,
154
+ };
155
+ function worseLayer1Verdict(a, b) {
156
+ return LAYER1_SEVERITY[b] > LAYER1_SEVERITY[a] ? b : a;
157
+ }
158
+ /** The whole conversation for the cloud client: history plus the current
159
+ * turn as its last entry. Empty object when there is no history, so a
160
+ * single-turn call still sends the bare `prompt` it always did. */
161
+ /** Trailing history argument for the local call — present ONLY when there is
162
+ * history, so a single-turn call keeps the exact arity it always had (mocks
163
+ * and harnesses that pin the argument list stay valid). */
164
+ /** Layer 1 classifies up to 4,000 chars in full and excerpts beyond that.
165
+ * History turns are cut into overlapping windows under that limit so every
166
+ * region of every turn is classified. Exported for tests. */
167
+ export const HISTORY_TURN_WINDOW_CHARS = 3_600;
168
+ export const HISTORY_TURN_WINDOW_OVERLAP = 200;
169
+ /** The deterministic co-occurrence rules (restraint+document, diagnos+determine…)
170
+ * are proximity rules: over a whole 20k-char pasted file, "diagnose" and
171
+ * "determine" 14k chars apart fired one (measured 2026-09-16, +20% of real
172
+ * source files refused). They run over 7,200-char windows advancing by
173
+ * 3,400 (the classifier stride), so ANY two terms up to 3,800 chars apart —
174
+ * more than one classifier window, about one prompt — share a window
175
+ * wherever they sit in the turn (a 7,000-char stride left a pair straddling
176
+ * the boundary in no window: round-5 review). Wider apart than that is not
177
+ * one intent. */
178
+ export const DETERMINISTIC_FLOOR_WINDOW_CHARS = 7_200;
179
+ export const DETERMINISTIC_FLOOR_WINDOW_OVERLAP = 3_800;
180
+ export function windowsOf(content, size, overlap) {
181
+ if (content.length <= size)
182
+ return [content];
183
+ const isHigh = (i) => { const c = content.charCodeAt(i); return c >= 0xd800 && c <= 0xdbff; };
184
+ const isLow = (i) => { const c = content.charCodeAt(i); return c >= 0xdc00 && c <= 0xdfff; };
185
+ const out = [];
186
+ const step = size - overlap;
187
+ for (let i = 0; i < content.length; i += step) {
188
+ // Never cut a surrogate pair: a window that starts on a low or ends
189
+ // on a high surrogate is malformed text for the classifier.
190
+ let start = i;
191
+ if (start > 0 && isLow(start))
192
+ start += 1;
193
+ let end = Math.min(content.length, start + size);
194
+ if (end < content.length && isHigh(end - 1))
195
+ end += 1;
196
+ out.push(content.slice(start, end));
197
+ if (end >= content.length)
198
+ break;
199
+ }
200
+ return out;
201
+ }
202
+ export function historyTurnWindows(content) {
203
+ return windowsOf(content, HISTORY_TURN_WINDOW_CHARS, HISTORY_TURN_WINDOW_OVERLAP);
204
+ }
205
+ /** Verdict cache for history windows, keyed by a hash of model + text — no
206
+ * turn text is retained. A follow-up re-sends the same accepted turns, so
207
+ * without this an n-turn conversation re-screens every prior turn on every
208
+ * call (quadratic classifier work; review 2026-09-16). ERROR verdicts are
209
+ * transient and never cached; a window classified WITH images never goes
210
+ * through here (the key has no image bytes in it). */
211
+ /** Aggregate classifier-call budget for one request's history screen — a
212
+ * safety net at the STRUCTURAL maximum (49 turns / 128k chars of history
213
+ * alone: 49 base windows + 37 extra for the long ones = 86; one context
214
+ * window per turn and one for the prompt = 50; 136 budgeted misses, 137
215
+ * calls with the prompt's own), not a plan-level limit: every shape the
216
+ * caps allow fits under it with 33 calls of headroom, so a paid call never
217
+ * trips it, and a runaway loop cannot exceed it. Beyond it the screen
218
+ * raises to UNCERTAIN. The real bounds are the plan caps (enterprise: 30
219
+ * turns alone + 31 context + 1 ≈ 62 calls on a cold cache; a follow-up that
220
+ * appends pays its new turns alone, the prompt alone and their context
221
+ * windows; one that evicts the oldest turn also pays every context window
222
+ * that turn was in) and the consecutive-ERROR breaker below (review rounds
223
+ * 13–23). */
224
+ export let LAYER1_SCREEN_CALL_BUDGET = 170;
225
+ export function _setScreenCallBudgetForTest(n) { LAYER1_SCREEN_CALL_BUDGET = n ?? 170; }
226
+ /** A dead or stalled classifier answers ERROR after its 1.5 s + 5 s retry
227
+ * budget; across a long history that is minutes of nothing. After this many
228
+ * consecutive uncached ERRORs the remaining windows are UNCERTAIN without a
229
+ * call — and the read that reaches the threshold trips it too (review
230
+ * round 23: checked only before a call, a third ERROR on the last read left
231
+ * the aggregate on the ERROR path) — UNCERTAIN for a text call (cloud when
232
+ * allowed, else refused; a call carrying an image keeps the image policy,
233
+ * local only), NOT the ERROR path:
234
+ * the regex-only keyword net must not become the sole guard for windows the
235
+ * classifier never read (review round 16). */
236
+ export const LAYER1_SCREEN_ERROR_BREAKER = 3;
237
+ const LAYER1_HISTORY_CACHE_MAX = 1_000;
238
+ /** Entries expire so a classifier alias updated in place (same name, new
239
+ * weights) cannot keep serving a clearance the old weights gave. */
240
+ export const LAYER1_HISTORY_CACHE_TTL_MS = 15 * 60_000;
241
+ const layer1HistoryCache = new Map();
242
+ export function _resetLayer1HistoryCacheForTest() { layer1HistoryCache.clear(); }
243
+ async function classifyHistoryWindow(l1fn, window, ollamaUrl, model, budget) {
244
+ const key = createHash("sha256").update(model).update("\0").update(window).digest("hex");
245
+ // performance.now() is monotonic: a wall-clock rollback must not extend
246
+ // a cached clearance (review round 3).
247
+ const hit = layer1HistoryCache.get(key);
248
+ if (hit && hit.expiresAt > performance.now())
249
+ return hit.verdict;
250
+ if (hit)
251
+ layer1HistoryCache.delete(key);
252
+ // Cache misses cost a model call; over budget the screen fails closed,
253
+ // and a classifier that keeps failing is not asked again this request.
254
+ if (budget && budget.consecutiveErrors >= LAYER1_SCREEN_ERROR_BREAKER) {
255
+ budget.tripped = true;
256
+ return "UNCERTAIN";
257
+ }
258
+ if (budget && ++budget.calls > LAYER1_SCREEN_CALL_BUDGET) {
259
+ budget.tripped = true;
260
+ return "UNCERTAIN";
261
+ }
262
+ const verdict = await l1fn(window, ollamaUrl, model, undefined, undefined, { deterministic: false });
263
+ if (budget) {
264
+ budget.consecutiveErrors = verdict === "ERROR" ? budget.consecutiveErrors + 1 : 0;
265
+ if (budget.consecutiveErrors >= LAYER1_SCREEN_ERROR_BREAKER) {
266
+ budget.tripped = true;
267
+ return "UNCERTAIN";
268
+ }
269
+ }
270
+ if (verdict !== "ERROR") {
271
+ if (layer1HistoryCache.size >= LAYER1_HISTORY_CACHE_MAX) {
272
+ const oldest = layer1HistoryCache.keys().next().value;
273
+ if (oldest !== undefined)
274
+ layer1HistoryCache.delete(oldest);
275
+ }
276
+ layer1HistoryCache.set(key, { verdict, expiresAt: performance.now() + LAYER1_HISTORY_CACHE_TTL_MS });
277
+ }
278
+ return verdict;
279
+ }
280
+ function historyArgs(args) {
281
+ return args.messages?.length ? [args.messages] : [];
282
+ }
283
+ /** Total history characters — what the plan cap and the truncation floor count. */
284
+ function historyChars(args) {
285
+ return (args.messages ?? []).reduce((n, m) => n + m.content.length, 0);
286
+ }
287
+ /** The portal's message-count cap on /api/v1/prism/inference. The current
288
+ * prompt is appended as the last message, so 50 prior turns make 51 and the
289
+ * portal answers 413 — the client must refuse first (review 2026-09-16). */
290
+ export const CLOUD_HISTORY_MAX_MESSAGES = 50;
291
+ /** Byte-exact mirror of how /api/v1/prism/inference flattens `messages`
292
+ * before its 32 KB check: role-labelled lines joined by newline plus the
293
+ * trailing "Assistant:" cue. Any drift here re-opens the 10-byte window in
294
+ * which the client accepts what the portal rejects. Exported for tests. */
295
+ export function portalFlattenedTranscript(messages) {
296
+ return messages
297
+ .map(m => `${m.role === "user" ? "User" : "Assistant"}: ${m.content}`)
298
+ .join("\n") + "\nAssistant:";
299
+ }
300
+ function cloudHistory(args) {
301
+ if (!args.messages?.length)
302
+ return {};
303
+ return { messages: [...args.messages, { role: "user", content: args.prompt }] };
304
+ }
95
305
  /** Bytes per supplied image. Beyond this the base64 blows request memory and
96
306
  * the tier timeout before the model ever sees it. */
97
307
  export const MAX_IMAGE_BYTES = 12 * 1024 * 1024;
@@ -183,7 +393,14 @@ export const PRISM_INFER_TOOL = {
183
393
  "When `project` is provided, loads the dashboard-configured quick/standard/deep handoff and bounded history " +
184
394
  "as untrusted historical context for a memory-aware local worker. " +
185
395
  "Use this for code generation, summarisation, classification, or any synth task you would " +
186
- "otherwise hand to the cloud model — it costs $0 when the local hit succeeds.",
396
+ "otherwise hand to the cloud model — it costs $0 when the local hit succeeds. " +
397
+ "For a FOLLOW-UP to an earlier prism_infer answer, pass the accepted prior turns as `messages` " +
398
+ "(paid plans): without them the worker answers the follow-up from nothing and fabricates. " +
399
+ "Every entitlement-resolved result reports `multi_turn` (your plan's caps) and `history_turns` (what was sent); " +
400
+ "the crisis intercept reports only `history_turns`. " +
401
+ "History over the plan's caps is refused (history_over_plan_cap), never trimmed; a free plan " +
402
+ "or a host with no portal is refused (multi_turn_not_in_plan). Hosts that compact large " +
403
+ "schemas may drop parameter text, so the contract lives here.",
187
404
  inputSchema: {
188
405
  type: "object",
189
406
  properties: {
@@ -191,17 +408,33 @@ export const PRISM_INFER_TOOL = {
191
408
  type: "array",
192
409
  items: { type: "string" },
193
410
  maxItems: MAX_INFER_IMAGES,
194
- description: "Screenshots or frames to analyse. Each entry is an absolute file path " +
195
- "or raw base64. Requires a vision-capable tier; tiers without vision are " +
196
- "skipped rather than shown the prompt without the image.",
411
+ description: "Screenshots or frames: absolute file paths or raw base64. Needs a vision-capable " +
412
+ "tier; tiers without vision are skipped, never shown the prompt without the image.",
197
413
  },
198
414
  prompt: {
199
415
  type: "string",
200
- description: "The user prompt. Required.",
416
+ description: "The user prompt.",
417
+ },
418
+ messages: {
419
+ type: "array",
420
+ description: "Prior turns of THIS conversation, oldest first; `prompt` stays the current turn. " +
421
+ "Send only accepted turns, as a brief, not a transcript; text only. A paid Synalux " +
422
+ "plan feature: the plan sets turn and character caps; free plan or no portal is " +
423
+ "refused (multi_turn_not_in_plan); over-cap or malformed history is refused with " +
424
+ "the caps named, never trimmed. Turns are safety-screened, counted against the " +
425
+ "tier's context, forwarded to the cloud on escalation (32 KB cap), never stored.",
426
+ items: {
427
+ type: "object",
428
+ properties: {
429
+ role: { type: "string", enum: ["user", "assistant"] },
430
+ content: { type: "string" },
431
+ },
432
+ required: ["role", "content"],
433
+ },
201
434
  },
202
435
  system: {
203
436
  type: "string",
204
- description: "Optional system instruction prepended to the prompt.",
437
+ description: "System instruction prepended to the prompt.",
205
438
  },
206
439
  max_tokens: {
207
440
  type: "number",
@@ -210,135 +443,167 @@ export const PRISM_INFER_TOOL = {
210
443
  },
211
444
  temperature: {
212
445
  type: "number",
213
- description: "Sampling temperature, 0 = deterministic (default 0).",
446
+ description: "Sampling temperature; default 0 = deterministic.",
214
447
  default: 0,
215
448
  },
216
449
  model_ceiling: {
217
450
  type: "string",
218
451
  enum: ["27b", "9b", "4b", "2b"],
219
- description: "Cap the largest tier the picker may select. e.g. '9b' forbids 27B even if RAM allows.",
452
+ description: "Largest tier the picker may select; '9b' forbids 27B even if RAM allows.",
220
453
  },
221
454
  task_complexity: {
222
455
  type: "number",
223
456
  minimum: 1,
224
457
  maximum: 10,
225
- description: "Optional deterministic 1-10 workload hint. prism_infer—not the task router—uses it " +
226
- "to choose the initial local tier and thinking mode. Explicit model_ceiling/think overrides win.",
458
+ description: "1-10 workload hint prism_infer (not the task router) uses to pick the initial " +
459
+ "local tier and thinking mode; explicit model_ceiling/think win.",
227
460
  },
228
461
  project: {
229
462
  type: "string",
230
- description: "Optional Prism project whose dashboard-depth handoff and recent session memory should be supplied " +
231
- "to the local worker as historical data.",
463
+ description: "Prism project whose dashboard-depth handoff and recent session memory go to the " +
464
+ "local worker as historical data.",
232
465
  },
233
466
  context_depth: {
234
467
  type: "string",
235
468
  enum: ["quick", "standard", "deep"],
236
- description: "Project-memory depth. Defaults to the Prism dashboard setting when `project` is provided.",
469
+ description: "Project-memory depth; defaults to the dashboard setting when `project` is given.",
237
470
  },
238
471
  conversation_id: {
239
472
  type: "string",
240
- description: "Conversation id returned by session_bootstrap. Used for inference telemetry and continuity.",
473
+ description: "Conversation id from session_bootstrap (telemetry, continuity).",
241
474
  },
242
475
  cloud_fallback: {
243
476
  type: "boolean",
244
- description: "If true, fall through to synalux portal cascade on local fail. Default false — token-saving mode is the point of this tool.",
477
+ description: "Fall through to the Synalux portal cascade on local failure. Default false: saving tokens is the point.",
245
478
  default: false,
246
479
  },
247
480
  timeout_ms: {
248
481
  type: "number",
249
- description: "Override per-call timeout. Default scales with model size: 27B=120s, 9B=60s, 4B=20s, 2B=15s.",
482
+ description: "Per-call timeout override. Default by tier: 27B 120s, 9B 60s, 4B 20s, 2B 15s.",
250
483
  },
251
484
  evidence: {
252
485
  type: "array",
253
- description: "Optional evidence snippets the model output must be grounded in. " +
254
- "When supplied with `verify: true`, every assertive claim in the draft " +
255
- "(numbers, names, dates, codes, $ amounts) must be ENTAILED by one of " +
256
- "these snippets or the draft is refused.",
486
+ description: "Snippets the output must be grounded in. With `verify: true`, every assertive " +
487
+ "claim (numbers, names, dates, codes, $ amounts) must be ENTAILED by a snippet " +
488
+ "or the draft is refused.",
257
489
  items: {
258
490
  type: "object",
259
491
  properties: {
260
- source: { type: "string", description: "Label for the snippet (e.g. 'tool:knowledge_search#3')." },
261
- content: { type: "string", description: "The evidence text itself." },
492
+ source: { type: "string", description: "Snippet label, e.g. 'tool:knowledge_search#3'." },
493
+ content: { type: "string", description: "The snippet text." },
262
494
  },
263
495
  required: ["source", "content"],
264
496
  },
265
497
  },
266
498
  verify: {
267
499
  type: "boolean",
268
- description: "Enable the L3 grounding verifier. Default: true when `evidence` is provided, " +
269
- "false otherwise. When enabled, the model's draft is checked by a different model " +
270
- "(qwen3.5:4b by default) against the supplied `evidence`. Drafts with " +
271
- "NEUTRAL or CONTRADICTED claims are refused.",
500
+ description: "L3 grounding verifier; default true when `evidence` is given. A second model " +
501
+ "(qwen3.5:4b by default) checks the draft against `evidence`; NEUTRAL or " +
502
+ "CONTRADICTED claims are refused.",
272
503
  },
273
504
  verifier_model: {
274
505
  type: "string",
275
- description: "Override the verifier model. Default: qwen3.5:4b.",
506
+ description: "Verifier model override. Default qwen3.5:4b.",
276
507
  },
277
508
  verifier_timeout_ms: {
278
509
  type: "number",
279
- description: "Override the verifier hard timeout. Default 2000 ms.",
510
+ description: "Verifier hard timeout override. Default 2000 ms.",
280
511
  default: 2000,
281
512
  },
282
513
  mode: {
283
514
  type: "string",
284
515
  enum: ["route", "chat", "code"],
285
- description: "Execution mode. 'route' (default) for MCP tool routing — fast, nothink. " +
286
- "'chat' for general conversation — uses thinking, escalates to cloud on failure. " +
287
- "'code' for code generation — uses thinking, larger context. " +
288
- "In chat/code modes, prefers the 27B tier and enables <think> reasoning.",
516
+ description: "'route' (default): MCP tool routing, fast, no thinking. 'chat': conversation, " +
517
+ "thinking on, cloud escalation on failure. 'code': code generation, thinking on, " +
518
+ "larger context. chat/code prefer the 27B tier.",
289
519
  default: "route",
290
520
  },
291
521
  allowed_tools: {
292
522
  type: "array",
293
523
  maxItems: MAX_ROUTE_TOOLS,
294
524
  items: { type: "string" },
295
- description: "Tool names actually advertised to the route model. In route mode, " +
296
- "well-formed calls outside this registry are suppressed before return. " +
297
- "Defaults to Prism's seven trained routing tools.",
525
+ description: "Tool names advertised to the route model; well-formed calls outside this list " +
526
+ "are suppressed in route mode. Default: Prism's seven trained routing tools.",
298
527
  },
299
528
  route_guard: {
300
529
  type: "string",
301
530
  enum: ["auto", "local"],
302
- description: "Route-output guard. 'auto' (default) applies the local advertised-tool " +
303
- "contract and, for authenticated paid plans, the private Synalux deterministic " +
304
- "route correction. 'local' keeps the prompt and draft entirely on-device.",
531
+ description: "'auto' (default): local advertised-tool contract plus, on paid plans, the private " +
532
+ "Synalux deterministic route correction. 'local': prompt and draft stay on-device.",
305
533
  default: "auto",
306
534
  },
307
535
  think: {
308
536
  type: "boolean",
309
- description: "Enable thinking mode (<think> blocks). Default: true for chat/code, false for route. " +
310
- "Thinking improves quality on complex tasks but adds latency (~2-5s).",
537
+ description: "<think> reasoning. Default true for chat/code, false for route; better on complex " +
538
+ "tasks, adds ~2-5s.",
311
539
  },
312
540
  strict_entitlements: {
313
541
  type: "boolean",
314
- description: "Fail loud instead of running with ASSUMED free-tier limits (plan v2 §5.5). " +
315
- "When true and entitlement resolution fell back to free because the portal " +
316
- "was unreachable (source='fallback_free'), the call throws instead of " +
317
- "silently applying free clamps. Portal-confirmed free plans and " +
318
- "unconfigured machines are unaffected. Default: false.",
542
+ description: "Fail loud instead of running with ASSUMED free-tier limits: when entitlements " +
543
+ "fell back to free because the portal was unreachable (source='fallback_free'), " +
544
+ "throw instead of silently applying free clamps. Portal-confirmed free plans and " +
545
+ "unconfigured machines are unaffected.",
319
546
  default: false,
320
547
  },
321
548
  escalation: {
322
549
  type: "string",
323
550
  enum: ["serve", "report"],
324
- description: "Failure contract (plan v2 §5.2). 'serve' (default) keeps legacy behavior: " +
325
- "safety refusals throw, gate-failed output may be served. 'report' returns a " +
326
- "structured gate_outcome on every terminal path — refused results come back as " +
327
- "{status:'refused', output:''} instead of an error, and degraded (gate-failed, " +
328
- "served-anyway) output is explicitly flagged so callers can distinguish " +
329
- "success / degraded / refused.",
551
+ description: "'serve' (default): safety refusals throw, gate-failed output may be served. " +
552
+ "'report': every terminal path returns a structured gate_outcome; refused results " +
553
+ "come back as {status:'refused', output:''} and degraded (served gate-failed) " +
554
+ "output is flagged.",
330
555
  default: "serve",
331
556
  },
332
557
  },
333
558
  required: ["prompt"],
334
559
  },
335
560
  };
561
+ /** Why a `messages` value fails the structural (absolute) contract, or null
562
+ * when it passes. The MCP handler surfaces this text so an over-ceiling or
563
+ * malformed history is refused with the ceiling named, not as a generic
564
+ * "invalid arguments" (review 2026-09-16). Plan caps are checked later. */
565
+ export function messagesProblem(messages) {
566
+ if (!Array.isArray(messages))
567
+ return "must be an array of {role, content} turns";
568
+ if (messages.length > ABSOLUTE_MULTI_TURN.max_turns) {
569
+ return `has ${messages.length} turns; the absolute ceiling is ${ABSOLUTE_MULTI_TURN.max_turns} (plans cap lower)`;
570
+ }
571
+ let chars = 0;
572
+ for (const [i, m] of messages.entries()) {
573
+ if (typeof m !== "object" || m === null)
574
+ return `turn ${i} must be an object {role, content}`;
575
+ const t = m;
576
+ // user/assistant only: a `system` turn here would be a second system
577
+ // prompt behind the safety-bearing one.
578
+ if (t.role !== "user" && t.role !== "assistant")
579
+ return `turn ${i} role must be 'user' or 'assistant'`;
580
+ if (typeof t.content !== "string" || !t.content.trim())
581
+ return `turn ${i} content must be a non-empty string`;
582
+ // text only: a turn carrying images (or anything else) would bypass
583
+ // the image screen, which sees the current call's images only.
584
+ if (Object.keys(t).some(k => k !== "role" && k !== "content"))
585
+ return `turn ${i} may carry only role and content (text only)`;
586
+ chars += t.content.length;
587
+ }
588
+ if (chars > ABSOLUTE_MULTI_TURN.max_chars) {
589
+ return `totals ${chars} chars; the absolute ceiling is ${ABSOLUTE_MULTI_TURN.max_chars} (plans cap lower)`;
590
+ }
591
+ return null;
592
+ }
593
+ /** With history, the current prompt is bounded like the history itself, so
594
+ * the transcript screen has a hard ceiling of classifier work (review
595
+ * round 12: an uncapped prompt made the window count unbounded). Anything
596
+ * this long is already over every local window; the cap changes no
597
+ * routing outcome. */
598
+ export const MULTI_TURN_PROMPT_MAX_CHARS = ABSOLUTE_MULTI_TURN.max_chars;
336
599
  export function isPrismInferArgs(args) {
337
600
  if (typeof args !== "object" || args === null)
338
601
  return false;
339
602
  const a = args;
340
603
  if (typeof a.prompt !== "string" || !a.prompt.trim())
341
604
  return false;
605
+ if (Array.isArray(a.messages) && a.messages.length > 0 && a.prompt.length > MULTI_TURN_PROMPT_MAX_CHARS)
606
+ return false;
342
607
  if (a.system !== undefined && typeof a.system !== "string")
343
608
  return false;
344
609
  if (a.images !== undefined) {
@@ -347,6 +612,8 @@ export function isPrismInferArgs(args) {
347
612
  if (a.images.some((i) => typeof i !== "string" || !i.trim()))
348
613
  return false;
349
614
  }
615
+ if (a.messages !== undefined && messagesProblem(a.messages) !== null)
616
+ return false;
350
617
  if (a.max_tokens !== undefined && typeof a.max_tokens !== "number")
351
618
  return false;
352
619
  if (a.temperature !== undefined && typeof a.temperature !== "number")
@@ -715,11 +982,16 @@ export async function probeVision(url, model) {
715
982
  * that copied one layer of production and assumed the rest; measuring against a
716
983
  * hand-rolled copy of this function would repeat that.
717
984
  */
718
- export async function callOllamaGenerate(url, model, prompt, system, maxTokens, temperature, timeoutMs, think, images) {
985
+ export async function callOllamaGenerate(url, model, prompt, system, maxTokens, temperature, timeoutMs, think, images, history) {
719
986
  try {
720
987
  const messages = [];
721
988
  if (system)
722
989
  messages.push({ role: "system", content: system });
990
+ // Prior turns sit between the system message and the current turn, in
991
+ // the model's own chat template — the form every installed tier read
992
+ // correctly in the 2026-09-15 probes, including turns another tier wrote.
993
+ for (const t of history ?? [])
994
+ messages.push({ role: t.role, content: t.content });
723
995
  messages.push({ role: "user", content: prompt, ...(images?.length ? { images } : {}) });
724
996
  const body = {
725
997
  model,
@@ -799,7 +1071,28 @@ function makeReservedRefusal(verdict, attempts, category = null, cloudWasAllowed
799
1071
  });
800
1072
  return new ReservedRefusalError(verdict, attempts, category, cloudWasAllowed);
801
1073
  }
802
- async function callSynaluxInference(prompt, maxTokens, timeoutMs, opts) {
1074
+ /** Portal cap on the flattened conversation (`ROLE: content` lines) — see
1075
+ * portal/src/app/api/v1/prism/inference/route.ts MAX_PROMPT_BYTES. */
1076
+ export const CLOUD_HISTORY_CAP_BYTES = 32 * 1024;
1077
+ /** Exported for tests: the cap check must be provable without a portal. */
1078
+ export async function callSynaluxInference(prompt, maxTokens, timeoutMs, opts) {
1079
+ // /api/v1/prism/inference accepts `messages` (≤ 50) OR `prompt`, flattens the
1080
+ // former to a role-labelled transcript, and rejects a flattened prompt over
1081
+ // 32 KB with 413. Pure validation, so it runs before the base-URL check and
1082
+ // the JWT exchange: an oversize conversation fails loud before any network
1083
+ // call and before anything is spent.
1084
+ if (opts?.messages?.length) {
1085
+ if (opts.messages.length > CLOUD_HISTORY_MAX_MESSAGES)
1086
+ return { ok: false, reason: "history_over_cloud_cap" };
1087
+ if (Buffer.byteLength(portalFlattenedTranscript(opts.messages), "utf8") > CLOUD_HISTORY_CAP_BYTES) {
1088
+ return { ok: false, reason: "history_over_cloud_cap" };
1089
+ }
1090
+ }
1091
+ else if (Buffer.byteLength(prompt, "utf8") > CLOUD_HISTORY_CAP_BYTES) {
1092
+ // Same portal cap on the single-prompt body; fail fast instead of a
1093
+ // doomed round trip that ends in 413 (review round 2, 2026-09-16).
1094
+ return { ok: false, reason: "prompt_over_cloud_cap" };
1095
+ }
803
1096
  if (!PRISM_SYNALUX_BASE_URL)
804
1097
  return { ok: false, reason: "no_synalux_base_url" };
805
1098
  const jwt = await getSynaluxJwt();
@@ -809,7 +1102,14 @@ async function callSynaluxInference(prompt, maxTokens, timeoutMs, opts) {
809
1102
  // reserved=true tells the portal this prompt was refused by local Layer-1
810
1103
  // as reserved clinical content: it must be served by the portal's
811
1104
  // reserved-capable cloud backend or refused — never by a local model.
812
- const reqBody = JSON.stringify({ prompt, max_tokens: maxTokens, ...(opts?.reserved ? { reserved: true } : {}) });
1105
+ const reqBody = JSON.stringify({
1106
+ // With history the conversation travels as `messages` (the current turn
1107
+ // is its last entry) and `prompt` is omitted, because the portal reads
1108
+ // `prompt` first and would ignore the turns.
1109
+ ...(opts?.messages?.length ? { messages: opts.messages } : { prompt }),
1110
+ max_tokens: maxTokens,
1111
+ ...(opts?.reserved ? { reserved: true } : {}),
1112
+ });
813
1113
  try {
814
1114
  let res = await fetch(url, {
815
1115
  method: "POST",
@@ -1040,7 +1340,15 @@ export async function runInfer(args, deps) {
1040
1340
  const t0 = Date.now();
1041
1341
  const temperature = args.temperature ?? 0;
1042
1342
  // ── L1 Safety — deterministic input interception ────────────
1043
- const safetyIntercept = checkInputSafety(args.prompt);
1343
+ // Over the current turn AND every history turn: a first-person crisis
1344
+ // disclosure in a prior turn must meet the same intercept the portal
1345
+ // applies to the flattened conversation (adversarial review 2026-09-16).
1346
+ // Per turn, not over a join: two adjacent turns must not synthesise a
1347
+ // phrase neither contains. USER turns only: the intercept models a
1348
+ // first-person disclosure, and the worker's own prior answer ("here is a
1349
+ // jumping-off point for the refactor") is not one (review round 2).
1350
+ const safetyIntercept = [...(args.messages ?? []).filter(t => t.role === "user").map(t => t.content), args.prompt]
1351
+ .map(checkInputSafety).find(Boolean) ?? null;
1044
1352
  if (safetyIntercept) {
1045
1353
  return {
1046
1354
  output: safetyIntercept,
@@ -1050,6 +1358,9 @@ export async function runInfer(args, deps) {
1050
1358
  latency_ms: Date.now() - t0,
1051
1359
  used_cloud: false,
1052
1360
  attempts: [{ tier: "l1_safety", reason: "crisis_or_medical_intercept" }],
1361
+ // Entitlements are not resolved yet on this path (no network before
1362
+ // the intercept), so `multi_turn` is absent; what was sent is not.
1363
+ history_turns: args.messages?.length ?? 0,
1053
1364
  };
1054
1365
  }
1055
1366
  // ── Entitlement enforcement ──────────────────────────────────
@@ -1086,8 +1397,8 @@ export async function runInfer(args, deps) {
1086
1397
  // Per-tier adjustment happens in the tier loop — a tier that reasons before
1087
1398
  // answering needs room for the reasoning as well as the answer.
1088
1399
  const localMaxTokens = Math.min(args.max_tokens ?? 1024, 8192);
1089
- // Retained for the log line and the layer-1 recursion guard, both of which
1090
- // describe the request rather than a specific backend.
1400
+ // Retained for the log line, which describes the request rather than a
1401
+ // specific backend.
1091
1402
  const maxTokens = cloudMaxTokens;
1092
1403
  // Cloud fallback only for paid plans
1093
1404
  // let, not const: the reserved-image branch pins this off mid-call so no
@@ -1121,7 +1432,12 @@ export async function runInfer(args, deps) {
1121
1432
  const wantReport = args.escalation === "report";
1122
1433
  // Shared per-result entitlement metadata (§5.5) — spread into every
1123
1434
  // terminal result so callers can audit which plan/provenance applied.
1124
- const entMeta = { plan: ent.plan, entitlements_source: entSource };
1435
+ const entMeta = {
1436
+ plan: ent.plan,
1437
+ entitlements_source: entSource,
1438
+ multi_turn: multiTurnPolicy(ent),
1439
+ history_turns: args.messages?.length ?? 0,
1440
+ };
1125
1441
  const refusedResult = (reason) => ({
1126
1442
  output: "",
1127
1443
  backend: "refused",
@@ -1135,6 +1451,33 @@ export async function runInfer(args, deps) {
1135
1451
  });
1136
1452
  debugLog(`[prism_infer] plan=${ent.plan} ceiling=${effectiveCeiling} max_tokens=${maxTokens} ` +
1137
1453
  `cloud=${allowCloud} verify=${canVerify} route_guard=${canUsePrivateRouteGuard}`);
1454
+ // Multi-turn policy — the portal's, not ours. Enforced here (not in the
1455
+ // validator) because the caps are entitlements, resolved per call.
1456
+ if (args.messages?.length) {
1457
+ const policy = multiTurnPolicy(ent);
1458
+ const turns = args.messages.length;
1459
+ const chars = args.messages.reduce((n, t) => n + t.content.length, 0);
1460
+ if (!policy.enabled) {
1461
+ attempts.push({ tier: "entitlements", reason: "multi_turn_not_in_plan" });
1462
+ if (wantReport)
1463
+ return refusedResult("multi_turn_not_in_plan");
1464
+ // A portal outage assumes free-plan limits; say so instead of
1465
+ // telling a paying customer to upgrade (review 2026-09-16).
1466
+ const why = entSource === "fallback_free"
1467
+ ? "the Synalux portal was unreachable, so free-plan limits are assumed " +
1468
+ "(entitlements_source=fallback_free); retry when it is back"
1469
+ : `multi-turn history is not included in the ${ent.plan} plan`;
1470
+ throw new Error(`prism_infer: ${why}. Send a single prompt, or upgrade: ${ent.upgrade_url}`);
1471
+ }
1472
+ if (turns > policy.max_turns || chars > policy.max_chars) {
1473
+ attempts.push({ tier: "entitlements", reason: "history_over_plan_cap" });
1474
+ if (wantReport)
1475
+ return refusedResult("history_over_plan_cap");
1476
+ throw new Error(`prism_infer: history of ${turns} turn(s) / ${chars} chars exceeds the ${ent.plan} plan's ` +
1477
+ `cap of ${policy.max_turns} turns / ${policy.max_chars} chars. Send fewer, shorter turns ` +
1478
+ `(a brief, not a transcript); nothing was trimmed for you.`);
1479
+ }
1480
+ }
1138
1481
  // Log tier enforcement to Datadog for monetization visibility
1139
1482
  const ceilingClamped = effectiveCeiling !== (requestedCeiling ?? ent.model_ceiling);
1140
1483
  const tokensClamped = maxTokens < (args.max_tokens ?? 1024);
@@ -1168,13 +1511,16 @@ export async function runInfer(args, deps) {
1168
1511
  attempts.push({ tier: "ollama_probe", reason: "unreachable" });
1169
1512
  }
1170
1513
  // ── §E Layer 1 semantic pre-classifier ──────────────────────────────────
1171
- // Runs for ALL tiers when Ollama is reachable. RESERVED prompts escalate
1172
- // to cloud if available; otherwise refuse (fail-closed). Free-tier users
1514
+ // Runs for ALL tiers when Ollama is reachable. RESERVED text escalates
1515
+ // to cloud if available; otherwise refuse (fail-closed); a request with
1516
+ // an image keeps the image policy below (local only). Free-tier users
1173
1517
  // without cloud still get classified — a RESERVED verdict refuses the
1174
1518
  // request rather than silently routing to local.
1175
- // Recursion guard: skip when this call IS the Layer 1 classification
1176
- // (mode="route" + max_tokens<=16 is the Layer 1 call signature).
1177
- const layer1RecursionGuard = mode === "route" && maxTokens <= 16;
1519
+ // No recursion guard: the classifier (layer1.ts) calls Ollama directly and
1520
+ // never re-enters runInfer, so the old "mode=route + max_tokens<=16 is the
1521
+ // classifier" skip only ever served as a caller-controlled bypass of the
1522
+ // safety screen (two independent reviews, 2026-09-16). Every call is
1523
+ // screened.
1178
1524
  // Resolved BEFORE Layer 1: the classifier must see the same images the
1179
1525
  // model will. Classifying only the text prompt let a screenshot of
1180
1526
  // clinical material through a gate that never looked at it.
@@ -1210,7 +1556,7 @@ export async function runInfer(args, deps) {
1210
1556
  attempts.push({ tier: "verifier", reason: "verifier_skipped_images_stay_local" });
1211
1557
  gatedArgs = { ...gatedArgs, route_guard: "local", verify: false };
1212
1558
  }
1213
- if (installed && !layer1RecursionGuard) {
1559
+ if (installed) {
1214
1560
  const l1fn = deps.callLayer1 ?? defaultCallLayer1;
1215
1561
  const l1Model = resolveOllamaName("prism-coder:4b", installed);
1216
1562
  // The classifier must be able to SEE what it is classifying. Ollama
@@ -1238,10 +1584,138 @@ export async function runInfer(args, deps) {
1238
1584
  }
1239
1585
  }
1240
1586
  // 4th arg is fetchImpl (default), 5th is the images the classifier must see.
1241
- const l1 = await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages);
1587
+ // Single turn: one call, unchanged. With history, three layers: the
1588
+ // deterministic floor per turn, every turn read alone (every verdict
1589
+ // kept: reserved and uncertain fail closed for text, error follows
1590
+ // the single-prompt error path), then each turn and the prompt in
1591
+ // context (raise only) — see below.
1592
+ let l1;
1593
+ if (!args.messages?.length) {
1594
+ // Single turn: the exact call it always was.
1595
+ l1 = await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages);
1596
+ }
1597
+ else {
1598
+ l1 = "OBVIOUS_NOT_RESERVED";
1599
+ // 1. Deterministic floor, per TURN and role-aware, regex only.
1600
+ for (const turn of args.messages) {
1601
+ // Role matters for the deterministic OPERATIONAL rules (write
1602
+ // auth code, auth bypass, ship/deploy, PHI exposure): they
1603
+ // classify a request, and by description they match ordinary
1604
+ // code — measured 2026-09-16, half of this repo's files and
1605
+ // the worker's own code answers refused the follow-up when
1606
+ // re-sent as an assistant turn. A USER turn is a request and
1607
+ // gets them; an ASSISTANT turn is the worker's prior output
1608
+ // and does not. Clinical rules run on every turn.
1609
+ // Roles come from the host's `messages`, not from the text:
1610
+ // the host is the trusted orchestrator and the alternative —
1611
+ // request rules over the worker's own answers — refused half
1612
+ // of this repo's files. The semantic classifier still reads
1613
+ // every window whatever the label says.
1614
+ const isUser = turn.role === "user";
1615
+ // Co-occurrence rules are proximity rules: 7,200-char windows
1616
+ // advancing by 3,400, so any two terms up to 3,800 chars apart
1617
+ // share a window wherever they sit (review rounds 2-5). The
1618
+ // artifact exemption ("add auth_bypass as a test fixture
1619
+ // label…") is decided per slice too: an exemption thousands
1620
+ // of chars away from a trigger is not the same clause.
1621
+ for (const slice of windowsOf(turn.content, DETERMINISTIC_FLOOR_WINDOW_CHARS, DETERMINISTIC_FLOOR_WINDOW_OVERLAP)) {
1622
+ const det = classifyDeterministicLayer1(slice, { operational: isUser });
1623
+ if (det)
1624
+ l1 = worseLayer1Verdict(l1, det);
1625
+ }
1626
+ }
1627
+ // 2. Semantic floor, per TURN in isolation; every verdict read
1628
+ // alone is kept: OBVIOUS_RESERVED is final (nothing
1629
+ // written later can lower it — measured 2026-09-16, a classifier-
1630
+ // directed note placed later cleared a reserved earlier turn when
1631
+ // the two shared one window); UNCERTAIN is kept (cloud when the
1632
+ // plan allows it, else refused — never local for a text-only call;
1633
+ // a call carrying an image keeps the image policy below, local
1634
+ // only); ERROR is kept and takes the path a single-prompt ERROR
1635
+ // always took (cloud when it is allowed and answers; otherwise the
1636
+ // keyword net over the whole conversation decides, and keyword-
1637
+ // clean text is served locally; three in a row trip to UNCERTAIN —
1638
+ // an availability policy the owner accepted for single turns, kept
1639
+ // identical here, so this one path is NOT fail-closed). Deferring
1640
+ // UNCERTAIN to "context" was
1641
+ // tried in four shapes and each was measured bypassable: a note in
1642
+ // whichever window decided flipped the classifier. A turn read
1643
+ // alone is the one read no later text can touch. The deterministic
1644
+ // rules stay role-aware (step 1); the semantic read is not.
1645
+ const budget = { calls: 0, consecutiveErrors: 0, tripped: false };
1646
+ // Skipped once the deterministic floor has already refused: the
1647
+ // verdict cannot move and every read would be spent for nothing.
1648
+ history: for (const turn of l1 === "OBVIOUS_RESERVED" ? [] : args.messages) {
1649
+ for (const window of historyTurnWindows(turn.content)) {
1650
+ if (!window.trim())
1651
+ continue;
1652
+ const alone = await classifyHistoryWindow(l1fn, window, deps.ollamaUrl, l1Model, budget);
1653
+ if (alone === "OBVIOUS_RESERVED") {
1654
+ l1 = "OBVIOUS_RESERVED";
1655
+ break history;
1656
+ }
1657
+ l1 = worseLayer1Verdict(l1, alone);
1658
+ }
1659
+ }
1660
+ // The current prompt is a request: its deterministic floor runs
1661
+ // here explicitly (not only inside the classifier entry point, so
1662
+ // an injected classifier cannot skip it), in the same proximity
1663
+ // slices as a turn; then, unless the routine fast path below
1664
+ // applies, it is read alone with its images and that verdict is
1665
+ // kept like a turn's.
1666
+ let promptRoutine = true;
1667
+ for (const slice of windowsOf(args.prompt, DETERMINISTIC_FLOOR_WINDOW_CHARS, DETERMINISTIC_FLOOR_WINDOW_OVERLAP)) {
1668
+ const promptDet = classifyDeterministicLayer1(slice);
1669
+ if (promptDet)
1670
+ l1 = worseLayer1Verdict(l1, promptDet);
1671
+ if (promptDet !== "OBVIOUS_NOT_RESERVED")
1672
+ promptRoutine = false;
1673
+ }
1674
+ // The classifier entry point's own whole-prompt deterministic pass
1675
+ // is switched off here — it would undo the slicing above (words
1676
+ // 14k chars apart firing one rule; review round 18). The routine
1677
+ // fast path it provided is kept explicitly and on ITS boundary: a
1678
+ // prompt of at most 4,000 chars whose rules verdict is routine,
1679
+ // with no images, skips the model. Longer prompts always reach
1680
+ // the entry point, whose full-text keyword floor must run
1681
+ // (review round 19: skipping it there bypassed that floor).
1682
+ const promptFastPath = promptRoutine && args.prompt.length <= MAX_CLASSIFIER_PROMPT_LENGTH && (resolvedImages?.length ?? 0) === 0;
1683
+ if (l1 !== "OBVIOUS_RESERVED" && !promptFastPath) {
1684
+ l1 = worseLayer1Verdict(l1, await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages, { deterministic: false }));
1685
+ }
1686
+ // 3. Context, raise only: one window per turn and one for the
1687
+ // prompt (see contextWindows), cached like any window. Skipped
1688
+ // once the verdict is UNCERTAIN or RESERVED: only a raise to
1689
+ // RESERVED is possible and the two take the same branch below;
1690
+ // the recorded label is then the isolated verdict, not the
1691
+ // strongest a context read might have returned.
1692
+ if (l1 === "UNCERTAIN" || l1 === "OBVIOUS_RESERVED") {
1693
+ // Audit: "context never read" is distinguishable from "context read clean".
1694
+ attempts.push({ tier: "layer1", reason: `layer1_context_skipped_${l1.toLowerCase()}` });
1695
+ }
1696
+ else {
1697
+ for (const window of contextWindows(args)) {
1698
+ if (!window.trim())
1699
+ continue;
1700
+ l1 = worseLayer1Verdict(l1, await classifyHistoryWindow(l1fn, window, deps.ollamaUrl, l1Model, budget));
1701
+ if (l1 === "UNCERTAIN" || l1 === "OBVIOUS_RESERVED")
1702
+ break;
1703
+ }
1704
+ }
1705
+ // A budget or breaker trip raises to UNCERTAIN whatever the cache
1706
+ // held (text: cloud or refused; with an image: local only).
1707
+ if (budget.tripped)
1708
+ l1 = worseLayer1Verdict(l1, "UNCERTAIN");
1709
+ if (budget.calls > LAYER1_SCREEN_CALL_BUDGET) {
1710
+ attempts.push({ tier: "layer1", reason: `layer1_screen_over_budget:${LAYER1_SCREEN_CALL_BUDGET}` });
1711
+ }
1712
+ if (budget.consecutiveErrors >= LAYER1_SCREEN_ERROR_BREAKER) {
1713
+ attempts.push({ tier: "layer1", reason: `layer1_screen_error_breaker:${LAYER1_SCREEN_ERROR_BREAKER}` });
1714
+ }
1715
+ }
1242
1716
  // Null when the deterministic floor did not fire — the verdict then came
1243
1717
  // from the semantic classifier, which has no per-rule attribution.
1244
- const reservedCat = reservedCategory(args.prompt);
1718
+ const reservedCat = reservedCategory(screenedText(args));
1245
1719
  if ((l1 === "OBVIOUS_RESERVED" || l1 === "UNCERTAIN")
1246
1720
  && (resolvedImages?.length ?? 0) > 0) {
1247
1721
  // Clinical images are PROCESSED, never refused (ruling 2026-08-18:
@@ -1283,7 +1757,7 @@ export async function runInfer(args, deps) {
1283
1757
  }
1284
1758
  if (allowCloud) {
1285
1759
  const cloudTimeout = args.timeout_ms ?? 90_000;
1286
- const cloud = await deps.callCloud(args.prompt, maxTokens, cloudTimeout, { reserved: true });
1760
+ const cloud = await deps.callCloud(args.prompt, maxTokens, cloudTimeout, { reserved: true, ...cloudHistory(args) });
1287
1761
  if (cloud.ok && cloud.output) {
1288
1762
  // Defense in depth (§5.1): the escalation target for reserved
1289
1763
  // content must be STRONGER than the local model that refused
@@ -1348,7 +1822,7 @@ export async function runInfer(args, deps) {
1348
1822
  }
1349
1823
  if (allowCloud) {
1350
1824
  const cloudTimeout = args.timeout_ms ?? 90_000;
1351
- const cloud = await deps.callCloud(args.prompt, maxTokens, cloudTimeout);
1825
+ const cloud = await deps.callCloud(args.prompt, maxTokens, cloudTimeout, cloudHistory(args));
1352
1826
  if (cloud.ok && cloud.output) {
1353
1827
  return await applyVerification(cloud.output, gatedArgs, deps, {
1354
1828
  backend: cloud.backend ?? "synalux",
@@ -1364,7 +1838,7 @@ export async function runInfer(args, deps) {
1364
1838
  }
1365
1839
  attempts.push({ tier: "synalux", reason: cloud.reason ?? "unknown" });
1366
1840
  }
1367
- const backstop = keywordBackstop(args.prompt);
1841
+ const backstop = keywordBackstop(screenedText(args));
1368
1842
  debugLog(`[prism_infer] keyword backstop verdict=${backstop}`);
1369
1843
  attempts.push({ tier: "keyword_backstop", reason: `backstop_${backstop.toLowerCase()}` });
1370
1844
  if (backstop === "OBVIOUS_RESERVED") {
@@ -1572,6 +2046,7 @@ export async function runInfer(args, deps) {
1572
2046
  // effectiveSystem, not args.system: the default vision prompt is 46
1573
2047
  // estimated tokens and the model is charged for them.
1574
2048
  const promptBodyEst = estimateImageTokens(resolvedImages?.length ?? 0) + estimateTokens(args.prompt)
2049
+ + historyTokenEstimate(args.messages)
1575
2050
  + (effectiveSystem ? estimateTokens(effectiveSystem) : 0);
1576
2051
  // Prefer what the model reports over what the table remembers. A
1577
2052
  // tag with a pinned num_ctx is authoritative; one without keeps the
@@ -1632,14 +2107,14 @@ export async function runInfer(args, deps) {
1632
2107
  const tierTokens = enableThink && tier.minLocalTokens
1633
2108
  ? Math.min(Math.max(localMaxTokens, tier.minLocalTokens), 8192)
1634
2109
  : localMaxTokens;
1635
- let result = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, enableThink, resolvedImages);
2110
+ let result = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, enableThink, resolvedImages, ...historyArgs(args));
1636
2111
  // Think-only retry: model burned all tokens on <think>, empty content.
1637
2112
  // Retry same model with think=false rather than falling to a smaller tier.
1638
2113
  // One-shot: think=false cannot re-trigger think_only (no thinking to burn).
1639
2114
  if (!result.ok && result.reason === "think_only" && enableThink) {
1640
2115
  debugLog(`[prism_infer] ${tier.tag} returned think-only — retrying with think=false`);
1641
2116
  recordThinkOnlyRetry();
1642
- result = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, false, resolvedImages);
2117
+ result = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, false, resolvedImages, ...historyArgs(args));
1643
2118
  }
1644
2119
  if (result.ok) {
1645
2120
  // Did ollama silently drop part of the prompt?
@@ -1701,10 +2176,15 @@ export async function runInfer(args, deps) {
1701
2176
  // and strictly more permissive — it keeps short prompts out
1702
2177
  // without inheriting the estimate's blind spots.
1703
2178
  const halfCtx = liveCtx != null ? Math.floor(liveCtx / 2) : null;
2179
+ // The floor is on the whole INPUT: with history, a short
2180
+ // "continue" behind 50k chars of turns is exactly the case that
2181
+ // collapses (adversarial review 2026-09-16), and the prompt
2182
+ // alone would never reach the floor.
2183
+ const inputChars = args.prompt.length + historyChars(args);
1704
2184
  const looksTruncated = halfCtx != null
1705
2185
  && result.promptTokens != null
1706
2186
  && Math.abs(result.promptTokens - halfCtx) <= 8
1707
- && args.prompt.length >= halfCtx;
2187
+ && inputChars >= halfCtx;
1708
2188
  if (looksTruncated) {
1709
2189
  debugLog(`[prism_infer] ${tier.tag} evaluated ${result.promptTokens} tokens ≈ num_ctx/2 on a ${promptTokensEst}-token estimate — prompt was truncated`);
1710
2190
  attempts.push({ tier: tier.tag, reason: `input_truncated:${result.promptTokens}_of_${liveCtx}` });
@@ -1734,7 +2214,7 @@ export async function runInfer(args, deps) {
1734
2214
  if (!gate.pass && gate.reason === "hard_truncation" && enableThink) {
1735
2215
  debugLog(`[prism_infer] ${tier.tag} truncated mid-answer — retrying with think=false`);
1736
2216
  attempts.push({ tier: tier.tag, reason: "hard_truncation_retry" });
1737
- const retried = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, false, resolvedImages);
2217
+ const retried = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, false, resolvedImages, ...historyArgs(args));
1738
2218
  if (retried.ok) {
1739
2219
  const retriedStrip = stripThink(retried.text);
1740
2220
  const retriedGate = passesQualityGate(retriedStrip.stripped, retriedStrip.thinkOnly, retried.doneReason, mode);
@@ -1778,15 +2258,24 @@ export async function runInfer(args, deps) {
1778
2258
  break;
1779
2259
  }
1780
2260
  const repair = buildCodingRepairPrompt(args.prompt, output, failedReason);
1781
- const repairSystem = args.system
1782
- ? `${args.system}\n\n${repair.system}`
2261
+ // effectiveSystem, not args.system: the repair carries the
2262
+ // first call's images, so it keeps the default vision
2263
+ // instruction too. Sized like the first call — history and
2264
+ // images counted, against the live window (review 2026-09-16).
2265
+ const repairSystem = effectiveSystem
2266
+ ? `${effectiveSystem}\n\n${repair.system}`
1783
2267
  : repair.system;
1784
- const repairPromptTokens = estimateTokens(repair.prompt) +
2268
+ const repairPromptTokens = estimateImageTokens(resolvedImages?.length ?? 0) +
2269
+ estimateTokens(repair.prompt) +
2270
+ historyTokenEstimate(args.messages) +
1785
2271
  estimateTokens(repairSystem) +
1786
2272
  CTX_TEMPLATE_MARGIN;
1787
- if (repairPromptTokens <= tier.ctxTokens) {
2273
+ if (repairPromptTokens <= effectiveCtx) {
1788
2274
  attempts.push({ tier: tier.tag, reason: `code_repair:${failedReason}` });
1789
- const repaired = await deps.callLocal(deps.ollamaUrl, ollamaName, repair.prompt, repairSystem, maxTokens, 0, timeout, false);
2275
+ // Same images and history as the first call: a repair
2276
+ // of a follow-up without its context "repairs" against
2277
+ // nothing (adversarial review 2026-09-16).
2278
+ const repaired = await deps.callLocal(deps.ollamaUrl, ollamaName, repair.prompt, repairSystem, maxTokens, 0, timeout, false, resolvedImages, ...historyArgs(args));
1790
2279
  if (repaired.ok) {
1791
2280
  const repairedStripped = stripThink(repaired.text);
1792
2281
  const repairedGenericGate = passesQualityGate(repairedStripped.stripped, repairedStripped.thinkOnly, repaired.doneReason, mode);
@@ -1872,7 +2361,7 @@ export async function runInfer(args, deps) {
1872
2361
  throw new Error(`prism_infer: no local vision tier could serve this image request, and cloud fallback is refused ` +
1873
2362
  `for image inputs (screenshots stay on this device). attempts=${JSON.stringify(attempts)}`);
1874
2363
  }
1875
- const cloud = await deps.callCloud(args.prompt, maxTokens, cloudTimeout);
2364
+ const cloud = await deps.callCloud(args.prompt, maxTokens, cloudTimeout, cloudHistory(args));
1876
2365
  if (cloud.ok && cloud.output) {
1877
2366
  return await applyVerification(cloud.output, gatedArgs, deps, {
1878
2367
  backend: cloud.backend ?? "synalux",
@@ -2080,7 +2569,14 @@ export async function inferText(prompt, opts = {}) {
2080
2569
  }
2081
2570
  export async function prismInferHandler(args) {
2082
2571
  if (!isPrismInferArgs(args)) {
2083
- throw new Error("Invalid arguments for prism_infer (need {prompt: string})");
2572
+ const raw = typeof args === "object" && args !== null ? args : {};
2573
+ const mp = raw.messages !== undefined ? messagesProblem(raw.messages) : null;
2574
+ const longPrompt = Array.isArray(raw.messages) && raw.messages.length > 0 && typeof raw.prompt === "string" && raw.prompt.length > MULTI_TURN_PROMPT_MAX_CHARS;
2575
+ throw new Error(mp
2576
+ ? `Invalid arguments for prism_infer: messages ${mp}`
2577
+ : longPrompt
2578
+ ? `Invalid arguments for prism_infer: with messages, prompt is capped at ${MULTI_TURN_PROMPT_MAX_CHARS} chars (got ${raw.prompt.length})`
2579
+ : "Invalid arguments for prism_infer (need {prompt: string})");
2084
2580
  }
2085
2581
  try {
2086
2582
  const prepared = await prepareMemoryAwareInferArgs(args);