prism-mcp-server 20.21.16 → 20.21.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -158,6 +158,15 @@ or by re-enabling after each run.
158
158
  <details>
159
159
  <summary>Release history (optional)</summary>
160
160
 
161
+ ## What's New in v20.21.17
162
+
163
+ - Fixed: the on-device screen refused some ordinary coding requests. With a
164
+ Synalux account (free included), it now reads each request through a policy
165
+ that Synalux serves, pinned by its SHA-256. Without an account, or with
166
+ images attached, the screen is unchanged.
167
+ - `prism_infer` route mode: set `allow_parallel_calls` to keep a reply that
168
+ calls several of the tools you offered in `allowed_tools`.
169
+
161
170
  ## What's New in v20.21.16
162
171
 
163
172
  ### Conversations: checked on your device, and free with an account
@@ -35,7 +35,7 @@ import { passesClinicalQualityGate, clinicalPlanScaffold, formatClinicalSections
35
35
  import { applyDeterministicCodingRepairs, buildCodingRepairPrompt, passesCodingQualityGate, } from "../utils/codingQualityPolicy.js";
36
36
  import { checkInputSafety, checkOutputSafety } from "../utils/safetyGate.js";
37
37
  import { callLayer1 as defaultCallLayer1, classifyDeterministicLayer1, keywordBackstop, reservedCategory, MAX_CLASSIFIER_PROMPT_LENGTH, layer1ClassifierContent, secondReadExclusion } from "../utils/layer1.js";
38
- import { getSecondReadPolicy, getAnswerCheckPolicy } from "../utils/inferencePolicy.js";
38
+ import { getSecondReadPolicy, getAnswerCheckPolicy, getClassifierInputPolicy } from "../utils/inferencePolicy.js";
39
39
  import { pseudonymizeForCheck } from "../utils/pseudonymize.js";
40
40
  import { answerGroundingBytes, answerGroundingContent, parseGroundingVerdict, arithmeticSlips, arithmeticCorrection, ANSWER_GROUNDING_OUTPUT_TOKENS, ANSWER_GROUNDING_THINK, ANSWER_GROUNDING_THINK_TOKENS, ANSWER_GROUNDING_TIMEOUT_MS, ANSWER_GROUNDING_RETRY_TIMEOUT_MS, ANSWER_GROUNDING_FOLLOW_UP_TOKENS } from "../utils/answerGrounding.js";
41
41
  import { recordInference, recordThinkOnlyRetry, formatInferenceMetrics, estimateTokens } from "../utils/inferenceMetrics.js";
@@ -269,6 +269,8 @@ function layer1HistoryCached(model, window) {
269
269
  const hit = layer1HistoryCache.get(key);
270
270
  return hit !== undefined && hit.expiresAt > performance.now();
271
271
  }
272
+ /** How long the screen waits for the classifier-input policy on its first load. */
273
+ const CLASSIFIER_INPUT_LOAD_MS = 3_000;
272
274
  /** Tokens the classifier may generate (callLayer1's num_predict); they share
273
275
  * the context with the request. */
274
276
  export const LAYER1_CLASSIFIER_OUTPUT_TOKENS = 16;
@@ -765,6 +767,11 @@ export const PRISM_INFER_TOOL = {
765
767
  "Synalux deterministic route correction. 'local': skips that correction only.",
766
768
  default: "auto",
767
769
  },
770
+ allow_parallel_calls: {
771
+ type: "boolean",
772
+ description: "Route: keep a reply of several calls only if every call is in allowed_tools.",
773
+ default: false,
774
+ },
768
775
  think: {
769
776
  type: "boolean",
770
777
  description: "<think> reasoning. Default true for chat/code, false for route; better on complex " +
@@ -772,10 +779,8 @@ export const PRISM_INFER_TOOL = {
772
779
  },
773
780
  strict_entitlements: {
774
781
  type: "boolean",
775
- description: "Fail loud instead of running with ASSUMED free-tier limits: when entitlements " +
776
- "fell back to free because the portal was unreachable (source='fallback_free'), " +
777
- "throw instead of silently applying free clamps. Portal-confirmed free plans and " +
778
- "unconfigured machines are unaffected.",
782
+ description: "Throw rather than apply assumed free limits when the portal was unreachable " +
783
+ "(source='fallback_free'). Confirmed-free and unconfigured setups are unaffected.",
779
784
  default: false,
780
785
  },
781
786
  escalation: {
@@ -874,6 +879,8 @@ export function isPrismInferArgs(args) {
874
879
  if (a.route_guard !== undefined &&
875
880
  !["auto", "local"].includes(a.route_guard))
876
881
  return false;
882
+ if (a.allow_parallel_calls !== undefined && typeof a.allow_parallel_calls !== "boolean")
883
+ return false;
877
884
  if (a.allowed_tools !== undefined) {
878
885
  if (!Array.isArray(a.allowed_tools) || a.allowed_tools.length > MAX_ROUTE_TOOLS)
879
886
  return false;
@@ -1878,6 +1885,7 @@ export async function runInfer(args, deps) {
1878
1885
  "Retry, or drop strict_entitlements to accept free clamps.");
1879
1886
  }
1880
1887
  const mode = args.mode ?? "route";
1888
+ const gateOptions = { allowParallelCalls: mode === "route" && args.allow_parallel_calls === true };
1881
1889
  // Model choice belongs here—not in session_task_route—because this layer
1882
1890
  // owns every viability input and the explicit caller override contract.
1883
1891
  const requestedCeiling = resolveRequestedModelCeiling(args);
@@ -2156,10 +2164,15 @@ export async function runInfer(args, deps) {
2156
2164
  // kept: reserved and uncertain fail closed for text, error follows
2157
2165
  // the single-prompt error path), then each turn and the prompt in
2158
2166
  // context (raise only) — see below.
2167
+ // The account's classifier-input policy; without one the classifier
2168
+ // reads the prompt as written.
2169
+ const classifierInput = await (deps.classifierInputPolicy ?? (() => getClassifierInputPolicy({ deadlineMs: CLASSIFIER_INPUT_LOAD_MS })))().catch(() => null);
2159
2170
  let l1;
2160
2171
  if (!args.messages?.length) {
2161
- // Single turn: the exact call it always was.
2162
- l1 = await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages);
2172
+ // Single turn: one call; the classifier-input policy is passed when there is one.
2173
+ l1 = classifierInput
2174
+ ? await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages, { classifierInput })
2175
+ : await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages);
2163
2176
  if (l1 !== "OBVIOUS_NOT_RESERVED")
2164
2177
  l1Layer = "prompt";
2165
2178
  }
@@ -2250,7 +2263,7 @@ export async function runInfer(args, deps) {
2250
2263
  // (review round 19: skipping it there bypassed that floor).
2251
2264
  const promptFastPath = promptRoutine && args.prompt.length <= MAX_CLASSIFIER_PROMPT_LENGTH && (resolvedImages?.length ?? 0) === 0;
2252
2265
  if (l1 !== "OBVIOUS_RESERVED" && !promptFastPath) {
2253
- l1 = raise(l1, await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages, { deterministic: false }), "prompt");
2266
+ l1 = raise(l1, await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages, { deterministic: false, ...(classifierInput ? { classifierInput } : {}) }), "prompt");
2254
2267
  }
2255
2268
  // 3. Context, raise only: one window per turn and one for the
2256
2269
  // prompt (see contextWindows), cached like any window. Skipped
@@ -2844,7 +2857,7 @@ export async function runInfer(args, deps) {
2844
2857
  let { stripped, thinkOnly } = stripThink(result.text);
2845
2858
  let output = stripped;
2846
2859
  // Quality gate — all modes. Route uses mode-aware empty floor (length===0).
2847
- let gate = passesQualityGate(output, thinkOnly, result.doneReason, mode);
2860
+ let gate = passesQualityGate(output, thinkOnly, result.doneReason, mode, gateOptions);
2848
2861
  if (gate.pass && mode === "code") {
2849
2862
  gate = passesCodingQualityGate(args.prompt, output);
2850
2863
  }
@@ -2883,7 +2896,7 @@ export async function runInfer(args, deps) {
2883
2896
  const retried = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, false, resolvedImages, ...historyArgs(args));
2884
2897
  if (retried.ok) {
2885
2898
  const retriedStrip = stripThink(retried.text);
2886
- const retriedGate = passesQualityGate(retriedStrip.stripped, retriedStrip.thinkOnly, retried.doneReason, mode);
2899
+ const retriedGate = passesQualityGate(retriedStrip.stripped, retriedStrip.thinkOnly, retried.doneReason, mode, gateOptions);
2887
2900
  // Keep the retry only if it is actually better — a retry that
2888
2901
  // truncates too must not overwrite the original with a
2889
2902
  // shorter fragment.
@@ -2913,7 +2926,7 @@ export async function runInfer(args, deps) {
2913
2926
  const deterministicRepair = applyDeterministicCodingRepairs(output, failedReason);
2914
2927
  if (deterministicRepair.changes.length > 0) {
2915
2928
  output = deterministicRepair.output;
2916
- gate = passesQualityGate(output, false, result.doneReason, mode);
2929
+ gate = passesQualityGate(output, false, result.doneReason, mode, gateOptions);
2917
2930
  if (gate.pass) {
2918
2931
  gate = passesCodingQualityGate(args.prompt, output);
2919
2932
  }
@@ -3181,6 +3194,7 @@ async function applyVerification(draft, args, deps, partial) {
3181
3194
  const mode = args.mode ?? "route";
3182
3195
  if (mode === "route") {
3183
3196
  const allowedTools = new Set(args.allowed_tools ?? DEFAULT_PRISM_ROUTE_TOOLS);
3197
+ const contractOptions = { allowParallel: args.allow_parallel_calls === true };
3184
3198
  const parsed = parseRouteOutput(draft);
3185
3199
  const shouldUsePortal = args.route_guard !== "local" &&
3186
3200
  partial.plan !== "free" &&
@@ -3198,7 +3212,7 @@ async function applyVerification(draft, args, deps, partial) {
3198
3212
  });
3199
3213
  const portalOutcome = validatePortalRouteGuardOutcome(untrustedPortalOutcome, draft, allowedTools, args.prompt);
3200
3214
  if (!portalOutcome) {
3201
- const localCheck = applyLocalRouteContract(draft, allowedTools);
3215
+ const localCheck = applyLocalRouteContract(draft, allowedTools, contractOptions);
3202
3216
  routeGuard = {
3203
3217
  ...localCheck,
3204
3218
  source: "local_fallback",
@@ -3220,7 +3234,7 @@ async function applyVerification(draft, args, deps, partial) {
3220
3234
  }
3221
3235
  }
3222
3236
  catch (error) {
3223
- const localFallback = applyLocalRouteContract(draft, allowedTools);
3237
+ const localFallback = applyLocalRouteContract(draft, allowedTools, contractOptions);
3224
3238
  routeGuard = {
3225
3239
  ...localFallback,
3226
3240
  source: "local_fallback",
@@ -3239,7 +3253,7 @@ async function applyVerification(draft, args, deps, partial) {
3239
3253
  }
3240
3254
  }
3241
3255
  else {
3242
- routeGuard = applyLocalRouteContract(draft, allowedTools);
3256
+ routeGuard = applyLocalRouteContract(draft, allowedTools, contractOptions);
3243
3257
  }
3244
3258
  routedDraft = routeGuard.output;
3245
3259
  }
@@ -1,7 +1,8 @@
1
1
  /**
2
- * Policies the on-device multi-turn features run with, served by Synalux to
3
- * plans with multi-turn: the second read's exclusion policy (the conversations
4
- * the 9b may not re-read after a 4b hedge) and the answer check's rules. Each
2
+ * Policies on-device features run with, served by Synalux to plans with
3
+ * multi-turn: the second read's exclusion policy (the conversations the 9b may
4
+ * not re-read after a 4b hedge), the answer check's rules, and the screen's
5
+ * classifier-input policy (layer1.ts classifierCopy). Each
5
6
  * release accepts exactly one artifact of each, pinned by its SHA-256, so the
6
7
  * policy a client runs is the one it was released and validated with.
7
8
  *
@@ -11,7 +12,8 @@
11
12
  * kept. Anything but the pinned, well-formed artifact is no policy: with
12
13
  * no second-read policy the second read does not run (the hedge stands); with
13
14
  * no answer-check policy a local answer to a conversation is unchecked (cloud,
14
- * else withheld).
15
+ * else withheld); with no classifier-input policy the classifier reads the
16
+ * request as written.
15
17
  */
16
18
  import { createHash } from "node:crypto";
17
19
  import { PRISM_SYNALUX_BASE_URL } from "../config.js";
@@ -25,10 +27,17 @@ export const SECOND_READ_POLICY_EVALUATOR = "second-read-exclusion/1";
25
27
  export const ANSWER_CHECK_POLICY_SHA256 = "ba12ab1f6858b68ed36b7c0551aa3381ffb45b6123eb0aacd09c9316efd27993";
26
28
  /** The mechanism this client implements (answerGrounding.ts, groundAnswer). */
27
29
  export const ANSWER_CHECK_POLICY_EVALUATOR = "answer-check/1";
30
+ /** The classifier-input artifact this release runs. */
31
+ export const CLASSIFIER_INPUT_POLICY_SHA256 = "6b215f9af8cb94c9467852d6bd20f93c3a5c33dd35c649bee77e5f69e0abe4b1";
32
+ /** The mechanism this client implements (layer1.ts classifierCopy). */
33
+ export const CLASSIFIER_INPUT_POLICY_EVALUATOR = "classifier-input/1";
28
34
  const MAX_ARTIFACT_BYTES = 64 * 1024;
29
35
  const MAX_PATTERN_CHARS = 1_024;
30
36
  const MIN_OPERATIONAL_TERMS = 8;
31
37
  const MIN_DEPLOY_DECISION = 2;
38
+ const MIN_DROP_WORDS = 8;
39
+ const MAX_WORD_CHARS = 40;
40
+ const MAX_REQUIRED_GROUPS = 4;
32
41
  /** A list longer than this is not a policy this client was released with. */
33
42
  const MAX_LIST_ENTRIES = 1_000;
34
43
  /** A group whose body repeats may be repeated at most this many times. */
@@ -274,6 +283,66 @@ export function parseAnswerCheckPolicy(bytes, expectSha256 = ANSWER_CHECK_POLICY
274
283
  return null;
275
284
  }
276
285
  }
286
+ /** The classifier-input policy from the artifact's exact bytes, or null for
287
+ * anything but the expected artifact: another hash, schema or evaluator, a
288
+ * word list below its floor or with an entry that is not one lowercase word,
289
+ * no required group or a group or qualifier naming an unlisted word, no
290
+ * words a kept sentence must offer, a token
291
+ * pattern that is oversized, refers back, repeats a varying group or does not
292
+ * compile. */
293
+ export function parseClassifierInputPolicy(bytes, expectSha256 = CLASSIFIER_INPUT_POLICY_SHA256) {
294
+ if (Buffer.byteLength(bytes, "utf8") > MAX_ARTIFACT_BYTES)
295
+ return null;
296
+ if (sha256(bytes) !== expectSha256)
297
+ return null;
298
+ let a;
299
+ try {
300
+ a = JSON.parse(bytes);
301
+ }
302
+ catch {
303
+ return null;
304
+ }
305
+ const art = a;
306
+ if (art?.schema !== 1 || art.evaluator !== CLASSIFIER_INPUT_POLICY_EVALUATOR || typeof art.classifier_input !== "object" || art.classifier_input === null)
307
+ return null;
308
+ const s = art.classifier_input;
309
+ const words = s.drop_sentence_words;
310
+ const word = (v) => typeof v === "string" && v.length > 0 && v.length <= MAX_WORD_CHARS && v === v.toLowerCase() && !/\s/.test(v);
311
+ if (!Array.isArray(words) || words.length < MIN_DROP_WORDS || words.length > MAX_LIST_ENTRIES || !words.every(word))
312
+ return null;
313
+ const listed = new Set(words);
314
+ const subset = (v) => Array.isArray(v) && v.length > 0 && v.length <= MAX_LIST_ENTRIES && v.every(w => typeof w === "string" && listed.has(w));
315
+ const groups = s.require_each;
316
+ if (!Array.isArray(groups) || groups.length === 0 || groups.length > MAX_REQUIRED_GROUPS || !groups.every(subset))
317
+ return null;
318
+ const after = s.only_after ?? {};
319
+ if (typeof after !== "object" || after === null || Array.isArray(after))
320
+ return null;
321
+ const afterEntries = Object.entries(after);
322
+ if (!afterEntries.every(([w, prev]) => listed.has(w) && subset(prev)))
323
+ return null;
324
+ const needed = s.kept_needs_one_of;
325
+ if (!Array.isArray(needed) || needed.length === 0 || needed.length > MAX_LIST_ENTRIES || !needed.every(word))
326
+ return null;
327
+ const tokenPattern = (v) => v === undefined || (typeof v === "string" && v.length > 0 && v.length <= MAX_PATTERN_CHARS && !/\\[1-9]|\\k</.test(v) && !hasNestedRepetition(v));
328
+ const also = s.also_match, afterPattern = s.only_after_pattern;
329
+ if (!tokenPattern(also) || !tokenPattern(afterPattern))
330
+ return null;
331
+ try {
332
+ // The client sets the flags (none); the artifact supplies the source only.
333
+ return {
334
+ dropWords: listed,
335
+ requireEach: groups.map(g => new Set(g)),
336
+ onlyAfter: new Map(afterEntries.map(([w, prev]) => [w, new Set(prev)])),
337
+ onlyAfterPattern: typeof afterPattern === "string" ? new RegExp(afterPattern) : null,
338
+ alsoMatch: typeof also === "string" ? new RegExp(also) : null,
339
+ keptNeedsOneOf: new Set(needed),
340
+ };
341
+ }
342
+ catch {
343
+ return null;
344
+ }
345
+ }
277
346
  /** One pinned artifact: loaded once, shared by concurrent callers, retried
278
347
  * after RETRY_AFTER_MS when a load fails, and dropped by reset() when the
279
348
  * account changes (a sign-in can switch the account and the portal). A load
@@ -358,15 +427,19 @@ async function load(o, sha, parse) {
358
427
  }
359
428
  const secondRead = pinned(SECOND_READ_POLICY_SHA256, parseSecondReadPolicy);
360
429
  const answerCheck = pinned(ANSWER_CHECK_POLICY_SHA256, parseAnswerCheckPolicy);
430
+ const classifierInput = pinned(CLASSIFIER_INPUT_POLICY_SHA256, parseClassifierInputPolicy);
361
431
  /** The pinned second-read policy, or null. */
362
432
  export const getSecondReadPolicy = (o = {}) => secondRead.get(o);
363
433
  /** The pinned answer-check policy, or null. */
364
434
  export const getAnswerCheckPolicy = (o = {}) => answerCheck.get(o);
365
- /** Drops both cached policies; the next conversation loads them again. Called
435
+ /** The pinned classifier-input policy, or null. */
436
+ export const getClassifierInputPolicy = (o = {}) => classifierInput.get(o);
437
+ /** Drops every cached policy; the next request loads them again. Called
366
438
  * when the account changes (dashboard sign-in and sign-out). */
367
439
  export function clearInferencePolicies() {
368
440
  secondRead.reset();
369
441
  answerCheck.reset();
442
+ classifierInput.reset();
370
443
  }
371
444
  /** Tests only. */
372
445
  export function _resetSecondReadPolicyForTest() {
@@ -66,6 +66,62 @@ Answer (one word):`;
66
66
  export function layer1ClassifierContent(input) {
67
67
  return LAYER1_PROMPT.replace("{prompt}", () => input);
68
68
  }
69
+ /** Question marks in the scripts prism serves: a sentence with one is never left out. */
70
+ const QUESTION_MARK = new RegExp("[?" + String.fromCharCode(0xff1f, 0xfe56, 0x061f, 0x037e, 0x00bf, 0x203d, 0x2047, 0x2048, 0x2049, 0x2e2e, 0x055e, 0x1367) + "]");
71
+ function words(sentence) {
72
+ return sentence.toLowerCase().split(/[\s,;:()]+/).map((w) => w.replace(TOKEN_EDGE, "")).filter(Boolean);
73
+ }
74
+ const TOKEN_EDGE = /^[^\w`'#.+-]+|[^\w`'#+-]+$/g;
75
+ function droppable(sentence, p) {
76
+ // A sentence with a question mark is never left out: it may be what the request asks.
77
+ if (QUESTION_MARK.test(sentence))
78
+ return false;
79
+ const hit = p.requireEach.map(() => false);
80
+ let prev = null;
81
+ for (const token of words(sentence)) {
82
+ const pattern = !!p.alsoMatch?.test(token);
83
+ if (p.dropWords.has(token)) {
84
+ const after = p.onlyAfter.get(token);
85
+ if (after && !(prev !== null && (after.has(prev) || !!p.onlyAfterPattern?.test(prev))))
86
+ return false;
87
+ p.requireEach.forEach((group, i) => { if (group.has(token))
88
+ hit[i] = true; });
89
+ }
90
+ else if (!pattern) {
91
+ return false;
92
+ }
93
+ prev = token;
94
+ }
95
+ return hit.length > 0 && hit.every(Boolean);
96
+ }
97
+ /**
98
+ * The classifier's copy of a request under a classifier-input policy. A
99
+ * sentence the policy allows is left out with one adjacent separator; every
100
+ * other character is kept, and text with nothing to leave out is returned as it is.
101
+ */
102
+ export function classifierCopy(text, p) {
103
+ // Even indexes are sentences, odd indexes the separators between them.
104
+ const parts = text.split(/((?<=[.!?])[^\S\r\n]+|[\r\n]+)/);
105
+ const keep = parts.map(() => true);
106
+ let dropped = false;
107
+ for (let i = 0; i < parts.length; i += 2) {
108
+ if (!droppable(parts[i], p))
109
+ continue;
110
+ keep[i] = false;
111
+ dropped = true;
112
+ if (i > 0 && keep[i - 1])
113
+ keep[i - 1] = false;
114
+ else if (i + 1 < parts.length)
115
+ keep[i + 1] = false;
116
+ }
117
+ if (!dropped)
118
+ return text;
119
+ // The sentences that stay must still say what is asked; otherwise the classifier reads it all.
120
+ if (!parts.some((part, i) => i % 2 === 0 && keep[i] && words(part).some((w) => p.keptNeedsOneOf.has(w))))
121
+ return text;
122
+ const out = parts.filter((_, i) => keep[i]).join("");
123
+ return out.trim() ? out : text;
124
+ }
69
125
  const VALID = new Set([
70
126
  "OBVIOUS_RESERVED",
71
127
  "OBVIOUS_NOT_RESERVED",
@@ -408,7 +464,8 @@ images, opts) {
408
464
  // short-circuits to reserved handling.
409
465
  return "OBVIOUS_RESERVED";
410
466
  }
411
- const classifierInput = oversize ? buildOversizeExcerpt(userPrompt) : userPrompt;
467
+ const excerpt = oversize ? buildOversizeExcerpt(userPrompt) : userPrompt;
468
+ const classifierInput = opts?.classifierInput && !hasImages ? classifierCopy(excerpt, opts.classifierInput) : excerpt;
412
469
  // A SYSTEM baked into the classifier model's Modelfile must not sit in front
413
470
  // of LAYER1_PROMPT. prism-coder:4b bakes a tool-routing prompt; with it the
414
471
  // private eval gate failed 5/5 runs (two hard negatives refused every run),
@@ -1,4 +1,14 @@
1
- import { parseRouteOutput } from "./routeContract.js";
1
+ import { parseRouteCalls, parseRouteOutput } from "./routeContract.js";
2
+ /** JSON with object keys sorted at every level: the same arguments in any order give one string. */
3
+ function canonicalJson(value) {
4
+ if (Array.isArray(value))
5
+ return `[${value.map(canonicalJson).join(",")}]`;
6
+ if (value !== null && typeof value === "object") {
7
+ const record = value;
8
+ return `{${Object.keys(record).sort().map((k) => `${JSON.stringify(k)}:${canonicalJson(record[k])}`).join(",")}}`;
9
+ }
10
+ return JSON.stringify(value) ?? "null";
11
+ }
2
12
  /**
3
13
  * Signal 5 — Tool-call bleed: pipe-delimited format leaking into non-tool turns.
4
14
  * Matches <|tool_call|> and <|tool_call_end|> only — NOT angle-bracket <tool_call> variants
@@ -11,8 +21,9 @@ export const TOOL_CALL_BLEED_RE = /<\|tool_call\|>|<\|tool_call_end\|>/;
11
21
  * @param thinkOnly True if the response was only <think> blocks with no answer
12
22
  * @param finishReason Ollama's finish_reason if available (e.g. "length" = truncated)
13
23
  * @param mode Inference mode — "route": empty only when blank; "chat": empty only with no letter or digit; "code"/unset: 4 chars or fewer
24
+ * @param options.allowParallelCalls Route mode: a reply of several complete calls is valid
14
25
  */
15
- export function passesQualityGate(stripped, thinkOnly, finishReason, mode) {
26
+ export function passesQualityGate(stripped, thinkOnly, finishReason, mode, options = {}) {
16
27
  // Signal 1: Think-only — model reasoned but produced no answer (check before empty)
17
28
  if (thinkOnly) {
18
29
  return { pass: false, reason: "think_only" };
@@ -37,6 +48,22 @@ export function passesQualityGate(stripped, thinkOnly, finishReason, mode) {
37
48
  if (finishReason === "length") {
38
49
  return { pass: false, reason: "hard_truncation" };
39
50
  }
51
+ // Several complete calls, when the caller asked for them (route mode). The
52
+ // envelopes repeat by design, so the prose loop checks below would fail any
53
+ // three calls; a loop here is the same call again and again.
54
+ if (mode === "route" && options.allowParallelCalls) {
55
+ const several = parseRouteCalls(stripped);
56
+ if (several.kind === "tool_calls") {
57
+ const seen = new Map();
58
+ for (const c of several.calls) {
59
+ const key = canonicalJson([c.name, c.args]);
60
+ seen.set(key, (seen.get(key) ?? 0) + 1);
61
+ if ((seen.get(key) ?? 0) >= 3)
62
+ return { pass: false, reason: "loop_detected" };
63
+ }
64
+ return { pass: true };
65
+ }
66
+ }
40
67
  // Signal 5: Tool-call bleed. The pipe envelope is invalid in chat/code,
41
68
  // but it is the canonical trained output in route mode. Route mode parses
42
69
  // the whole envelope and fails only when the contract is malformed.
@@ -273,11 +273,76 @@ export function routeServesProse(output) {
273
273
  const last = ends.reduce((a, b) => (b.at > a.at ? b : a));
274
274
  return t.slice(last.at + last.e.length).trim() !== ""; // text after the envelope
275
275
  }
276
- export function applyLocalRouteContract(draft, allowedTools = DEFAULT_PRISM_ROUTE_TOOLS) {
276
+ const OPENERS = [PIPE_START, ANGLE_START];
277
+ const MAX_PARALLEL_CALLS = 64;
278
+ /**
279
+ * A reply of complete tool-call envelopes one after another, with only
280
+ * whitespace between them. Every envelope must be closed and hold a valid
281
+ * call; any text, an unclosed envelope or a bad body makes the whole reply
282
+ * malformed. A single envelope is left to parseRouteOutput.
283
+ */
284
+ export function parseRouteCalls(output) {
285
+ if (output.length > MAX_ROUTE_OUTPUT_CHARS)
286
+ return { kind: "malformed" };
287
+ const calls = [];
288
+ let rest = output.trim();
289
+ while (rest.length > 0) {
290
+ const opener = OPENERS.find(o => rest.startsWith(o));
291
+ if (!opener || calls.length >= MAX_PARALLEL_CALLS)
292
+ return { kind: "malformed" };
293
+ let end = -1;
294
+ let endToken = "";
295
+ for (const token of END_TOKENS) {
296
+ const at = rest.indexOf(token, opener.length);
297
+ if (at >= 0 && (end < 0 || at < end)) {
298
+ end = at;
299
+ endToken = token;
300
+ }
301
+ }
302
+ if (end < 0)
303
+ return { kind: "malformed" };
304
+ const body = rest.slice(opener.length, end).trim();
305
+ if (OPENERS.some(o => body.includes(o)))
306
+ return { kind: "malformed" };
307
+ const parsed = parseToolJson(body);
308
+ if (parsed.kind !== "tool_call")
309
+ return { kind: "malformed" };
310
+ calls.push({ name: parsed.name, args: parsed.args });
311
+ rest = rest.slice(end + endToken.length).trim();
312
+ }
313
+ return calls.length > 1 ? { kind: "tool_calls", calls } : { kind: "malformed" };
314
+ }
315
+ export function applyLocalRouteContract(draft, allowedTools = DEFAULT_PRISM_ROUTE_TOOLS, options = {}) {
277
316
  const parsed = parseRouteOutput(draft);
278
317
  if (parsed.kind === "plain_text") {
279
318
  return { output: draft, action: "plain_text", source: "local" };
280
319
  }
320
+ if (parsed.kind === "malformed" && options.allowParallel) {
321
+ const several = parseRouteCalls(draft);
322
+ if (several.kind === "tool_calls") {
323
+ if (several.calls.some(c => c.name === "NO_TOOL")) {
324
+ return { output: "NO_TOOL", action: "suppressed", source: "local", reason: "malformed_tool_call" };
325
+ }
326
+ const unadvertised = several.calls.find(c => !allowedTools.has(c.name));
327
+ if (unadvertised) {
328
+ return {
329
+ output: "NO_TOOL",
330
+ action: "suppressed",
331
+ source: "local",
332
+ original_tool: unadvertised.name,
333
+ reason: "unadvertised_tool",
334
+ };
335
+ }
336
+ return {
337
+ output: draft,
338
+ action: "preserved",
339
+ source: "local",
340
+ original_tool: several.calls[0].name,
341
+ final_tool: several.calls[0].name,
342
+ calls: several.calls.map(c => c.name),
343
+ };
344
+ }
345
+ }
281
346
  if (parsed.kind === "malformed") {
282
347
  return {
283
348
  output: "NO_TOOL",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "prism-mcp-server",
3
- "version": "20.21.16",
3
+ "version": "20.21.17",
4
4
  "mcpName": "io.github.dcostenco/prism-coder",
5
5
  "description": "Persistent session memory for AI coding agents that never leaves your machine — including the on-device model that reasons over it. Restores your prior decisions, open TODOs, and changed files across sessions; adds associative recall of related past work, semantic drift detection, and local inference. Local-first by default. Works with Claude Code, Cursor, and Codex.",
6
6
  "module": "index.ts",