prism-mcp-server 20.21.16 → 20.21.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -0
- package/dist/tools/prismInferHandler.js +28 -14
- package/dist/utils/inferencePolicy.js +78 -5
- package/dist/utils/layer1.js +58 -1
- package/dist/utils/qualityGate.js +29 -2
- package/dist/utils/routeContract.js +66 -1
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -158,6 +158,15 @@ or by re-enabling after each run.
|
|
|
158
158
|
<details>
|
|
159
159
|
<summary>Release history (optional)</summary>
|
|
160
160
|
|
|
161
|
+
## What's New in v20.21.17
|
|
162
|
+
|
|
163
|
+
- Fixed: the on-device screen refused some ordinary coding requests. With a
|
|
164
|
+
Synalux account (free included), it now reads each request through a policy
|
|
165
|
+
that Synalux serves, pinned by its SHA-256. Without an account, or with
|
|
166
|
+
images attached, the screen is unchanged.
|
|
167
|
+
- `prism_infer` route mode: set `allow_parallel_calls` to keep a reply that
|
|
168
|
+
calls several of the tools you offered in `allowed_tools`.
|
|
169
|
+
|
|
161
170
|
## What's New in v20.21.16
|
|
162
171
|
|
|
163
172
|
### Conversations: checked on your device, and free with an account
|
|
@@ -35,7 +35,7 @@ import { passesClinicalQualityGate, clinicalPlanScaffold, formatClinicalSections
|
|
|
35
35
|
import { applyDeterministicCodingRepairs, buildCodingRepairPrompt, passesCodingQualityGate, } from "../utils/codingQualityPolicy.js";
|
|
36
36
|
import { checkInputSafety, checkOutputSafety } from "../utils/safetyGate.js";
|
|
37
37
|
import { callLayer1 as defaultCallLayer1, classifyDeterministicLayer1, keywordBackstop, reservedCategory, MAX_CLASSIFIER_PROMPT_LENGTH, layer1ClassifierContent, secondReadExclusion } from "../utils/layer1.js";
|
|
38
|
-
import { getSecondReadPolicy, getAnswerCheckPolicy } from "../utils/inferencePolicy.js";
|
|
38
|
+
import { getSecondReadPolicy, getAnswerCheckPolicy, getClassifierInputPolicy } from "../utils/inferencePolicy.js";
|
|
39
39
|
import { pseudonymizeForCheck } from "../utils/pseudonymize.js";
|
|
40
40
|
import { answerGroundingBytes, answerGroundingContent, parseGroundingVerdict, arithmeticSlips, arithmeticCorrection, ANSWER_GROUNDING_OUTPUT_TOKENS, ANSWER_GROUNDING_THINK, ANSWER_GROUNDING_THINK_TOKENS, ANSWER_GROUNDING_TIMEOUT_MS, ANSWER_GROUNDING_RETRY_TIMEOUT_MS, ANSWER_GROUNDING_FOLLOW_UP_TOKENS } from "../utils/answerGrounding.js";
|
|
41
41
|
import { recordInference, recordThinkOnlyRetry, formatInferenceMetrics, estimateTokens } from "../utils/inferenceMetrics.js";
|
|
@@ -269,6 +269,8 @@ function layer1HistoryCached(model, window) {
|
|
|
269
269
|
const hit = layer1HistoryCache.get(key);
|
|
270
270
|
return hit !== undefined && hit.expiresAt > performance.now();
|
|
271
271
|
}
|
|
272
|
+
/** How long the screen waits for the classifier-input policy on its first load. */
|
|
273
|
+
const CLASSIFIER_INPUT_LOAD_MS = 3_000;
|
|
272
274
|
/** Tokens the classifier may generate (callLayer1's num_predict); they share
|
|
273
275
|
* the context with the request. */
|
|
274
276
|
export const LAYER1_CLASSIFIER_OUTPUT_TOKENS = 16;
|
|
@@ -765,6 +767,11 @@ export const PRISM_INFER_TOOL = {
|
|
|
765
767
|
"Synalux deterministic route correction. 'local': skips that correction only.",
|
|
766
768
|
default: "auto",
|
|
767
769
|
},
|
|
770
|
+
allow_parallel_calls: {
|
|
771
|
+
type: "boolean",
|
|
772
|
+
description: "Route: keep a reply of several calls only if every call is in allowed_tools.",
|
|
773
|
+
default: false,
|
|
774
|
+
},
|
|
768
775
|
think: {
|
|
769
776
|
type: "boolean",
|
|
770
777
|
description: "<think> reasoning. Default true for chat/code, false for route; better on complex " +
|
|
@@ -772,10 +779,8 @@ export const PRISM_INFER_TOOL = {
|
|
|
772
779
|
},
|
|
773
780
|
strict_entitlements: {
|
|
774
781
|
type: "boolean",
|
|
775
|
-
description: "
|
|
776
|
-
"
|
|
777
|
-
"throw instead of silently applying free clamps. Portal-confirmed free plans and " +
|
|
778
|
-
"unconfigured machines are unaffected.",
|
|
782
|
+
description: "Throw rather than apply assumed free limits when the portal was unreachable " +
|
|
783
|
+
"(source='fallback_free'). Confirmed-free and unconfigured setups are unaffected.",
|
|
779
784
|
default: false,
|
|
780
785
|
},
|
|
781
786
|
escalation: {
|
|
@@ -874,6 +879,8 @@ export function isPrismInferArgs(args) {
|
|
|
874
879
|
if (a.route_guard !== undefined &&
|
|
875
880
|
!["auto", "local"].includes(a.route_guard))
|
|
876
881
|
return false;
|
|
882
|
+
if (a.allow_parallel_calls !== undefined && typeof a.allow_parallel_calls !== "boolean")
|
|
883
|
+
return false;
|
|
877
884
|
if (a.allowed_tools !== undefined) {
|
|
878
885
|
if (!Array.isArray(a.allowed_tools) || a.allowed_tools.length > MAX_ROUTE_TOOLS)
|
|
879
886
|
return false;
|
|
@@ -1878,6 +1885,7 @@ export async function runInfer(args, deps) {
|
|
|
1878
1885
|
"Retry, or drop strict_entitlements to accept free clamps.");
|
|
1879
1886
|
}
|
|
1880
1887
|
const mode = args.mode ?? "route";
|
|
1888
|
+
const gateOptions = { allowParallelCalls: mode === "route" && args.allow_parallel_calls === true };
|
|
1881
1889
|
// Model choice belongs here—not in session_task_route—because this layer
|
|
1882
1890
|
// owns every viability input and the explicit caller override contract.
|
|
1883
1891
|
const requestedCeiling = resolveRequestedModelCeiling(args);
|
|
@@ -2156,10 +2164,15 @@ export async function runInfer(args, deps) {
|
|
|
2156
2164
|
// kept: reserved and uncertain fail closed for text, error follows
|
|
2157
2165
|
// the single-prompt error path), then each turn and the prompt in
|
|
2158
2166
|
// context (raise only) — see below.
|
|
2167
|
+
// The account's classifier-input policy; without one the classifier
|
|
2168
|
+
// reads the prompt as written.
|
|
2169
|
+
const classifierInput = await (deps.classifierInputPolicy ?? (() => getClassifierInputPolicy({ deadlineMs: CLASSIFIER_INPUT_LOAD_MS })))().catch(() => null);
|
|
2159
2170
|
let l1;
|
|
2160
2171
|
if (!args.messages?.length) {
|
|
2161
|
-
// Single turn: the
|
|
2162
|
-
l1 =
|
|
2172
|
+
// Single turn: one call; the classifier-input policy is passed when there is one.
|
|
2173
|
+
l1 = classifierInput
|
|
2174
|
+
? await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages, { classifierInput })
|
|
2175
|
+
: await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages);
|
|
2163
2176
|
if (l1 !== "OBVIOUS_NOT_RESERVED")
|
|
2164
2177
|
l1Layer = "prompt";
|
|
2165
2178
|
}
|
|
@@ -2250,7 +2263,7 @@ export async function runInfer(args, deps) {
|
|
|
2250
2263
|
// (review round 19: skipping it there bypassed that floor).
|
|
2251
2264
|
const promptFastPath = promptRoutine && args.prompt.length <= MAX_CLASSIFIER_PROMPT_LENGTH && (resolvedImages?.length ?? 0) === 0;
|
|
2252
2265
|
if (l1 !== "OBVIOUS_RESERVED" && !promptFastPath) {
|
|
2253
|
-
l1 = raise(l1, await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages, { deterministic: false }), "prompt");
|
|
2266
|
+
l1 = raise(l1, await l1fn(args.prompt, deps.ollamaUrl, l1Model, undefined, resolvedImages, { deterministic: false, ...(classifierInput ? { classifierInput } : {}) }), "prompt");
|
|
2254
2267
|
}
|
|
2255
2268
|
// 3. Context, raise only: one window per turn and one for the
|
|
2256
2269
|
// prompt (see contextWindows), cached like any window. Skipped
|
|
@@ -2844,7 +2857,7 @@ export async function runInfer(args, deps) {
|
|
|
2844
2857
|
let { stripped, thinkOnly } = stripThink(result.text);
|
|
2845
2858
|
let output = stripped;
|
|
2846
2859
|
// Quality gate — all modes. Route uses mode-aware empty floor (length===0).
|
|
2847
|
-
let gate = passesQualityGate(output, thinkOnly, result.doneReason, mode);
|
|
2860
|
+
let gate = passesQualityGate(output, thinkOnly, result.doneReason, mode, gateOptions);
|
|
2848
2861
|
if (gate.pass && mode === "code") {
|
|
2849
2862
|
gate = passesCodingQualityGate(args.prompt, output);
|
|
2850
2863
|
}
|
|
@@ -2883,7 +2896,7 @@ export async function runInfer(args, deps) {
|
|
|
2883
2896
|
const retried = await deps.callLocal(deps.ollamaUrl, ollamaName, args.prompt, effectiveSystem, tierTokens, temperature, timeout, false, resolvedImages, ...historyArgs(args));
|
|
2884
2897
|
if (retried.ok) {
|
|
2885
2898
|
const retriedStrip = stripThink(retried.text);
|
|
2886
|
-
const retriedGate = passesQualityGate(retriedStrip.stripped, retriedStrip.thinkOnly, retried.doneReason, mode);
|
|
2899
|
+
const retriedGate = passesQualityGate(retriedStrip.stripped, retriedStrip.thinkOnly, retried.doneReason, mode, gateOptions);
|
|
2887
2900
|
// Keep the retry only if it is actually better — a retry that
|
|
2888
2901
|
// truncates too must not overwrite the original with a
|
|
2889
2902
|
// shorter fragment.
|
|
@@ -2913,7 +2926,7 @@ export async function runInfer(args, deps) {
|
|
|
2913
2926
|
const deterministicRepair = applyDeterministicCodingRepairs(output, failedReason);
|
|
2914
2927
|
if (deterministicRepair.changes.length > 0) {
|
|
2915
2928
|
output = deterministicRepair.output;
|
|
2916
|
-
gate = passesQualityGate(output, false, result.doneReason, mode);
|
|
2929
|
+
gate = passesQualityGate(output, false, result.doneReason, mode, gateOptions);
|
|
2917
2930
|
if (gate.pass) {
|
|
2918
2931
|
gate = passesCodingQualityGate(args.prompt, output);
|
|
2919
2932
|
}
|
|
@@ -3181,6 +3194,7 @@ async function applyVerification(draft, args, deps, partial) {
|
|
|
3181
3194
|
const mode = args.mode ?? "route";
|
|
3182
3195
|
if (mode === "route") {
|
|
3183
3196
|
const allowedTools = new Set(args.allowed_tools ?? DEFAULT_PRISM_ROUTE_TOOLS);
|
|
3197
|
+
const contractOptions = { allowParallel: args.allow_parallel_calls === true };
|
|
3184
3198
|
const parsed = parseRouteOutput(draft);
|
|
3185
3199
|
const shouldUsePortal = args.route_guard !== "local" &&
|
|
3186
3200
|
partial.plan !== "free" &&
|
|
@@ -3198,7 +3212,7 @@ async function applyVerification(draft, args, deps, partial) {
|
|
|
3198
3212
|
});
|
|
3199
3213
|
const portalOutcome = validatePortalRouteGuardOutcome(untrustedPortalOutcome, draft, allowedTools, args.prompt);
|
|
3200
3214
|
if (!portalOutcome) {
|
|
3201
|
-
const localCheck = applyLocalRouteContract(draft, allowedTools);
|
|
3215
|
+
const localCheck = applyLocalRouteContract(draft, allowedTools, contractOptions);
|
|
3202
3216
|
routeGuard = {
|
|
3203
3217
|
...localCheck,
|
|
3204
3218
|
source: "local_fallback",
|
|
@@ -3220,7 +3234,7 @@ async function applyVerification(draft, args, deps, partial) {
|
|
|
3220
3234
|
}
|
|
3221
3235
|
}
|
|
3222
3236
|
catch (error) {
|
|
3223
|
-
const localFallback = applyLocalRouteContract(draft, allowedTools);
|
|
3237
|
+
const localFallback = applyLocalRouteContract(draft, allowedTools, contractOptions);
|
|
3224
3238
|
routeGuard = {
|
|
3225
3239
|
...localFallback,
|
|
3226
3240
|
source: "local_fallback",
|
|
@@ -3239,7 +3253,7 @@ async function applyVerification(draft, args, deps, partial) {
|
|
|
3239
3253
|
}
|
|
3240
3254
|
}
|
|
3241
3255
|
else {
|
|
3242
|
-
routeGuard = applyLocalRouteContract(draft, allowedTools);
|
|
3256
|
+
routeGuard = applyLocalRouteContract(draft, allowedTools, contractOptions);
|
|
3243
3257
|
}
|
|
3244
3258
|
routedDraft = routeGuard.output;
|
|
3245
3259
|
}
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Policies
|
|
3
|
-
*
|
|
4
|
-
*
|
|
2
|
+
* Policies on-device features run with, served by Synalux to plans with
|
|
3
|
+
* multi-turn: the second read's exclusion policy (the conversations the 9b may
|
|
4
|
+
* not re-read after a 4b hedge), the answer check's rules, and the screen's
|
|
5
|
+
* classifier-input policy (layer1.ts classifierCopy). Each
|
|
5
6
|
* release accepts exactly one artifact of each, pinned by its SHA-256, so the
|
|
6
7
|
* policy a client runs is the one it was released and validated with.
|
|
7
8
|
*
|
|
@@ -11,7 +12,8 @@
|
|
|
11
12
|
* kept. Anything but the pinned, well-formed artifact is no policy: with
|
|
12
13
|
* no second-read policy the second read does not run (the hedge stands); with
|
|
13
14
|
* no answer-check policy a local answer to a conversation is unchecked (cloud,
|
|
14
|
-
* else withheld)
|
|
15
|
+
* else withheld); with no classifier-input policy the classifier reads the
|
|
16
|
+
* request as written.
|
|
15
17
|
*/
|
|
16
18
|
import { createHash } from "node:crypto";
|
|
17
19
|
import { PRISM_SYNALUX_BASE_URL } from "../config.js";
|
|
@@ -25,10 +27,17 @@ export const SECOND_READ_POLICY_EVALUATOR = "second-read-exclusion/1";
|
|
|
25
27
|
export const ANSWER_CHECK_POLICY_SHA256 = "ba12ab1f6858b68ed36b7c0551aa3381ffb45b6123eb0aacd09c9316efd27993";
|
|
26
28
|
/** The mechanism this client implements (answerGrounding.ts, groundAnswer). */
|
|
27
29
|
export const ANSWER_CHECK_POLICY_EVALUATOR = "answer-check/1";
|
|
30
|
+
/** The classifier-input artifact this release runs. */
|
|
31
|
+
export const CLASSIFIER_INPUT_POLICY_SHA256 = "6b215f9af8cb94c9467852d6bd20f93c3a5c33dd35c649bee77e5f69e0abe4b1";
|
|
32
|
+
/** The mechanism this client implements (layer1.ts classifierCopy). */
|
|
33
|
+
export const CLASSIFIER_INPUT_POLICY_EVALUATOR = "classifier-input/1";
|
|
28
34
|
const MAX_ARTIFACT_BYTES = 64 * 1024;
|
|
29
35
|
const MAX_PATTERN_CHARS = 1_024;
|
|
30
36
|
const MIN_OPERATIONAL_TERMS = 8;
|
|
31
37
|
const MIN_DEPLOY_DECISION = 2;
|
|
38
|
+
const MIN_DROP_WORDS = 8;
|
|
39
|
+
const MAX_WORD_CHARS = 40;
|
|
40
|
+
const MAX_REQUIRED_GROUPS = 4;
|
|
32
41
|
/** A list longer than this is not a policy this client was released with. */
|
|
33
42
|
const MAX_LIST_ENTRIES = 1_000;
|
|
34
43
|
/** A group whose body repeats may be repeated at most this many times. */
|
|
@@ -274,6 +283,66 @@ export function parseAnswerCheckPolicy(bytes, expectSha256 = ANSWER_CHECK_POLICY
|
|
|
274
283
|
return null;
|
|
275
284
|
}
|
|
276
285
|
}
|
|
286
|
+
/** The classifier-input policy from the artifact's exact bytes, or null for
|
|
287
|
+
* anything but the expected artifact: another hash, schema or evaluator, a
|
|
288
|
+
* word list below its floor or with an entry that is not one lowercase word,
|
|
289
|
+
* no required group or a group or qualifier naming an unlisted word, no
|
|
290
|
+
* words a kept sentence must offer, a token
|
|
291
|
+
* pattern that is oversized, refers back, repeats a varying group or does not
|
|
292
|
+
* compile. */
|
|
293
|
+
export function parseClassifierInputPolicy(bytes, expectSha256 = CLASSIFIER_INPUT_POLICY_SHA256) {
|
|
294
|
+
if (Buffer.byteLength(bytes, "utf8") > MAX_ARTIFACT_BYTES)
|
|
295
|
+
return null;
|
|
296
|
+
if (sha256(bytes) !== expectSha256)
|
|
297
|
+
return null;
|
|
298
|
+
let a;
|
|
299
|
+
try {
|
|
300
|
+
a = JSON.parse(bytes);
|
|
301
|
+
}
|
|
302
|
+
catch {
|
|
303
|
+
return null;
|
|
304
|
+
}
|
|
305
|
+
const art = a;
|
|
306
|
+
if (art?.schema !== 1 || art.evaluator !== CLASSIFIER_INPUT_POLICY_EVALUATOR || typeof art.classifier_input !== "object" || art.classifier_input === null)
|
|
307
|
+
return null;
|
|
308
|
+
const s = art.classifier_input;
|
|
309
|
+
const words = s.drop_sentence_words;
|
|
310
|
+
const word = (v) => typeof v === "string" && v.length > 0 && v.length <= MAX_WORD_CHARS && v === v.toLowerCase() && !/\s/.test(v);
|
|
311
|
+
if (!Array.isArray(words) || words.length < MIN_DROP_WORDS || words.length > MAX_LIST_ENTRIES || !words.every(word))
|
|
312
|
+
return null;
|
|
313
|
+
const listed = new Set(words);
|
|
314
|
+
const subset = (v) => Array.isArray(v) && v.length > 0 && v.length <= MAX_LIST_ENTRIES && v.every(w => typeof w === "string" && listed.has(w));
|
|
315
|
+
const groups = s.require_each;
|
|
316
|
+
if (!Array.isArray(groups) || groups.length === 0 || groups.length > MAX_REQUIRED_GROUPS || !groups.every(subset))
|
|
317
|
+
return null;
|
|
318
|
+
const after = s.only_after ?? {};
|
|
319
|
+
if (typeof after !== "object" || after === null || Array.isArray(after))
|
|
320
|
+
return null;
|
|
321
|
+
const afterEntries = Object.entries(after);
|
|
322
|
+
if (!afterEntries.every(([w, prev]) => listed.has(w) && subset(prev)))
|
|
323
|
+
return null;
|
|
324
|
+
const needed = s.kept_needs_one_of;
|
|
325
|
+
if (!Array.isArray(needed) || needed.length === 0 || needed.length > MAX_LIST_ENTRIES || !needed.every(word))
|
|
326
|
+
return null;
|
|
327
|
+
const tokenPattern = (v) => v === undefined || (typeof v === "string" && v.length > 0 && v.length <= MAX_PATTERN_CHARS && !/\\[1-9]|\\k</.test(v) && !hasNestedRepetition(v));
|
|
328
|
+
const also = s.also_match, afterPattern = s.only_after_pattern;
|
|
329
|
+
if (!tokenPattern(also) || !tokenPattern(afterPattern))
|
|
330
|
+
return null;
|
|
331
|
+
try {
|
|
332
|
+
// The client sets the flags (none); the artifact supplies the source only.
|
|
333
|
+
return {
|
|
334
|
+
dropWords: listed,
|
|
335
|
+
requireEach: groups.map(g => new Set(g)),
|
|
336
|
+
onlyAfter: new Map(afterEntries.map(([w, prev]) => [w, new Set(prev)])),
|
|
337
|
+
onlyAfterPattern: typeof afterPattern === "string" ? new RegExp(afterPattern) : null,
|
|
338
|
+
alsoMatch: typeof also === "string" ? new RegExp(also) : null,
|
|
339
|
+
keptNeedsOneOf: new Set(needed),
|
|
340
|
+
};
|
|
341
|
+
}
|
|
342
|
+
catch {
|
|
343
|
+
return null;
|
|
344
|
+
}
|
|
345
|
+
}
|
|
277
346
|
/** One pinned artifact: loaded once, shared by concurrent callers, retried
|
|
278
347
|
* after RETRY_AFTER_MS when a load fails, and dropped by reset() when the
|
|
279
348
|
* account changes (a sign-in can switch the account and the portal). A load
|
|
@@ -358,15 +427,19 @@ async function load(o, sha, parse) {
|
|
|
358
427
|
}
|
|
359
428
|
const secondRead = pinned(SECOND_READ_POLICY_SHA256, parseSecondReadPolicy);
|
|
360
429
|
const answerCheck = pinned(ANSWER_CHECK_POLICY_SHA256, parseAnswerCheckPolicy);
|
|
430
|
+
const classifierInput = pinned(CLASSIFIER_INPUT_POLICY_SHA256, parseClassifierInputPolicy);
|
|
361
431
|
/** The pinned second-read policy, or null. */
|
|
362
432
|
export const getSecondReadPolicy = (o = {}) => secondRead.get(o);
|
|
363
433
|
/** The pinned answer-check policy, or null. */
|
|
364
434
|
export const getAnswerCheckPolicy = (o = {}) => answerCheck.get(o);
|
|
365
|
-
/**
|
|
435
|
+
/** The pinned classifier-input policy, or null. */
|
|
436
|
+
export const getClassifierInputPolicy = (o = {}) => classifierInput.get(o);
|
|
437
|
+
/** Drops every cached policy; the next request loads them again. Called
|
|
366
438
|
* when the account changes (dashboard sign-in and sign-out). */
|
|
367
439
|
export function clearInferencePolicies() {
|
|
368
440
|
secondRead.reset();
|
|
369
441
|
answerCheck.reset();
|
|
442
|
+
classifierInput.reset();
|
|
370
443
|
}
|
|
371
444
|
/** Tests only. */
|
|
372
445
|
export function _resetSecondReadPolicyForTest() {
|
package/dist/utils/layer1.js
CHANGED
|
@@ -66,6 +66,62 @@ Answer (one word):`;
|
|
|
66
66
|
export function layer1ClassifierContent(input) {
|
|
67
67
|
return LAYER1_PROMPT.replace("{prompt}", () => input);
|
|
68
68
|
}
|
|
69
|
+
/** Question marks in the scripts prism serves: a sentence with one is never left out. */
|
|
70
|
+
const QUESTION_MARK = new RegExp("[?" + String.fromCharCode(0xff1f, 0xfe56, 0x061f, 0x037e, 0x00bf, 0x203d, 0x2047, 0x2048, 0x2049, 0x2e2e, 0x055e, 0x1367) + "]");
|
|
71
|
+
function words(sentence) {
|
|
72
|
+
return sentence.toLowerCase().split(/[\s,;:()]+/).map((w) => w.replace(TOKEN_EDGE, "")).filter(Boolean);
|
|
73
|
+
}
|
|
74
|
+
const TOKEN_EDGE = /^[^\w`'#.+-]+|[^\w`'#+-]+$/g;
|
|
75
|
+
function droppable(sentence, p) {
|
|
76
|
+
// A sentence with a question mark is never left out: it may be what the request asks.
|
|
77
|
+
if (QUESTION_MARK.test(sentence))
|
|
78
|
+
return false;
|
|
79
|
+
const hit = p.requireEach.map(() => false);
|
|
80
|
+
let prev = null;
|
|
81
|
+
for (const token of words(sentence)) {
|
|
82
|
+
const pattern = !!p.alsoMatch?.test(token);
|
|
83
|
+
if (p.dropWords.has(token)) {
|
|
84
|
+
const after = p.onlyAfter.get(token);
|
|
85
|
+
if (after && !(prev !== null && (after.has(prev) || !!p.onlyAfterPattern?.test(prev))))
|
|
86
|
+
return false;
|
|
87
|
+
p.requireEach.forEach((group, i) => { if (group.has(token))
|
|
88
|
+
hit[i] = true; });
|
|
89
|
+
}
|
|
90
|
+
else if (!pattern) {
|
|
91
|
+
return false;
|
|
92
|
+
}
|
|
93
|
+
prev = token;
|
|
94
|
+
}
|
|
95
|
+
return hit.length > 0 && hit.every(Boolean);
|
|
96
|
+
}
|
|
97
|
+
/**
|
|
98
|
+
* The classifier's copy of a request under a classifier-input policy. A
|
|
99
|
+
* sentence the policy allows is left out with one adjacent separator; every
|
|
100
|
+
* other character is kept, and text with nothing to leave out is returned as it is.
|
|
101
|
+
*/
|
|
102
|
+
export function classifierCopy(text, p) {
|
|
103
|
+
// Even indexes are sentences, odd indexes the separators between them.
|
|
104
|
+
const parts = text.split(/((?<=[.!?])[^\S\r\n]+|[\r\n]+)/);
|
|
105
|
+
const keep = parts.map(() => true);
|
|
106
|
+
let dropped = false;
|
|
107
|
+
for (let i = 0; i < parts.length; i += 2) {
|
|
108
|
+
if (!droppable(parts[i], p))
|
|
109
|
+
continue;
|
|
110
|
+
keep[i] = false;
|
|
111
|
+
dropped = true;
|
|
112
|
+
if (i > 0 && keep[i - 1])
|
|
113
|
+
keep[i - 1] = false;
|
|
114
|
+
else if (i + 1 < parts.length)
|
|
115
|
+
keep[i + 1] = false;
|
|
116
|
+
}
|
|
117
|
+
if (!dropped)
|
|
118
|
+
return text;
|
|
119
|
+
// The sentences that stay must still say what is asked; otherwise the classifier reads it all.
|
|
120
|
+
if (!parts.some((part, i) => i % 2 === 0 && keep[i] && words(part).some((w) => p.keptNeedsOneOf.has(w))))
|
|
121
|
+
return text;
|
|
122
|
+
const out = parts.filter((_, i) => keep[i]).join("");
|
|
123
|
+
return out.trim() ? out : text;
|
|
124
|
+
}
|
|
69
125
|
const VALID = new Set([
|
|
70
126
|
"OBVIOUS_RESERVED",
|
|
71
127
|
"OBVIOUS_NOT_RESERVED",
|
|
@@ -408,7 +464,8 @@ images, opts) {
|
|
|
408
464
|
// short-circuits to reserved handling.
|
|
409
465
|
return "OBVIOUS_RESERVED";
|
|
410
466
|
}
|
|
411
|
-
const
|
|
467
|
+
const excerpt = oversize ? buildOversizeExcerpt(userPrompt) : userPrompt;
|
|
468
|
+
const classifierInput = opts?.classifierInput && !hasImages ? classifierCopy(excerpt, opts.classifierInput) : excerpt;
|
|
412
469
|
// A SYSTEM baked into the classifier model's Modelfile must not sit in front
|
|
413
470
|
// of LAYER1_PROMPT. prism-coder:4b bakes a tool-routing prompt; with it the
|
|
414
471
|
// private eval gate failed 5/5 runs (two hard negatives refused every run),
|
|
@@ -1,4 +1,14 @@
|
|
|
1
|
-
import { parseRouteOutput } from "./routeContract.js";
|
|
1
|
+
import { parseRouteCalls, parseRouteOutput } from "./routeContract.js";
|
|
2
|
+
/** JSON with object keys sorted at every level: the same arguments in any order give one string. */
|
|
3
|
+
function canonicalJson(value) {
|
|
4
|
+
if (Array.isArray(value))
|
|
5
|
+
return `[${value.map(canonicalJson).join(",")}]`;
|
|
6
|
+
if (value !== null && typeof value === "object") {
|
|
7
|
+
const record = value;
|
|
8
|
+
return `{${Object.keys(record).sort().map((k) => `${JSON.stringify(k)}:${canonicalJson(record[k])}`).join(",")}}`;
|
|
9
|
+
}
|
|
10
|
+
return JSON.stringify(value) ?? "null";
|
|
11
|
+
}
|
|
2
12
|
/**
|
|
3
13
|
* Signal 5 — Tool-call bleed: pipe-delimited format leaking into non-tool turns.
|
|
4
14
|
* Matches <|tool_call|> and <|tool_call_end|> only — NOT angle-bracket <tool_call> variants
|
|
@@ -11,8 +21,9 @@ export const TOOL_CALL_BLEED_RE = /<\|tool_call\|>|<\|tool_call_end\|>/;
|
|
|
11
21
|
* @param thinkOnly True if the response was only <think> blocks with no answer
|
|
12
22
|
* @param finishReason Ollama's finish_reason if available (e.g. "length" = truncated)
|
|
13
23
|
* @param mode Inference mode — "route": empty only when blank; "chat": empty only with no letter or digit; "code"/unset: 4 chars or fewer
|
|
24
|
+
* @param options.allowParallelCalls Route mode: a reply of several complete calls is valid
|
|
14
25
|
*/
|
|
15
|
-
export function passesQualityGate(stripped, thinkOnly, finishReason, mode) {
|
|
26
|
+
export function passesQualityGate(stripped, thinkOnly, finishReason, mode, options = {}) {
|
|
16
27
|
// Signal 1: Think-only — model reasoned but produced no answer (check before empty)
|
|
17
28
|
if (thinkOnly) {
|
|
18
29
|
return { pass: false, reason: "think_only" };
|
|
@@ -37,6 +48,22 @@ export function passesQualityGate(stripped, thinkOnly, finishReason, mode) {
|
|
|
37
48
|
if (finishReason === "length") {
|
|
38
49
|
return { pass: false, reason: "hard_truncation" };
|
|
39
50
|
}
|
|
51
|
+
// Several complete calls, when the caller asked for them (route mode). The
|
|
52
|
+
// envelopes repeat by design, so the prose loop checks below would fail any
|
|
53
|
+
// three calls; a loop here is the same call again and again.
|
|
54
|
+
if (mode === "route" && options.allowParallelCalls) {
|
|
55
|
+
const several = parseRouteCalls(stripped);
|
|
56
|
+
if (several.kind === "tool_calls") {
|
|
57
|
+
const seen = new Map();
|
|
58
|
+
for (const c of several.calls) {
|
|
59
|
+
const key = canonicalJson([c.name, c.args]);
|
|
60
|
+
seen.set(key, (seen.get(key) ?? 0) + 1);
|
|
61
|
+
if ((seen.get(key) ?? 0) >= 3)
|
|
62
|
+
return { pass: false, reason: "loop_detected" };
|
|
63
|
+
}
|
|
64
|
+
return { pass: true };
|
|
65
|
+
}
|
|
66
|
+
}
|
|
40
67
|
// Signal 5: Tool-call bleed. The pipe envelope is invalid in chat/code,
|
|
41
68
|
// but it is the canonical trained output in route mode. Route mode parses
|
|
42
69
|
// the whole envelope and fails only when the contract is malformed.
|
|
@@ -273,11 +273,76 @@ export function routeServesProse(output) {
|
|
|
273
273
|
const last = ends.reduce((a, b) => (b.at > a.at ? b : a));
|
|
274
274
|
return t.slice(last.at + last.e.length).trim() !== ""; // text after the envelope
|
|
275
275
|
}
|
|
276
|
-
|
|
276
|
+
const OPENERS = [PIPE_START, ANGLE_START];
|
|
277
|
+
const MAX_PARALLEL_CALLS = 64;
|
|
278
|
+
/**
|
|
279
|
+
* A reply of complete tool-call envelopes one after another, with only
|
|
280
|
+
* whitespace between them. Every envelope must be closed and hold a valid
|
|
281
|
+
* call; any text, an unclosed envelope or a bad body makes the whole reply
|
|
282
|
+
* malformed. A single envelope is left to parseRouteOutput.
|
|
283
|
+
*/
|
|
284
|
+
export function parseRouteCalls(output) {
|
|
285
|
+
if (output.length > MAX_ROUTE_OUTPUT_CHARS)
|
|
286
|
+
return { kind: "malformed" };
|
|
287
|
+
const calls = [];
|
|
288
|
+
let rest = output.trim();
|
|
289
|
+
while (rest.length > 0) {
|
|
290
|
+
const opener = OPENERS.find(o => rest.startsWith(o));
|
|
291
|
+
if (!opener || calls.length >= MAX_PARALLEL_CALLS)
|
|
292
|
+
return { kind: "malformed" };
|
|
293
|
+
let end = -1;
|
|
294
|
+
let endToken = "";
|
|
295
|
+
for (const token of END_TOKENS) {
|
|
296
|
+
const at = rest.indexOf(token, opener.length);
|
|
297
|
+
if (at >= 0 && (end < 0 || at < end)) {
|
|
298
|
+
end = at;
|
|
299
|
+
endToken = token;
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
if (end < 0)
|
|
303
|
+
return { kind: "malformed" };
|
|
304
|
+
const body = rest.slice(opener.length, end).trim();
|
|
305
|
+
if (OPENERS.some(o => body.includes(o)))
|
|
306
|
+
return { kind: "malformed" };
|
|
307
|
+
const parsed = parseToolJson(body);
|
|
308
|
+
if (parsed.kind !== "tool_call")
|
|
309
|
+
return { kind: "malformed" };
|
|
310
|
+
calls.push({ name: parsed.name, args: parsed.args });
|
|
311
|
+
rest = rest.slice(end + endToken.length).trim();
|
|
312
|
+
}
|
|
313
|
+
return calls.length > 1 ? { kind: "tool_calls", calls } : { kind: "malformed" };
|
|
314
|
+
}
|
|
315
|
+
export function applyLocalRouteContract(draft, allowedTools = DEFAULT_PRISM_ROUTE_TOOLS, options = {}) {
|
|
277
316
|
const parsed = parseRouteOutput(draft);
|
|
278
317
|
if (parsed.kind === "plain_text") {
|
|
279
318
|
return { output: draft, action: "plain_text", source: "local" };
|
|
280
319
|
}
|
|
320
|
+
if (parsed.kind === "malformed" && options.allowParallel) {
|
|
321
|
+
const several = parseRouteCalls(draft);
|
|
322
|
+
if (several.kind === "tool_calls") {
|
|
323
|
+
if (several.calls.some(c => c.name === "NO_TOOL")) {
|
|
324
|
+
return { output: "NO_TOOL", action: "suppressed", source: "local", reason: "malformed_tool_call" };
|
|
325
|
+
}
|
|
326
|
+
const unadvertised = several.calls.find(c => !allowedTools.has(c.name));
|
|
327
|
+
if (unadvertised) {
|
|
328
|
+
return {
|
|
329
|
+
output: "NO_TOOL",
|
|
330
|
+
action: "suppressed",
|
|
331
|
+
source: "local",
|
|
332
|
+
original_tool: unadvertised.name,
|
|
333
|
+
reason: "unadvertised_tool",
|
|
334
|
+
};
|
|
335
|
+
}
|
|
336
|
+
return {
|
|
337
|
+
output: draft,
|
|
338
|
+
action: "preserved",
|
|
339
|
+
source: "local",
|
|
340
|
+
original_tool: several.calls[0].name,
|
|
341
|
+
final_tool: several.calls[0].name,
|
|
342
|
+
calls: several.calls.map(c => c.name),
|
|
343
|
+
};
|
|
344
|
+
}
|
|
345
|
+
}
|
|
281
346
|
if (parsed.kind === "malformed") {
|
|
282
347
|
return {
|
|
283
348
|
output: "NO_TOOL",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "prism-mcp-server",
|
|
3
|
-
"version": "20.21.
|
|
3
|
+
"version": "20.21.17",
|
|
4
4
|
"mcpName": "io.github.dcostenco/prism-coder",
|
|
5
5
|
"description": "Persistent session memory for AI coding agents that never leaves your machine — including the on-device model that reasons over it. Restores your prior decisions, open TODOs, and changed files across sessions; adds associative recall of related past work, semantic drift detection, and local inference. Local-first by default. Works with Claude Code, Cursor, and Codex.",
|
|
6
6
|
"module": "index.ts",
|