prism-mcp-server 20.21.2 → 20.21.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -30,6 +30,7 @@ import { getEntitlements, clampCeiling, multiTurnPolicy, ABSOLUTE_MULTI_TURN } f
|
|
|
30
30
|
import { ddLog } from "../utils/ddLogger.js";
|
|
31
31
|
import { stripThink } from "../utils/thinkStrip.js";
|
|
32
32
|
import { passesQualityGate } from "../utils/qualityGate.js";
|
|
33
|
+
import { passesClinicalQualityGate, clinicalPlanScaffold, formatClinicalSections, } from "../utils/clinicalQualityPolicy.js";
|
|
33
34
|
import { applyDeterministicCodingRepairs, buildCodingRepairPrompt, passesCodingQualityGate, } from "../utils/codingQualityPolicy.js";
|
|
34
35
|
import { checkInputSafety, checkOutputSafety } from "../utils/safetyGate.js";
|
|
35
36
|
import { callLayer1 as defaultCallLayer1, classifyDeterministicLayer1, keywordBackstop, reservedCategory, MAX_CLASSIFIER_PROMPT_LENGTH } from "../utils/layer1.js";
|
|
@@ -402,7 +403,9 @@ export const PRISM_INFER_TOOL = {
|
|
|
402
403
|
"For a FOLLOW-UP to an earlier prism_infer answer, pass the accepted prior turns as `messages` " +
|
|
403
404
|
"(paid plans): without them the worker answers the follow-up from nothing and fabricates. " +
|
|
404
405
|
"Every entitlement-resolved result reports `multi_turn` (your plan's caps) and `history_turns` (what was sent); " +
|
|
405
|
-
"the crisis intercept reports only `history_turns`. "
|
|
406
|
+
"the crisis intercept reports only `history_turns`. "
|
|
407
|
+
+
|
|
408
|
+
"A behaviour-plan request also reports `clinical_sections` — how many required sections were found and which were not. That is a structural census, never a clinical endorsement: a section can be present and still be wrong, and a credentialed BCBA decides whether a plan is adequate. " +
|
|
406
409
|
"History over the plan's caps is refused (history_over_plan_cap), never trimmed; a free plan " +
|
|
407
410
|
"or a host with no portal is refused (multi_turn_not_in_plan). Hosts that compact large " +
|
|
408
411
|
"schemas may drop parameter text, so the contract lives here.",
|
|
@@ -2015,9 +2018,13 @@ export async function runInfer(args, deps) {
|
|
|
2015
2018
|
// their own — never override an explicit instruction.
|
|
2016
2019
|
// `=== undefined`, not falsy: `system: ""` is a caller explicitly asking
|
|
2017
2020
|
// for no system prompt, and overriding that is still an override.
|
|
2018
|
-
|
|
2019
|
-
|
|
2020
|
-
|
|
2021
|
+
// A caller's own `system` always wins, including `system: ""`, which is an
|
|
2022
|
+
// explicit request for none. Defaults apply only when it is undefined.
|
|
2023
|
+
const defaultSystem = [
|
|
2024
|
+
(resolvedImages?.length ?? 0) > 0 ? VISION_SYSTEM_PROMPT : undefined,
|
|
2025
|
+
clinicalPlanScaffold(args.prompt),
|
|
2026
|
+
].filter(Boolean).join("\n\n") || undefined;
|
|
2027
|
+
const effectiveSystem = args.system === undefined ? defaultSystem : args.system;
|
|
2021
2028
|
// Walk order for images.
|
|
2022
2029
|
//
|
|
2023
2030
|
// Smallest-first was tried and REVERTED on 2026-08-15. It is correct
|
|
@@ -2239,6 +2246,21 @@ export async function runInfer(args, deps) {
|
|
|
2239
2246
|
if (gate.pass && mode === "code") {
|
|
2240
2247
|
gate = passesCodingQualityGate(args.prompt, output);
|
|
2241
2248
|
}
|
|
2249
|
+
// Clinical structural check runs in EVERY mode. A behaviour plan
|
|
2250
|
+
// arrives as chat as readily as code, and the mode a caller picked
|
|
2251
|
+
// must not decide whether clinical output is inspected. The gate
|
|
2252
|
+
// self-gates on the prompt, so it is a no-op for everything else.
|
|
2253
|
+
// A clinical reason does not match the repair loop's code_/python_
|
|
2254
|
+
// prefixes, so it escalates instead of being locally patched —
|
|
2255
|
+
// deliberate: a local model inventing a missing decision-rules
|
|
2256
|
+
// section produces plausible unratified clinical text.
|
|
2257
|
+
let clinicalSections;
|
|
2258
|
+
if (gate.pass) {
|
|
2259
|
+
const clinical = passesClinicalQualityGate(args.prompt, output);
|
|
2260
|
+
clinicalSections = clinical.sections;
|
|
2261
|
+
if (!clinical.pass)
|
|
2262
|
+
gate = { pass: false, reason: clinical.reason };
|
|
2263
|
+
}
|
|
2242
2264
|
// Hard-truncation retry: the budget went on <think> and the answer
|
|
2243
2265
|
// was cut mid-emission. Previously this only escalated to cloud, or
|
|
2244
2266
|
// served the truncated text when no cloud was available — neither
|
|
@@ -2281,7 +2303,8 @@ export async function runInfer(args, deps) {
|
|
|
2281
2303
|
const codingGateFailure = !gate.pass &&
|
|
2282
2304
|
mode === "code" &&
|
|
2283
2305
|
(gate.reason?.startsWith("code_") === true ||
|
|
2284
|
-
gate.reason?.startsWith("python_") === true
|
|
2306
|
+
gate.reason?.startsWith("python_") === true ||
|
|
2307
|
+
gate.reason?.startsWith("ts_") === true);
|
|
2285
2308
|
if (!codingGateFailure)
|
|
2286
2309
|
break;
|
|
2287
2310
|
const failedReason = gate.reason ?? "code_quality";
|
|
@@ -2377,6 +2400,7 @@ export async function runInfer(args, deps) {
|
|
|
2377
2400
|
prompt_tokens: result.promptTokens,
|
|
2378
2401
|
completion_tokens: result.completionTokens,
|
|
2379
2402
|
quality_gate_failed: gate.pass ? undefined : true,
|
|
2403
|
+
clinical_sections: clinicalSections,
|
|
2380
2404
|
gate_outcome: gate.pass
|
|
2381
2405
|
? { status: "success", served_anyway: false }
|
|
2382
2406
|
: { status: "degraded", reason: gate.reason, served_anyway: true },
|
|
@@ -2609,6 +2633,55 @@ export async function inferText(prompt, opts = {}) {
|
|
|
2609
2633
|
return null;
|
|
2610
2634
|
}
|
|
2611
2635
|
}
|
|
2636
|
+
/** The one-line header the host sees above the model output.
|
|
2637
|
+
*
|
|
2638
|
+
* Pure and exported so the reporting contract in PRISM_INFER_TOOL.description
|
|
2639
|
+
* ("every entitlement-resolved result reports multi_turn and history_turns")
|
|
2640
|
+
* is assertable without standing up Ollama.
|
|
2641
|
+
*
|
|
2642
|
+
* Both fields were set on the result and written to the ledger for a release
|
|
2643
|
+
* before anything rendered them here, so the only way to learn what a call
|
|
2644
|
+
* carried was to open the SQLite ledger. An agent benchmarking multi-turn
|
|
2645
|
+
* sent no `messages` across three turns, saw nothing in the response saying
|
|
2646
|
+
* so, and published the resulting degradation as a model defect. */
|
|
2647
|
+
export function inferResponseHeader(result, memory) {
|
|
2648
|
+
const tokenStr = result.prompt_tokens != null || result.completion_tokens != null
|
|
2649
|
+
? ` tokens=${result.prompt_tokens ?? "?"}in/${result.completion_tokens ?? "?"}out`
|
|
2650
|
+
: "";
|
|
2651
|
+
return (`[prism_infer] backend=${result.backend}` +
|
|
2652
|
+
` model=${result.model_picked ?? "n/a"}` +
|
|
2653
|
+
` plan=${result.plan ?? "unknown"}` +
|
|
2654
|
+
` free_ram=${result.ram_free_mb}MB` +
|
|
2655
|
+
` latency=${result.latency_ms}ms` +
|
|
2656
|
+
` used_cloud=${result.used_cloud}` +
|
|
2657
|
+
tokenStr +
|
|
2658
|
+
// What this call actually carried, on every response including zero.
|
|
2659
|
+
// A caller that meant to send history and did not must be able to see
|
|
2660
|
+
// that here; omitting the zero is what made the failure silent.
|
|
2661
|
+
(result.history_turns != null ? ` history_turns=${result.history_turns}` : "") +
|
|
2662
|
+
(result.multi_turn
|
|
2663
|
+
? ` multi_turn=${result.multi_turn.enabled
|
|
2664
|
+
? `${result.multi_turn.max_turns}/${result.multi_turn.max_chars}`
|
|
2665
|
+
: "off"}`
|
|
2666
|
+
: "") +
|
|
2667
|
+
// Raise-only: a count of the sections found, and the names of those that
|
|
2668
|
+
// were not. Never a pass/fail word — presence is not clinical soundness.
|
|
2669
|
+
(result.clinical_sections ? ` ${formatClinicalSections(result.clinical_sections)}` : "") +
|
|
2670
|
+
(result.quality_gate_failed ? ` quality_gate_failed=true` : "") +
|
|
2671
|
+
(result.gate_outcome && result.gate_outcome.status !== "success"
|
|
2672
|
+
? ` gate=${result.gate_outcome.status}${result.gate_outcome.reason ? `:${result.gate_outcome.reason}` : ""}`
|
|
2673
|
+
: "") +
|
|
2674
|
+
(result.entitlements_source && result.entitlements_source !== "portal"
|
|
2675
|
+
? ` ent_source=${result.entitlements_source}`
|
|
2676
|
+
: "") +
|
|
2677
|
+
(result.verification ? ` verify=${result.verification.action}` : "") +
|
|
2678
|
+
(result.route_guard
|
|
2679
|
+
? ` route_guard=${result.route_guard.source}:${result.route_guard.action}` +
|
|
2680
|
+
(result.route_guard.reason ? `:${result.route_guard.reason}` : "")
|
|
2681
|
+
: "") +
|
|
2682
|
+
(memory ? ` memory=${memory.project}:${memory.depth}` : "") +
|
|
2683
|
+
(result.attempts.length ? ` attempts=${JSON.stringify(result.attempts)}` : ""));
|
|
2684
|
+
}
|
|
2612
2685
|
export async function prismInferHandler(args) {
|
|
2613
2686
|
if (!isPrismInferArgs(args)) {
|
|
2614
2687
|
const raw = typeof args === "object" && args !== null ? args : {};
|
|
@@ -2653,30 +2726,7 @@ export async function prismInferHandler(args) {
|
|
|
2653
2726
|
latency_ms: result.latency_ms,
|
|
2654
2727
|
});
|
|
2655
2728
|
}
|
|
2656
|
-
const
|
|
2657
|
-
? ` tokens=${result.prompt_tokens ?? "?"}in/${result.completion_tokens ?? "?"}out`
|
|
2658
|
-
: "";
|
|
2659
|
-
const headerBase = `[prism_infer] backend=${result.backend}` +
|
|
2660
|
-
` model=${result.model_picked ?? "n/a"}` +
|
|
2661
|
-
` plan=${result.plan ?? "unknown"}` +
|
|
2662
|
-
` free_ram=${result.ram_free_mb}MB` +
|
|
2663
|
-
` latency=${result.latency_ms}ms` +
|
|
2664
|
-
` used_cloud=${result.used_cloud}` +
|
|
2665
|
-
tokenStr +
|
|
2666
|
-
(result.quality_gate_failed ? ` quality_gate_failed=true` : "") +
|
|
2667
|
-
(result.gate_outcome && result.gate_outcome.status !== "success"
|
|
2668
|
-
? ` gate=${result.gate_outcome.status}${result.gate_outcome.reason ? `:${result.gate_outcome.reason}` : ""}`
|
|
2669
|
-
: "") +
|
|
2670
|
-
(result.entitlements_source && result.entitlements_source !== "portal"
|
|
2671
|
-
? ` ent_source=${result.entitlements_source}`
|
|
2672
|
-
: "") +
|
|
2673
|
-
(result.verification ? ` verify=${result.verification.action}` : "") +
|
|
2674
|
-
(result.route_guard
|
|
2675
|
-
? ` route_guard=${result.route_guard.source}:${result.route_guard.action}` +
|
|
2676
|
-
(result.route_guard.reason ? `:${result.route_guard.reason}` : "")
|
|
2677
|
-
: "") +
|
|
2678
|
-
(prepared.memory ? ` memory=${prepared.memory.project}:${prepared.memory.depth}` : "") +
|
|
2679
|
-
(result.attempts.length ? ` attempts=${JSON.stringify(result.attempts)}` : "");
|
|
2729
|
+
const headerBase = inferResponseHeader(result, prepared.memory);
|
|
2680
2730
|
// Append periodic session-level stats to the header line.
|
|
2681
2731
|
// compact=true is threshold-gated (PRISM_METRICS_EVERY, default every 5 calls)
|
|
2682
2732
|
// so it doesn't appear on every response — only as a rolling summary.
|
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Structural gate for clinical behaviour-analytic output.
|
|
3
|
+
*
|
|
4
|
+
* The coding gate already proves the shape of this idea: a deterministic check
|
|
5
|
+
* names a concrete defect, and the named reason drives what happens next.
|
|
6
|
+
* Python has three static passes behind it. Clinical output had none, so a
|
|
7
|
+
* behaviour plan missing its decision rules or its data-collection procedure
|
|
8
|
+
* was served exactly like a complete one.
|
|
9
|
+
*
|
|
10
|
+
* Two hard constraints, both deliberate:
|
|
11
|
+
*
|
|
12
|
+
* 1. RAISE ONLY — it never certifies. A full section count is a statement
|
|
13
|
+
* about presence, not about clinical soundness: a section can be present
|
|
14
|
+
* and wrong. Nothing here may be read as "this plan is safe to implement".
|
|
15
|
+
* The `bcba_ai_assistant` standard is that the checklist reports what is
|
|
16
|
+
* present; a credentialed BCBA decides whether the plan is adequate.
|
|
17
|
+
*
|
|
18
|
+
* 2. IT DOES NOT TOUCH THE RESERVED LIST — crisis de-escalation, restraint,
|
|
19
|
+
* SIB with injury history and risk assessment never reach a local model at
|
|
20
|
+
* all; that boundary is enforced upstream in the Layer 1 screen and is not
|
|
21
|
+
* relaxed, widened or re-implemented here. This gate governs the routine
|
|
22
|
+
* band that is already local-eligible: operational definitions,
|
|
23
|
+
* measurement, antecedent strategies, caregiver training.
|
|
24
|
+
*
|
|
25
|
+
* A failure is NOT auto-repaired. The coding repair loop re-prompts the same
|
|
26
|
+
* tier to fix a syntax defect, which is safe for code; asking a local model to
|
|
27
|
+
* invent a missing decision-rules section produces plausible unratified
|
|
28
|
+
* clinical text, which is worse than a visibly incomplete draft. A clinical
|
|
29
|
+
* reason therefore falls out of the repair loop and escalates instead.
|
|
30
|
+
*/
|
|
31
|
+
/** A full written plan was asked for — the whole section list applies. */
|
|
32
|
+
const CLINICAL_PLAN_REQUEST_RE = /\b(bip\b|behavi(?:o|ou)r(?:al)?[ -](?:intervention|support|management)[ -]plan|behavi(?:o|ou)r plan|treatment plan|intervention plan)\b/i;
|
|
33
|
+
/** An operational definition specifically was asked for. */
|
|
34
|
+
const OPERATIONAL_DEFINITION_REQUEST_RE = /\boperational(?:ly)?[ -]?(?:defin\w*)|\bdefine the (?:target )?behaviou?r\b/i;
|
|
35
|
+
/** Any behaviour-analytic context at all — gates the AAC safety check. */
|
|
36
|
+
const CLINICAL_CONTEXT_RE = /\b(aba\b|bcba\b|behaviou?r analyst|functional behaviou?r assessment|\bfba\b|\bbip\b|replacement behaviou?r|target behaviou?r|reinforcement schedule|\bfct\b|\bdro\b|\bdra\b|\bncr\b)/i;
|
|
37
|
+
/** Ordered so the report reads the way a plan is written. */
|
|
38
|
+
const PLAN_SECTIONS = [
|
|
39
|
+
{ name: "operational_definition", requirement: "an operational definition that is observable and measurable, with examples AND non-examples", pattern: /operational(?:ly)?[ -]?defin|\bdefinition\b[\s\S]{0,80}\b(observable|measurable)\b/i },
|
|
40
|
+
{ name: "function_hypothesis", requirement: "a hypothesised function supported by A-B-C data", pattern: /\b(hypothesi[sz]ed function|function of the behaviou?r|maintained by|\ba-?b-?c\b|antecedent[\s\S]{0,40}consequence)\b/i },
|
|
41
|
+
{ name: "antecedent_strategies", requirement: "antecedent and prevention strategies", pattern: /\b(antecedent (?:strateg|modificat|intervention|procedure|support)|prevention strateg|proactive strateg|pre-?work strateg|pre-?correct|setting event|environmental modificat|visual (?:schedule|timer|cue)|priming)/i },
|
|
42
|
+
{ name: "replacement_behaviour", requirement: "a functionally equivalent replacement behaviour", pattern: /\b(replacement behaviou?r|functional communication training|\bfct\b|alternative behaviou?r|\bdra\b)/i },
|
|
43
|
+
{ name: "consequence_strategies", requirement: "consequence strategies, including what reinforces the replacement", pattern: /\b(consequence|reinforcement (?:schedule|procedure|strateg|plan|system|for\b)|reinforc\w+ the (?:replacement|desired|appropriate|target)|planned ignoring|response to (?:the )?behaviou?r|redirect\w*|\bpraise\b|\bdro\b|\bncr\b|extinction)/i },
|
|
44
|
+
{ name: "data_collection", requirement: "a data collection method", pattern: /\b(data collection|data sheet|measurement (?:system|procedure)|frequency count|partial interval|momentary time sampling|\bioa\b|interobserver)/i },
|
|
45
|
+
{ name: "decision_rules", requirement: "decision rules and a review schedule", pattern: /\b(decision rule|mastery criteri|criteri\w+ for (?:change|modificat|advancement)|evaluation criteri|review (?:schedule|trigger|date)|plan review|progress monitor\w*|plan will be (?:adjusted|modified|revised|changed))/i },
|
|
46
|
+
{ name: "generalisation_maintenance", requirement: "generalisation and maintenance", pattern: /\b(generali[sz]|maintenance)\b/i },
|
|
47
|
+
{ name: "caregiver_training", requirement: "caregiver and staff training", pattern: /\b((?:caregiver|staff|parent|family|teacher|team)[ -]?training|train(?:ing|ed)? (?:the )?(?:caregivers?|staff|parents?|team)|(?:staff|caregivers?|parents?|team|teachers?)\b[^.\n]{0,30}\btrain\w+|train\w+ on the plan)/i },
|
|
48
|
+
{ name: "bcba_review_disclaimer", requirement: "a statement that a credentialed BCBA must review and individualise the plan before implementation", pattern: /\b(reviewed and individuali[sz]ed|credentialed bcba|licensed behaviou?r analyst|must be reviewed)\b/i },
|
|
49
|
+
];
|
|
50
|
+
/**
|
|
51
|
+
* AAC access may never be removed, withheld or delayed as a consequence.
|
|
52
|
+
*
|
|
53
|
+
* A correct plan states this rule explicitly ("AAC access is never restricted"),
|
|
54
|
+
* so a bare co-occurrence of an AAC term and a restriction verb fires on GOOD
|
|
55
|
+
* text. The lookback suppresses a match when the clause is negated. It is
|
|
56
|
+
* approximate by construction, which is acceptable only because this raises and
|
|
57
|
+
* never clears: an escalation costs one call, and no output is marked safe here.
|
|
58
|
+
*/
|
|
59
|
+
const AAC_TERM = /\b(aac\b|speech[- ]generating device|\bsgd\b|communication device|communication board|\bpecs\b|talker\b)/i;
|
|
60
|
+
const RESTRICT_VERB = /\b(remov\w+|withh\w+|restrict\w+|tak\w+ away|deni\w+|deny|block\w*|delay\w*|confiscat\w+|limit\w*)\b/i;
|
|
61
|
+
const NEGATOR = /\b(never|not|n't|no|avoid\w*|prohibit\w*|must not|cannot|can't|without)\b/i;
|
|
62
|
+
// Asymmetric on purpose. "remove the AAC device" puts the verb BEFORE the term,
|
|
63
|
+
// so a forward-only window misses the most direct phrasing of the thing this
|
|
64
|
+
// check exists to catch. The backward reach is kept short because a removal
|
|
65
|
+
// sentence about something else ("remove the token board") sitting a paragraph
|
|
66
|
+
// above an AAC mention is not a restriction of AAC.
|
|
67
|
+
const AAC_WINDOW_AFTER = 120;
|
|
68
|
+
const AAC_WINDOW_BEFORE = 40;
|
|
69
|
+
const NEGATION_LOOKBACK = 60;
|
|
70
|
+
/** Clause boundaries. The verb must act on the AAC term, not merely sit near it:
|
|
71
|
+
* "AAC remains available at all times; remove the token board" removes a token
|
|
72
|
+
* board, and scanning past the semicolon read it as removing AAC. */
|
|
73
|
+
const CLAUSE_BREAK = /[.;:\n]|\bhowever\b|\bwhereas\b/i;
|
|
74
|
+
function clauseAfter(text, from, limit) {
|
|
75
|
+
const slice = text.slice(from, from + limit);
|
|
76
|
+
const brk = CLAUSE_BREAK.exec(slice);
|
|
77
|
+
return brk ? slice.slice(0, brk.index) : slice;
|
|
78
|
+
}
|
|
79
|
+
function clauseBefore(text, end, limit) {
|
|
80
|
+
const slice = text.slice(Math.max(0, end - limit), end);
|
|
81
|
+
let last = -1;
|
|
82
|
+
for (const m of slice.matchAll(new RegExp(CLAUSE_BREAK.source, "gi"))) {
|
|
83
|
+
last = (m.index ?? 0) + m[0].length;
|
|
84
|
+
}
|
|
85
|
+
return last >= 0 ? slice.slice(last) : slice;
|
|
86
|
+
}
|
|
87
|
+
function aacRestrictedAsConsequence(output) {
|
|
88
|
+
for (const m of output.matchAll(new RegExp(AAC_TERM.source, "gi"))) {
|
|
89
|
+
const start = m.index ?? 0;
|
|
90
|
+
const before = clauseBefore(output, start, AAC_WINDOW_BEFORE);
|
|
91
|
+
const after = clauseAfter(output, start, AAC_WINDOW_AFTER);
|
|
92
|
+
const from = start - before.length;
|
|
93
|
+
const window = before + after;
|
|
94
|
+
const verb = RESTRICT_VERB.exec(window);
|
|
95
|
+
if (!verb)
|
|
96
|
+
continue;
|
|
97
|
+
const absolute = from + (verb.index ?? 0);
|
|
98
|
+
const lookback = output.slice(Math.max(0, absolute - NEGATION_LOOKBACK), absolute);
|
|
99
|
+
if (NEGATOR.test(lookback))
|
|
100
|
+
continue; // "AAC access is never removed"
|
|
101
|
+
return true;
|
|
102
|
+
}
|
|
103
|
+
return false;
|
|
104
|
+
}
|
|
105
|
+
/** Characters of real prose required near a section marker for it to count. */
|
|
106
|
+
const SECTION_CONTENT_CHARS = 30;
|
|
107
|
+
const SECTION_WINDOW = 400;
|
|
108
|
+
/** Drop whole heading lines. Stripping only the `#` turns the NEXT heading into
|
|
109
|
+
* prose, which is why ten empty headings first scored 8 of 10. */
|
|
110
|
+
function proseOnly(text) {
|
|
111
|
+
return text
|
|
112
|
+
.split("\n")
|
|
113
|
+
.filter(line => !/^\s*#{1,6}\s/.test(line)) // markdown headings
|
|
114
|
+
.filter(line => !/^\s*\*\*[^*]+\*\*\s*:?\s*$/.test(line)) // bold-only lines
|
|
115
|
+
.join(" ")
|
|
116
|
+
.replace(/[>#*_|]+/g, "")
|
|
117
|
+
.replace(/\[[^\]]*\]/g, "") // [placeholders]
|
|
118
|
+
.replace(/[-\s]+/g, " ")
|
|
119
|
+
.trim();
|
|
120
|
+
}
|
|
121
|
+
/**
|
|
122
|
+
* A section counts only when there is real prose NEAR its marker.
|
|
123
|
+
*
|
|
124
|
+
* Without this, ten empty headings scored 8 of 10: an output with no clinical
|
|
125
|
+
* content looked nearly complete, because the census matched vocabulary rather
|
|
126
|
+
* than substance. The scaffold already demands "substantive content rather than
|
|
127
|
+
* a heading alone"; this is the census checking the same thing.
|
|
128
|
+
*
|
|
129
|
+
* The window spans both directions. A first version looked only forward and
|
|
130
|
+
* dropped a legitimate credit — "we will write down how often it happens on a
|
|
131
|
+
* data sheet" puts the content BEFORE the keyword.
|
|
132
|
+
*
|
|
133
|
+
* Two limits, both measured rather than assumed.
|
|
134
|
+
*
|
|
135
|
+
* The window is wide, so in a dense document a marker finds prose belonging to
|
|
136
|
+
* a NEIGHBOURING section and is credited for it. This is therefore closer to a
|
|
137
|
+
* document-level check than a per-section one; what it reliably catches is the
|
|
138
|
+
* empty or near-empty output, which is what it was added for.
|
|
139
|
+
*
|
|
140
|
+
* And a model that echoes the section list back as prose still scores full
|
|
141
|
+
* marks, because a description of what a plan must contain is, at this level of
|
|
142
|
+
* analysis, indistinguishable from a plan. That is a limit of the approach, not
|
|
143
|
+
* something to regex away, and one more reason nothing here is an endorsement.
|
|
144
|
+
*
|
|
145
|
+
* The floor is deliberately low. At 50 characters a legitimately terse section
|
|
146
|
+
* — "Frequency count / Daily tally / Weekly IOA" — was refused, and a false
|
|
147
|
+
* negative on real content is the more expensive error for a gap report.
|
|
148
|
+
*/
|
|
149
|
+
function sectionHasContent(output, pattern) {
|
|
150
|
+
const m = new RegExp(pattern.source, pattern.flags.replace("g", "")).exec(output);
|
|
151
|
+
if (!m)
|
|
152
|
+
return false;
|
|
153
|
+
const at = m.index ?? 0;
|
|
154
|
+
const window = output.slice(Math.max(0, at - SECTION_WINDOW), at + m[0].length + SECTION_WINDOW);
|
|
155
|
+
return proseOnly(window).length >= SECTION_CONTENT_CHARS;
|
|
156
|
+
}
|
|
157
|
+
/**
|
|
158
|
+
* Raise-only structural check. `pass: true` means nothing was detected as
|
|
159
|
+
* missing — it is not a clinical endorsement.
|
|
160
|
+
*/
|
|
161
|
+
export function passesClinicalQualityGate(prompt, output) {
|
|
162
|
+
// An operational-definition request is clinical on its own: "write an
|
|
163
|
+
// operational definition of elopement" names no ABA vocabulary the broad
|
|
164
|
+
// pattern looks for, and was silently skipped before.
|
|
165
|
+
const clinicalContext = CLINICAL_CONTEXT_RE.test(prompt)
|
|
166
|
+
|| CLINICAL_PLAN_REQUEST_RE.test(prompt)
|
|
167
|
+
|| OPERATIONAL_DEFINITION_REQUEST_RE.test(prompt);
|
|
168
|
+
if (!clinicalContext)
|
|
169
|
+
return { pass: true };
|
|
170
|
+
if (aacRestrictedAsConsequence(output)) {
|
|
171
|
+
return { pass: false, reason: "clinical_aac_restricted_as_consequence" };
|
|
172
|
+
}
|
|
173
|
+
if (OPERATIONAL_DEFINITION_REQUEST_RE.test(prompt)) {
|
|
174
|
+
const hasExamples = /\bexamples?\b/i.test(output);
|
|
175
|
+
const hasNonExamples = /\bnon-?examples?\b/i.test(output);
|
|
176
|
+
if (!hasExamples || !hasNonExamples) {
|
|
177
|
+
return { pass: false, reason: "clinical_operational_definition_incomplete" };
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
if (!CLINICAL_PLAN_REQUEST_RE.test(prompt))
|
|
181
|
+
return { pass: true };
|
|
182
|
+
const missing = PLAN_SECTIONS
|
|
183
|
+
.filter(s => !sectionHasContent(output, s.pattern))
|
|
184
|
+
.map(s => s.name);
|
|
185
|
+
const sections = {
|
|
186
|
+
required: PLAN_SECTIONS.length,
|
|
187
|
+
present: PLAN_SECTIONS.length - missing.length,
|
|
188
|
+
missing,
|
|
189
|
+
};
|
|
190
|
+
// An incomplete plan REPORTS; it does not fail. Failing the gate rejects the
|
|
191
|
+
// output, and when escalation is unavailable the caller receives nothing at
|
|
192
|
+
// all — measured in review: a 3-of-10 plan with cloud_fallback:true and an
|
|
193
|
+
// unreachable portal returned "no backend produced output". A draft labelled
|
|
194
|
+
// `clinical_sections=3/10 missing:...` is strictly more useful to a clinician
|
|
195
|
+
// than silence, and suppressing it contradicts the raise-only rule above.
|
|
196
|
+
//
|
|
197
|
+
// The two findings ABOVE do fail, because they are defects rather than
|
|
198
|
+
// incompleteness: AAC restricted as a consequence is a safety violation, and
|
|
199
|
+
// an operational definition without non-examples is wrong, not unfinished.
|
|
200
|
+
return { pass: true, sections };
|
|
201
|
+
}
|
|
202
|
+
/** Compact, raise-only header fragment. Counts only — never a verdict word. */
|
|
203
|
+
export function formatClinicalSections(s) {
|
|
204
|
+
const base = `clinical_sections=${s.present}/${s.required}`;
|
|
205
|
+
return s.missing.length ? `${base} missing:${s.missing.join(",")}` : base;
|
|
206
|
+
}
|
|
207
|
+
/**
|
|
208
|
+
* A system instruction naming every section a plan must contain, generated from
|
|
209
|
+
* PLAN_SECTIONS so the list that INSTRUCTS is the list that VERIFIES.
|
|
210
|
+
*
|
|
211
|
+
* Measured on prism-coder:9b: the same plan request scored 7/10 unscaffolded and
|
|
212
|
+
* 10/10 scaffolded, in FEWER characters — it restructured rather than padded,
|
|
213
|
+
* and the previously absent sections came back with substantive content
|
|
214
|
+
* (a real observable definition with non-examples, real generalisation content,
|
|
215
|
+
* a correctly worded review statement).
|
|
216
|
+
*
|
|
217
|
+
* KNOWN EPISTEMIC COST, recorded rather than hidden: once the model is told the
|
|
218
|
+
* list, the census stops being independent confirmation and becomes a check
|
|
219
|
+
* that the instruction was followed. A scaffolded 10/10 is weaker evidence than
|
|
220
|
+
* an unscaffolded one. Sharing one list is still the right trade — two lists
|
|
221
|
+
* drift, and a census that disagrees with the instruction is worse than a
|
|
222
|
+
* census that merely confirms it — but nothing here should be read as evidence
|
|
223
|
+
* that the model knows what a plan needs.
|
|
224
|
+
*
|
|
225
|
+
* LOCAL ONLY. `callCloud` takes the prompt and no system argument, so nothing
|
|
226
|
+
* here reaches an escalated request — true of VISION_SYSTEM_PROMPT as well, and
|
|
227
|
+
* pre-existing rather than introduced with this scaffold. The consequence is
|
|
228
|
+
* that a cloud-served plan is measured by the census WITHOUT having been given
|
|
229
|
+
* the list, so it can score lower than a local one for reasons that have
|
|
230
|
+
* nothing to do with the model. Threading `system` through the portal API is
|
|
231
|
+
* the real fix and is deliberately out of scope here.
|
|
232
|
+
*
|
|
233
|
+
* Returns undefined unless a full plan was requested, so it never touches the
|
|
234
|
+
* prompt for ordinary work.
|
|
235
|
+
*/
|
|
236
|
+
export function clinicalPlanScaffold(prompt) {
|
|
237
|
+
if (!CLINICAL_PLAN_REQUEST_RE.test(prompt))
|
|
238
|
+
return undefined;
|
|
239
|
+
const items = PLAN_SECTIONS.map(s => `- ${s.requirement}`).join("\n");
|
|
240
|
+
return ("A behaviour plan must contain all of the following, each with substantive "
|
|
241
|
+
+ "content rather than a heading alone:\n" + items
|
|
242
|
+
+ "\n\nUse least restrictive, dignity-preserving, function-based procedures. "
|
|
243
|
+
+ "Never restrict, remove or delay access to an AAC or communication device "
|
|
244
|
+
+ "as a consequence.");
|
|
245
|
+
}
|
|
@@ -304,7 +304,98 @@ function pythonStaticContractFailure(code) {
|
|
|
304
304
|
? `python_static_contract:${[...issues].sort().join(",")}`
|
|
305
305
|
: undefined;
|
|
306
306
|
}
|
|
307
|
+
/**
|
|
308
|
+
* A generic used with no type argument, e.g. `Map<string, Array>`.
|
|
309
|
+
*
|
|
310
|
+
* prism-coder:9b emits this repeatedly — observed in four separate generations
|
|
311
|
+
* of the same EventEmitter task — and it is a hard compile error (TS2314), so
|
|
312
|
+
* the file never builds. Detecting it needs no TypeScript dependency: the shape
|
|
313
|
+
* is unambiguous when the bare name sits inside a type-argument list or
|
|
314
|
+
* directly after a type annotation.
|
|
315
|
+
*
|
|
316
|
+
* Deliberately narrow. Prose mentioning "the Array, then the Map" must not
|
|
317
|
+
* match, so a bare name is only a finding when it is syntactically in a type
|
|
318
|
+
* position; a bare `Promise` as a lone return type with no delimiter after it
|
|
319
|
+
* is missed, which is the conservative direction.
|
|
320
|
+
*/
|
|
321
|
+
const TS_GENERIC = "(?:Array|Map|Set|Promise|Record|Partial|Readonly|WeakMap|WeakSet)";
|
|
322
|
+
const TS_BARE_GENERIC_RE = new RegExp(`<[^<>]*\\b${TS_GENERIC}\\b(?!\\s*<)[^<>]*>` +
|
|
323
|
+
`|:\\s*${TS_GENERIC}\\b(?!\\s*<)\\s*[={;,)\\]]`);
|
|
324
|
+
/** Null when the code carries no TypeScript static-contract defect. */
|
|
325
|
+
function tsStaticContractFailure(code) {
|
|
326
|
+
return TS_BARE_GENERIC_RE.test(code) ? "ts_static_contract:bare_generic" : null;
|
|
327
|
+
}
|
|
328
|
+
/** Character ranges a repair must not touch on one line.
|
|
329
|
+
*
|
|
330
|
+
* Strings, because rewriting `"use Map<string, Array> carefully"` changes a
|
|
331
|
+
* RUNTIME VALUE rather than a type. And comments, because `// don't use a bare
|
|
332
|
+
* Map<string, Array>` is a warning against the very thing the repair would
|
|
333
|
+
* write, so editing it inverts the author's meaning. Both found in review. */
|
|
334
|
+
function protectedSpans(line) {
|
|
335
|
+
const spans = [];
|
|
336
|
+
const strings = /"(?:[^"\\]|\\.)*"|'(?:[^'\\]|\\.)*'|`(?:[^`\\]|\\.)*`/g;
|
|
337
|
+
for (const m of line.matchAll(strings))
|
|
338
|
+
spans.push([m.index ?? 0, (m.index ?? 0) + m[0].length]);
|
|
339
|
+
// A line comment runs to end of line. An apostrophe in prose ("don't")
|
|
340
|
+
// breaks the string scan, which is how comments slipped through before.
|
|
341
|
+
const comment = /\/\/|\/\*/.exec(line);
|
|
342
|
+
if (comment)
|
|
343
|
+
spans.push([comment.index, line.length]);
|
|
344
|
+
return spans;
|
|
345
|
+
}
|
|
346
|
+
/** Fence languages whose contents are TypeScript. A bare generic inside a
|
|
347
|
+
* python or json block is not a type error to fix — the first version rewrote
|
|
348
|
+
* a comment inside a python block in a multi-language answer. */
|
|
349
|
+
const TS_FENCE_LANG = /^\s*```\s*(ts|typescript|tsx)?\s*$/i;
|
|
350
|
+
/** `Map<string, Array>` -> `Map<string, Array<any>>`, within code only.
|
|
351
|
+
*
|
|
352
|
+
* `any` rather than `unknown` on purpose: `unknown` makes the file compile and
|
|
353
|
+
* then breaks every use of the value, which trades one compile error for
|
|
354
|
+
* several. This makes the code build; it does not make it well typed.
|
|
355
|
+
*
|
|
356
|
+
* Scoped twice, both from adversarial review. Fenced output is repaired only
|
|
357
|
+
* INSIDE its fences, because rewriting the surrounding prose inverts sentences
|
|
358
|
+
* like "do not write Map<string, Array>". And no match inside a string literal
|
|
359
|
+
* is touched, because that is a value, not a type. */
|
|
360
|
+
function repairBareGenerics(code) {
|
|
361
|
+
let changed = false;
|
|
362
|
+
const spanRe = new RegExp(TS_BARE_GENERIC_RE.source, "g");
|
|
363
|
+
const repairLine = (line) => {
|
|
364
|
+
const off_limits = protectedSpans(line);
|
|
365
|
+
return line.replace(spanRe, (span, offset) => {
|
|
366
|
+
if (off_limits.some(([a, b]) => offset >= a && offset < b))
|
|
367
|
+
return span;
|
|
368
|
+
const fixed = span.replace(new RegExp(`\\b(${TS_GENERIC})\\b(?!\\s*<)`, "g"), "$1<any>");
|
|
369
|
+
if (fixed !== span)
|
|
370
|
+
changed = true;
|
|
371
|
+
return fixed;
|
|
372
|
+
});
|
|
373
|
+
};
|
|
374
|
+
const lines = code.split("\n");
|
|
375
|
+
const hasFences = /^\s*```/m.test(code);
|
|
376
|
+
let inFence = false;
|
|
377
|
+
let fenceIsTs = false;
|
|
378
|
+
const out = lines.map((line) => {
|
|
379
|
+
if (/^\s*```/.test(line)) {
|
|
380
|
+
if (!inFence)
|
|
381
|
+
fenceIsTs = TS_FENCE_LANG.test(line);
|
|
382
|
+
inFence = !inFence;
|
|
383
|
+
return line;
|
|
384
|
+
}
|
|
385
|
+
// No fences at all: the whole output is the code block.
|
|
386
|
+
return (!hasFences || (inFence && fenceIsTs)) ? repairLine(line) : line;
|
|
387
|
+
});
|
|
388
|
+
return { code: out.join("\n"), changed };
|
|
389
|
+
}
|
|
307
390
|
export function applyDeterministicCodingRepairs(output, reason) {
|
|
391
|
+
if (reason.startsWith("ts_static_contract:")) {
|
|
392
|
+
if (!reason.includes("bare_generic"))
|
|
393
|
+
return { output, changes: [] };
|
|
394
|
+
const repaired = repairBareGenerics(output);
|
|
395
|
+
return repaired.changed
|
|
396
|
+
? { output: repaired.code, changes: ["bare_generic"] }
|
|
397
|
+
: { output, changes: [] };
|
|
398
|
+
}
|
|
308
399
|
if (!reason.startsWith("python_static_contract:")) {
|
|
309
400
|
return { output, changes: [] };
|
|
310
401
|
}
|
|
@@ -369,6 +460,9 @@ export function passesCodingQualityGate(prompt, output) {
|
|
|
369
460
|
if (pythonFailure)
|
|
370
461
|
return { pass: false, reason: pythonFailure };
|
|
371
462
|
}
|
|
463
|
+
const tsFailure = tsStaticContractFailure(code);
|
|
464
|
+
if (tsFailure)
|
|
465
|
+
return { pass: false, reason: tsFailure };
|
|
372
466
|
return { pass: true };
|
|
373
467
|
}
|
|
374
468
|
const CODING_REPAIR_SYSTEM_INSTRUCTION = "Repair the supplied implementation. Return one complete replacement implementation with no prose, " +
|
|
@@ -383,6 +477,7 @@ const CODING_REPAIR_GUIDANCE = {
|
|
|
383
477
|
python_method_missing_receiver: "Instance methods must take self first; class methods must take cls first unless decorated staticmethod.",
|
|
384
478
|
python_undefined_private_helper: "Define every directly called private self helper or replace the call with the correct defined helper.",
|
|
385
479
|
constructor_attribute_missing_receiver: "In __init__, persist instance state as self.<attribute>; do not assign it to a discarded local variable.",
|
|
480
|
+
bare_generic: "Every generic needs its type argument: write Array<T>, Map<K, V>, Set<T>, Promise<T> — never a bare Array, Map, Set or Promise in a type position.",
|
|
386
481
|
dict_keys_unpack: "When unpacking key and value, iterate dictionary .items(); .keys() yields one key per iteration.",
|
|
387
482
|
};
|
|
388
483
|
function repairGuidance(reason) {
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "prism-mcp-server",
|
|
3
|
-
"version": "20.21.
|
|
3
|
+
"version": "20.21.4",
|
|
4
4
|
"mcpName": "io.github.dcostenco/prism-coder",
|
|
5
5
|
"description": "Persistent session memory for AI coding agents that never leaves your machine — including the on-device model that reasons over it. Restores your prior decisions, open TODOs, and changed files across sessions; adds associative recall of related past work, semantic drift detection, and local inference. Local-first by default. Works with Claude Code, Cursor, and Codex.",
|
|
6
6
|
"module": "index.ts",
|