prism-mcp-server 20.21.2 → 20.21.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -30,6 +30,7 @@ import { getEntitlements, clampCeiling, multiTurnPolicy, ABSOLUTE_MULTI_TURN } f
30
30
  import { ddLog } from "../utils/ddLogger.js";
31
31
  import { stripThink } from "../utils/thinkStrip.js";
32
32
  import { passesQualityGate } from "../utils/qualityGate.js";
33
+ import { passesClinicalQualityGate, formatClinicalSections, } from "../utils/clinicalQualityPolicy.js";
33
34
  import { applyDeterministicCodingRepairs, buildCodingRepairPrompt, passesCodingQualityGate, } from "../utils/codingQualityPolicy.js";
34
35
  import { checkInputSafety, checkOutputSafety } from "../utils/safetyGate.js";
35
36
  import { callLayer1 as defaultCallLayer1, classifyDeterministicLayer1, keywordBackstop, reservedCategory, MAX_CLASSIFIER_PROMPT_LENGTH } from "../utils/layer1.js";
@@ -402,7 +403,9 @@ export const PRISM_INFER_TOOL = {
402
403
  "For a FOLLOW-UP to an earlier prism_infer answer, pass the accepted prior turns as `messages` " +
403
404
  "(paid plans): without them the worker answers the follow-up from nothing and fabricates. " +
404
405
  "Every entitlement-resolved result reports `multi_turn` (your plan's caps) and `history_turns` (what was sent); " +
405
- "the crisis intercept reports only `history_turns`. " +
406
+ "the crisis intercept reports only `history_turns`. "
407
+ +
408
+ "A behaviour-plan request also reports `clinical_sections` — how many required sections were found and which were not. That is a structural census, never a clinical endorsement: a section can be present and still be wrong, and a credentialed BCBA decides whether a plan is adequate. " +
406
409
  "History over the plan's caps is refused (history_over_plan_cap), never trimmed; a free plan " +
407
410
  "or a host with no portal is refused (multi_turn_not_in_plan). Hosts that compact large " +
408
411
  "schemas may drop parameter text, so the contract lives here.",
@@ -2239,6 +2242,21 @@ export async function runInfer(args, deps) {
2239
2242
  if (gate.pass && mode === "code") {
2240
2243
  gate = passesCodingQualityGate(args.prompt, output);
2241
2244
  }
2245
+ // Clinical structural check runs in EVERY mode. A behaviour plan
2246
+ // arrives as chat as readily as code, and the mode a caller picked
2247
+ // must not decide whether clinical output is inspected. The gate
2248
+ // self-gates on the prompt, so it is a no-op for everything else.
2249
+ // A clinical reason does not match the repair loop's code_/python_
2250
+ // prefixes, so it escalates instead of being locally patched —
2251
+ // deliberate: a local model inventing a missing decision-rules
2252
+ // section produces plausible unratified clinical text.
2253
+ let clinicalSections;
2254
+ if (gate.pass) {
2255
+ const clinical = passesClinicalQualityGate(args.prompt, output);
2256
+ clinicalSections = clinical.sections;
2257
+ if (!clinical.pass)
2258
+ gate = { pass: false, reason: clinical.reason };
2259
+ }
2242
2260
  // Hard-truncation retry: the budget went on <think> and the answer
2243
2261
  // was cut mid-emission. Previously this only escalated to cloud, or
2244
2262
  // served the truncated text when no cloud was available — neither
@@ -2377,6 +2395,7 @@ export async function runInfer(args, deps) {
2377
2395
  prompt_tokens: result.promptTokens,
2378
2396
  completion_tokens: result.completionTokens,
2379
2397
  quality_gate_failed: gate.pass ? undefined : true,
2398
+ clinical_sections: clinicalSections,
2380
2399
  gate_outcome: gate.pass
2381
2400
  ? { status: "success", served_anyway: false }
2382
2401
  : { status: "degraded", reason: gate.reason, served_anyway: true },
@@ -2609,6 +2628,55 @@ export async function inferText(prompt, opts = {}) {
2609
2628
  return null;
2610
2629
  }
2611
2630
  }
2631
+ /** The one-line header the host sees above the model output.
2632
+ *
2633
+ * Pure and exported so the reporting contract in PRISM_INFER_TOOL.description
2634
+ * ("every entitlement-resolved result reports multi_turn and history_turns")
2635
+ * is assertable without standing up Ollama.
2636
+ *
2637
+ * Both fields were set on the result and written to the ledger for a release
2638
+ * before anything rendered them here, so the only way to learn what a call
2639
+ * carried was to open the SQLite ledger. An agent benchmarking multi-turn
2640
+ * sent no `messages` across three turns, saw nothing in the response saying
2641
+ * so, and published the resulting degradation as a model defect. */
2642
+ export function inferResponseHeader(result, memory) {
2643
+ const tokenStr = result.prompt_tokens != null || result.completion_tokens != null
2644
+ ? ` tokens=${result.prompt_tokens ?? "?"}in/${result.completion_tokens ?? "?"}out`
2645
+ : "";
2646
+ return (`[prism_infer] backend=${result.backend}` +
2647
+ ` model=${result.model_picked ?? "n/a"}` +
2648
+ ` plan=${result.plan ?? "unknown"}` +
2649
+ ` free_ram=${result.ram_free_mb}MB` +
2650
+ ` latency=${result.latency_ms}ms` +
2651
+ ` used_cloud=${result.used_cloud}` +
2652
+ tokenStr +
2653
+ // What this call actually carried, on every response including zero.
2654
+ // A caller that meant to send history and did not must be able to see
2655
+ // that here; omitting the zero is what made the failure silent.
2656
+ (result.history_turns != null ? ` history_turns=${result.history_turns}` : "") +
2657
+ (result.multi_turn
2658
+ ? ` multi_turn=${result.multi_turn.enabled
2659
+ ? `${result.multi_turn.max_turns}/${result.multi_turn.max_chars}`
2660
+ : "off"}`
2661
+ : "") +
2662
+ // Raise-only: a count of the sections found, and the names of those that
2663
+ // were not. Never a pass/fail word — presence is not clinical soundness.
2664
+ (result.clinical_sections ? ` ${formatClinicalSections(result.clinical_sections)}` : "") +
2665
+ (result.quality_gate_failed ? ` quality_gate_failed=true` : "") +
2666
+ (result.gate_outcome && result.gate_outcome.status !== "success"
2667
+ ? ` gate=${result.gate_outcome.status}${result.gate_outcome.reason ? `:${result.gate_outcome.reason}` : ""}`
2668
+ : "") +
2669
+ (result.entitlements_source && result.entitlements_source !== "portal"
2670
+ ? ` ent_source=${result.entitlements_source}`
2671
+ : "") +
2672
+ (result.verification ? ` verify=${result.verification.action}` : "") +
2673
+ (result.route_guard
2674
+ ? ` route_guard=${result.route_guard.source}:${result.route_guard.action}` +
2675
+ (result.route_guard.reason ? `:${result.route_guard.reason}` : "")
2676
+ : "") +
2677
+ (memory ? ` memory=${memory.project}:${memory.depth}` : "") +
2678
+ (result.attempts.length ? ` attempts=${JSON.stringify(result.attempts)}` : ""));
2679
+ }
2612
2680
  export async function prismInferHandler(args) {
2613
2681
  if (!isPrismInferArgs(args)) {
2614
2682
  const raw = typeof args === "object" && args !== null ? args : {};
@@ -2653,30 +2721,7 @@ export async function prismInferHandler(args) {
2653
2721
  latency_ms: result.latency_ms,
2654
2722
  });
2655
2723
  }
2656
- const tokenStr = result.prompt_tokens != null || result.completion_tokens != null
2657
- ? ` tokens=${result.prompt_tokens ?? "?"}in/${result.completion_tokens ?? "?"}out`
2658
- : "";
2659
- const headerBase = `[prism_infer] backend=${result.backend}` +
2660
- ` model=${result.model_picked ?? "n/a"}` +
2661
- ` plan=${result.plan ?? "unknown"}` +
2662
- ` free_ram=${result.ram_free_mb}MB` +
2663
- ` latency=${result.latency_ms}ms` +
2664
- ` used_cloud=${result.used_cloud}` +
2665
- tokenStr +
2666
- (result.quality_gate_failed ? ` quality_gate_failed=true` : "") +
2667
- (result.gate_outcome && result.gate_outcome.status !== "success"
2668
- ? ` gate=${result.gate_outcome.status}${result.gate_outcome.reason ? `:${result.gate_outcome.reason}` : ""}`
2669
- : "") +
2670
- (result.entitlements_source && result.entitlements_source !== "portal"
2671
- ? ` ent_source=${result.entitlements_source}`
2672
- : "") +
2673
- (result.verification ? ` verify=${result.verification.action}` : "") +
2674
- (result.route_guard
2675
- ? ` route_guard=${result.route_guard.source}:${result.route_guard.action}` +
2676
- (result.route_guard.reason ? `:${result.route_guard.reason}` : "")
2677
- : "") +
2678
- (prepared.memory ? ` memory=${prepared.memory.project}:${prepared.memory.depth}` : "") +
2679
- (result.attempts.length ? ` attempts=${JSON.stringify(result.attempts)}` : "");
2724
+ const headerBase = inferResponseHeader(result, prepared.memory);
2680
2725
  // Append periodic session-level stats to the header line.
2681
2726
  // compact=true is threshold-gated (PRISM_METRICS_EVERY, default every 5 calls)
2682
2727
  // so it doesn't appear on every response — only as a rolling summary.
@@ -0,0 +1,152 @@
1
+ /**
2
+ * Structural gate for clinical behaviour-analytic output.
3
+ *
4
+ * The coding gate already proves the shape of this idea: a deterministic check
5
+ * names a concrete defect, and the named reason drives what happens next.
6
+ * Python has three static passes behind it. Clinical output had none, so a
7
+ * behaviour plan missing its decision rules or its data-collection procedure
8
+ * was served exactly like a complete one.
9
+ *
10
+ * Two hard constraints, both deliberate:
11
+ *
12
+ * 1. RAISE ONLY — it never certifies. A full section count is a statement
13
+ * about presence, not about clinical soundness: a section can be present
14
+ * and wrong. Nothing here may be read as "this plan is safe to implement".
15
+ * The `bcba_ai_assistant` standard is that the checklist reports what is
16
+ * present; a credentialed BCBA decides whether the plan is adequate.
17
+ *
18
+ * 2. IT DOES NOT TOUCH THE RESERVED LIST — crisis de-escalation, restraint,
19
+ * SIB with injury history and risk assessment never reach a local model at
20
+ * all; that boundary is enforced upstream in the Layer 1 screen and is not
21
+ * relaxed, widened or re-implemented here. This gate governs the routine
22
+ * band that is already local-eligible: operational definitions,
23
+ * measurement, antecedent strategies, caregiver training.
24
+ *
25
+ * A failure is NOT auto-repaired. The coding repair loop re-prompts the same
26
+ * tier to fix a syntax defect, which is safe for code; asking a local model to
27
+ * invent a missing decision-rules section produces plausible unratified
28
+ * clinical text, which is worse than a visibly incomplete draft. A clinical
29
+ * reason therefore falls out of the repair loop and escalates instead.
30
+ */
31
+ /** A full written plan was asked for — the whole section list applies. */
32
+ const CLINICAL_PLAN_REQUEST_RE = /\b(bip\b|behavi(?:o|ou)r(?:al)?[ -](?:intervention|support|management)[ -]plan|behavi(?:o|ou)r plan|treatment plan|intervention plan)\b/i;
33
+ /** An operational definition specifically was asked for. */
34
+ const OPERATIONAL_DEFINITION_REQUEST_RE = /\boperational(?:ly)?[ -]?(?:defin\w*)|\bdefine the (?:target )?behaviou?r\b/i;
35
+ /** Any behaviour-analytic context at all — gates the AAC safety check. */
36
+ const CLINICAL_CONTEXT_RE = /\b(aba\b|bcba\b|behaviou?r analyst|functional behaviou?r assessment|\bfba\b|\bbip\b|replacement behaviou?r|target behaviou?r|reinforcement schedule|\bfct\b|\bdro\b|\bdra\b|\bncr\b)/i;
37
+ /** Ordered so the report reads the way a plan is written. */
38
+ const PLAN_SECTIONS = [
39
+ { name: "operational_definition", pattern: /operational(?:ly)?[ -]?defin|\bdefinition\b[\s\S]{0,80}\b(observable|measurable)\b/i },
40
+ { name: "function_hypothesis", pattern: /\b(hypothesi[sz]ed function|function of the behaviou?r|maintained by|\ba-?b-?c\b|antecedent[\s\S]{0,40}consequence)\b/i },
41
+ { name: "antecedent_strategies", pattern: /\b(antecedent (?:strateg|modificat|intervention)|prevention strateg|setting event|environmental modificat)/i },
42
+ { name: "replacement_behaviour", pattern: /\b(replacement behaviou?r|functional communication training|\bfct\b|alternative behaviou?r|\bdra\b)/i },
43
+ { name: "consequence_strategies", pattern: /\b(consequence (?:strateg|procedure)|reinforcement (?:schedule|procedure|strateg)|\bdro\b|\bncr\b|extinction)/i },
44
+ { name: "data_collection", pattern: /\b(data collection|data sheet|measurement (?:system|procedure)|frequency count|partial interval|momentary time sampling|\bioa\b|interobserver)/i },
45
+ { name: "decision_rules", pattern: /\b(decision rule|mastery criteri|criteri\w+ for (?:change|modificat|advancement)|review (?:schedule|trigger)|plan review)/i },
46
+ { name: "generalisation_maintenance", pattern: /\b(generali[sz]|maintenance)\b/i },
47
+ { name: "caregiver_training", pattern: /\b((?:caregiver|staff|parent|family)[ -]?training|train(?:ing)? (?:the )?(?:caregivers?|staff|parents?))/i },
48
+ { name: "bcba_review_disclaimer", pattern: /\b(reviewed and individuali[sz]ed|credentialed bcba|licensed behaviou?r analyst|must be reviewed)\b/i },
49
+ ];
50
+ /**
51
+ * AAC access may never be removed, withheld or delayed as a consequence.
52
+ *
53
+ * A correct plan states this rule explicitly ("AAC access is never restricted"),
54
+ * so a bare co-occurrence of an AAC term and a restriction verb fires on GOOD
55
+ * text. The lookback suppresses a match when the clause is negated. It is
56
+ * approximate by construction, which is acceptable only because this raises and
57
+ * never clears: an escalation costs one call, and no output is marked safe here.
58
+ */
59
+ const AAC_TERM = /\b(aac\b|speech[- ]generating device|\bsgd\b|communication device|communication board|\bpecs\b|talker\b)/i;
60
+ const RESTRICT_VERB = /\b(remov\w+|withh\w+|restrict\w+|tak\w+ away|deni\w+|deny|block\w*|delay\w*|confiscat\w+|limit\w*)\b/i;
61
+ const NEGATOR = /\b(never|not|n't|no|avoid\w*|prohibit\w*|must not|cannot|can't|without)\b/i;
62
+ // Asymmetric on purpose. "remove the AAC device" puts the verb BEFORE the term,
63
+ // so a forward-only window misses the most direct phrasing of the thing this
64
+ // check exists to catch. The backward reach is kept short because a removal
65
+ // sentence about something else ("remove the token board") sitting a paragraph
66
+ // above an AAC mention is not a restriction of AAC.
67
+ const AAC_WINDOW_AFTER = 120;
68
+ const AAC_WINDOW_BEFORE = 40;
69
+ const NEGATION_LOOKBACK = 60;
70
+ /** Clause boundaries. The verb must act on the AAC term, not merely sit near it:
71
+ * "AAC remains available at all times; remove the token board" removes a token
72
+ * board, and scanning past the semicolon read it as removing AAC. */
73
+ const CLAUSE_BREAK = /[.;:\n]|\bhowever\b|\bwhereas\b/i;
74
+ function clauseAfter(text, from, limit) {
75
+ const slice = text.slice(from, from + limit);
76
+ const brk = CLAUSE_BREAK.exec(slice);
77
+ return brk ? slice.slice(0, brk.index) : slice;
78
+ }
79
+ function clauseBefore(text, end, limit) {
80
+ const slice = text.slice(Math.max(0, end - limit), end);
81
+ let last = -1;
82
+ for (const m of slice.matchAll(new RegExp(CLAUSE_BREAK.source, "gi"))) {
83
+ last = (m.index ?? 0) + m[0].length;
84
+ }
85
+ return last >= 0 ? slice.slice(last) : slice;
86
+ }
87
+ function aacRestrictedAsConsequence(output) {
88
+ for (const m of output.matchAll(new RegExp(AAC_TERM.source, "gi"))) {
89
+ const start = m.index ?? 0;
90
+ const before = clauseBefore(output, start, AAC_WINDOW_BEFORE);
91
+ const after = clauseAfter(output, start, AAC_WINDOW_AFTER);
92
+ const from = start - before.length;
93
+ const window = before + after;
94
+ const verb = RESTRICT_VERB.exec(window);
95
+ if (!verb)
96
+ continue;
97
+ const absolute = from + (verb.index ?? 0);
98
+ const lookback = output.slice(Math.max(0, absolute - NEGATION_LOOKBACK), absolute);
99
+ if (NEGATOR.test(lookback))
100
+ continue; // "AAC access is never removed"
101
+ return true;
102
+ }
103
+ return false;
104
+ }
105
+ /**
106
+ * Raise-only structural check. `pass: true` means nothing was detected as
107
+ * missing — it is not a clinical endorsement.
108
+ */
109
+ export function passesClinicalQualityGate(prompt, output) {
110
+ // An operational-definition request is clinical on its own: "write an
111
+ // operational definition of elopement" names no ABA vocabulary the broad
112
+ // pattern looks for, and was silently skipped before.
113
+ const clinicalContext = CLINICAL_CONTEXT_RE.test(prompt)
114
+ || CLINICAL_PLAN_REQUEST_RE.test(prompt)
115
+ || OPERATIONAL_DEFINITION_REQUEST_RE.test(prompt);
116
+ if (!clinicalContext)
117
+ return { pass: true };
118
+ if (aacRestrictedAsConsequence(output)) {
119
+ return { pass: false, reason: "clinical_aac_restricted_as_consequence" };
120
+ }
121
+ if (OPERATIONAL_DEFINITION_REQUEST_RE.test(prompt)) {
122
+ const hasExamples = /\bexamples?\b/i.test(output);
123
+ const hasNonExamples = /\bnon-?examples?\b/i.test(output);
124
+ if (!hasExamples || !hasNonExamples) {
125
+ return { pass: false, reason: "clinical_operational_definition_incomplete" };
126
+ }
127
+ }
128
+ if (!CLINICAL_PLAN_REQUEST_RE.test(prompt))
129
+ return { pass: true };
130
+ const missing = PLAN_SECTIONS.filter(s => !s.pattern.test(output)).map(s => s.name);
131
+ const sections = {
132
+ required: PLAN_SECTIONS.length,
133
+ present: PLAN_SECTIONS.length - missing.length,
134
+ missing,
135
+ };
136
+ // An incomplete plan REPORTS; it does not fail. Failing the gate rejects the
137
+ // output, and when escalation is unavailable the caller receives nothing at
138
+ // all — measured in review: a 3-of-10 plan with cloud_fallback:true and an
139
+ // unreachable portal returned "no backend produced output". A draft labelled
140
+ // `clinical_sections=3/10 missing:...` is strictly more useful to a clinician
141
+ // than silence, and suppressing it contradicts the raise-only rule above.
142
+ //
143
+ // The two findings ABOVE do fail, because they are defects rather than
144
+ // incompleteness: AAC restricted as a consequence is a safety violation, and
145
+ // an operational definition without non-examples is wrong, not unfinished.
146
+ return { pass: true, sections };
147
+ }
148
+ /** Compact, raise-only header fragment. Counts only — never a verdict word. */
149
+ export function formatClinicalSections(s) {
150
+ const base = `clinical_sections=${s.present}/${s.required}`;
151
+ return s.missing.length ? `${base} missing:${s.missing.join(",")}` : base;
152
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "prism-mcp-server",
3
- "version": "20.21.2",
3
+ "version": "20.21.3",
4
4
  "mcpName": "io.github.dcostenco/prism-coder",
5
5
  "description": "Persistent session memory for AI coding agents that never leaves your machine — including the on-device model that reasons over it. Restores your prior decisions, open TODOs, and changed files across sessions; adds associative recall of related past work, semantic drift detection, and local inference. Local-first by default. Works with Claude Code, Cursor, and Codex.",
6
6
  "module": "index.ts",