prism-mcp-server 20.21.3 → 20.21.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -30,7 +30,7 @@ import { getEntitlements, clampCeiling, multiTurnPolicy, ABSOLUTE_MULTI_TURN } f
|
|
|
30
30
|
import { ddLog } from "../utils/ddLogger.js";
|
|
31
31
|
import { stripThink } from "../utils/thinkStrip.js";
|
|
32
32
|
import { passesQualityGate } from "../utils/qualityGate.js";
|
|
33
|
-
import { passesClinicalQualityGate, formatClinicalSections, } from "../utils/clinicalQualityPolicy.js";
|
|
33
|
+
import { passesClinicalQualityGate, clinicalPlanScaffold, formatClinicalSections, } from "../utils/clinicalQualityPolicy.js";
|
|
34
34
|
import { applyDeterministicCodingRepairs, buildCodingRepairPrompt, passesCodingQualityGate, } from "../utils/codingQualityPolicy.js";
|
|
35
35
|
import { checkInputSafety, checkOutputSafety } from "../utils/safetyGate.js";
|
|
36
36
|
import { callLayer1 as defaultCallLayer1, classifyDeterministicLayer1, keywordBackstop, reservedCategory, MAX_CLASSIFIER_PROMPT_LENGTH } from "../utils/layer1.js";
|
|
@@ -2018,9 +2018,13 @@ export async function runInfer(args, deps) {
|
|
|
2018
2018
|
// their own — never override an explicit instruction.
|
|
2019
2019
|
// `=== undefined`, not falsy: `system: ""` is a caller explicitly asking
|
|
2020
2020
|
// for no system prompt, and overriding that is still an override.
|
|
2021
|
-
|
|
2022
|
-
|
|
2023
|
-
|
|
2021
|
+
// A caller's own `system` always wins, including `system: ""`, which is an
|
|
2022
|
+
// explicit request for none. Defaults apply only when it is undefined.
|
|
2023
|
+
const defaultSystem = [
|
|
2024
|
+
(resolvedImages?.length ?? 0) > 0 ? VISION_SYSTEM_PROMPT : undefined,
|
|
2025
|
+
clinicalPlanScaffold(args.prompt),
|
|
2026
|
+
].filter(Boolean).join("\n\n") || undefined;
|
|
2027
|
+
const effectiveSystem = args.system === undefined ? defaultSystem : args.system;
|
|
2024
2028
|
// Walk order for images.
|
|
2025
2029
|
//
|
|
2026
2030
|
// Smallest-first was tried and REVERTED on 2026-08-15. It is correct
|
|
@@ -2299,7 +2303,8 @@ export async function runInfer(args, deps) {
|
|
|
2299
2303
|
const codingGateFailure = !gate.pass &&
|
|
2300
2304
|
mode === "code" &&
|
|
2301
2305
|
(gate.reason?.startsWith("code_") === true ||
|
|
2302
|
-
gate.reason?.startsWith("python_") === true
|
|
2306
|
+
gate.reason?.startsWith("python_") === true ||
|
|
2307
|
+
gate.reason?.startsWith("ts_") === true);
|
|
2303
2308
|
if (!codingGateFailure)
|
|
2304
2309
|
break;
|
|
2305
2310
|
const failedReason = gate.reason ?? "code_quality";
|
|
@@ -36,16 +36,16 @@ const OPERATIONAL_DEFINITION_REQUEST_RE = /\boperational(?:ly)?[ -]?(?:defin\w*)
|
|
|
36
36
|
const CLINICAL_CONTEXT_RE = /\b(aba\b|bcba\b|behaviou?r analyst|functional behaviou?r assessment|\bfba\b|\bbip\b|replacement behaviou?r|target behaviou?r|reinforcement schedule|\bfct\b|\bdro\b|\bdra\b|\bncr\b)/i;
|
|
37
37
|
/** Ordered so the report reads the way a plan is written. */
|
|
38
38
|
const PLAN_SECTIONS = [
|
|
39
|
-
{ name: "operational_definition", pattern: /operational(?:ly)?[ -]?defin|\bdefinition\b[\s\S]{0,80}\b(observable|measurable)\b/i },
|
|
40
|
-
{ name: "function_hypothesis", pattern: /\b(hypothesi[sz]ed function|function of the behaviou?r|maintained by|\ba-?b-?c\b|antecedent[\s\S]{0,40}consequence)\b/i },
|
|
41
|
-
{ name: "antecedent_strategies", pattern: /\b(antecedent (?:strateg|modificat|intervention)|prevention strateg|setting event|environmental modificat)/i },
|
|
42
|
-
{ name: "replacement_behaviour", pattern: /\b(replacement behaviou?r|functional communication training|\bfct\b|alternative behaviou?r|\bdra\b)/i },
|
|
43
|
-
{ name: "consequence_strategies", pattern: /\b(consequence (?:strateg|
|
|
44
|
-
{ name: "data_collection", pattern: /\b(data collection|data sheet|measurement (?:system|procedure)|frequency count|partial interval|momentary time sampling|\bioa\b|interobserver)/i },
|
|
45
|
-
{ name: "decision_rules", pattern: /\b(decision rule|mastery criteri|criteri\w+ for (?:change|modificat|advancement)|review (?:schedule|trigger)|plan review)/i },
|
|
46
|
-
{ name: "generalisation_maintenance", pattern: /\b(generali[sz]|maintenance)\b/i },
|
|
47
|
-
{ name: "caregiver_training", pattern: /\b((?:caregiver|staff|parent|family)[ -]?training|train(?:ing)? (?:the )?(?:caregivers?|staff|parents?))/i },
|
|
48
|
-
{ name: "bcba_review_disclaimer", pattern: /\b(reviewed and individuali[sz]ed|credentialed bcba|licensed behaviou?r analyst|must be reviewed)\b/i },
|
|
39
|
+
{ name: "operational_definition", requirement: "an operational definition that is observable and measurable, with examples AND non-examples", pattern: /operational(?:ly)?[ -]?defin|\bdefinition\b[\s\S]{0,80}\b(observable|measurable)\b/i },
|
|
40
|
+
{ name: "function_hypothesis", requirement: "a hypothesised function supported by A-B-C data", pattern: /\b(hypothesi[sz]ed function|function of the behaviou?r|maintained by|\ba-?b-?c\b|antecedent[\s\S]{0,40}consequence)\b/i },
|
|
41
|
+
{ name: "antecedent_strategies", requirement: "antecedent and prevention strategies", pattern: /\b(antecedent (?:strateg|modificat|intervention|procedure|support)|prevention strateg|proactive strateg|pre-?work strateg|pre-?correct|setting event|environmental modificat|visual (?:schedule|timer|cue)|priming)/i },
|
|
42
|
+
{ name: "replacement_behaviour", requirement: "a functionally equivalent replacement behaviour", pattern: /\b(replacement behaviou?r|functional communication training|\bfct\b|alternative behaviou?r|\bdra\b)/i },
|
|
43
|
+
{ name: "consequence_strategies", requirement: "consequence strategies, including what reinforces the replacement", pattern: /\b(consequence|reinforcement (?:schedule|procedure|strateg|plan|system|for\b)|reinforc\w+ the (?:replacement|desired|appropriate|target)|planned ignoring|response to (?:the )?behaviou?r|redirect\w*|\bpraise\b|\bdro\b|\bncr\b|extinction)/i },
|
|
44
|
+
{ name: "data_collection", requirement: "a data collection method", pattern: /\b(data collection|data sheet|measurement (?:system|procedure)|frequency count|partial interval|momentary time sampling|\bioa\b|interobserver)/i },
|
|
45
|
+
{ name: "decision_rules", requirement: "decision rules and a review schedule", pattern: /\b(decision rule|mastery criteri|criteri\w+ for (?:change|modificat|advancement)|evaluation criteri|review (?:schedule|trigger|date)|plan review|progress monitor\w*|plan will be (?:adjusted|modified|revised|changed))/i },
|
|
46
|
+
{ name: "generalisation_maintenance", requirement: "generalisation and maintenance", pattern: /\b(generali[sz]|maintenance)\b/i },
|
|
47
|
+
{ name: "caregiver_training", requirement: "caregiver and staff training", pattern: /\b((?:caregiver|staff|parent|family|teacher|team)[ -]?training|train(?:ing|ed)? (?:the )?(?:caregivers?|staff|parents?|team)|(?:staff|caregivers?|parents?|team|teachers?)\b[^.\n]{0,30}\btrain\w+|train\w+ on the plan)/i },
|
|
48
|
+
{ name: "bcba_review_disclaimer", requirement: "a statement that a credentialed BCBA must review and individualise the plan before implementation", pattern: /\b(reviewed and individuali[sz]ed|credentialed bcba|licensed behaviou?r analyst|must be reviewed)\b/i },
|
|
49
49
|
];
|
|
50
50
|
/**
|
|
51
51
|
* AAC access may never be removed, withheld or delayed as a consequence.
|
|
@@ -102,6 +102,58 @@ function aacRestrictedAsConsequence(output) {
|
|
|
102
102
|
}
|
|
103
103
|
return false;
|
|
104
104
|
}
|
|
105
|
+
/** Characters of real prose required near a section marker for it to count. */
|
|
106
|
+
const SECTION_CONTENT_CHARS = 30;
|
|
107
|
+
const SECTION_WINDOW = 400;
|
|
108
|
+
/** Drop whole heading lines. Stripping only the `#` turns the NEXT heading into
|
|
109
|
+
* prose, which is why ten empty headings first scored 8 of 10. */
|
|
110
|
+
function proseOnly(text) {
|
|
111
|
+
return text
|
|
112
|
+
.split("\n")
|
|
113
|
+
.filter(line => !/^\s*#{1,6}\s/.test(line)) // markdown headings
|
|
114
|
+
.filter(line => !/^\s*\*\*[^*]+\*\*\s*:?\s*$/.test(line)) // bold-only lines
|
|
115
|
+
.join(" ")
|
|
116
|
+
.replace(/[>#*_|]+/g, "")
|
|
117
|
+
.replace(/\[[^\]]*\]/g, "") // [placeholders]
|
|
118
|
+
.replace(/[-\s]+/g, " ")
|
|
119
|
+
.trim();
|
|
120
|
+
}
|
|
121
|
+
/**
|
|
122
|
+
* A section counts only when there is real prose NEAR its marker.
|
|
123
|
+
*
|
|
124
|
+
* Without this, ten empty headings scored 8 of 10: an output with no clinical
|
|
125
|
+
* content looked nearly complete, because the census matched vocabulary rather
|
|
126
|
+
* than substance. The scaffold already demands "substantive content rather than
|
|
127
|
+
* a heading alone"; this is the census checking the same thing.
|
|
128
|
+
*
|
|
129
|
+
* The window spans both directions. A first version looked only forward and
|
|
130
|
+
* dropped a legitimate credit — "we will write down how often it happens on a
|
|
131
|
+
* data sheet" puts the content BEFORE the keyword.
|
|
132
|
+
*
|
|
133
|
+
* Two limits, both measured rather than assumed.
|
|
134
|
+
*
|
|
135
|
+
* The window is wide, so in a dense document a marker finds prose belonging to
|
|
136
|
+
* a NEIGHBOURING section and is credited for it. This is therefore closer to a
|
|
137
|
+
* document-level check than a per-section one; what it reliably catches is the
|
|
138
|
+
* empty or near-empty output, which is what it was added for.
|
|
139
|
+
*
|
|
140
|
+
* And a model that echoes the section list back as prose still scores full
|
|
141
|
+
* marks, because a description of what a plan must contain is, at this level of
|
|
142
|
+
* analysis, indistinguishable from a plan. That is a limit of the approach, not
|
|
143
|
+
* something to regex away, and one more reason nothing here is an endorsement.
|
|
144
|
+
*
|
|
145
|
+
* The floor is deliberately low. At 50 characters a legitimately terse section
|
|
146
|
+
* — "Frequency count / Daily tally / Weekly IOA" — was refused, and a false
|
|
147
|
+
* negative on real content is the more expensive error for a gap report.
|
|
148
|
+
*/
|
|
149
|
+
function sectionHasContent(output, pattern) {
|
|
150
|
+
const m = new RegExp(pattern.source, pattern.flags.replace("g", "")).exec(output);
|
|
151
|
+
if (!m)
|
|
152
|
+
return false;
|
|
153
|
+
const at = m.index ?? 0;
|
|
154
|
+
const window = output.slice(Math.max(0, at - SECTION_WINDOW), at + m[0].length + SECTION_WINDOW);
|
|
155
|
+
return proseOnly(window).length >= SECTION_CONTENT_CHARS;
|
|
156
|
+
}
|
|
105
157
|
/**
|
|
106
158
|
* Raise-only structural check. `pass: true` means nothing was detected as
|
|
107
159
|
* missing — it is not a clinical endorsement.
|
|
@@ -127,7 +179,9 @@ export function passesClinicalQualityGate(prompt, output) {
|
|
|
127
179
|
}
|
|
128
180
|
if (!CLINICAL_PLAN_REQUEST_RE.test(prompt))
|
|
129
181
|
return { pass: true };
|
|
130
|
-
const missing = PLAN_SECTIONS
|
|
182
|
+
const missing = PLAN_SECTIONS
|
|
183
|
+
.filter(s => !sectionHasContent(output, s.pattern))
|
|
184
|
+
.map(s => s.name);
|
|
131
185
|
const sections = {
|
|
132
186
|
required: PLAN_SECTIONS.length,
|
|
133
187
|
present: PLAN_SECTIONS.length - missing.length,
|
|
@@ -150,3 +204,42 @@ export function formatClinicalSections(s) {
|
|
|
150
204
|
const base = `clinical_sections=${s.present}/${s.required}`;
|
|
151
205
|
return s.missing.length ? `${base} missing:${s.missing.join(",")}` : base;
|
|
152
206
|
}
|
|
207
|
+
/**
|
|
208
|
+
* A system instruction naming every section a plan must contain, generated from
|
|
209
|
+
* PLAN_SECTIONS so the list that INSTRUCTS is the list that VERIFIES.
|
|
210
|
+
*
|
|
211
|
+
* Measured on prism-coder:9b: the same plan request scored 7/10 unscaffolded and
|
|
212
|
+
* 10/10 scaffolded, in FEWER characters — it restructured rather than padded,
|
|
213
|
+
* and the previously absent sections came back with substantive content
|
|
214
|
+
* (a real observable definition with non-examples, real generalisation content,
|
|
215
|
+
* a correctly worded review statement).
|
|
216
|
+
*
|
|
217
|
+
* KNOWN EPISTEMIC COST, recorded rather than hidden: once the model is told the
|
|
218
|
+
* list, the census stops being independent confirmation and becomes a check
|
|
219
|
+
* that the instruction was followed. A scaffolded 10/10 is weaker evidence than
|
|
220
|
+
* an unscaffolded one. Sharing one list is still the right trade — two lists
|
|
221
|
+
* drift, and a census that disagrees with the instruction is worse than a
|
|
222
|
+
* census that merely confirms it — but nothing here should be read as evidence
|
|
223
|
+
* that the model knows what a plan needs.
|
|
224
|
+
*
|
|
225
|
+
* LOCAL ONLY. `callCloud` takes the prompt and no system argument, so nothing
|
|
226
|
+
* here reaches an escalated request — true of VISION_SYSTEM_PROMPT as well, and
|
|
227
|
+
* pre-existing rather than introduced with this scaffold. The consequence is
|
|
228
|
+
* that a cloud-served plan is measured by the census WITHOUT having been given
|
|
229
|
+
* the list, so it can score lower than a local one for reasons that have
|
|
230
|
+
* nothing to do with the model. Threading `system` through the portal API is
|
|
231
|
+
* the real fix and is deliberately out of scope here.
|
|
232
|
+
*
|
|
233
|
+
* Returns undefined unless a full plan was requested, so it never touches the
|
|
234
|
+
* prompt for ordinary work.
|
|
235
|
+
*/
|
|
236
|
+
export function clinicalPlanScaffold(prompt) {
|
|
237
|
+
if (!CLINICAL_PLAN_REQUEST_RE.test(prompt))
|
|
238
|
+
return undefined;
|
|
239
|
+
const items = PLAN_SECTIONS.map(s => `- ${s.requirement}`).join("\n");
|
|
240
|
+
return ("A behaviour plan must contain all of the following, each with substantive "
|
|
241
|
+
+ "content rather than a heading alone:\n" + items
|
|
242
|
+
+ "\n\nUse least restrictive, dignity-preserving, function-based procedures. "
|
|
243
|
+
+ "Never restrict, remove or delay access to an AAC or communication device "
|
|
244
|
+
+ "as a consequence.");
|
|
245
|
+
}
|
|
@@ -2,6 +2,7 @@ import { spawnSync } from "node:child_process";
|
|
|
2
2
|
const IMPLEMENTATION_REQUEST_RE = /\b(?:implement|write|create|generate|complete|finish|fix)\b[\s\S]{0,160}\b(?:code|source|function|method|class|interface|struct|enum|implementation|algorithm|component|endpoint)\b/i;
|
|
3
3
|
const STRICT_SOURCE_REQUEST_RE = /\b(?:return|output|respond with)\s+only\s+(?:the\s+)?(?:implementation\s+)?(?:source\s+)?code\b/i;
|
|
4
4
|
const CODE_SHAPE_RE = /(?:^|\n)\s*(?:(?:export|public|private|protected|internal|open|pub|static|final|abstract|async)\s+)*(?:class|interface|struct|enum|function|def|func|fun|fn|type)\s+[A-Za-z_$][\w$]*|(?:^|\n)\s*(?:const|let|var)\s+[A-Za-z_$][\w$]*\s*=|(?:^|\n)\s*(?:[A-Za-z_$][\w$:<>,.?*[\]&]*\s+)+[A-Za-z_$][\w$]*\s*\([^;\n]*\)\s*(?:const\s*)?(?:noexcept\s*)?(?:\{|=>)|=>\s*[{(]/m;
|
|
5
|
+
import { analyzeTypeScript } from "./typescriptDiagnostics.js";
|
|
5
6
|
export const INCOMPLETE_IMPLEMENTATION_PATTERNS = [
|
|
6
7
|
{
|
|
7
8
|
reason: "code_placeholder",
|
|
@@ -46,6 +47,8 @@ const PYTHON_AST_SCRIPT = `import ast,sys; print('${PYTHON_READY_SENTINEL}', flu
|
|
|
46
47
|
"tree=ast.parse(sys.stdin.read()); compile(tree, '<prism-coding-gate>', 'exec')";
|
|
47
48
|
const PYTHON_COMMANDS = ["python3", "python"];
|
|
48
49
|
const PYTHON_CHILDREN_KEYS_UNPACK_RE = /\bfor\s+[A-Za-z_]\w*\s*,\s*child(?:_node)?\s+in\s+(?:sorted\(\s*)?[A-Za-z_][\w.]*\.children\.keys\(\)\s*\)?\s*:/;
|
|
50
|
+
/** Enough of a type-annotation or declaration signal to call a block TypeScript. */
|
|
51
|
+
const TS_SIGNAL_RE = /:\s*(?:string|number|boolean|void|any|unknown|never|Promise<)\b|\binterface\s+[A-Z]|\bexport\s+(?:class|interface|type|abstract)\b|\b(?:private|public|protected|readonly)\s+\w+\s*[:=]|<[A-Z]\w*(?:\s*,\s*[A-Z]\w*)*>/;
|
|
49
52
|
function extractUnfencedPythonCode(output) {
|
|
50
53
|
const lines = output.trim().split(/\r?\n/);
|
|
51
54
|
const start = lines.findIndex((line) => UNFENCED_PYTHON_START_RE.test(line));
|
|
@@ -75,9 +78,11 @@ function extractCode(output) {
|
|
|
75
78
|
})).filter((block) => block.code.length > 0);
|
|
76
79
|
if (blocks.length === 0) {
|
|
77
80
|
const python = extractUnfencedPythonCode(output);
|
|
81
|
+
const bare = output.trim();
|
|
78
82
|
return {
|
|
79
|
-
all:
|
|
83
|
+
all: bare,
|
|
80
84
|
...(python ? { python } : {}),
|
|
85
|
+
...(!python && TS_SIGNAL_RE.test(bare) ? { typescript: bare } : {}),
|
|
81
86
|
hasFences: false,
|
|
82
87
|
};
|
|
83
88
|
}
|
|
@@ -86,11 +91,18 @@ function extractCode(output) {
|
|
|
86
91
|
block.language === "py" ||
|
|
87
92
|
(!block.language && PYTHON_SIGNAL_RE.test(block.code))))
|
|
88
93
|
.map((block) => block.code);
|
|
94
|
+
const tsBlocks = blocks
|
|
95
|
+
.filter((block) => (block.language === "typescript" ||
|
|
96
|
+
block.language === "ts" ||
|
|
97
|
+
block.language === "tsx" ||
|
|
98
|
+
(!block.language && TS_SIGNAL_RE.test(block.code))))
|
|
99
|
+
.map((block) => block.code);
|
|
89
100
|
return {
|
|
90
101
|
all: blocks.map((block) => block.code).join("\n\n"),
|
|
91
102
|
...(pythonBlocks.length > 0
|
|
92
103
|
? { python: pythonBlocks.join("\n\n") }
|
|
93
104
|
: {}),
|
|
105
|
+
...(tsBlocks.length > 0 ? { typescript: tsBlocks.join("\n\n") } : {}),
|
|
94
106
|
hasFences: true,
|
|
95
107
|
};
|
|
96
108
|
}
|
|
@@ -304,7 +316,98 @@ function pythonStaticContractFailure(code) {
|
|
|
304
316
|
? `python_static_contract:${[...issues].sort().join(",")}`
|
|
305
317
|
: undefined;
|
|
306
318
|
}
|
|
319
|
+
/**
|
|
320
|
+
* A generic used with no type argument, e.g. `Map<string, Array>`.
|
|
321
|
+
*
|
|
322
|
+
* prism-coder:9b emits this repeatedly — observed in four separate generations
|
|
323
|
+
* of the same EventEmitter task — and it is a hard compile error (TS2314), so
|
|
324
|
+
* the file never builds. Detecting it needs no TypeScript dependency: the shape
|
|
325
|
+
* is unambiguous when the bare name sits inside a type-argument list or
|
|
326
|
+
* directly after a type annotation.
|
|
327
|
+
*
|
|
328
|
+
* Deliberately narrow. Prose mentioning "the Array, then the Map" must not
|
|
329
|
+
* match, so a bare name is only a finding when it is syntactically in a type
|
|
330
|
+
* position; a bare `Promise` as a lone return type with no delimiter after it
|
|
331
|
+
* is missed, which is the conservative direction.
|
|
332
|
+
*/
|
|
333
|
+
const TS_GENERIC = "(?:Array|Map|Set|Promise|Record|Partial|Readonly|WeakMap|WeakSet)";
|
|
334
|
+
const TS_BARE_GENERIC_RE = new RegExp(`<[^<>]*\\b${TS_GENERIC}\\b(?!\\s*<)[^<>]*>` +
|
|
335
|
+
`|:\\s*${TS_GENERIC}\\b(?!\\s*<)\\s*[={;,)\\]]`);
|
|
336
|
+
/** Null when the code carries no TypeScript static-contract defect. */
|
|
337
|
+
function tsStaticContractFailure(code) {
|
|
338
|
+
return TS_BARE_GENERIC_RE.test(code) ? "ts_static_contract:bare_generic" : null;
|
|
339
|
+
}
|
|
340
|
+
/** Character ranges a repair must not touch on one line.
|
|
341
|
+
*
|
|
342
|
+
* Strings, because rewriting `"use Map<string, Array> carefully"` changes a
|
|
343
|
+
* RUNTIME VALUE rather than a type. And comments, because `// don't use a bare
|
|
344
|
+
* Map<string, Array>` is a warning against the very thing the repair would
|
|
345
|
+
* write, so editing it inverts the author's meaning. Both found in review. */
|
|
346
|
+
function protectedSpans(line) {
|
|
347
|
+
const spans = [];
|
|
348
|
+
const strings = /"(?:[^"\\]|\\.)*"|'(?:[^'\\]|\\.)*'|`(?:[^`\\]|\\.)*`/g;
|
|
349
|
+
for (const m of line.matchAll(strings))
|
|
350
|
+
spans.push([m.index ?? 0, (m.index ?? 0) + m[0].length]);
|
|
351
|
+
// A line comment runs to end of line. An apostrophe in prose ("don't")
|
|
352
|
+
// breaks the string scan, which is how comments slipped through before.
|
|
353
|
+
const comment = /\/\/|\/\*/.exec(line);
|
|
354
|
+
if (comment)
|
|
355
|
+
spans.push([comment.index, line.length]);
|
|
356
|
+
return spans;
|
|
357
|
+
}
|
|
358
|
+
/** Fence languages whose contents are TypeScript. A bare generic inside a
|
|
359
|
+
* python or json block is not a type error to fix — the first version rewrote
|
|
360
|
+
* a comment inside a python block in a multi-language answer. */
|
|
361
|
+
const TS_FENCE_LANG = /^\s*```\s*(ts|typescript|tsx)?\s*$/i;
|
|
362
|
+
/** `Map<string, Array>` -> `Map<string, Array<any>>`, within code only.
|
|
363
|
+
*
|
|
364
|
+
* `any` rather than `unknown` on purpose: `unknown` makes the file compile and
|
|
365
|
+
* then breaks every use of the value, which trades one compile error for
|
|
366
|
+
* several. This makes the code build; it does not make it well typed.
|
|
367
|
+
*
|
|
368
|
+
* Scoped twice, both from adversarial review. Fenced output is repaired only
|
|
369
|
+
* INSIDE its fences, because rewriting the surrounding prose inverts sentences
|
|
370
|
+
* like "do not write Map<string, Array>". And no match inside a string literal
|
|
371
|
+
* is touched, because that is a value, not a type. */
|
|
372
|
+
function repairBareGenerics(code) {
|
|
373
|
+
let changed = false;
|
|
374
|
+
const spanRe = new RegExp(TS_BARE_GENERIC_RE.source, "g");
|
|
375
|
+
const repairLine = (line) => {
|
|
376
|
+
const off_limits = protectedSpans(line);
|
|
377
|
+
return line.replace(spanRe, (span, offset) => {
|
|
378
|
+
if (off_limits.some(([a, b]) => offset >= a && offset < b))
|
|
379
|
+
return span;
|
|
380
|
+
const fixed = span.replace(new RegExp(`\\b(${TS_GENERIC})\\b(?!\\s*<)`, "g"), "$1<any>");
|
|
381
|
+
if (fixed !== span)
|
|
382
|
+
changed = true;
|
|
383
|
+
return fixed;
|
|
384
|
+
});
|
|
385
|
+
};
|
|
386
|
+
const lines = code.split("\n");
|
|
387
|
+
const hasFences = /^\s*```/m.test(code);
|
|
388
|
+
let inFence = false;
|
|
389
|
+
let fenceIsTs = false;
|
|
390
|
+
const out = lines.map((line) => {
|
|
391
|
+
if (/^\s*```/.test(line)) {
|
|
392
|
+
if (!inFence)
|
|
393
|
+
fenceIsTs = TS_FENCE_LANG.test(line);
|
|
394
|
+
inFence = !inFence;
|
|
395
|
+
return line;
|
|
396
|
+
}
|
|
397
|
+
// No fences at all: the whole output is the code block.
|
|
398
|
+
return (!hasFences || (inFence && fenceIsTs)) ? repairLine(line) : line;
|
|
399
|
+
});
|
|
400
|
+
return { code: out.join("\n"), changed };
|
|
401
|
+
}
|
|
307
402
|
export function applyDeterministicCodingRepairs(output, reason) {
|
|
403
|
+
if (reason.startsWith("ts_static_contract:")) {
|
|
404
|
+
if (!reason.includes("bare_generic"))
|
|
405
|
+
return { output, changes: [] };
|
|
406
|
+
const repaired = repairBareGenerics(output);
|
|
407
|
+
return repaired.changed
|
|
408
|
+
? { output: repaired.code, changes: ["bare_generic"] }
|
|
409
|
+
: { output, changes: [] };
|
|
410
|
+
}
|
|
308
411
|
if (!reason.startsWith("python_static_contract:")) {
|
|
309
412
|
return { output, changes: [] };
|
|
310
413
|
}
|
|
@@ -369,6 +472,18 @@ export function passesCodingQualityGate(prompt, output) {
|
|
|
369
472
|
if (pythonFailure)
|
|
370
473
|
return { pass: false, reason: pythonFailure };
|
|
371
474
|
}
|
|
475
|
+
// The regex floor runs first: it is the one finding with a deterministic
|
|
476
|
+
// repair, and it works even if the compiler cannot be loaded.
|
|
477
|
+
const tsFailure = tsStaticContractFailure(code);
|
|
478
|
+
if (tsFailure)
|
|
479
|
+
return { pass: false, reason: tsFailure };
|
|
480
|
+
// Then the compiler, over TypeScript blocks only.
|
|
481
|
+
if (extracted.typescript) {
|
|
482
|
+
const findings = analyzeTypeScript(extracted.typescript, extracted.hasFences);
|
|
483
|
+
if (findings.length > 0) {
|
|
484
|
+
return { pass: false, reason: `ts_static_contract:${findings.join(",")}` };
|
|
485
|
+
}
|
|
486
|
+
}
|
|
372
487
|
return { pass: true };
|
|
373
488
|
}
|
|
374
489
|
const CODING_REPAIR_SYSTEM_INSTRUCTION = "Repair the supplied implementation. Return one complete replacement implementation with no prose, " +
|
|
@@ -383,6 +498,12 @@ const CODING_REPAIR_GUIDANCE = {
|
|
|
383
498
|
python_method_missing_receiver: "Instance methods must take self first; class methods must take cls first unless decorated staticmethod.",
|
|
384
499
|
python_undefined_private_helper: "Define every directly called private self helper or replace the call with the correct defined helper.",
|
|
385
500
|
constructor_attribute_missing_receiver: "In __init__, persist instance state as self.<attribute>; do not assign it to a discarded local variable.",
|
|
501
|
+
syntax_error: "The code does not parse. Balance every brace, bracket and parenthesis, and finish every statement.",
|
|
502
|
+
optional_chain_assignment: "Optional chaining cannot appear on the left of an assignment. Guard with an if, or assert the value is present, before assigning to the property.",
|
|
503
|
+
type_not_assignable: "Make the returned value match the declared return type, or widen the declaration to what the implementation actually produces.",
|
|
504
|
+
implicit_any_param: "Annotate every parameter; under strict mode an inferred any is an error.",
|
|
505
|
+
deferred_mutation_returned_sync: "A value mutated inside a then/catch/setTimeout callback is returned before that callback runs. Await the work, or return a Promise that resolves after it.",
|
|
506
|
+
bare_generic: "Every generic needs its type argument: write Array<T>, Map<K, V>, Set<T>, Promise<T> — never a bare Array, Map, Set or Promise in a type position.",
|
|
386
507
|
dict_keys_unpack: "When unpacking key and value, iterate dictionary .items(); .keys() yields one key per iteration.",
|
|
387
508
|
};
|
|
388
509
|
function repairGuidance(reason) {
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Real type checking for generated TypeScript, plus one AST rule a type checker
|
|
3
|
+
* cannot express.
|
|
4
|
+
*
|
|
5
|
+
* The regex gate that shipped in 20.21.4 catches exactly one defect class
|
|
6
|
+
* (TS2314, a generic with no type argument). The very next sample defeated it:
|
|
7
|
+
* a doubly-linked-list splice written as `node.next?.prev = this.head`, which is
|
|
8
|
+
* TS2779 and does not compile. Extending the regex per error code is an arms
|
|
9
|
+
* race the compiler already wins.
|
|
10
|
+
*
|
|
11
|
+
* WHY THIS IS NOT JUST `tsc`. A generated snippet has no tsconfig, no resolved
|
|
12
|
+
* imports, and no way to declare whether it targets the DOM or Node. Checking a
|
|
13
|
+
* fragment therefore produces errors about the HARNESS rather than the code.
|
|
14
|
+
* Measured while building this: checking one LRU cache against the DOM lib made
|
|
15
|
+
* the snippet's own `Node` class collide with the DOM's, producing twelve
|
|
16
|
+
* phantom errors beside the two real ones. A gate that fails correct code gets
|
|
17
|
+
* switched off, so the design is an ALLOWLIST: a diagnostic is reported only if
|
|
18
|
+
* its code appears in ALLOWED_CODES. That, and nothing else, is what prevents a
|
|
19
|
+
* context failure from being reported as a defect.
|
|
20
|
+
*
|
|
21
|
+
* Two things that look load-bearing and are not — established by mutation, and
|
|
22
|
+
* recorded so nobody trusts them for safety:
|
|
23
|
+
*
|
|
24
|
+
* IGNORED_CODES does NOT filter anything. It marks codes already triaged as
|
|
25
|
+
* context failures, so the debug log can flag genuinely unclassified ones.
|
|
26
|
+
* Emptying it changes no reported finding.
|
|
27
|
+
*
|
|
28
|
+
* The narrow default library (`lib.es2022.d.ts`, no DOM) is defence in depth,
|
|
29
|
+
* not protection. It was chosen after the DOM lib made a snippet's own `Node`
|
|
30
|
+
* class collide with the DOM's and produce twelve extra diagnostics — but all
|
|
31
|
+
* twelve were outside the allowlist, so none would have been reported anyway.
|
|
32
|
+
* No constructed case makes the library choice change a reported finding. It
|
|
33
|
+
* is kept because less noise and less work are both worth having.
|
|
34
|
+
*/
|
|
35
|
+
import { createRequire } from "node:module";
|
|
36
|
+
import { dirname, join } from "node:path";
|
|
37
|
+
import { readFileSync } from "node:fs";
|
|
38
|
+
import { debugLog } from "./logger.js";
|
|
39
|
+
/** Semantic diagnostics that indict the SNIPPET. Each observed on real output. */
|
|
40
|
+
const ALLOWED_CODES = new Map([
|
|
41
|
+
[2314, "bare_generic"], // Map<string, Array>
|
|
42
|
+
[2779, "optional_chain_assignment"], // node.next?.prev = x
|
|
43
|
+
[2322, "type_not_assignable"], // Promise<PromiseSettledResult[]> as Promise<void>
|
|
44
|
+
[7006, "implicit_any_param"], // (listener) => ... under strict — CONDITIONAL, see below
|
|
45
|
+
]);
|
|
46
|
+
/**
|
|
47
|
+
* An implicit `any` only indicts the snippet once everything else resolved.
|
|
48
|
+
*
|
|
49
|
+
* `app.get("/", (req, res) => ...)` is correct Express, and `req` is implicitly
|
|
50
|
+
* any ONLY because a fragment cannot resolve `express`. Reporting that blames
|
|
51
|
+
* the harness. So TS7006 is suppressed whenever a module or name failed to
|
|
52
|
+
* resolve — found by attacking the allowlist rather than by review.
|
|
53
|
+
*/
|
|
54
|
+
const CONDITIONAL_ON_RESOLUTION = 7006;
|
|
55
|
+
const RESOLUTION_FAILURE_CODES = new Set([2304, 2307, 2792, 2583]);
|
|
56
|
+
/**
|
|
57
|
+
* Codes already triaged as context failures rather than defects.
|
|
58
|
+
*
|
|
59
|
+
* NOT a filter — the allowlist above is what decides what is reported. This
|
|
60
|
+
* exists so the debug log can distinguish "known to be noise" from "never seen
|
|
61
|
+
* before", which is how the allowlist gets extended from evidence.
|
|
62
|
+
*
|
|
63
|
+
* 2304/2583/2584 unresolved name, 2307 unresolved module, 2300 duplicate
|
|
64
|
+
* identifier, 2315/2554 a snippet type shadowed by a lib type (the `Node`
|
|
65
|
+
* collision), 6053 a lib file we did not serve, 2686/2695 UMD and expression
|
|
66
|
+
* complaints that only make sense inside a real project.
|
|
67
|
+
*/
|
|
68
|
+
const IGNORED_CODES = new Set([2300, 2304, 2307, 2315, 2554, 2583, 2584, 2686, 2695, 2792, 6053]);
|
|
69
|
+
let tsModule;
|
|
70
|
+
/** Loaded once, synchronously, so the quality gate stays synchronous. */
|
|
71
|
+
function loadTypeScript() {
|
|
72
|
+
if (tsModule !== undefined)
|
|
73
|
+
return tsModule;
|
|
74
|
+
try {
|
|
75
|
+
tsModule = createRequire(import.meta.url)("typescript");
|
|
76
|
+
}
|
|
77
|
+
catch (e) {
|
|
78
|
+
// A declared dependency, so this means a broken install. The regex floor
|
|
79
|
+
// in codingQualityPolicy still runs; this enhancement simply does not.
|
|
80
|
+
debugLog(`[ts-diagnostics] typescript unavailable: ${e instanceof Error ? e.message : e}`);
|
|
81
|
+
tsModule = null;
|
|
82
|
+
}
|
|
83
|
+
return tsModule;
|
|
84
|
+
}
|
|
85
|
+
const libCache = new Map();
|
|
86
|
+
function readLib(libDir, file) {
|
|
87
|
+
if (!libCache.has(file)) {
|
|
88
|
+
try {
|
|
89
|
+
libCache.set(file, readFileSync(join(libDir, file), "utf8"));
|
|
90
|
+
}
|
|
91
|
+
catch {
|
|
92
|
+
libCache.set(file, undefined);
|
|
93
|
+
}
|
|
94
|
+
}
|
|
95
|
+
return libCache.get(file);
|
|
96
|
+
}
|
|
97
|
+
/**
|
|
98
|
+
* Allowed findings only, de-duplicated, stable order. Empty when unavailable.
|
|
99
|
+
*
|
|
100
|
+
* `fenced` says the text was inside a ``` block, i.e. the model MEANT it as
|
|
101
|
+
* code. Only then is a parse failure attributable to the snippet. Unfenced
|
|
102
|
+
* output is ambiguous: "the function takes a value: string and returns a
|
|
103
|
+
* formatted result" is a sentence, and reporting it as a syntax error rejects a
|
|
104
|
+
* correct answer. Ambiguity favours the caller.
|
|
105
|
+
*/
|
|
106
|
+
export function typecheckSnippet(code, fenced = true) {
|
|
107
|
+
const ts = loadTypeScript();
|
|
108
|
+
if (!ts)
|
|
109
|
+
return [];
|
|
110
|
+
const libDir = dirname(createRequire(import.meta.url).resolve("typescript"));
|
|
111
|
+
const name = "snippet.ts";
|
|
112
|
+
const sourceOf = (file) => file === name ? code : readLib(libDir, file);
|
|
113
|
+
const host = {
|
|
114
|
+
getSourceFile: (file) => {
|
|
115
|
+
const text = sourceOf(file);
|
|
116
|
+
return text === undefined
|
|
117
|
+
? undefined
|
|
118
|
+
: ts.createSourceFile(file, text, ts.ScriptTarget.ES2022, true);
|
|
119
|
+
},
|
|
120
|
+
getDefaultLibFileName: () => "lib.es2022.d.ts",
|
|
121
|
+
writeFile: () => { },
|
|
122
|
+
getCurrentDirectory: () => "/",
|
|
123
|
+
getCanonicalFileName: (file) => file,
|
|
124
|
+
useCaseSensitiveFileNames: () => true,
|
|
125
|
+
getNewLine: () => "\n",
|
|
126
|
+
fileExists: (file) => sourceOf(file) !== undefined,
|
|
127
|
+
readFile: sourceOf,
|
|
128
|
+
};
|
|
129
|
+
let syntactic;
|
|
130
|
+
let semantic;
|
|
131
|
+
try {
|
|
132
|
+
const program = ts.createProgram([name], {
|
|
133
|
+
strict: true,
|
|
134
|
+
target: ts.ScriptTarget.ES2022,
|
|
135
|
+
noEmit: true,
|
|
136
|
+
types: [],
|
|
137
|
+
skipLibCheck: true,
|
|
138
|
+
}, host);
|
|
139
|
+
syntactic = program.getSyntacticDiagnostics();
|
|
140
|
+
semantic = program.getSemanticDiagnostics();
|
|
141
|
+
}
|
|
142
|
+
catch (e) {
|
|
143
|
+
debugLog(`[ts-diagnostics] check failed: ${e instanceof Error ? e.message : e}`);
|
|
144
|
+
return [];
|
|
145
|
+
}
|
|
146
|
+
// A parse failure needs no allowlist. Nothing about a missing library or an
|
|
147
|
+
// unresolved import can make a brace go missing, so a syntactic diagnostic
|
|
148
|
+
// always indicts the snippet. And once the file does not parse, the semantic
|
|
149
|
+
// results describe a tree that was never valid, so they are not consulted.
|
|
150
|
+
if (syntactic.length > 0)
|
|
151
|
+
return fenced ? ["syntax_error"] : [];
|
|
152
|
+
const unresolved = semantic.some(d => RESOLUTION_FAILURE_CODES.has(d.code));
|
|
153
|
+
const found = new Set();
|
|
154
|
+
for (const d of semantic) {
|
|
155
|
+
if (d.code === CONDITIONAL_ON_RESOLUTION && unresolved)
|
|
156
|
+
continue;
|
|
157
|
+
const name_ = ALLOWED_CODES.get(d.code);
|
|
158
|
+
if (name_)
|
|
159
|
+
found.add(name_);
|
|
160
|
+
else if (!IGNORED_CODES.has(d.code)) {
|
|
161
|
+
// Neither indicted nor excused. Logged so the lists can be extended
|
|
162
|
+
// from evidence rather than guessed at; never reported as a finding,
|
|
163
|
+
// because an unclassified code is exactly the kind that turns out to
|
|
164
|
+
// be about the harness.
|
|
165
|
+
debugLog(`[ts-diagnostics] unclassified TS${d.code}`);
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
return [...found].sort();
|
|
169
|
+
}
|
|
170
|
+
const DEFERRED_METHOD = /^(then|catch|finally)$/;
|
|
171
|
+
const DEFERRED_FN = /^(setTimeout|setInterval|setImmediate|queueMicrotask)$/;
|
|
172
|
+
/**
|
|
173
|
+
* A value mutated inside a promise or timer callback and then returned
|
|
174
|
+
* synchronously by the enclosing function.
|
|
175
|
+
*
|
|
176
|
+
* Valid TypeScript, and wrong: the callback runs in a later microtask, so the
|
|
177
|
+
* returned value never includes it. From the first multi-turn benchmark, where
|
|
178
|
+
* `emit()` incremented its counter inside `result.then(() => count++)` and
|
|
179
|
+
* returned 1 for three listeners. No type checker expresses this, but the AST
|
|
180
|
+
* is already loaded, so the rule is nearly free.
|
|
181
|
+
*/
|
|
182
|
+
export function findDeferredMutationReturnedSync(code) {
|
|
183
|
+
const ts = loadTypeScript();
|
|
184
|
+
if (!ts)
|
|
185
|
+
return [];
|
|
186
|
+
const sf = ts.createSourceFile("snippet.ts", code, ts.ScriptTarget.ES2022, true);
|
|
187
|
+
const hits = new Set();
|
|
188
|
+
const insideDeferredCallback = (node) => {
|
|
189
|
+
for (let p = node.parent; p; p = p.parent) {
|
|
190
|
+
if (ts.isCallExpression(p)) {
|
|
191
|
+
const callee = p.expression;
|
|
192
|
+
if (ts.isPropertyAccessExpression(callee) && DEFERRED_METHOD.test(callee.name.text))
|
|
193
|
+
return true;
|
|
194
|
+
if (ts.isIdentifier(callee) && DEFERRED_FN.test(callee.text))
|
|
195
|
+
return true;
|
|
196
|
+
}
|
|
197
|
+
// Stop at the enclosing function: a mutation in a sibling function
|
|
198
|
+
// says nothing about this one's return value.
|
|
199
|
+
if (ts.isFunctionDeclaration(p) || ts.isMethodDeclaration(p))
|
|
200
|
+
return false;
|
|
201
|
+
}
|
|
202
|
+
return false;
|
|
203
|
+
};
|
|
204
|
+
const inspect = (fn) => {
|
|
205
|
+
const mutatedLate = new Set();
|
|
206
|
+
const returned = new Set();
|
|
207
|
+
const visit = (n) => {
|
|
208
|
+
const target = (ts.isPostfixUnaryExpression(n) || ts.isPrefixUnaryExpression(n)) && ts.isIdentifier(n.operand)
|
|
209
|
+
? n.operand.text
|
|
210
|
+
: ts.isBinaryExpression(n) && ts.isIdentifier(n.left) && [
|
|
211
|
+
ts.SyntaxKind.EqualsToken,
|
|
212
|
+
ts.SyntaxKind.PlusEqualsToken,
|
|
213
|
+
ts.SyntaxKind.MinusEqualsToken,
|
|
214
|
+
].includes(n.operatorToken.kind)
|
|
215
|
+
? n.left.text
|
|
216
|
+
: null;
|
|
217
|
+
if (target && insideDeferredCallback(n))
|
|
218
|
+
mutatedLate.add(target);
|
|
219
|
+
if (ts.isReturnStatement(n) && n.expression && ts.isIdentifier(n.expression)) {
|
|
220
|
+
returned.add(n.expression.text);
|
|
221
|
+
}
|
|
222
|
+
// Do not descend into a NESTED named function or method: it has its
|
|
223
|
+
// own scope and its own `count`, and `scan` visits it separately.
|
|
224
|
+
// Without this, `inner`'s deferred mutation was credited to `outer`,
|
|
225
|
+
// flagging correct code. Arrow functions and anonymous function
|
|
226
|
+
// expressions ARE entered, because that is what a callback is.
|
|
227
|
+
if (ts.isFunctionDeclaration(n) || ts.isMethodDeclaration(n))
|
|
228
|
+
return;
|
|
229
|
+
ts.forEachChild(n, visit);
|
|
230
|
+
};
|
|
231
|
+
ts.forEachChild(fn, visit);
|
|
232
|
+
for (const v of mutatedLate)
|
|
233
|
+
if (returned.has(v))
|
|
234
|
+
hits.add(v);
|
|
235
|
+
return;
|
|
236
|
+
};
|
|
237
|
+
const scan = (n) => {
|
|
238
|
+
if (ts.isMethodDeclaration(n) || ts.isFunctionDeclaration(n) || ts.isFunctionExpression(n))
|
|
239
|
+
inspect(n);
|
|
240
|
+
ts.forEachChild(n, scan);
|
|
241
|
+
};
|
|
242
|
+
ts.forEachChild(sf, scan);
|
|
243
|
+
return hits.size ? ["deferred_mutation_returned_sync"] : [];
|
|
244
|
+
}
|
|
245
|
+
/** Every TypeScript finding for a snippet, type errors and the AST rule. */
|
|
246
|
+
export function analyzeTypeScript(code, fenced = true) {
|
|
247
|
+
return [...typecheckSnippet(code, fenced), ...findDeferredMutationReturnedSync(code)].sort();
|
|
248
|
+
}
|
package/package.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "prism-mcp-server",
|
|
3
|
-
"version": "20.21.
|
|
3
|
+
"version": "20.21.5",
|
|
4
4
|
"mcpName": "io.github.dcostenco/prism-coder",
|
|
5
|
-
"description": "Persistent session memory for AI coding agents that never leaves your machine
|
|
5
|
+
"description": "Persistent session memory for AI coding agents that never leaves your machine \u2014 including the on-device model that reasons over it. Restores your prior decisions, open TODOs, and changed files across sessions; adds associative recall of related past work, semantic drift detection, and local inference. Local-first by default. Works with Claude Code, Cursor, and Codex.",
|
|
6
6
|
"module": "index.ts",
|
|
7
7
|
"type": "module",
|
|
8
8
|
"main": "dist/server.js",
|
|
@@ -114,6 +114,7 @@
|
|
|
114
114
|
"stream-json": "^3.6.0",
|
|
115
115
|
"tldts": "^7.0.27",
|
|
116
116
|
"turndown": "^7.2.2",
|
|
117
|
+
"typescript": "^5.9.3",
|
|
117
118
|
"zod": "^4.3.6"
|
|
118
119
|
}
|
|
119
120
|
}
|