pi-tool-repair 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,980 @@
1
+ import { existsSync, readFileSync } from "node:fs";
2
+ import { homedir } from "node:os";
3
+ import { join } from "node:path";
4
+
5
+ export const GRAMMAR_NAMES = [
6
+ "dsml",
7
+ "invoke",
8
+ "qwen",
9
+ "kimi",
10
+ "mistral",
11
+ "llama",
12
+ "glm",
13
+ "granite",
14
+ "minimax-text",
15
+ "olmo",
16
+ ] as const;
17
+
18
+ export type GrammarName = typeof GRAMMAR_NAMES[number];
19
+ export type GrammarRepairMode = "recover" | "strip";
20
+
21
+ export interface GrammarRepairConfig {
22
+ enabled: boolean;
23
+ grammars: GrammarName[];
24
+ mode: GrammarRepairMode;
25
+ requireKnownTool: boolean;
26
+ debug: boolean;
27
+ }
28
+
29
+ export interface ExtensionFileConfig {
30
+ grammarRepair?: Partial<GrammarRepairConfig>;
31
+ }
32
+
33
+ export interface RecoveredToolCall {
34
+ name: string;
35
+ arguments: Record<string, unknown>;
36
+ grammar: GrammarName;
37
+ }
38
+
39
+ interface Candidate extends RecoveredToolCall {
40
+ range: Range;
41
+ }
42
+
43
+ interface Range {
44
+ start: number;
45
+ end: number;
46
+ }
47
+
48
+ interface MinimalTextContent {
49
+ type: "text";
50
+ text: string;
51
+ [key: string]: unknown;
52
+ }
53
+
54
+ interface MinimalThinkingContent {
55
+ type: "thinking";
56
+ thinking: string;
57
+ [key: string]: unknown;
58
+ }
59
+
60
+ interface MinimalToolCallContent {
61
+ type: "toolCall";
62
+ id: string;
63
+ name: string;
64
+ arguments: Record<string, unknown>;
65
+ [key: string]: unknown;
66
+ }
67
+
68
+ type MinimalAssistantContent = MinimalTextContent | MinimalThinkingContent | MinimalToolCallContent | Record<string, unknown>;
69
+
70
+ export interface MinimalAssistantMessage {
71
+ role: "assistant";
72
+ content: MinimalAssistantContent[];
73
+ stopReason?: string;
74
+ diagnostics?: unknown[];
75
+ [key: string]: unknown;
76
+ }
77
+
78
+ export interface GrammarRepairResult {
79
+ changed: boolean;
80
+ recoveredCalls: RecoveredToolCall[];
81
+ strippedRanges: number;
82
+ message: MinimalAssistantMessage;
83
+ }
84
+
85
+ const ALL_GRAMMARS = [...GRAMMAR_NAMES];
86
+
87
+ export const DEFAULT_GRAMMAR_REPAIR_CONFIG: GrammarRepairConfig = {
88
+ enabled: false,
89
+ grammars: ALL_GRAMMARS,
90
+ mode: "recover",
91
+ requireKnownTool: true,
92
+ debug: Boolean(process.env.PI_TOOL_REPAIR_DEBUG),
93
+ };
94
+
95
+ export function loadGrammarRepairConfig(path = defaultConfigPath()): GrammarRepairConfig {
96
+ if (!existsSync(path)) return { ...DEFAULT_GRAMMAR_REPAIR_CONFIG };
97
+
98
+ try {
99
+ const parsed = JSON.parse(readFileSync(path, "utf8")) as ExtensionFileConfig | Partial<GrammarRepairConfig>;
100
+ const raw = isObject(parsed) && "grammarRepair" in parsed
101
+ ? (parsed as ExtensionFileConfig).grammarRepair ?? {}
102
+ : parsed;
103
+ return normalizeGrammarRepairConfig(raw as Partial<GrammarRepairConfig>);
104
+ } catch (error) {
105
+ const message = error instanceof Error ? error.message : String(error);
106
+ process.stderr.write(`[pi-tool-repair] Failed to read grammar repair config at ${path}: ${message}\n`);
107
+ return { ...DEFAULT_GRAMMAR_REPAIR_CONFIG };
108
+ }
109
+ }
110
+
111
+ export function defaultConfigPath(): string {
112
+ return join(homedir(), ".pi", "agent", "extensions", "pi-tool-repair.json");
113
+ }
114
+
115
+ export function normalizeGrammarRepairConfig(raw: Partial<GrammarRepairConfig> = {}): GrammarRepairConfig {
116
+ const grammarSet = new Set<GrammarName>(ALL_GRAMMARS);
117
+ const grammars = Array.isArray(raw.grammars)
118
+ ? raw.grammars.filter((name): name is GrammarName => grammarSet.has(name as GrammarName))
119
+ : ALL_GRAMMARS;
120
+
121
+ return {
122
+ enabled: raw.enabled ?? DEFAULT_GRAMMAR_REPAIR_CONFIG.enabled,
123
+ grammars: grammars.length > 0 ? grammars : ALL_GRAMMARS,
124
+ mode: raw.mode === "strip" ? "strip" : "recover",
125
+ requireKnownTool: raw.requireKnownTool ?? DEFAULT_GRAMMAR_REPAIR_CONFIG.requireKnownTool,
126
+ debug: raw.debug ?? DEFAULT_GRAMMAR_REPAIR_CONFIG.debug,
127
+ };
128
+ }
129
+
130
+ export function repairAssistantMessageGrammarLeaks(
131
+ message: MinimalAssistantMessage,
132
+ config: GrammarRepairConfig,
133
+ knownTools: Set<string> = new Set(),
134
+ ): GrammarRepairResult {
135
+ if (!config.enabled || message.role !== "assistant" || !Array.isArray(message.content)) {
136
+ return { changed: false, recoveredCalls: [], strippedRanges: 0, message };
137
+ }
138
+
139
+ const enabled = new Set(config.grammars);
140
+ const existingToolCalls = message.content.filter(isToolCallContent) as MinimalToolCallContent[];
141
+ const recoveredCalls: RecoveredToolCall[] = [];
142
+ let strippedRanges = 0;
143
+ let changed = false;
144
+
145
+ const nextContent = message.content.map((part) => {
146
+ const text = getPartText(part);
147
+ if (text === undefined) return part;
148
+
149
+ const candidates = selectCandidates(parseToolGrammarCandidates(text, enabled))
150
+ .filter((candidate) => isAllowedTool(candidate.name, config, knownTools));
151
+
152
+ if (candidates.length === 0) return part;
153
+
154
+ strippedRanges += candidates.length;
155
+ changed = true;
156
+ for (const candidate of candidates) {
157
+ recoveredCalls.push({
158
+ name: candidate.name,
159
+ arguments: candidate.arguments,
160
+ grammar: candidate.grammar,
161
+ });
162
+ }
163
+
164
+ const strippedText = removeRanges(text, candidates.map((candidate) => candidate.range));
165
+ return setPartText(part, strippedText);
166
+ });
167
+
168
+ const shouldRecover = config.mode === "recover" && existingToolCalls.length === 0 && recoveredCalls.length > 0;
169
+ if (!changed && !shouldRecover) {
170
+ return { changed: false, recoveredCalls: [], strippedRanges: 0, message };
171
+ }
172
+
173
+ if (shouldRecover) {
174
+ let index = 0;
175
+ for (const call of recoveredCalls) {
176
+ nextContent.push({
177
+ type: "toolCall",
178
+ id: makeRecoveredToolCallId(call.grammar, index++),
179
+ name: call.name,
180
+ arguments: call.arguments,
181
+ });
182
+ }
183
+ }
184
+
185
+ const nextMessage: MinimalAssistantMessage = {
186
+ ...message,
187
+ content: nextContent,
188
+ };
189
+
190
+ if (shouldRecover) {
191
+ nextMessage.stopReason = "toolUse";
192
+ }
193
+
194
+ return {
195
+ changed: changed || shouldRecover,
196
+ recoveredCalls: shouldRecover ? recoveredCalls : [],
197
+ strippedRanges,
198
+ message: nextMessage,
199
+ };
200
+ }
201
+
202
+ export function parseToolGrammarLeaks(text: string, grammars: Iterable<GrammarName> = ALL_GRAMMARS): RecoveredToolCall[] {
203
+ const enabled = new Set(grammars);
204
+ return selectCandidates(parseToolGrammarCandidates(text, enabled)).map((candidate) => ({
205
+ name: candidate.name,
206
+ arguments: candidate.arguments,
207
+ grammar: candidate.grammar,
208
+ }));
209
+ }
210
+
211
+ function parseToolGrammarCandidates(text: string, enabled: Set<GrammarName>): Candidate[] {
212
+ const candidates: Candidate[] = [];
213
+ if (enabled.has("dsml")) candidates.push(...parseDsml(text));
214
+ if (enabled.has("kimi")) candidates.push(...parseKimi(text));
215
+ if (enabled.has("mistral")) {
216
+ candidates.push(...parseMistral(text));
217
+ candidates.push(...parseBareJsonToolCalls(text, "mistral"));
218
+ }
219
+ if (enabled.has("minimax-text")) candidates.push(...parseMiniMaxText01(text));
220
+ if (enabled.has("invoke")) candidates.push(...parseInvokeXml(text));
221
+ if (enabled.has("qwen") || enabled.has("glm") || enabled.has("granite")) {
222
+ candidates.push(...parseToolCallXml(text, enabled));
223
+ }
224
+ if (enabled.has("granite")) candidates.push(...parseBarePythonicToolCalls(text, "granite"));
225
+ if (enabled.has("llama")) {
226
+ candidates.push(...parseLlamaPythonTag(text));
227
+ candidates.push(...parseBareJsonToolCalls(text, "llama"));
228
+ }
229
+ if (enabled.has("olmo")) candidates.push(...parseOlmo(text));
230
+ return candidates.filter((candidate) => candidate.range.end > candidate.range.start);
231
+ }
232
+
233
+ function parseDsml(text: string): Candidate[] {
234
+ const candidates: Candidate[] = [];
235
+ const prefix = "(?:|{1,2}DSML|{1,2}|DSML||\\s*\\|\\s*DSML\\s*\\|\\s*)";
236
+ const outerOpen = new RegExp(`<${prefix}(?:tool_calls|function_calls)>`, "giu");
237
+
238
+ for (const match of text.matchAll(outerOpen)) {
239
+ if (match.index === undefined || isInsideCodeFence(text, match.index)) continue;
240
+ const start = match.index;
241
+ const bodyStart = start + match[0].length;
242
+ const close = findDsmlClose(text, bodyStart, "tool_calls") ?? findDsmlClose(text, bodyStart, "function_calls");
243
+ const end = close ? close.end : findBestUnclosedDsmlEnd(text, bodyStart);
244
+ if (end === undefined) continue;
245
+ const body = text.slice(bodyStart, close ? close.start : end);
246
+ const calls = parseDsmlInvokes(body);
247
+ for (const call of calls) {
248
+ candidates.push({ ...call, grammar: "dsml", range: { start, end } });
249
+ }
250
+ }
251
+
252
+ return candidates;
253
+ }
254
+
255
+ function findDsmlClose(text: string, from: number, outerName: string): Range | undefined {
256
+ const prefix = "(?:|{1,2}DSML|{1,2}|DSML||\\s*\\|\\s*DSML\\s*\\|\\s*)";
257
+ const closeRe = new RegExp(`</${prefix}${outerName}>`, "giu");
258
+ closeRe.lastIndex = from;
259
+ const match = closeRe.exec(text);
260
+ return match && match.index >= from ? { start: match.index, end: match.index + match[0].length } : undefined;
261
+ }
262
+
263
+ function findBestUnclosedDsmlEnd(text: string, from: number): number | undefined {
264
+ const invokeClose = /<\/(?:|{1,2}DSML|{1,2}|DSML||\s*\|\s*DSML\s*\|\s*)invoke>/giu;
265
+ invokeClose.lastIndex = from;
266
+ let end: number | undefined;
267
+ for (;;) {
268
+ const match = invokeClose.exec(text);
269
+ if (!match) break;
270
+ end = match.index + match[0].length;
271
+ }
272
+ return end;
273
+ }
274
+
275
+ function parseDsmlInvokes(body: string): Array<Omit<Candidate, "range" | "grammar">> {
276
+ const calls: Array<Omit<Candidate, "range" | "grammar">> = [];
277
+ const prefix = "(?:|{1,2}DSML|{1,2}|DSML||\\s*\\|\\s*DSML\\s*\\|\\s*)";
278
+ const invokeRe = new RegExp(`<${prefix}invoke\\s+name=["']([^"']+)["']\\s*>`, "giu");
279
+
280
+ for (const match of body.matchAll(invokeRe)) {
281
+ if (match.index === undefined) continue;
282
+ const name = match[1]?.trim();
283
+ if (!name) continue;
284
+ const invokeBodyStart = match.index + match[0].length;
285
+ const close = findPattern(body, new RegExp(`</${prefix}invoke>`, "iu"), invokeBodyStart);
286
+ if (!close) continue;
287
+ const invokeBody = body.slice(invokeBodyStart, close.start);
288
+ calls.push({ name, arguments: parseDsmlArguments(invokeBody) });
289
+ }
290
+
291
+ return calls;
292
+ }
293
+
294
+ function parseDsmlArguments(body: string): Record<string, unknown> {
295
+ const args: Record<string, unknown> = {};
296
+ const prefix = "(?:|{1,2}DSML|{1,2}|DSML||\\s*\\|\\s*DSML\\s*\\|\\s*)";
297
+ const paramRe = new RegExp(
298
+ `<${prefix}parameter\\s+name=["']([^"']+)["'](?:\\s+string=["'](true|false)["'])?\\s*>([\\s\\S]*?)</${prefix}parameter>`,
299
+ "giu",
300
+ );
301
+
302
+ for (const match of body.matchAll(paramRe)) {
303
+ const key = match[1]?.trim();
304
+ if (!key) continue;
305
+ const stringAttr = match[2];
306
+ const rawValue = match[3] ?? "";
307
+ args[key] = stringAttr === "false" ? parseJsonValueOrString(rawValue.trim()) : rawValue;
308
+ }
309
+
310
+ if (Object.keys(args).length > 0) return args;
311
+
312
+ const direct = parseJsonObject(extractFirstBalancedJson(body.trim())?.json ?? body.trim());
313
+ return normalizeArgumentsObject(direct) ?? {};
314
+ }
315
+
316
+ function parseKimi(text: string): Candidate[] {
317
+ const candidates: Candidate[] = [];
318
+ const sectionRe = /<\|tool_calls?_section_begin\|>([\s\S]*?)<\|tool_calls?_section_end\|>/gi;
319
+
320
+ for (const section of text.matchAll(sectionRe)) {
321
+ if (section.index === undefined || isInsideCodeFence(text, section.index)) continue;
322
+ const sectionStart = section.index;
323
+ const body = section[1] ?? "";
324
+ const callRe = /<\|tool_call_begin\|>([^<]*?)<\|tool_call_argument_begin\|>([\s\S]*?)<\|tool_call_end\|>/gi;
325
+ for (const call of body.matchAll(callRe)) {
326
+ const idText = (call[1] ?? "").trim();
327
+ const name = parseKimiToolName(idText);
328
+ if (!name) continue;
329
+ const args = parseJsonObject(call[2]?.trim() ?? "") ?? {};
330
+ candidates.push({
331
+ name,
332
+ arguments: args,
333
+ grammar: "kimi",
334
+ range: { start: sectionStart, end: sectionStart + section[0].length },
335
+ });
336
+ }
337
+ }
338
+
339
+ return candidates;
340
+ }
341
+
342
+ function parseKimiToolName(idText: string): string | undefined {
343
+ const canonical = /^functions\.([A-Za-z_][\w.-]*):\d+$/.exec(idText);
344
+ if (canonical) return canonical[1];
345
+ const relaxed = /^(?:functions\.)?([A-Za-z_][\w.-]*)(?::\d+)?$/.exec(idText);
346
+ if (relaxed && !/^call[_-]?\d+$/i.test(relaxed[1] ?? "")) return relaxed[1];
347
+ return undefined;
348
+ }
349
+
350
+ function parseMistral(text: string): Candidate[] {
351
+ const candidates: Candidate[] = [];
352
+ const marker = "[TOOL_CALLS]";
353
+ let index = 0;
354
+
355
+ while ((index = text.indexOf(marker, index)) !== -1) {
356
+ if (isInsideCodeFence(text, index)) {
357
+ index += marker.length;
358
+ continue;
359
+ }
360
+
361
+ const afterMarker = index + marker.length;
362
+ const rest = text.slice(afterMarker).trimStart();
363
+ const whitespace = text.slice(afterMarker).length - rest.length;
364
+ const jsonStart = afterMarker + whitespace;
365
+
366
+ if (rest.startsWith("[")) {
367
+ const extracted = extractFirstBalancedJson(text.slice(jsonStart));
368
+ if (extracted?.json.startsWith("[")) {
369
+ for (const item of parseJsonArrayObjects(extracted.json)) {
370
+ const call = callFromJsonObject(item);
371
+ if (call) {
372
+ candidates.push({ ...call, grammar: "mistral", range: { start: index, end: jsonStart + extracted.end } });
373
+ }
374
+ }
375
+ index = jsonStart + extracted.end;
376
+ continue;
377
+ }
378
+ }
379
+
380
+ const v11 = /^([A-Za-z_][\w.-]*)\[CALL_ID\]([^\[]*)\[ARGS\]/.exec(rest);
381
+ if (v11) {
382
+ const name = v11[1];
383
+ const argsStart = jsonStart + v11[0].length;
384
+ const extracted = extractFirstBalancedJson(text.slice(argsStart));
385
+ if (extracted) {
386
+ candidates.push({
387
+ name,
388
+ arguments: normalizeArgumentsObject(parseJsonObject(extracted.json)) ?? {},
389
+ grammar: "mistral",
390
+ range: { start: index, end: argsStart + extracted.end },
391
+ });
392
+ index = argsStart + extracted.end;
393
+ continue;
394
+ }
395
+ }
396
+
397
+ index += marker.length;
398
+ }
399
+
400
+ return candidates;
401
+ }
402
+
403
+ function parseMiniMaxText01(text: string): Candidate[] {
404
+ const candidates: Candidate[] = [];
405
+ const re = /<function_call>[\s\S]*?functions\.([A-Za-z_][\w.-]*)\s*\(/gi;
406
+
407
+ for (const match of text.matchAll(re)) {
408
+ if (match.index === undefined || isInsideCodeFence(text, match.index)) continue;
409
+ const name = match[1];
410
+ const openParen = match.index + match[0].length - 1;
411
+ const closeParen = findMatching(text, openParen, "(", ")");
412
+ if (closeParen === undefined) continue;
413
+ const rawArgs = text.slice(openParen + 1, closeParen).trim();
414
+ const args = normalizeArgumentsObject(parseJsonObject(rawArgs)) ?? {};
415
+ const fenceEnd = text.indexOf("```", closeParen);
416
+ const end = fenceEnd === -1 ? closeParen + 1 : fenceEnd + 3;
417
+ candidates.push({ name, arguments: args, grammar: "minimax-text", range: { start: match.index, end } });
418
+ }
419
+
420
+ return candidates;
421
+ }
422
+
423
+ function parseInvokeXml(text: string): Candidate[] {
424
+ const candidates: Candidate[] = [];
425
+ const wrappedRe = /<(?:[A-Za-z][\w.-]*:)?tool_call>([\s\S]*?)<\/(?:[A-Za-z][\w.-]*:)?tool_call>/gi;
426
+
427
+ for (const wrapper of text.matchAll(wrappedRe)) {
428
+ if (wrapper.index === undefined || isInsideCodeFence(text, wrapper.index)) continue;
429
+ const calls = parseInvokeBody(wrapper[1] ?? "");
430
+ for (const call of calls) {
431
+ candidates.push({ ...call, grammar: "invoke", range: { start: wrapper.index, end: wrapper.index + wrapper[0].length } });
432
+ }
433
+ }
434
+
435
+ const standaloneRe = /<invoke\s+name=["']([^"']+)["']\s*>[\s\S]*?<\/invoke>/gi;
436
+ for (const match of text.matchAll(standaloneRe)) {
437
+ if (match.index === undefined || isInsideCodeFence(text, match.index)) continue;
438
+ const calls = parseInvokeBody(match[0]);
439
+ for (const call of calls) {
440
+ candidates.push({ ...call, grammar: "invoke", range: { start: match.index, end: match.index + match[0].length } });
441
+ }
442
+ }
443
+
444
+ candidates.push(...parseMalformedMiniMaxInvoke(text));
445
+ return candidates;
446
+ }
447
+
448
+ function parseInvokeBody(body: string): Array<Omit<Candidate, "range" | "grammar">> {
449
+ const calls: Array<Omit<Candidate, "range" | "grammar">> = [];
450
+ const invokeRe = /<invoke\s+name=["']([^"']+)["']\s*>([\s\S]*?)<\/invoke>/gi;
451
+
452
+ for (const match of body.matchAll(invokeRe)) {
453
+ const name = match[1]?.trim();
454
+ if (!name) continue;
455
+ calls.push({ name, arguments: parseInvokeArguments(match[2] ?? "") });
456
+ }
457
+
458
+ return calls;
459
+ }
460
+
461
+ function parseInvokeArguments(body: string): Record<string, unknown> {
462
+ const args: Record<string, unknown> = {};
463
+ const paramRe = /<parameter\s+name=["']([^"']+)["'](?:\s+string=["'](true|false)["'])?\s*>([\s\S]*?)<\/parameter>/gi;
464
+
465
+ for (const match of body.matchAll(paramRe)) {
466
+ const key = match[1]?.trim();
467
+ if (!key) continue;
468
+ const raw = match[3] ?? "";
469
+ args[key] = match[2] === "false" ? parseJsonValueOrString(raw.trim()) : maybeParseJsonValue(raw.trim());
470
+ }
471
+
472
+ if (Object.keys(args).length > 0) return args;
473
+ return normalizeArgumentsObject(parseJsonObject(body.trim())) ?? {};
474
+ }
475
+
476
+ function parseMalformedMiniMaxInvoke(text: string): Candidate[] {
477
+ const candidates: Candidate[] = [];
478
+ const re = /(?:^|\n)(\s*)invoke\s+name=["']([^"']+)["']\s*>([\s\S]*?)(?:\n\s*(?:\/invoke|invoke)>|$)/gi;
479
+
480
+ for (const match of text.matchAll(re)) {
481
+ if (match.index === undefined || isInsideCodeFence(text, match.index)) continue;
482
+ const start = match.index + (match[0].startsWith("\n") ? 1 : 0);
483
+ const name = match[2]?.trim();
484
+ if (!name) continue;
485
+ const args: Record<string, unknown> = {};
486
+ const paramRe = /parameter\s+name=["']([^"']+)["']\s*>([\s\S]*?)\s*parameter>/gi;
487
+ for (const param of (match[3] ?? "").matchAll(paramRe)) {
488
+ const key = param[1]?.trim();
489
+ if (key) args[key] = maybeParseJsonValue((param[2] ?? "").trim());
490
+ }
491
+ candidates.push({ name, arguments: args, grammar: "invoke", range: { start, end: match.index + match[0].length } });
492
+ }
493
+
494
+ return candidates;
495
+ }
496
+
497
+ function parseToolCallXml(text: string, enabled: Set<GrammarName>): Candidate[] {
498
+ const candidates: Candidate[] = [];
499
+ const wrapperRe = /<(tool_call|tools)>[\s\S]*?<\/\1>/gi;
500
+
501
+ for (const match of text.matchAll(wrapperRe)) {
502
+ if (match.index === undefined || isInsideCodeFence(text, match.index)) continue;
503
+ const tag = match[1]?.toLowerCase();
504
+ const openTagEnd = match[0].indexOf(">") + 1;
505
+ const body = match[0].slice(openTagEnd, match[0].length - (`</${tag}>`).length);
506
+
507
+ if (enabled.has("granite") || enabled.has("qwen")) {
508
+ const jsonGrammar = tag === "tools" || !enabled.has("granite") ? "qwen" : "granite";
509
+ const jsonCalls = parseToolCallJsonBody(body, jsonGrammar);
510
+ for (const call of jsonCalls) {
511
+ candidates.push({ ...call, range: { start: match.index, end: match.index + match[0].length } });
512
+ }
513
+ }
514
+
515
+ if (enabled.has("glm")) {
516
+ const glmCall = parseGlmToolCallBody(body);
517
+ if (glmCall) candidates.push({ ...glmCall, range: { start: match.index, end: match.index + match[0].length } });
518
+ }
519
+
520
+ if (enabled.has("qwen")) {
521
+ const qwenCalls = parseQwenFunctionBody(body);
522
+ for (const call of qwenCalls) {
523
+ candidates.push({ ...call, range: { start: match.index, end: match.index + match[0].length } });
524
+ }
525
+ }
526
+ }
527
+
528
+ if (enabled.has("qwen")) {
529
+ const bareFunctionRe = /<function=([A-Za-z_][\w.-]*)>[\s\S]*?<\/function>/gi;
530
+ for (const match of text.matchAll(bareFunctionRe)) {
531
+ if (match.index === undefined || isInsideCodeFence(text, match.index)) continue;
532
+ const calls = parseQwenFunctionBody(match[0]);
533
+ for (const call of calls) {
534
+ candidates.push({ ...call, range: { start: match.index, end: match.index + match[0].length } });
535
+ }
536
+ }
537
+ }
538
+
539
+ return candidates;
540
+ }
541
+
542
+ function parseToolCallJsonBody(body: string, grammar: GrammarName): Array<Omit<Candidate, "range">> {
543
+ const calls: Array<Omit<Candidate, "range">> = [];
544
+ const trimmed = unwrapMarkdownFence(body.trim());
545
+ const json = extractFirstBalancedJson(trimmed)?.json ?? trimmed;
546
+ const parsed = parseJsonValue(json);
547
+
548
+ if (Array.isArray(parsed)) {
549
+ for (const item of parsed) {
550
+ if (isObject(item)) {
551
+ const call = callFromJsonObject(item);
552
+ if (call) calls.push({ ...call, grammar });
553
+ }
554
+ }
555
+ return calls;
556
+ }
557
+
558
+ if (isObject(parsed)) {
559
+ const call = callFromJsonObject(parsed);
560
+ if (call) calls.push({ ...call, grammar });
561
+ }
562
+
563
+ return calls;
564
+ }
565
+
566
+ function parseQwenFunctionBody(body: string): Array<Omit<Candidate, "range">> {
567
+ const calls: Array<Omit<Candidate, "range">> = [];
568
+ const functionRe = /<function=([A-Za-z_][\w.-]*)>\s*([\s\S]*?)<\/function>/gi;
569
+
570
+ for (const match of body.matchAll(functionRe)) {
571
+ const name = match[1]?.trim();
572
+ if (!name) continue;
573
+ const args: Record<string, unknown> = {};
574
+ const paramRe = /<parameter=([^>]+)>([\s\S]*?)<\/parameter>/gi;
575
+ for (const param of (match[2] ?? "").matchAll(paramRe)) {
576
+ const key = param[1]?.trim();
577
+ if (!key) continue;
578
+ args[key] = maybeParseJsonValue((param[2] ?? "").trim());
579
+ }
580
+ calls.push({ name, arguments: args, grammar: "qwen" });
581
+ }
582
+
583
+ return calls;
584
+ }
585
+
586
+ function parseGlmToolCallBody(body: string): Omit<Candidate, "range"> | undefined {
587
+ const keyRe = /<arg_key>([\s\S]*?)<\/arg_key>/gi;
588
+ const valueRe = /<arg_value>([\s\S]*?)<\/arg_value>/gi;
589
+ const keys = [...body.matchAll(keyRe)].map((m) => (m[1] ?? "").trim()).filter(Boolean);
590
+ const values = [...body.matchAll(valueRe)].map((m) => (m[1] ?? "").trim());
591
+ const nameEnd = keys.length === 0 ? body.length : body.search(/<arg_key>/i);
592
+ const name = body.slice(0, nameEnd).trim().split(/\s+/)[0];
593
+ if (!name || !/^[A-Za-z_][\w.-]*$/.test(name)) return undefined;
594
+
595
+ const args: Record<string, unknown> = {};
596
+ keys.forEach((key, i) => {
597
+ args[key] = maybeParseJsonValue(values[i] ?? "");
598
+ });
599
+ return { name, arguments: args, grammar: "glm" };
600
+ }
601
+
602
+ function parseLlamaPythonTag(text: string): Candidate[] {
603
+ const candidates: Candidate[] = [];
604
+ const marker = "<|python_tag|>";
605
+ let index = 0;
606
+
607
+ while ((index = text.indexOf(marker, index)) !== -1) {
608
+ if (isInsideCodeFence(text, index)) {
609
+ index += marker.length;
610
+ continue;
611
+ }
612
+ const bodyStart = index + marker.length;
613
+ const rest = text.slice(bodyStart).trimStart();
614
+ const whitespace = text.slice(bodyStart).length - rest.length;
615
+ const payloadStart = bodyStart + whitespace;
616
+
617
+ const extracted = extractFirstBalancedJson(text.slice(payloadStart));
618
+ if (extracted) {
619
+ const parsed = parseJsonValue(extracted.json);
620
+ for (const call of callsFromJsonValue(parsed)) {
621
+ candidates.push({ ...call, grammar: "llama", range: { start: index, end: payloadStart + extracted.end } });
622
+ }
623
+ index = payloadStart + extracted.end;
624
+ continue;
625
+ }
626
+
627
+ const lineEnd = findLineEnd(text, payloadStart);
628
+ for (const call of parsePythonicCalls(text.slice(payloadStart, lineEnd))) {
629
+ candidates.push({ ...call, grammar: "llama", range: { start: index, end: lineEnd } });
630
+ }
631
+ index = lineEnd;
632
+ }
633
+
634
+ return candidates;
635
+ }
636
+
637
+ function parseBareJsonToolCalls(text: string, grammar: GrammarName): Candidate[] {
638
+ const candidates: Candidate[] = [];
639
+ const objectRe = /\{\s*"(?:name|function_name|function)"/g;
640
+
641
+ for (const match of text.matchAll(objectRe)) {
642
+ if (match.index === undefined || isInsideCodeFence(text, match.index)) continue;
643
+ const extracted = extractFirstBalancedJson(text.slice(match.index));
644
+ if (!extracted) continue;
645
+ for (const call of callsFromJsonValue(parseJsonValue(extracted.json))) {
646
+ candidates.push({ ...call, grammar, range: { start: match.index, end: match.index + extracted.end } });
647
+ }
648
+ }
649
+
650
+ const arrayRe = /\[\s*\{\s*"(?:name|function_name|function)"/g;
651
+ for (const match of text.matchAll(arrayRe)) {
652
+ if (match.index === undefined || isInsideCodeFence(text, match.index)) continue;
653
+ const extracted = extractFirstBalancedJson(text.slice(match.index));
654
+ if (!extracted) continue;
655
+ for (const call of callsFromJsonValue(parseJsonValue(extracted.json))) {
656
+ candidates.push({ ...call, grammar, range: { start: match.index, end: match.index + extracted.end } });
657
+ }
658
+ }
659
+
660
+ return candidates;
661
+ }
662
+
663
+ function parseBarePythonicToolCalls(text: string, grammar: GrammarName): Candidate[] {
664
+ const candidates: Candidate[] = [];
665
+ const lineRe = /(?:^|\n)\s*([A-Za-z_][\w.-]*)\s*\(/g;
666
+
667
+ for (const match of text.matchAll(lineRe)) {
668
+ if (match.index === undefined || isInsideCodeFence(text, match.index)) continue;
669
+ const start = match.index + (match[0].startsWith("\n") ? 1 : 0);
670
+ const openParen = match.index + match[0].lastIndexOf("(");
671
+ const closeParen = findMatching(text, openParen, "(", ")");
672
+ if (closeParen === undefined) continue;
673
+ const lineStart = text.lastIndexOf("\n", openParen) + 1;
674
+ if (text.slice(lineStart, match.index).trim() !== "") continue;
675
+ const [call] = parsePythonicCalls(text.slice(start, closeParen + 1));
676
+ if (call) candidates.push({ ...call, grammar, range: { start, end: closeParen + 1 } });
677
+ }
678
+
679
+ return candidates;
680
+ }
681
+
682
+ function parseOlmo(text: string): Candidate[] {
683
+ const candidates: Candidate[] = [];
684
+ const re = /<function_calls>([\s\S]*?)<\/function_calls>/gi;
685
+ for (const match of text.matchAll(re)) {
686
+ if (match.index === undefined || isInsideCodeFence(text, match.index)) continue;
687
+ for (const call of parsePythonicCalls(match[1] ?? "")) {
688
+ candidates.push({ ...call, grammar: "olmo", range: { start: match.index, end: match.index + match[0].length } });
689
+ }
690
+ }
691
+ return candidates;
692
+ }
693
+
694
+ function parsePythonicCalls(text: string): Array<Omit<Candidate, "range" | "grammar">> {
695
+ const calls: Array<Omit<Candidate, "range" | "grammar">> = [];
696
+ const re = /(?:^|\n)\s*([A-Za-z_][\w.-]*)\s*\(/g;
697
+ for (const match of text.matchAll(re)) {
698
+ if (match.index === undefined) continue;
699
+ const name = match[1];
700
+ const openParen = match.index + match[0].lastIndexOf("(");
701
+ const closeParen = findMatching(text, openParen, "(", ")");
702
+ if (closeParen === undefined) continue;
703
+ const argsText = text.slice(openParen + 1, closeParen);
704
+ calls.push({ name, arguments: parseKeywordArguments(argsText) });
705
+ }
706
+ return calls;
707
+ }
708
+
709
+ function parseKeywordArguments(text: string): Record<string, unknown> {
710
+ const args: Record<string, unknown> = {};
711
+ for (const part of splitTopLevel(text, ",")) {
712
+ const eq = findTopLevelChar(part, "=");
713
+ if (eq === -1) continue;
714
+ const key = part.slice(0, eq).trim();
715
+ if (!/^[A-Za-z_][\w.-]*$/.test(key)) continue;
716
+ args[key] = parsePythonishValue(part.slice(eq + 1).trim());
717
+ }
718
+ return args;
719
+ }
720
+
721
+ function parsePythonishValue(raw: string): unknown {
722
+ const trimmed = raw.trim();
723
+ if (trimmed === "True") return true;
724
+ if (trimmed === "False") return false;
725
+ if (trimmed === "None") return null;
726
+ if (/^[-+]?\d+(?:\.\d+)?$/.test(trimmed)) return Number(trimmed);
727
+ if ((trimmed.startsWith("'") && trimmed.endsWith("'")) || (trimmed.startsWith('"') && trimmed.endsWith('"'))) {
728
+ return trimmed.slice(1, -1).replace(/\\(['"\\])/g, "$1");
729
+ }
730
+ return parseJsonValueOrString(trimmed.replace(/\bTrue\b/g, "true").replace(/\bFalse\b/g, "false").replace(/\bNone\b/g, "null"));
731
+ }
732
+
733
+ function callsFromJsonValue(value: unknown): Array<Omit<Candidate, "range" | "grammar">> {
734
+ if (Array.isArray(value)) {
735
+ return value.flatMap((item) => isObject(item) ? callsFromJsonValue(item) : []);
736
+ }
737
+ if (!isObject(value)) return [];
738
+ const call = callFromJsonObject(value);
739
+ return call ? [call] : [];
740
+ }
741
+
742
+ function callFromJsonObject(value: Record<string, unknown>): Omit<Candidate, "range" | "grammar"> | undefined {
743
+ const name = value.name ?? value.function_name ?? (isObject(value.function) ? value.function.name : undefined);
744
+ if (typeof name !== "string" || !name.trim()) return undefined;
745
+
746
+ let args: unknown = value.arguments ?? value.args ?? value.parameters;
747
+ if (args === undefined && isObject(value.function)) args = value.function.arguments;
748
+ const normalized = normalizeArgumentsObject(args) ?? {};
749
+ return { name: name.trim(), arguments: normalized };
750
+ }
751
+
752
+ function normalizeArgumentsObject(value: unknown): Record<string, unknown> | undefined {
753
+ if (typeof value === "string") {
754
+ return normalizeArgumentsObject(parseJsonValue(value));
755
+ }
756
+ if (isObject(value)) {
757
+ const nested = value.arguments;
758
+ if (typeof nested === "string" || isObject(nested)) {
759
+ const unwrapped = normalizeArgumentsObject(nested);
760
+ if (unwrapped) return unwrapped;
761
+ }
762
+ return value;
763
+ }
764
+ return undefined;
765
+ }
766
+
767
+ function parseJsonArrayObjects(json: string): Record<string, unknown>[] {
768
+ const parsed = parseJsonValue(json);
769
+ return Array.isArray(parsed) ? parsed.filter(isObject) : [];
770
+ }
771
+
772
+ function parseJsonObject(json: string): Record<string, unknown> | undefined {
773
+ const parsed = parseJsonValue(json);
774
+ return isObject(parsed) ? parsed : undefined;
775
+ }
776
+
777
+ function parseJsonValue(json: string): unknown {
778
+ try {
779
+ return JSON.parse(json);
780
+ } catch {
781
+ return undefined;
782
+ }
783
+ }
784
+
785
+ function parseJsonValueOrString(value: string): unknown {
786
+ const parsed = parseJsonValue(value);
787
+ return parsed === undefined ? value : parsed;
788
+ }
789
+
790
+ function maybeParseJsonValue(value: string): unknown {
791
+ if (value === "") return "";
792
+ if (/^(?:true|false|null|-?\d|[\[{]|\")/.test(value)) {
793
+ return parseJsonValueOrString(value);
794
+ }
795
+ return value;
796
+ }
797
+
798
+ function selectCandidates(candidates: Candidate[]): Candidate[] {
799
+ const selected: Candidate[] = [];
800
+ const sorted = [...candidates].sort((a, b) => {
801
+ if (a.range.start !== b.range.start) return a.range.start - b.range.start;
802
+ return (b.range.end - b.range.start) - (a.range.end - a.range.start);
803
+ });
804
+
805
+ for (const candidate of sorted) {
806
+ const duplicate = selected.some((existing) => {
807
+ const sameRange = existing.range.start === candidate.range.start && existing.range.end === candidate.range.end;
808
+ return !sameRange && rangesOverlap(existing.range, candidate.range);
809
+ });
810
+ if (!duplicate) selected.push(candidate);
811
+ }
812
+
813
+ return selected;
814
+ }
815
+
816
+ function rangesOverlap(a: Range, b: Range): boolean {
817
+ return a.start < b.end && b.start < a.end;
818
+ }
819
+
820
+ function isAllowedTool(candidateName: string, config: GrammarRepairConfig, knownTools: Set<string>): boolean {
821
+ if (!config.requireKnownTool) return true;
822
+ return knownTools.size > 0 && knownTools.has(candidateName);
823
+ }
824
+
825
+ function removeRanges(text: string, ranges: Range[]): string {
826
+ const sorted = [...ranges].sort((a, b) => a.start - b.start);
827
+ let result = "";
828
+ let cursor = 0;
829
+ for (const range of sorted) {
830
+ result += text.slice(cursor, range.start);
831
+ cursor = Math.max(cursor, range.end);
832
+ }
833
+ result += text.slice(cursor);
834
+ return result.replace(/[ \t]+\n/g, "\n").replace(/\n{3,}/g, "\n\n").trim();
835
+ }
836
+
837
+ function getPartText(part: MinimalAssistantContent): string | undefined {
838
+ if (isObject(part) && part.type === "text" && typeof part.text === "string") return part.text;
839
+ if (isObject(part) && part.type === "thinking" && typeof part.thinking === "string") return part.thinking;
840
+ return undefined;
841
+ }
842
+
843
+ function setPartText(part: MinimalAssistantContent, text: string): MinimalAssistantContent {
844
+ if (isObject(part) && part.type === "text") return { ...part, text };
845
+ if (isObject(part) && part.type === "thinking") return { ...part, thinking: text };
846
+ return part;
847
+ }
848
+
849
+ function isToolCallContent(part: MinimalAssistantContent): part is MinimalToolCallContent {
850
+ return isObject(part) && part.type === "toolCall" && typeof part.name === "string";
851
+ }
852
+
853
+ function makeRecoveredToolCallId(grammar: GrammarName, index: number): string {
854
+ return `tool_repair_${grammar.replace(/[^a-z0-9]/gi, "_")}_${Date.now().toString(36)}_${index}`;
855
+ }
856
+
857
+ function findPattern(text: string, pattern: RegExp, from: number): Range | undefined {
858
+ pattern.lastIndex = 0;
859
+ const chunk = text.slice(from);
860
+ const match = pattern.exec(chunk);
861
+ return match ? { start: from + match.index, end: from + match.index + match[0].length } : undefined;
862
+ }
863
+
864
+ function extractFirstBalancedJson(text: string): { json: string; start: number; end: number } | undefined {
865
+ const start = text.search(/[\[{]/);
866
+ if (start === -1) return undefined;
867
+ const opener = text[start];
868
+ const closer = opener === "{" ? "}" : "]";
869
+ const end = findMatching(text, start, opener, closer);
870
+ if (end === undefined) return undefined;
871
+ return { json: text.slice(start, end + 1), start, end: end + 1 };
872
+ }
873
+
874
+ function findMatching(text: string, openIndex: number, opener: string, closer: string): number | undefined {
875
+ let depth = 0;
876
+ let quote: string | undefined;
877
+ let escaped = false;
878
+
879
+ for (let i = openIndex; i < text.length; i++) {
880
+ const ch = text[i];
881
+ if (quote) {
882
+ if (escaped) {
883
+ escaped = false;
884
+ } else if (ch === "\\") {
885
+ escaped = true;
886
+ } else if (ch === quote) {
887
+ quote = undefined;
888
+ }
889
+ continue;
890
+ }
891
+
892
+ if (ch === '"' || ch === "'") {
893
+ quote = ch;
894
+ continue;
895
+ }
896
+ if (ch === opener) depth++;
897
+ if (ch === closer) {
898
+ depth--;
899
+ if (depth === 0) return i;
900
+ }
901
+ }
902
+
903
+ return undefined;
904
+ }
905
+
906
+ function splitTopLevel(text: string, delimiter: string): string[] {
907
+ const parts: string[] = [];
908
+ let start = 0;
909
+ let depth = 0;
910
+ let quote: string | undefined;
911
+ let escaped = false;
912
+
913
+ for (let i = 0; i < text.length; i++) {
914
+ const ch = text[i];
915
+ if (quote) {
916
+ if (escaped) escaped = false;
917
+ else if (ch === "\\") escaped = true;
918
+ else if (ch === quote) quote = undefined;
919
+ continue;
920
+ }
921
+ if (ch === '"' || ch === "'") {
922
+ quote = ch;
923
+ continue;
924
+ }
925
+ if ("([{".includes(ch)) depth++;
926
+ if (")]}".includes(ch)) depth--;
927
+ if (ch === delimiter && depth === 0) {
928
+ parts.push(text.slice(start, i));
929
+ start = i + 1;
930
+ }
931
+ }
932
+
933
+ parts.push(text.slice(start));
934
+ return parts.map((part) => part.trim()).filter(Boolean);
935
+ }
936
+
937
+ function findTopLevelChar(text: string, target: string): number {
938
+ let depth = 0;
939
+ let quote: string | undefined;
940
+ let escaped = false;
941
+
942
+ for (let i = 0; i < text.length; i++) {
943
+ const ch = text[i];
944
+ if (quote) {
945
+ if (escaped) escaped = false;
946
+ else if (ch === "\\") escaped = true;
947
+ else if (ch === quote) quote = undefined;
948
+ continue;
949
+ }
950
+ if (ch === '"' || ch === "'") {
951
+ quote = ch;
952
+ continue;
953
+ }
954
+ if ("([{".includes(ch)) depth++;
955
+ if (")]}".includes(ch)) depth--;
956
+ if (ch === target && depth === 0) return i;
957
+ }
958
+
959
+ return -1;
960
+ }
961
+
962
+ function findLineEnd(text: string, start: number): number {
963
+ const newline = text.indexOf("\n", start);
964
+ return newline === -1 ? text.length : newline;
965
+ }
966
+
967
+ function isInsideCodeFence(text: string, index: number): boolean {
968
+ const before = text.slice(0, index);
969
+ const fences = before.match(/```/g);
970
+ return Boolean(fences && fences.length % 2 === 1);
971
+ }
972
+
973
+ function unwrapMarkdownFence(text: string): string {
974
+ const match = /^```\w*\s*([\s\S]*?)\s*```$/.exec(text);
975
+ return match ? match[1] ?? "" : text;
976
+ }
977
+
978
+ function isObject(value: unknown): value is Record<string, unknown> {
979
+ return value !== null && typeof value === "object" && !Array.isArray(value);
980
+ }