token-goat 2.9.12 → 2.9.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +20 -1
- package/dist/token-goat-chunk-2ESBO4IN.mjs +209 -0
- package/dist/token-goat-chunk-3BTK54F3.mjs +1733 -0
- package/dist/{token-goat-chunk-6NIGPPN6.mjs → token-goat-chunk-3XQPEJMV.mjs} +1 -1
- package/dist/token-goat-chunk-4SDX3QP3.mjs +122 -0
- package/dist/token-goat-chunk-7OGKZ7AP.mjs +142 -0
- package/dist/token-goat-chunk-AH6QILZM.mjs +26 -0
- package/dist/token-goat-chunk-AMYCQJX4.mjs +1548 -0
- package/dist/{token-goat-chunk-OTW7LC4Y.mjs → token-goat-chunk-ASVVF4JV.mjs} +13 -10
- package/dist/{token-goat-chunk-GCXX67HM.mjs → token-goat-chunk-DAMXYVIW.mjs} +12477 -12156
- package/dist/token-goat-chunk-DBNY4RLN.mjs +308 -0
- package/dist/token-goat-chunk-EG3663UT.mjs +226 -0
- package/dist/token-goat-chunk-EZNVAIR3.mjs +21 -0
- package/dist/token-goat-chunk-GMOUBOX4.mjs +386 -0
- package/dist/{token-goat-chunk-6F5TLJC7.mjs → token-goat-chunk-HF6H7RNK.mjs} +1 -1
- package/dist/{token-goat-chunk-GM6QZWCF.mjs → token-goat-chunk-IT6O3PNN.mjs} +2660 -6877
- package/dist/token-goat-chunk-NEI4NC54.mjs +424 -0
- package/dist/token-goat-chunk-O5WAMISC.mjs +167 -0
- package/dist/{token-goat-chunk-LHLQFGWQ.mjs → token-goat-chunk-OMNQUUIT.mjs} +9 -27
- package/dist/{token-goat-chunk-SF2CKCFQ.mjs → token-goat-chunk-PGGDW7DZ.mjs} +30 -13
- package/dist/token-goat-chunk-POBYR64E.mjs +1632 -0
- package/dist/token-goat-chunk-PRJVGIC5.mjs +58 -0
- package/dist/{token-goat-chunk-4CK445AW.mjs → token-goat-chunk-RUDOKYPJ.mjs} +10074 -9765
- package/dist/token-goat-chunk-SAQ5PG4L.mjs +2634 -0
- package/dist/token-goat-chunk-T2IWWTHB.mjs +3155 -0
- package/dist/{token-goat-chunk-TOGYS5A7.mjs → token-goat-chunk-VZYD4OZB.mjs} +1469 -5752
- package/dist/token-goat-chunk-XEH6KBWW.mjs +23 -0
- package/dist/token-goat-chunk-XQF5J25J.mjs +33 -0
- package/dist/token-goat-chunk-XTQAOTSO.mjs +89 -0
- package/dist/{token-goat-chunk-THLHC6QJ.mjs → token-goat-chunk-XVZ4MNQC.mjs} +614 -603
- package/dist/token-goat-chunk-Y4AFKTHK.mjs +22 -0
- package/dist/token-goat-chunk-YHGTGG6K.mjs +396 -0
- package/dist/{token-goat-chunk-SBYRP3X4.mjs → token-goat-chunk-YTQHZJXW.mjs} +61 -27
- package/dist/token-goat-chunk-Z6UXPYJA.mjs +62 -0
- package/dist/token-goat-chunk-ZD4EM4LR.mjs +30 -0
- package/dist/token-goat-chunk-ZFM4PWXL.mjs +585 -0
- package/dist/{token-goat-chunk-A6QTLAWO.mjs → token-goat-chunk-ZYNNQ36L.mjs} +128 -328
- package/dist/token-goat-hook.mjs +12 -6
- package/dist/token-goat.core.mjs +22 -7
- package/docs/cli.md +12 -8
- package/docs/security.md +1 -1
- package/package.json +4 -2
- package/dist/token-goat-chunk-HKFOH6JH.mjs +0 -28
- package/dist/token-goat-chunk-LOCOX2ML.mjs +0 -3564
- package/dist/token-goat-chunk-VRWX6QYW.mjs +0 -24
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
import { createRequire as __cjsRequire } from 'node:module';
|
|
2
|
+
const require = __cjsRequire(import.meta.url);
|
|
3
|
+
import {
|
|
4
|
+
init_define_import_meta_env
|
|
5
|
+
} from "./token-goat-chunk-A37V4PBF.mjs";
|
|
6
|
+
|
|
7
|
+
// src/document_refusal.ts
|
|
8
|
+
init_define_import_meta_env();
|
|
9
|
+
var MAX_DOCUMENT_WORK_MILLIS = 6e4;
|
|
10
|
+
var DocumentRefusedError = class extends Error {
|
|
11
|
+
transient;
|
|
12
|
+
constructor(message, name, transient = false) {
|
|
13
|
+
super(message);
|
|
14
|
+
this.name = name;
|
|
15
|
+
this.transient = transient;
|
|
16
|
+
}
|
|
17
|
+
};
|
|
18
|
+
|
|
19
|
+
export {
|
|
20
|
+
MAX_DOCUMENT_WORK_MILLIS,
|
|
21
|
+
DocumentRefusedError
|
|
22
|
+
};
|
|
@@ -0,0 +1,396 @@
|
|
|
1
|
+
import { createRequire as __cjsRequire } from 'node:module';
|
|
2
|
+
const require = __cjsRequire(import.meta.url);
|
|
3
|
+
import {
|
|
4
|
+
MAX_ZIP_INPUT_BYTES,
|
|
5
|
+
MAX_ZIP_OUTPUT_BYTES,
|
|
6
|
+
ZipInputTooLargeError,
|
|
7
|
+
ZipOutputTooLargeError,
|
|
8
|
+
unzipBounded
|
|
9
|
+
} from "./token-goat-chunk-XTQAOTSO.mjs";
|
|
10
|
+
import {
|
|
11
|
+
createLazyModuleLoader
|
|
12
|
+
} from "./token-goat-chunk-AH6QILZM.mjs";
|
|
13
|
+
import {
|
|
14
|
+
DocumentRefusedError,
|
|
15
|
+
MAX_DOCUMENT_WORK_MILLIS
|
|
16
|
+
} from "./token-goat-chunk-Y4AFKTHK.mjs";
|
|
17
|
+
import {
|
|
18
|
+
pushAll
|
|
19
|
+
} from "./token-goat-chunk-AMYCQJX4.mjs";
|
|
20
|
+
import {
|
|
21
|
+
init_define_import_meta_env
|
|
22
|
+
} from "./token-goat-chunk-A37V4PBF.mjs";
|
|
23
|
+
|
|
24
|
+
// src/ooxml_extract.ts
|
|
25
|
+
init_define_import_meta_env();
|
|
26
|
+
import * as fs from "node:fs";
|
|
27
|
+
|
|
28
|
+
// src/xml_parser.ts
|
|
29
|
+
init_define_import_meta_env();
|
|
30
|
+
var MAX_XML_DEPTH = 512;
|
|
31
|
+
var XmlTooDeepError = class extends DocumentRefusedError {
|
|
32
|
+
constructor(message) {
|
|
33
|
+
super(message, "XmlTooDeepError");
|
|
34
|
+
}
|
|
35
|
+
};
|
|
36
|
+
var NAMED_ENTITIES = {
|
|
37
|
+
lt: "<",
|
|
38
|
+
gt: ">",
|
|
39
|
+
amp: "&",
|
|
40
|
+
quot: '"',
|
|
41
|
+
apos: "'"
|
|
42
|
+
};
|
|
43
|
+
function decodeXmlEntities(text) {
|
|
44
|
+
if (!text.includes("&")) return text;
|
|
45
|
+
return text.replace(/&(#[xX][0-9a-fA-F]+|#[0-9]+|[a-zA-Z_][\w.:-]*);/g, (whole, body) => {
|
|
46
|
+
if (body.charCodeAt(0) === 35) {
|
|
47
|
+
const isHex = body.charCodeAt(1) === 120 || body.charCodeAt(1) === 88;
|
|
48
|
+
const digits = isHex ? body.slice(2) : body.slice(1);
|
|
49
|
+
const code = parseInt(digits, isHex ? 16 : 10);
|
|
50
|
+
if (!Number.isFinite(code) || code < 0 || code > 1114111) return whole;
|
|
51
|
+
if (code >= 55296 && code <= 57343) return whole;
|
|
52
|
+
return String.fromCodePoint(code);
|
|
53
|
+
}
|
|
54
|
+
const named = NAMED_ENTITIES[body];
|
|
55
|
+
return named ?? whole;
|
|
56
|
+
});
|
|
57
|
+
}
|
|
58
|
+
function newFrame(name) {
|
|
59
|
+
return { name, children: /* @__PURE__ */ new Map(), text: [], attrs: [] };
|
|
60
|
+
}
|
|
61
|
+
function finishFrame(frame) {
|
|
62
|
+
const text = frame.text.join("");
|
|
63
|
+
if (frame.children.size === 0 && frame.attrs.length === 0) return text;
|
|
64
|
+
const obj = {};
|
|
65
|
+
for (const [key, value] of frame.children) obj[key] = value;
|
|
66
|
+
if (text.length > 0) obj["#text"] = text;
|
|
67
|
+
for (const [key, value] of frame.attrs) obj[`@_${key}`] = value;
|
|
68
|
+
return obj;
|
|
69
|
+
}
|
|
70
|
+
function addChild(parent, name, value) {
|
|
71
|
+
const existing = parent.children.get(name);
|
|
72
|
+
if (existing === void 0 && !parent.children.has(name)) {
|
|
73
|
+
parent.children.set(name, value);
|
|
74
|
+
return;
|
|
75
|
+
}
|
|
76
|
+
if (Array.isArray(existing)) {
|
|
77
|
+
existing.push(value);
|
|
78
|
+
return;
|
|
79
|
+
}
|
|
80
|
+
parent.children.set(name, [existing, value]);
|
|
81
|
+
}
|
|
82
|
+
var WHITESPACE = /* @__PURE__ */ new Set([" ", " ", "\n", "\r"]);
|
|
83
|
+
function findTagEnd(src, from) {
|
|
84
|
+
let quote = "";
|
|
85
|
+
for (let i = from; i < src.length; i++) {
|
|
86
|
+
const ch = src[i];
|
|
87
|
+
if (quote !== "") {
|
|
88
|
+
if (ch === quote) quote = "";
|
|
89
|
+
continue;
|
|
90
|
+
}
|
|
91
|
+
if (ch === '"' || ch === "'") {
|
|
92
|
+
quote = ch;
|
|
93
|
+
continue;
|
|
94
|
+
}
|
|
95
|
+
if (ch === ">") return i;
|
|
96
|
+
}
|
|
97
|
+
return -1;
|
|
98
|
+
}
|
|
99
|
+
function skipDeclaration(src, from) {
|
|
100
|
+
let quote = "";
|
|
101
|
+
let inSubset = false;
|
|
102
|
+
for (let i = from + 2; i < src.length; i++) {
|
|
103
|
+
const ch = src[i];
|
|
104
|
+
if (quote !== "") {
|
|
105
|
+
if (ch === quote) quote = "";
|
|
106
|
+
continue;
|
|
107
|
+
}
|
|
108
|
+
if (ch === '"' || ch === "'") {
|
|
109
|
+
quote = ch;
|
|
110
|
+
continue;
|
|
111
|
+
}
|
|
112
|
+
if (ch === "[") inSubset = true;
|
|
113
|
+
else if (ch === "]") inSubset = false;
|
|
114
|
+
else if (ch === ">" && !inSubset) return i + 1;
|
|
115
|
+
}
|
|
116
|
+
return src.length;
|
|
117
|
+
}
|
|
118
|
+
function parseTagBody(body) {
|
|
119
|
+
let i = 0;
|
|
120
|
+
while (i < body.length && !WHITESPACE.has(body[i])) i++;
|
|
121
|
+
const name = body.slice(0, i);
|
|
122
|
+
const attrs = [];
|
|
123
|
+
while (i < body.length) {
|
|
124
|
+
while (i < body.length && WHITESPACE.has(body[i])) i++;
|
|
125
|
+
if (i >= body.length) break;
|
|
126
|
+
const nameStart = i;
|
|
127
|
+
while (i < body.length && !WHITESPACE.has(body[i]) && body[i] !== "=") i++;
|
|
128
|
+
const attrName = body.slice(nameStart, i);
|
|
129
|
+
if (attrName.length === 0) {
|
|
130
|
+
i++;
|
|
131
|
+
continue;
|
|
132
|
+
}
|
|
133
|
+
while (i < body.length && WHITESPACE.has(body[i])) i++;
|
|
134
|
+
if (body[i] !== "=") {
|
|
135
|
+
attrs.push([attrName, ""]);
|
|
136
|
+
continue;
|
|
137
|
+
}
|
|
138
|
+
i++;
|
|
139
|
+
while (i < body.length && WHITESPACE.has(body[i])) i++;
|
|
140
|
+
const quote = body[i];
|
|
141
|
+
if (quote === '"' || quote === "'") {
|
|
142
|
+
const valueStart = i + 1;
|
|
143
|
+
const valueEnd = body.indexOf(quote, valueStart);
|
|
144
|
+
const end = valueEnd === -1 ? body.length : valueEnd;
|
|
145
|
+
attrs.push([attrName, decodeXmlEntities(body.slice(valueStart, end))]);
|
|
146
|
+
i = end + 1;
|
|
147
|
+
} else {
|
|
148
|
+
const valueStart = i;
|
|
149
|
+
while (i < body.length && !WHITESPACE.has(body[i])) i++;
|
|
150
|
+
attrs.push([attrName, decodeXmlEntities(body.slice(valueStart, i))]);
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
return { name, attrs };
|
|
154
|
+
}
|
|
155
|
+
function parseXml(xml) {
|
|
156
|
+
const src = xml.charCodeAt(0) === 65279 ? xml.slice(1) : xml;
|
|
157
|
+
const root = newFrame("");
|
|
158
|
+
const stack = [root];
|
|
159
|
+
const len = src.length;
|
|
160
|
+
let i = 0;
|
|
161
|
+
const top = () => stack[stack.length - 1];
|
|
162
|
+
const pushText = (from, to) => {
|
|
163
|
+
if (to > from) top().text.push(decodeXmlEntities(src.slice(from, to)));
|
|
164
|
+
};
|
|
165
|
+
const closeElement = (name) => {
|
|
166
|
+
let target = -1;
|
|
167
|
+
for (let k = stack.length - 1; k >= 1; k--) {
|
|
168
|
+
if (stack[k].name === name) {
|
|
169
|
+
target = k;
|
|
170
|
+
break;
|
|
171
|
+
}
|
|
172
|
+
}
|
|
173
|
+
if (target === -1) return;
|
|
174
|
+
while (stack.length > target) {
|
|
175
|
+
const frame = stack.pop();
|
|
176
|
+
addChild(top(), frame.name, finishFrame(frame));
|
|
177
|
+
}
|
|
178
|
+
};
|
|
179
|
+
while (i < len) {
|
|
180
|
+
const lt = src.indexOf("<", i);
|
|
181
|
+
if (lt === -1) {
|
|
182
|
+
pushText(i, len);
|
|
183
|
+
break;
|
|
184
|
+
}
|
|
185
|
+
pushText(i, lt);
|
|
186
|
+
if (src.startsWith("<!--", lt)) {
|
|
187
|
+
const end2 = src.indexOf("-->", lt + 4);
|
|
188
|
+
i = end2 === -1 ? len : end2 + 3;
|
|
189
|
+
continue;
|
|
190
|
+
}
|
|
191
|
+
if (src.startsWith("<![CDATA[", lt)) {
|
|
192
|
+
const end2 = src.indexOf("]]>", lt + 9);
|
|
193
|
+
top().text.push(src.slice(lt + 9, end2 === -1 ? len : end2));
|
|
194
|
+
i = end2 === -1 ? len : end2 + 3;
|
|
195
|
+
continue;
|
|
196
|
+
}
|
|
197
|
+
if (src.startsWith("<!", lt)) {
|
|
198
|
+
i = skipDeclaration(src, lt);
|
|
199
|
+
continue;
|
|
200
|
+
}
|
|
201
|
+
if (src.startsWith("<?", lt)) {
|
|
202
|
+
const end2 = src.indexOf("?>", lt + 2);
|
|
203
|
+
const { name: name2, attrs: attrs2 } = parseTagBody(src.slice(lt + 2, end2 === -1 ? len : end2));
|
|
204
|
+
if (name2.length > 0) {
|
|
205
|
+
const frame2 = newFrame(`?${name2}`);
|
|
206
|
+
frame2.attrs = attrs2;
|
|
207
|
+
addChild(top(), frame2.name, finishFrame(frame2));
|
|
208
|
+
}
|
|
209
|
+
i = end2 === -1 ? len : end2 + 2;
|
|
210
|
+
continue;
|
|
211
|
+
}
|
|
212
|
+
if (src.startsWith("</", lt)) {
|
|
213
|
+
const end2 = src.indexOf(">", lt + 2);
|
|
214
|
+
closeElement(src.slice(lt + 2, end2 === -1 ? len : end2).trim());
|
|
215
|
+
i = end2 === -1 ? len : end2 + 1;
|
|
216
|
+
continue;
|
|
217
|
+
}
|
|
218
|
+
const end = findTagEnd(src, lt + 1);
|
|
219
|
+
const tagEnd = end === -1 ? len : end;
|
|
220
|
+
let body = src.slice(lt + 1, tagEnd);
|
|
221
|
+
const selfClosing = body.endsWith("/");
|
|
222
|
+
if (selfClosing) body = body.slice(0, -1);
|
|
223
|
+
const { name, attrs } = parseTagBody(body);
|
|
224
|
+
i = end === -1 ? len : end + 1;
|
|
225
|
+
if (name.length === 0) continue;
|
|
226
|
+
const frame = newFrame(name);
|
|
227
|
+
frame.attrs = attrs;
|
|
228
|
+
if (selfClosing) {
|
|
229
|
+
addChild(top(), name, finishFrame(frame));
|
|
230
|
+
continue;
|
|
231
|
+
}
|
|
232
|
+
if (stack.length > MAX_XML_DEPTH) {
|
|
233
|
+
throw new XmlTooDeepError(`XML nesting deeper than ${MAX_XML_DEPTH} elements; refusing to parse (this file is not something any office application produces)`);
|
|
234
|
+
}
|
|
235
|
+
stack.push(frame);
|
|
236
|
+
}
|
|
237
|
+
while (stack.length > 1) {
|
|
238
|
+
const frame = stack.pop();
|
|
239
|
+
addChild(top(), frame.name, finishFrame(frame));
|
|
240
|
+
}
|
|
241
|
+
return Object.fromEntries(root.children);
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
// src/ooxml_extract.ts
|
|
245
|
+
var MAX_OOXML_WORK_MILLIS = MAX_DOCUMENT_WORK_MILLIS;
|
|
246
|
+
var OoxmlTookTooLongError = class extends DocumentRefusedError {
|
|
247
|
+
constructor(message) {
|
|
248
|
+
super(message, "OoxmlTookTooLongError", true);
|
|
249
|
+
}
|
|
250
|
+
};
|
|
251
|
+
var NotAnOfficeDocumentError = class extends DocumentRefusedError {
|
|
252
|
+
constructor(message, cause) {
|
|
253
|
+
super(message, "NotAnOfficeDocumentError");
|
|
254
|
+
if (cause !== void 0) this.cause = cause;
|
|
255
|
+
}
|
|
256
|
+
};
|
|
257
|
+
function ooxmlWorkDeadline() {
|
|
258
|
+
return Date.now() + MAX_OOXML_WORK_MILLIS;
|
|
259
|
+
}
|
|
260
|
+
function assertOoxmlWithinDeadline(deadline, hint) {
|
|
261
|
+
if (Date.now() > deadline) {
|
|
262
|
+
throw new OoxmlTookTooLongError(`reading this document's slides/sheets passed the ${MAX_OOXML_WORK_MILLIS}ms limit. ${hint}`);
|
|
263
|
+
}
|
|
264
|
+
}
|
|
265
|
+
var loadFflate = createLazyModuleLoader(
|
|
266
|
+
async () => await import("fflate"),
|
|
267
|
+
"office-file reading disabled (fflate unavailable)"
|
|
268
|
+
);
|
|
269
|
+
function accessFailureMessage(err, filePath) {
|
|
270
|
+
const code = err?.code;
|
|
271
|
+
if (code === "ENOENT") return `File not found: ${filePath}`;
|
|
272
|
+
return `could not read ${filePath} (${code ?? "unknown error"})`;
|
|
273
|
+
}
|
|
274
|
+
async function readOoxmlZip(filePath, kind) {
|
|
275
|
+
const fflate = await loadFflate();
|
|
276
|
+
if (!fflate) throw new Error("fflate is not installed; run `npm install fflate` to enable this command");
|
|
277
|
+
let stat;
|
|
278
|
+
try {
|
|
279
|
+
stat = fs.statSync(filePath);
|
|
280
|
+
} catch (err) {
|
|
281
|
+
throw new Error(accessFailureMessage(err, filePath), { cause: err });
|
|
282
|
+
}
|
|
283
|
+
if (!stat.isFile()) throw new NotAnOfficeDocumentError(`not a valid ${kind} file: ${filePath}`);
|
|
284
|
+
if (stat.size > MAX_ZIP_INPUT_BYTES) throw new ZipInputTooLargeError(filePath, stat.size, MAX_ZIP_INPUT_BYTES);
|
|
285
|
+
let data;
|
|
286
|
+
try {
|
|
287
|
+
data = fs.readFileSync(filePath);
|
|
288
|
+
} catch (err) {
|
|
289
|
+
throw new Error(accessFailureMessage(err, filePath), { cause: err });
|
|
290
|
+
}
|
|
291
|
+
try {
|
|
292
|
+
return unzipBounded(fflate, new Uint8Array(data), { limitBytes: MAX_ZIP_OUTPUT_BYTES, shouldExtract: () => true });
|
|
293
|
+
} catch (err) {
|
|
294
|
+
if (err instanceof ZipOutputTooLargeError) throw err;
|
|
295
|
+
throw new NotAnOfficeDocumentError(`not a valid ${kind} file: ${filePath}`, err);
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
var MAX_OOXML_PART_BYTES = 32 * 1024 * 1024;
|
|
299
|
+
var OoxmlPartTooLargeError = class extends DocumentRefusedError {
|
|
300
|
+
constructor(entryPath, bytes) {
|
|
301
|
+
super(`${entryPath} is ${bytes} bytes, past the ${MAX_OOXML_PART_BYTES}-byte limit for one part of an office file. Split the document, or extract from a smaller copy.`, "OoxmlPartTooLargeError");
|
|
302
|
+
}
|
|
303
|
+
};
|
|
304
|
+
var MAX_OOXML_DOCUMENT_PART_BYTES = 64 * 1024 * 1024;
|
|
305
|
+
var OoxmlDocumentTooLargeError = class extends DocumentRefusedError {
|
|
306
|
+
constructor(entryPath, spentBytes, partBytes) {
|
|
307
|
+
super(`decoding ${entryPath} (${partBytes} bytes, after ${spentBytes} already decoded) would take this office file past the ${MAX_OOXML_DOCUMENT_PART_BYTES}-byte limit on the XML one document may have decoded at once. Split the document, or extract from a smaller copy.`, "OoxmlDocumentTooLargeError");
|
|
308
|
+
}
|
|
309
|
+
};
|
|
310
|
+
function ooxmlPartBudget() {
|
|
311
|
+
return { spent: 0 };
|
|
312
|
+
}
|
|
313
|
+
function decodeZipEntry(entries, entryPath, budget) {
|
|
314
|
+
const bytes = entries[entryPath];
|
|
315
|
+
if (bytes === void 0) return null;
|
|
316
|
+
if (bytes.length > MAX_OOXML_PART_BYTES) throw new OoxmlPartTooLargeError(entryPath, bytes.length);
|
|
317
|
+
if (budget.spent + bytes.length > MAX_OOXML_DOCUMENT_PART_BYTES) throw new OoxmlDocumentTooLargeError(entryPath, budget.spent, bytes.length);
|
|
318
|
+
budget.spent += bytes.length;
|
|
319
|
+
return new TextDecoder("utf-8").decode(bytes);
|
|
320
|
+
}
|
|
321
|
+
async function parseOoxmlPart(xmlText) {
|
|
322
|
+
return parseXml(xmlText);
|
|
323
|
+
}
|
|
324
|
+
function pushTextValue(runs, val) {
|
|
325
|
+
if (Array.isArray(val)) {
|
|
326
|
+
for (const v of val) pushTextValue(runs, v);
|
|
327
|
+
} else if (typeof val === "string") {
|
|
328
|
+
runs.push(val);
|
|
329
|
+
} else if (typeof val === "number" || typeof val === "boolean") {
|
|
330
|
+
runs.push(String(val));
|
|
331
|
+
} else if (val !== null && typeof val === "object" && "#text" in val) {
|
|
332
|
+
runs.push(String(val["#text"]));
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
function collectTextRuns(node, tag) {
|
|
336
|
+
const runs = [];
|
|
337
|
+
function walk(n) {
|
|
338
|
+
if (Array.isArray(n)) {
|
|
339
|
+
n.forEach(walk);
|
|
340
|
+
return;
|
|
341
|
+
}
|
|
342
|
+
if (n !== null && typeof n === "object") {
|
|
343
|
+
const obj = n;
|
|
344
|
+
for (const [key, val] of Object.entries(obj)) {
|
|
345
|
+
if (key === tag) {
|
|
346
|
+
pushTextValue(runs, val);
|
|
347
|
+
} else if (val !== null && typeof val === "object") {
|
|
348
|
+
walk(val);
|
|
349
|
+
}
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
}
|
|
353
|
+
walk(node);
|
|
354
|
+
return runs;
|
|
355
|
+
}
|
|
356
|
+
function collectElements(node, tag) {
|
|
357
|
+
const out = [];
|
|
358
|
+
function walk(n) {
|
|
359
|
+
if (Array.isArray(n)) {
|
|
360
|
+
n.forEach(walk);
|
|
361
|
+
return;
|
|
362
|
+
}
|
|
363
|
+
if (n !== null && typeof n === "object") {
|
|
364
|
+
const obj = n;
|
|
365
|
+
for (const [key, val] of Object.entries(obj)) {
|
|
366
|
+
if (key === tag) {
|
|
367
|
+
if (Array.isArray(val)) pushAll(out, val);
|
|
368
|
+
else out.push(val);
|
|
369
|
+
} else if (val !== null && typeof val === "object") {
|
|
370
|
+
walk(val);
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
}
|
|
374
|
+
}
|
|
375
|
+
walk(node);
|
|
376
|
+
return out;
|
|
377
|
+
}
|
|
378
|
+
function sortNumberedParts(paths, pattern) {
|
|
379
|
+
return paths.map((p) => {
|
|
380
|
+
const m = pattern.exec(p);
|
|
381
|
+
return { p, n: m?.[1] !== void 0 ? parseInt(m[1], 10) : Number.MAX_SAFE_INTEGER };
|
|
382
|
+
}).sort((a, b) => a.n - b.n).map((x) => x.p);
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
export {
|
|
386
|
+
NotAnOfficeDocumentError,
|
|
387
|
+
ooxmlWorkDeadline,
|
|
388
|
+
assertOoxmlWithinDeadline,
|
|
389
|
+
readOoxmlZip,
|
|
390
|
+
ooxmlPartBudget,
|
|
391
|
+
decodeZipEntry,
|
|
392
|
+
parseOoxmlPart,
|
|
393
|
+
collectTextRuns,
|
|
394
|
+
collectElements,
|
|
395
|
+
sortNumberedParts
|
|
396
|
+
};
|
|
@@ -25,8 +25,7 @@ import {
|
|
|
25
25
|
runSkeleton,
|
|
26
26
|
runSymbol,
|
|
27
27
|
withPinnedReads
|
|
28
|
-
} from "./token-goat-chunk-
|
|
29
|
-
import "./token-goat-chunk-6F5TLJC7.mjs";
|
|
28
|
+
} from "./token-goat-chunk-RUDOKYPJ.mjs";
|
|
30
29
|
import {
|
|
31
30
|
buildProjectMap,
|
|
32
31
|
embeddingsDepsAvailable,
|
|
@@ -35,16 +34,28 @@ import {
|
|
|
35
34
|
getProjectIndexCounts,
|
|
36
35
|
isWorkerRunning,
|
|
37
36
|
mapLookupBytesSaved
|
|
38
|
-
} from "./token-goat-chunk-
|
|
37
|
+
} from "./token-goat-chunk-IT6O3PNN.mjs";
|
|
38
|
+
import "./token-goat-chunk-VZYD4OZB.mjs";
|
|
39
|
+
import "./token-goat-chunk-NEI4NC54.mjs";
|
|
40
|
+
import "./token-goat-chunk-EG3663UT.mjs";
|
|
41
|
+
import "./token-goat-chunk-3BTK54F3.mjs";
|
|
42
|
+
import "./token-goat-chunk-XTQAOTSO.mjs";
|
|
43
|
+
import "./token-goat-chunk-AH6QILZM.mjs";
|
|
44
|
+
import "./token-goat-chunk-Y4AFKTHK.mjs";
|
|
45
|
+
import "./token-goat-chunk-DBNY4RLN.mjs";
|
|
39
46
|
import {
|
|
40
|
-
envStrList,
|
|
41
47
|
getDb,
|
|
42
|
-
loadConfig,
|
|
43
48
|
recordStat,
|
|
44
|
-
resolveProjectRoot,
|
|
45
49
|
savedTokensFromBytes
|
|
46
|
-
} from "./token-goat-chunk-
|
|
50
|
+
} from "./token-goat-chunk-T2IWWTHB.mjs";
|
|
51
|
+
import {
|
|
52
|
+
envStrList,
|
|
53
|
+
loadConfig,
|
|
54
|
+
resolveProjectRoot
|
|
55
|
+
} from "./token-goat-chunk-SAQ5PG4L.mjs";
|
|
47
56
|
import "./token-goat-chunk-EEIDFMEM.mjs";
|
|
57
|
+
import "./token-goat-chunk-HF6H7RNK.mjs";
|
|
58
|
+
import "./token-goat-chunk-POBYR64E.mjs";
|
|
48
59
|
import {
|
|
49
60
|
ENV_KEYS,
|
|
50
61
|
VERSION,
|
|
@@ -54,7 +65,8 @@ import {
|
|
|
54
65
|
foldCaseForContainment,
|
|
55
66
|
globalDbPath,
|
|
56
67
|
normalizePath
|
|
57
|
-
} from "./token-goat-chunk-
|
|
68
|
+
} from "./token-goat-chunk-AMYCQJX4.mjs";
|
|
69
|
+
import "./token-goat-chunk-GMOUBOX4.mjs";
|
|
58
70
|
import {
|
|
59
71
|
init_define_import_meta_env
|
|
60
72
|
} from "./token-goat-chunk-A37V4PBF.mjs";
|
|
@@ -242,7 +254,7 @@ function mcpToolAllowlist() {
|
|
|
242
254
|
return names.length === 0 ? null : new Set(names);
|
|
243
255
|
}
|
|
244
256
|
async function createMcpServer() {
|
|
245
|
-
const [{ McpServer }, { z }] = await Promise.all([import("./token-goat-chunk-
|
|
257
|
+
const [{ McpServer }, { z }] = await Promise.all([import("./token-goat-chunk-OMNQUUIT.mjs"), import("./token-goat-chunk-T6M7DAW3.mjs")]);
|
|
246
258
|
const server = new McpServer({ name: "token-goat", version: VERSION });
|
|
247
259
|
const allowlist = mcpToolAllowlist();
|
|
248
260
|
if (allowlist !== null) {
|
|
@@ -276,7 +288,8 @@ async function createMcpServer() {
|
|
|
276
288
|
kind: z.string().optional().describe("restrict to one kind (function, class, ...)"),
|
|
277
289
|
json: z.boolean().optional().describe("output as JSON"),
|
|
278
290
|
projectRoot: projectRootField
|
|
279
|
-
}
|
|
291
|
+
},
|
|
292
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
280
293
|
},
|
|
281
294
|
(args) => {
|
|
282
295
|
const { name, limit, file, kind, json, projectRoot } = args;
|
|
@@ -314,7 +327,9 @@ async function createMcpServer() {
|
|
|
314
327
|
forceRefresh: z.boolean().optional().describe("reparse file from disk before querying (ignore stale index)"),
|
|
315
328
|
stats: z.boolean().optional().describe("add per-symbol reference count and doc-coverage flag"),
|
|
316
329
|
projectRoot: projectRootField
|
|
317
|
-
}
|
|
330
|
+
},
|
|
331
|
+
// What readOnlyHint claims here, once, for all fifteen tools that carry it: the caller's own environment is untouched -- the project's files, and anything the caller would notice missing. Three things these tools do are deliberately outside that. They append a row to token-goat's `stats` table. They may reparse a file whose index entry is stale, and enqueue it for the background indexer. Opening a database for the first time creates its schema. All three are token-goat's own derived state, rebuilt from the project on demand and worth nothing if deleted; calling a symbol lookup a writing tool on their account would cost the caller a confirmation prompt for every read while protecting nothing. A tool that touches anything the caller owns says so instead, and the guard beside this file measures which tools those are rather than trusting the claim.
|
|
332
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
318
333
|
},
|
|
319
334
|
(args) => {
|
|
320
335
|
const { spec, json, forceRefresh, stats, projectRoot } = args;
|
|
@@ -343,7 +358,8 @@ async function createMcpServer() {
|
|
|
343
358
|
spec: z.string().describe("file::Heading"),
|
|
344
359
|
json: z.boolean().optional().describe("output as JSON"),
|
|
345
360
|
projectRoot: projectRootField
|
|
346
|
-
}
|
|
361
|
+
},
|
|
362
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
347
363
|
},
|
|
348
364
|
(args) => {
|
|
349
365
|
const { spec, json, projectRoot } = args;
|
|
@@ -373,7 +389,8 @@ async function createMcpServer() {
|
|
|
373
389
|
forceRefresh: z.boolean().optional().describe("reparse file from disk before querying (ignore stale index)"),
|
|
374
390
|
stats: z.boolean().optional().describe("add per-symbol reference count and doc-coverage flag"),
|
|
375
391
|
projectRoot: projectRootField
|
|
376
|
-
}
|
|
392
|
+
},
|
|
393
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
377
394
|
},
|
|
378
395
|
(args) => {
|
|
379
396
|
const { file, json, minLines, forceRefresh, stats, projectRoot } = args;
|
|
@@ -406,7 +423,8 @@ async function createMcpServer() {
|
|
|
406
423
|
forceRefresh: z.boolean().optional().describe("reparse file from disk before querying (ignore stale index)"),
|
|
407
424
|
stats: z.boolean().optional().describe("add per-symbol reference count and doc-coverage flag"),
|
|
408
425
|
projectRoot: projectRootField
|
|
409
|
-
}
|
|
426
|
+
},
|
|
427
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
410
428
|
},
|
|
411
429
|
(args) => {
|
|
412
430
|
const { file, json, minLines, forceRefresh, stats, projectRoot } = args;
|
|
@@ -439,7 +457,9 @@ async function createMcpServer() {
|
|
|
439
457
|
excludeTests: z.boolean().optional().describe("hide hits whose file is a test file (opt-in; default output is unchanged)"),
|
|
440
458
|
json: z.boolean().optional().describe("output as JSON"),
|
|
441
459
|
projectRoot: makeProjectRootField("search")
|
|
442
|
-
}
|
|
460
|
+
},
|
|
461
|
+
// openWorldHint stays false although a first call on a machine with no cached model downloads one over the network: what this tool interacts with is the local index, and the download provisions the tool rather than being the tool reaching out. A client reading `true` here would take it as "this searches the internet", which is the wrong warning.
|
|
462
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
443
463
|
},
|
|
444
464
|
async (args) => {
|
|
445
465
|
const { query, limit, grep, excludeTests, json, projectRoot } = args;
|
|
@@ -461,7 +481,8 @@ async function createMcpServer() {
|
|
|
461
481
|
description: 'Report whether the index for a project can be trusted right now: whether it has ever been indexed at all, current file/symbol counts, dirty-reindex-queue depth, whether the background worker is alive, and whether embeddings are available (semantic silently degrades to full-text search without them). Call this after an unexpectedly empty result from another token-goat tool (symbol/read/semantic/refs/brief/...) to tell apart "no match" from "the index is not ready yet" -- an MCP-only client has no hook layer to warn about this on its own, so an empty tool result and a stale/unindexed project look identical without this check.',
|
|
462
482
|
inputSchema: {
|
|
463
483
|
projectRoot: makeProjectRootField("check")
|
|
464
|
-
}
|
|
484
|
+
},
|
|
485
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
465
486
|
},
|
|
466
487
|
(args) => {
|
|
467
488
|
const { projectRoot } = args;
|
|
@@ -520,7 +541,8 @@ async function createMcpServer() {
|
|
|
520
541
|
),
|
|
521
542
|
json: z.boolean().optional().describe("output as JSON"),
|
|
522
543
|
projectRoot: projectRootField
|
|
523
|
-
}
|
|
544
|
+
},
|
|
545
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
524
546
|
},
|
|
525
547
|
(args) => {
|
|
526
548
|
const { spec, callers, limit, top, json, projectRoot } = args;
|
|
@@ -554,7 +576,8 @@ async function createMcpServer() {
|
|
|
554
576
|
excludeTests: z.boolean().optional().describe("hide callers whose call site lives in a test file (opt-in; default output is unchanged)"),
|
|
555
577
|
grep: z.string().optional().describe("only show callers whose enclosing symbol name matches this regex (literal substring if it is not valid regex)"),
|
|
556
578
|
projectRoot: makeProjectRootField("orient")
|
|
557
|
-
}
|
|
579
|
+
},
|
|
580
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
558
581
|
},
|
|
559
582
|
(args) => {
|
|
560
583
|
const { spec, limit, json, context, excludeTests, grep, projectRoot } = args;
|
|
@@ -584,7 +607,8 @@ async function createMcpServer() {
|
|
|
584
607
|
inputSchema: {
|
|
585
608
|
compact: z.boolean().optional().describe("compact, low-token summary"),
|
|
586
609
|
projectRoot: makeProjectRootField("overview")
|
|
587
|
-
}
|
|
610
|
+
},
|
|
611
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
588
612
|
},
|
|
589
613
|
(args) => {
|
|
590
614
|
const { compact, projectRoot } = args;
|
|
@@ -604,7 +628,8 @@ async function createMcpServer() {
|
|
|
604
628
|
symbolMode: z.boolean().optional().describe("list symbols instead of files"),
|
|
605
629
|
json: z.boolean().optional().describe("output as JSON"),
|
|
606
630
|
projectRoot: projectRootField
|
|
607
|
-
}
|
|
631
|
+
},
|
|
632
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
608
633
|
},
|
|
609
634
|
(args) => {
|
|
610
635
|
const { ref, symbolMode, json, projectRoot } = args;
|
|
@@ -631,7 +656,8 @@ async function createMcpServer() {
|
|
|
631
656
|
context: z.number().int().nonnegative().max(MCP_MAX_CONTEXT_LINES).optional().describe("lines of context to show before and after each match"),
|
|
632
657
|
// runGrep takes no projectRoot of its own (its `path` array is its scope), so this field only names the root the confinement check is made against -- without it, a search rooted anywhere but the server process's cwd is refused.
|
|
633
658
|
projectRoot: makeProjectRootField("search")
|
|
634
|
-
}
|
|
659
|
+
},
|
|
660
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
635
661
|
},
|
|
636
662
|
(args) => {
|
|
637
663
|
const { pattern, path: searchPath, maxLines, json, recursive, context, projectRoot } = args;
|
|
@@ -663,7 +689,8 @@ async function createMcpServer() {
|
|
|
663
689
|
file: z.string().describe("file path"),
|
|
664
690
|
json: z.boolean().optional().describe("output as JSON"),
|
|
665
691
|
projectRoot: projectRootField
|
|
666
|
-
}
|
|
692
|
+
},
|
|
693
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
667
694
|
},
|
|
668
695
|
(args) => {
|
|
669
696
|
const { file, json, projectRoot } = args;
|
|
@@ -686,7 +713,8 @@ async function createMcpServer() {
|
|
|
686
713
|
file: z.string().describe("file path"),
|
|
687
714
|
json: z.boolean().optional().describe("output as JSON"),
|
|
688
715
|
projectRoot: projectRootField
|
|
689
|
-
}
|
|
716
|
+
},
|
|
717
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
690
718
|
},
|
|
691
719
|
(args) => {
|
|
692
720
|
const { file, json, projectRoot } = args;
|
|
@@ -707,7 +735,9 @@ async function createMcpServer() {
|
|
|
707
735
|
description: "Compress arbitrary local text, persist it in the bounded local cache, and return an opaque recovery ID plus metadata.",
|
|
708
736
|
inputSchema: {
|
|
709
737
|
text: z.string().max(CONTENT_MAX_INPUT_CHARS).describe("text to compress")
|
|
710
|
-
}
|
|
738
|
+
},
|
|
739
|
+
// destructiveHint is true for every tool that writes here, including the ones that only ever add: the store is bounded by item count and by total bytes, so a write can evict an earlier entry and leave a recovery id the caller is holding unredeemable. idempotentHint survives a repeat writing a fresh timestamp and another stats row, on the same reading of "its environment" set out on the read tool above.
|
|
740
|
+
annotations: { readOnlyHint: false, destructiveHint: true, idempotentHint: true, openWorldHint: false }
|
|
711
741
|
},
|
|
712
742
|
(args) => toCallToolResult({ text: displaySafeJson(compressionPayload(compressText(args.text))), code: 0 })
|
|
713
743
|
);
|
|
@@ -717,7 +747,8 @@ async function createMcpServer() {
|
|
|
717
747
|
description: "Retrieve original text from a token-goat compression ID.",
|
|
718
748
|
inputSchema: {
|
|
719
749
|
id: z.string().regex(/^tg_[0-9a-f]{16}$/).describe("opaque token-goat content ID")
|
|
720
|
-
}
|
|
750
|
+
},
|
|
751
|
+
annotations: { readOnlyHint: true, openWorldHint: false }
|
|
721
752
|
},
|
|
722
753
|
(args) => {
|
|
723
754
|
const text = retrieveText(args.id);
|
|
@@ -732,7 +763,8 @@ async function createMcpServer() {
|
|
|
732
763
|
name: z.string().regex(/^[A-Za-z0-9._-]{1,128}$/).describe("handoff name"),
|
|
733
764
|
text: z.string().max(CONTENT_MAX_INPUT_CHARS).describe("handoff text"),
|
|
734
765
|
projectRoot: makeProjectRootField("scope")
|
|
735
|
-
}
|
|
766
|
+
},
|
|
767
|
+
annotations: { readOnlyHint: false, destructiveHint: true, idempotentHint: true, openWorldHint: false }
|
|
736
768
|
},
|
|
737
769
|
(args) => toCallToolResult({
|
|
738
770
|
text: displaySafeJson(createHandoff(args.name, args.text, resolveToolRoot(args.projectRoot))),
|
|
@@ -747,7 +779,9 @@ async function createMcpServer() {
|
|
|
747
779
|
name: z.string().regex(/^[A-Za-z0-9._-]{1,128}$/).describe("handoff name"),
|
|
748
780
|
full: z.boolean().optional().describe("return full text instead of a compact payload"),
|
|
749
781
|
projectRoot: makeProjectRootField("scope")
|
|
750
|
-
}
|
|
782
|
+
},
|
|
783
|
+
// Reads like a read, and is not one: resolving compactly runs the handoff text back through compressText, which stores the compact payload so the recovery id it hands back can be redeemed. A measurement, not a reading of this code -- the guard hashes the store's bytes around every tool call, and this is the tool it caught.
|
|
784
|
+
annotations: { readOnlyHint: false, destructiveHint: true, idempotentHint: true, openWorldHint: false }
|
|
751
785
|
},
|
|
752
786
|
(args) => {
|
|
753
787
|
const result = resolveHandoff(args.name, {
|