token-goat 2.9.18 → 2.9.20
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -1
- package/dist/token-goat-chunk-2CLWWRFR.mjs +26 -0
- package/dist/{token-goat-chunk-LTYWLKAG.mjs → token-goat-chunk-2Q3N5KUQ.mjs} +1283 -1012
- package/dist/{token-goat-chunk-6LGVCCMY.mjs → token-goat-chunk-3OGZOFLS.mjs} +26 -16
- package/dist/{token-goat-chunk-QML7ITHJ.mjs → token-goat-chunk-3VTJC7KU.mjs} +693 -438
- package/dist/{token-goat-chunk-QIDLJGXZ.mjs → token-goat-chunk-6LJNM4ZZ.mjs} +29 -5
- package/dist/{token-goat-chunk-CA6V36KW.mjs → token-goat-chunk-74ADIZZX.mjs} +68 -50
- package/dist/{token-goat-chunk-52IMXJ2L.mjs → token-goat-chunk-ASTBLE2R.mjs} +4 -4
- package/dist/token-goat-chunk-AU2K2QHR.mjs +20 -0
- package/dist/token-goat-chunk-BQIC3AVL.mjs +292 -0
- package/dist/token-goat-chunk-CBXLE54Q.mjs +30 -0
- package/dist/{token-goat-chunk-LMFO66YD.mjs → token-goat-chunk-DA43OE7L.mjs} +1 -1
- package/dist/{token-goat-chunk-ARNRFR6M.mjs → token-goat-chunk-DNAF26L4.mjs} +1 -1
- package/dist/{token-goat-chunk-BFEWOMA3.mjs → token-goat-chunk-DPHJU3P7.mjs} +1186 -4573
- package/dist/{token-goat-chunk-OWSJJY5R.mjs → token-goat-chunk-DW4VKU7D.mjs} +5 -5
- package/dist/token-goat-chunk-DYUBZZNP.mjs +198 -0
- package/dist/{token-goat-chunk-W3E65QOW.mjs → token-goat-chunk-ERGHDBC6.mjs} +310 -47
- package/dist/token-goat-chunk-EUZKG7ZO.mjs +304 -0
- package/dist/{token-goat-chunk-5MMJAJF3.mjs → token-goat-chunk-FWYZVS3E.mjs} +324 -31
- package/dist/token-goat-chunk-G6SYQQO7.mjs +41 -0
- package/dist/{token-goat-chunk-YQZE5NS6.mjs → token-goat-chunk-GIPMOGOI.mjs} +6 -6
- package/dist/{token-goat-chunk-BWCRC23L.mjs → token-goat-chunk-H4VC5TR6.mjs} +29 -20
- package/dist/{token-goat-chunk-V3ZWBEHM.mjs → token-goat-chunk-I5PQRHFE.mjs} +103 -24
- package/dist/token-goat-chunk-ITWYDOKP.mjs +341 -0
- package/dist/{token-goat-chunk-I67CPTDD.mjs → token-goat-chunk-K3ZNBN4U.mjs} +19 -14
- package/dist/{token-goat-chunk-HPOFR6QX.mjs → token-goat-chunk-K5XPIUVB.mjs} +222 -61
- package/dist/token-goat-chunk-KT75DZK4.mjs +1989 -0
- package/dist/{token-goat-chunk-H3FONFBO.mjs → token-goat-chunk-MNJWG2ZO.mjs} +3 -3
- package/dist/{token-goat-chunk-GRI6HWX3.mjs → token-goat-chunk-MUJV7BR3.mjs} +1 -1
- package/dist/{token-goat-chunk-ERTXEKB6.mjs → token-goat-chunk-NMTKNYGF.mjs} +2 -0
- package/dist/{token-goat-chunk-LA3O4JGD.mjs → token-goat-chunk-ODF7H7WS.mjs} +3 -3
- package/dist/{token-goat-chunk-K4DG6EEF.mjs → token-goat-chunk-ONMU7BON.mjs} +7 -8
- package/dist/{token-goat-chunk-4EXFN2AW.mjs → token-goat-chunk-OZRSREKR.mjs} +2 -2
- package/dist/{token-goat-chunk-EPXNOKIV.mjs → token-goat-chunk-P4XBHZXU.mjs} +2 -2
- package/dist/{token-goat-chunk-CTOYJ232.mjs → token-goat-chunk-PAVWB4L4.mjs} +173 -1840
- package/dist/{token-goat-chunk-UENEL6KX.mjs → token-goat-chunk-PTW22MRC.mjs} +2 -2
- package/dist/token-goat-chunk-SL454EN5.mjs +1816 -0
- package/dist/token-goat-chunk-SP5QR2HN.mjs +1379 -0
- package/dist/{token-goat-chunk-7VLWNQDQ.mjs → token-goat-chunk-T46UKRFM.mjs} +1 -1
- package/dist/{token-goat-chunk-L7VCX33I.mjs → token-goat-chunk-T63VHOL3.mjs} +40 -10
- package/dist/token-goat-chunk-VSJGPEBB.mjs +248 -0
- package/dist/{token-goat-chunk-HIZXQVNU.mjs → token-goat-chunk-VYXGHGUY.mjs} +4 -4
- package/dist/token-goat-chunk-W6QYCWAT.mjs +128 -0
- package/dist/{token-goat-chunk-WFX2V3OF.mjs → token-goat-chunk-Z4ALQADF.mjs} +579 -422
- package/dist/token-goat-chunk-ZHFM4DHJ.mjs +324 -0
- package/dist/token-goat-hook.mjs +18 -12
- package/dist/token-goat.core.mjs +29 -20
- package/docs/C4_RUNTIME_ARCHITECTURE.md +1 -1
- package/docs/cli.md +15 -13
- package/docs/install.md +1 -1
- package/docs/security.md +2 -2
- package/package.json +3 -2
- package/dist/token-goat-chunk-2KTZ6J7T.mjs +0 -173
- package/dist/token-goat-chunk-34OUJ2IE.mjs +0 -35
- package/dist/token-goat-chunk-RHHFEXKF.mjs +0 -93
|
@@ -0,0 +1,324 @@
|
|
|
1
|
+
import { createRequire as __cjsRequire } from 'node:module';
|
|
2
|
+
const require = __cjsRequire(import.meta.url);
|
|
3
|
+
import {
|
|
4
|
+
classifyContent,
|
|
5
|
+
guardDivisor,
|
|
6
|
+
stripAnsiEscapes
|
|
7
|
+
} from "./token-goat-chunk-K5XPIUVB.mjs";
|
|
8
|
+
import {
|
|
9
|
+
safeSlice
|
|
10
|
+
} from "./token-goat-chunk-6LJNM4ZZ.mjs";
|
|
11
|
+
import {
|
|
12
|
+
displaySafeText
|
|
13
|
+
} from "./token-goat-chunk-NMTKNYGF.mjs";
|
|
14
|
+
import {
|
|
15
|
+
init_define_import_meta_env
|
|
16
|
+
} from "./token-goat-chunk-A37V4PBF.mjs";
|
|
17
|
+
|
|
18
|
+
// src/claude_config_dir.ts
|
|
19
|
+
init_define_import_meta_env();
|
|
20
|
+
import * as os from "node:os";
|
|
21
|
+
import * as path from "node:path";
|
|
22
|
+
function claudeConfigDir(homeDir = os.homedir()) {
|
|
23
|
+
const override = process.env["CLAUDE_CONFIG_DIR"];
|
|
24
|
+
if (override !== void 0 && override !== "") return override;
|
|
25
|
+
return path.join(homeDir, ".claude");
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
// src/overflow_guard.ts
|
|
29
|
+
init_define_import_meta_env();
|
|
30
|
+
function estimateTokensFromLength(length, cls = "text") {
|
|
31
|
+
return Math.max(1, Math.floor(Math.max(0, length) / guardDivisor(cls)) + 1);
|
|
32
|
+
}
|
|
33
|
+
function estimateTokens(text) {
|
|
34
|
+
const stripped = stripAnsiEscapes(text);
|
|
35
|
+
return estimateTokensFromLength(stripped.length, classifyContent(stripped));
|
|
36
|
+
}
|
|
37
|
+
function trimToBudget(text, budgetTokens, command) {
|
|
38
|
+
const markerMarginTokens = 64;
|
|
39
|
+
const strippedAll = stripAnsiEscapes(text);
|
|
40
|
+
const contentClass = classifyContent(strippedAll);
|
|
41
|
+
const totalTokens = estimateTokensFromLength(strippedAll.length, contentClass);
|
|
42
|
+
if (totalTokens <= budgetTokens) {
|
|
43
|
+
return text;
|
|
44
|
+
}
|
|
45
|
+
const lines = text.split("\n");
|
|
46
|
+
if (lines.length > 1 && lines[lines.length - 1] === "") lines.pop();
|
|
47
|
+
const totalLines = lines.length;
|
|
48
|
+
const bodyBudget = Math.max(1, budgetTokens - markerMarginTokens);
|
|
49
|
+
const charBudget = bodyBudget * guardDivisor(contentClass);
|
|
50
|
+
const kept = [];
|
|
51
|
+
let used = 0;
|
|
52
|
+
for (const ln of lines) {
|
|
53
|
+
const stripped = stripAnsiEscapes(ln);
|
|
54
|
+
const cost = ln.length + 1;
|
|
55
|
+
if (kept.length === 0 && cost > charBudget) {
|
|
56
|
+
const truncated = safeSlice(stripped, charBudget);
|
|
57
|
+
kept.push(truncated);
|
|
58
|
+
break;
|
|
59
|
+
}
|
|
60
|
+
if (kept.length > 0 && used + cost > charBudget) {
|
|
61
|
+
break;
|
|
62
|
+
}
|
|
63
|
+
kept.push(ln);
|
|
64
|
+
used += cost;
|
|
65
|
+
}
|
|
66
|
+
const shown = kept.length;
|
|
67
|
+
const hint = getHintFor(command);
|
|
68
|
+
const marker = `[token-goat: output capped at ~${budgetTokens} tokens to protect context \u2014 showing ${shown} of ${totalLines} lines. ${hint}]`;
|
|
69
|
+
return kept.join("\n") + "\n" + marker;
|
|
70
|
+
}
|
|
71
|
+
function capJsonRows(items, budgetTokens) {
|
|
72
|
+
const totalCount = items.length;
|
|
73
|
+
const charBudget = Math.max(1, budgetTokens * guardDivisor());
|
|
74
|
+
const kept = [];
|
|
75
|
+
let used = 0;
|
|
76
|
+
for (const item of items) {
|
|
77
|
+
const cost = JSON.stringify(item).length + 2;
|
|
78
|
+
if (kept.length > 0 && used + cost > charBudget) break;
|
|
79
|
+
kept.push(item);
|
|
80
|
+
used += cost;
|
|
81
|
+
}
|
|
82
|
+
return { items: kept, truncated: kept.length < totalCount, totalCount };
|
|
83
|
+
}
|
|
84
|
+
function getHintFor(command) {
|
|
85
|
+
const cmd = (command || "").toLowerCase().trim();
|
|
86
|
+
if (cmd === "symbol") {
|
|
87
|
+
return "Request a specific method (file.py::Class.method) or use --json for structured access.";
|
|
88
|
+
}
|
|
89
|
+
if (cmd === "heading" || cmd === "section") {
|
|
90
|
+
return "Request a narrower sub-heading, e.g. 'doc.md::Section#2'.";
|
|
91
|
+
}
|
|
92
|
+
if (cmd === "lines") {
|
|
93
|
+
return "Request a smaller line range, e.g. 'file.py@100-150'.";
|
|
94
|
+
}
|
|
95
|
+
if (cmd === "bash-output" || cmd === "web-output") {
|
|
96
|
+
return "Use --grep PATTERN, --section HEADING, or --tail N to narrow the cached output.";
|
|
97
|
+
}
|
|
98
|
+
if (cmd === "semantic") {
|
|
99
|
+
return "Narrow your query text or pass --limit to reduce the number of matches returned.";
|
|
100
|
+
}
|
|
101
|
+
return "Narrow your query or raise overflow_guard max_tokens in config.";
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
// src/resident_context.ts
|
|
105
|
+
init_define_import_meta_env();
|
|
106
|
+
import * as fs from "node:fs";
|
|
107
|
+
var LARGE_TASK_LIST_BYTES = 2e4;
|
|
108
|
+
var LARGE_SKILL_BODY_BYTES = 2e4;
|
|
109
|
+
var SKILL_BODY_REPEAT_THRESHOLD = 2;
|
|
110
|
+
var RESIDENT_TAIL_MAX_BYTES = 1048576;
|
|
111
|
+
function createResidentContextStats() {
|
|
112
|
+
return {
|
|
113
|
+
attachmentsByType: /* @__PURE__ */ new Map(),
|
|
114
|
+
latestTaskList: null,
|
|
115
|
+
taskReminderCount: 0,
|
|
116
|
+
taskReminderBytes: 0,
|
|
117
|
+
skillBodies: /* @__PURE__ */ new Map(),
|
|
118
|
+
compactionCount: 0
|
|
119
|
+
};
|
|
120
|
+
}
|
|
121
|
+
function bump(map, key, bytes) {
|
|
122
|
+
const entry = map.get(key);
|
|
123
|
+
if (entry === void 0) map.set(key, { count: 1, bytes });
|
|
124
|
+
else {
|
|
125
|
+
entry.count += 1;
|
|
126
|
+
entry.bytes += bytes;
|
|
127
|
+
}
|
|
128
|
+
}
|
|
129
|
+
function messageText(message) {
|
|
130
|
+
if (message === null || typeof message !== "object") return "";
|
|
131
|
+
const content = message["content"];
|
|
132
|
+
if (typeof content === "string") return content;
|
|
133
|
+
if (!Array.isArray(content)) return "";
|
|
134
|
+
let out = "";
|
|
135
|
+
for (const block of content) {
|
|
136
|
+
if (block === null || typeof block !== "object") continue;
|
|
137
|
+
const text = block["text"];
|
|
138
|
+
if (typeof text === "string") out += text;
|
|
139
|
+
}
|
|
140
|
+
return out;
|
|
141
|
+
}
|
|
142
|
+
function skillNameFromBody(text) {
|
|
143
|
+
const dir = /^Base directory for this skill:\s*(.+?)\s*$/m.exec(text);
|
|
144
|
+
if (dir?.[1] !== void 0) {
|
|
145
|
+
const segments = dir[1].split(/[\\/]/).filter((s) => s.length > 0);
|
|
146
|
+
const last = segments[segments.length - 1];
|
|
147
|
+
if (last !== void 0 && last.length > 0) return last;
|
|
148
|
+
}
|
|
149
|
+
const heading = /^#\s+(.+?)\s*$/m.exec(text);
|
|
150
|
+
if (heading?.[1] !== void 0 && heading[1].length > 0) return heading[1];
|
|
151
|
+
return null;
|
|
152
|
+
}
|
|
153
|
+
function readTaskList(attachment, bytes) {
|
|
154
|
+
const snapshot = {
|
|
155
|
+
bytes,
|
|
156
|
+
itemCount: typeof attachment["itemCount"] === "number" ? attachment["itemCount"] : 0,
|
|
157
|
+
completed: 0,
|
|
158
|
+
inProgress: 0,
|
|
159
|
+
pending: 0,
|
|
160
|
+
descriptionBytes: 0,
|
|
161
|
+
completedDescriptionBytes: 0
|
|
162
|
+
};
|
|
163
|
+
const content = attachment["content"];
|
|
164
|
+
if (!Array.isArray(content)) return snapshot;
|
|
165
|
+
if (content.length > snapshot.itemCount) snapshot.itemCount = content.length;
|
|
166
|
+
for (const item of content) {
|
|
167
|
+
if (item === null || typeof item !== "object") continue;
|
|
168
|
+
const record = item;
|
|
169
|
+
const status = typeof record["status"] === "string" ? record["status"] : "";
|
|
170
|
+
const description = typeof record["description"] === "string" ? record["description"] : "";
|
|
171
|
+
snapshot.descriptionBytes += description.length;
|
|
172
|
+
if (status === "completed") {
|
|
173
|
+
snapshot.completed += 1;
|
|
174
|
+
snapshot.completedDescriptionBytes += description.length;
|
|
175
|
+
} else if (status === "in_progress") snapshot.inProgress += 1;
|
|
176
|
+
else if (status === "pending") snapshot.pending += 1;
|
|
177
|
+
}
|
|
178
|
+
return snapshot;
|
|
179
|
+
}
|
|
180
|
+
function collectInvokedSkills(acc, record) {
|
|
181
|
+
const skills = record["skills"];
|
|
182
|
+
if (!Array.isArray(skills)) return;
|
|
183
|
+
for (const entry of skills) {
|
|
184
|
+
if (entry === null || typeof entry !== "object") continue;
|
|
185
|
+
const skill = entry;
|
|
186
|
+
const content = skill["content"];
|
|
187
|
+
if (typeof content !== "string" || content.length < LARGE_SKILL_BODY_BYTES) continue;
|
|
188
|
+
const name = skillName(skill);
|
|
189
|
+
if (name !== null) bump(acc.skillBodies, name, content.length);
|
|
190
|
+
}
|
|
191
|
+
}
|
|
192
|
+
function skillName(skill) {
|
|
193
|
+
const name = skill["name"];
|
|
194
|
+
if (typeof name === "string" && name.trim() !== "") return name.trim();
|
|
195
|
+
const path2 = skill["path"];
|
|
196
|
+
if (typeof path2 !== "string") return null;
|
|
197
|
+
const segments = path2.split(/[\\/]/).filter((part) => part !== "");
|
|
198
|
+
const last = segments[segments.length - 1];
|
|
199
|
+
return last === void 0 || last === "" ? null : last;
|
|
200
|
+
}
|
|
201
|
+
function accumulateResidentLine(acc, parsed, bytes) {
|
|
202
|
+
try {
|
|
203
|
+
if (parsed === null || typeof parsed !== "object") return;
|
|
204
|
+
const line = parsed;
|
|
205
|
+
if (line["type"] === "system" && line["subtype"] === "compact_boundary") acc.compactionCount += 1;
|
|
206
|
+
const attachment = line["attachment"];
|
|
207
|
+
if (attachment !== null && typeof attachment === "object") {
|
|
208
|
+
const record = attachment;
|
|
209
|
+
const type = typeof record["type"] === "string" ? record["type"] : "unknown";
|
|
210
|
+
bump(acc.attachmentsByType, type, bytes);
|
|
211
|
+
if (type === "task_reminder") {
|
|
212
|
+
acc.taskReminderCount += 1;
|
|
213
|
+
acc.taskReminderBytes += bytes;
|
|
214
|
+
acc.latestTaskList = readTaskList(record, bytes);
|
|
215
|
+
}
|
|
216
|
+
if (type === "invoked_skills") collectInvokedSkills(acc, record);
|
|
217
|
+
}
|
|
218
|
+
if (line["isMeta"] === true && line["type"] === "user") {
|
|
219
|
+
const text = messageText(line["message"]);
|
|
220
|
+
if (text.length >= LARGE_SKILL_BODY_BYTES) {
|
|
221
|
+
const skill = skillNameFromBody(text);
|
|
222
|
+
if (skill !== null) bump(acc.skillBodies, skill, text.length);
|
|
223
|
+
}
|
|
224
|
+
}
|
|
225
|
+
} catch {
|
|
226
|
+
}
|
|
227
|
+
}
|
|
228
|
+
function summarizeResidentContext(acc) {
|
|
229
|
+
const attachmentClasses = [...acc.attachmentsByType.entries()].map(([type, v]) => ({ type, count: v.count, bytes: v.bytes })).sort((a, b) => b.bytes - a.bytes || a.type.localeCompare(b.type));
|
|
230
|
+
const repeatedSkillBodies = [...acc.skillBodies.entries()].filter(([, v]) => v.count >= SKILL_BODY_REPEAT_THRESHOLD).map(([skill, v]) => ({ skill, count: v.count, bytes: v.bytes })).sort((a, b) => b.bytes - a.bytes || a.skill.localeCompare(b.skill));
|
|
231
|
+
return {
|
|
232
|
+
attachmentClasses,
|
|
233
|
+
totalAttachmentBytes: attachmentClasses.reduce((n, c) => n + c.bytes, 0),
|
|
234
|
+
latestTaskList: acc.latestTaskList,
|
|
235
|
+
taskReminderCount: acc.taskReminderCount,
|
|
236
|
+
taskReminderBytes: acc.taskReminderBytes,
|
|
237
|
+
repeatedSkillBodies,
|
|
238
|
+
compactionCount: acc.compactionCount
|
|
239
|
+
};
|
|
240
|
+
}
|
|
241
|
+
function formatBytes(bytes) {
|
|
242
|
+
if (bytes >= 1e6) return `${(bytes / 1048576).toFixed(1)} MB`;
|
|
243
|
+
if (bytes >= 1e3) return `${Math.round(bytes / 1024)} KB`;
|
|
244
|
+
return `${bytes} B`;
|
|
245
|
+
}
|
|
246
|
+
function formatTokenEstimate(tokens) {
|
|
247
|
+
return tokens >= 1e3 ? `${Math.round(tokens / 1e3)}K` : String(tokens);
|
|
248
|
+
}
|
|
249
|
+
function taskListPruneHint(snapshot) {
|
|
250
|
+
if (snapshot === null) return null;
|
|
251
|
+
if (snapshot.bytes < LARGE_TASK_LIST_BYTES) return null;
|
|
252
|
+
if (snapshot.completed === 0) return null;
|
|
253
|
+
const tokens = formatTokenEstimate(estimateTokensFromLength(snapshot.bytes));
|
|
254
|
+
const prunable = formatBytes(snapshot.completedDescriptionBytes);
|
|
255
|
+
return `Your task list is ${formatBytes(snapshot.bytes)} (~${tokens} tok est) and is re-injected in full whenever it changes; ${snapshot.completed} of ${snapshot.itemCount} items are already completed and their descriptions alone are ${prunable}. Prune completed items or shorten their descriptions with TaskUpdate. Check ownership first if other agents share this list.`;
|
|
256
|
+
}
|
|
257
|
+
function repeatedSkillBodyHint(injections) {
|
|
258
|
+
const worst = injections[0];
|
|
259
|
+
if (worst === void 0) return null;
|
|
260
|
+
if (worst.count < SKILL_BODY_REPEAT_THRESHOLD || worst.bytes < LARGE_SKILL_BODY_BYTES) return null;
|
|
261
|
+
const skill = displaySafeText(worst.skill);
|
|
262
|
+
const tokens = formatTokenEstimate(estimateTokensFromLength(worst.bytes));
|
|
263
|
+
return `The \`${skill}\` skill body has been injected ${worst.count} times this session (${formatBytes(worst.bytes)} total, ~${tokens} tok est). Slash expansion and the Skill tool both send the whole body every time, and no hook can intercept either. If it is already loaded, work from it instead of re-invoking; to reread one part, use \`token-goat skill-section ${skill} '<heading>'\`.`;
|
|
264
|
+
}
|
|
265
|
+
function readTranscriptTail(transcriptPath, maxBytes = RESIDENT_TAIL_MAX_BYTES) {
|
|
266
|
+
let fd = null;
|
|
267
|
+
try {
|
|
268
|
+
const stat = fs.statSync(transcriptPath);
|
|
269
|
+
if (!stat.isFile() || stat.size === 0) return [];
|
|
270
|
+
const start = Math.max(0, stat.size - maxBytes);
|
|
271
|
+
const length = stat.size - start;
|
|
272
|
+
fd = fs.openSync(transcriptPath, "r");
|
|
273
|
+
const buf = Buffer.allocUnsafe(length);
|
|
274
|
+
const read = fs.readSync(fd, buf, 0, length, start);
|
|
275
|
+
const lines = buf.subarray(0, read).toString("utf8").split("\n");
|
|
276
|
+
if (start > 0) lines.shift();
|
|
277
|
+
return lines;
|
|
278
|
+
} catch {
|
|
279
|
+
return [];
|
|
280
|
+
} finally {
|
|
281
|
+
if (fd !== null) {
|
|
282
|
+
try {
|
|
283
|
+
fs.closeSync(fd);
|
|
284
|
+
} catch {
|
|
285
|
+
}
|
|
286
|
+
}
|
|
287
|
+
}
|
|
288
|
+
}
|
|
289
|
+
function lineMayCarryResidentSignal(line) {
|
|
290
|
+
return line.includes('"task_reminder"') || line.includes('"compact_boundary"') || line.includes('"invoked_skills"') || line.includes('"isMeta"');
|
|
291
|
+
}
|
|
292
|
+
function accumulateResidentLines(lines) {
|
|
293
|
+
const acc = createResidentContextStats();
|
|
294
|
+
for (const line of lines) {
|
|
295
|
+
const trimmed = line.trim();
|
|
296
|
+
if (trimmed === "") continue;
|
|
297
|
+
let parsed;
|
|
298
|
+
try {
|
|
299
|
+
parsed = JSON.parse(trimmed);
|
|
300
|
+
} catch {
|
|
301
|
+
continue;
|
|
302
|
+
}
|
|
303
|
+
accumulateResidentLine(acc, parsed, trimmed.length);
|
|
304
|
+
}
|
|
305
|
+
return acc;
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
export {
|
|
309
|
+
claudeConfigDir,
|
|
310
|
+
estimateTokensFromLength,
|
|
311
|
+
estimateTokens,
|
|
312
|
+
trimToBudget,
|
|
313
|
+
capJsonRows,
|
|
314
|
+
createResidentContextStats,
|
|
315
|
+
accumulateResidentLine,
|
|
316
|
+
summarizeResidentContext,
|
|
317
|
+
formatBytes,
|
|
318
|
+
formatTokenEstimate,
|
|
319
|
+
taskListPruneHint,
|
|
320
|
+
repeatedSkillBodyHint,
|
|
321
|
+
readTranscriptTail,
|
|
322
|
+
lineMayCarryResidentSignal,
|
|
323
|
+
accumulateResidentLines
|
|
324
|
+
};
|
package/dist/token-goat-hook.mjs
CHANGED
|
@@ -2,21 +2,27 @@ import { createRequire as __cjsRequire } from 'node:module';
|
|
|
2
2
|
const require = __cjsRequire(import.meta.url);
|
|
3
3
|
import {
|
|
4
4
|
relayInProcess
|
|
5
|
-
} from "./token-goat-chunk-
|
|
6
|
-
import "./token-goat-chunk-
|
|
7
|
-
import "./token-goat-chunk-
|
|
8
|
-
import "./token-goat-chunk-
|
|
9
|
-
import "./token-goat-chunk-
|
|
10
|
-
import "./token-goat-chunk-
|
|
5
|
+
} from "./token-goat-chunk-Z4ALQADF.mjs";
|
|
6
|
+
import "./token-goat-chunk-ITWYDOKP.mjs";
|
|
7
|
+
import "./token-goat-chunk-I5PQRHFE.mjs";
|
|
8
|
+
import "./token-goat-chunk-BQIC3AVL.mjs";
|
|
9
|
+
import "./token-goat-chunk-VSJGPEBB.mjs";
|
|
10
|
+
import "./token-goat-chunk-DPHJU3P7.mjs";
|
|
11
|
+
import "./token-goat-chunk-PAVWB4L4.mjs";
|
|
12
|
+
import "./token-goat-chunk-SP5QR2HN.mjs";
|
|
13
|
+
import "./token-goat-chunk-KT75DZK4.mjs";
|
|
14
|
+
import "./token-goat-chunk-ZHFM4DHJ.mjs";
|
|
15
|
+
import "./token-goat-chunk-SL454EN5.mjs";
|
|
16
|
+
import "./token-goat-chunk-K5XPIUVB.mjs";
|
|
11
17
|
import "./token-goat-chunk-3BTK54F3.mjs";
|
|
12
|
-
import "./token-goat-chunk-
|
|
13
|
-
import "./token-goat-chunk-
|
|
14
|
-
import "./token-goat-chunk-
|
|
15
|
-
import "./token-goat-chunk-
|
|
18
|
+
import "./token-goat-chunk-ERGHDBC6.mjs";
|
|
19
|
+
import "./token-goat-chunk-K3ZNBN4U.mjs";
|
|
20
|
+
import "./token-goat-chunk-T63VHOL3.mjs";
|
|
21
|
+
import "./token-goat-chunk-74ADIZZX.mjs";
|
|
16
22
|
import "./token-goat-chunk-EEIDFMEM.mjs";
|
|
17
23
|
import "./token-goat-chunk-OSUFN2FV.mjs";
|
|
18
|
-
import "./token-goat-chunk-
|
|
19
|
-
import "./token-goat-chunk-
|
|
24
|
+
import "./token-goat-chunk-6LJNM4ZZ.mjs";
|
|
25
|
+
import "./token-goat-chunk-NMTKNYGF.mjs";
|
|
20
26
|
import "./token-goat-chunk-Y4AFKTHK.mjs";
|
|
21
27
|
import "./token-goat-chunk-GMOUBOX4.mjs";
|
|
22
28
|
import "./token-goat-chunk-RRNZMM3A.mjs";
|
package/dist/token-goat.core.mjs
CHANGED
|
@@ -2,34 +2,43 @@ import { createRequire as __cjsRequire } from 'node:module';
|
|
|
2
2
|
const require = __cjsRequire(import.meta.url);
|
|
3
3
|
import {
|
|
4
4
|
run
|
|
5
|
-
} from "./token-goat-chunk-
|
|
6
|
-
import "./token-goat-chunk-
|
|
7
|
-
import "./token-goat-chunk-
|
|
8
|
-
import "./token-goat-chunk-
|
|
9
|
-
import "./token-goat-chunk-
|
|
10
|
-
import "./token-goat-chunk-
|
|
11
|
-
import "./token-goat-chunk-
|
|
12
|
-
import "./token-goat-chunk-
|
|
5
|
+
} from "./token-goat-chunk-2Q3N5KUQ.mjs";
|
|
6
|
+
import "./token-goat-chunk-ONMU7BON.mjs";
|
|
7
|
+
import "./token-goat-chunk-EUZKG7ZO.mjs";
|
|
8
|
+
import "./token-goat-chunk-W6QYCWAT.mjs";
|
|
9
|
+
import "./token-goat-chunk-3VTJC7KU.mjs";
|
|
10
|
+
import "./token-goat-chunk-MNJWG2ZO.mjs";
|
|
11
|
+
import "./token-goat-chunk-DA43OE7L.mjs";
|
|
12
|
+
import "./token-goat-chunk-I5PQRHFE.mjs";
|
|
13
|
+
import "./token-goat-chunk-BQIC3AVL.mjs";
|
|
14
|
+
import "./token-goat-chunk-VSJGPEBB.mjs";
|
|
15
|
+
import "./token-goat-chunk-DPHJU3P7.mjs";
|
|
16
|
+
import "./token-goat-chunk-PAVWB4L4.mjs";
|
|
17
|
+
import "./token-goat-chunk-SP5QR2HN.mjs";
|
|
18
|
+
import "./token-goat-chunk-KT75DZK4.mjs";
|
|
19
|
+
import "./token-goat-chunk-ZHFM4DHJ.mjs";
|
|
20
|
+
import "./token-goat-chunk-SL454EN5.mjs";
|
|
21
|
+
import "./token-goat-chunk-K5XPIUVB.mjs";
|
|
13
22
|
import "./token-goat-chunk-LKSXAMJB.mjs";
|
|
14
|
-
import "./token-goat-chunk-
|
|
15
|
-
import "./token-goat-chunk-
|
|
16
|
-
import "./token-goat-chunk-
|
|
17
|
-
import "./token-goat-chunk-
|
|
23
|
+
import "./token-goat-chunk-PTW22MRC.mjs";
|
|
24
|
+
import "./token-goat-chunk-DNAF26L4.mjs";
|
|
25
|
+
import "./token-goat-chunk-ODF7H7WS.mjs";
|
|
26
|
+
import "./token-goat-chunk-P4XBHZXU.mjs";
|
|
18
27
|
import "./token-goat-chunk-3BTK54F3.mjs";
|
|
19
|
-
import "./token-goat-chunk-
|
|
28
|
+
import "./token-goat-chunk-T46UKRFM.mjs";
|
|
20
29
|
import "./token-goat-chunk-XTQAOTSO.mjs";
|
|
21
30
|
import "./token-goat-chunk-AH6QILZM.mjs";
|
|
22
|
-
import "./token-goat-chunk-
|
|
23
|
-
import "./token-goat-chunk-
|
|
24
|
-
import "./token-goat-chunk-
|
|
25
|
-
import "./token-goat-chunk-
|
|
26
|
-
import "./token-goat-chunk-
|
|
31
|
+
import "./token-goat-chunk-MUJV7BR3.mjs";
|
|
32
|
+
import "./token-goat-chunk-ERGHDBC6.mjs";
|
|
33
|
+
import "./token-goat-chunk-K3ZNBN4U.mjs";
|
|
34
|
+
import "./token-goat-chunk-T63VHOL3.mjs";
|
|
35
|
+
import "./token-goat-chunk-74ADIZZX.mjs";
|
|
27
36
|
import "./token-goat-chunk-EEIDFMEM.mjs";
|
|
28
37
|
import "./token-goat-chunk-OSUFN2FV.mjs";
|
|
29
38
|
import {
|
|
30
39
|
installEpipeGuard
|
|
31
|
-
} from "./token-goat-chunk-
|
|
32
|
-
import "./token-goat-chunk-
|
|
40
|
+
} from "./token-goat-chunk-6LJNM4ZZ.mjs";
|
|
41
|
+
import "./token-goat-chunk-NMTKNYGF.mjs";
|
|
33
42
|
import "./token-goat-chunk-Y4AFKTHK.mjs";
|
|
34
43
|
import "./token-goat-chunk-GMOUBOX4.mjs";
|
|
35
44
|
import "./token-goat-chunk-RRNZMM3A.mjs";
|
|
@@ -191,7 +191,7 @@ flowchart TB
|
|
|
191
191
|
subgraph ParserAdapters ["Parser Language Adapters"]
|
|
192
192
|
TreeSitter["Inline Tree-Sitter Extractors<br/><small>TS, JS, Python, Go, Rust, Java, C/C++, Ruby</small>"]:::adapter
|
|
193
193
|
RegexInline["Inline Regex Extractors<br/><small>Markdown, JSON, YAML, TOML, CSS, Dockerfile</small>"]:::adapter
|
|
194
|
-
LangAdapters["Regex Adapters, inline and src/languages/ (
|
|
194
|
+
LangAdapters["Regex Adapters, inline and src/languages/ (77 non-tree-sitter languages)<br/><small>C#, PHP, Kotlin, GraphQL, SQL, Proto, Apex, Nginx, Caddyfile, Apache, etc.</small>"]:::adapter
|
|
195
195
|
end
|
|
196
196
|
|
|
197
197
|
Parser --> TreeSitter
|
package/docs/cli.md
CHANGED
|
@@ -58,6 +58,7 @@ token-goat pdf-extract manual.pdf --pages 12-15 --layout --head 120
|
|
|
58
58
|
| `token-goat call-chain <symbol>` | Trace every caller layer from a symbol back to the entry points — one step deeper than `callers`. Use when you need to know what reaches a function across the whole call graph, not only who invokes it directly. Pairs with `impact` for the downstream direction. Accepts `file::symbol` to disambiguate WHICH same-named definition the chain starts from — the file only narrows which definition, callers can still be found in any file. `--exclude-tests` prunes callers whose call site is a test file BEFORE they're admitted to the traversal, so nothing walks through a test node either (opt-in — omitted, output is unchanged); when every caller was a test, the "no callers" line names how many were hidden instead of reading as genuinely unreferenced. `Symbol not found: <symbol>` for an unindexed name (bare or `file::symbol`) now carries a `Did you mean:` suggestion when a near-name candidate is indexed. `--grep <pattern>` keeps only completed chains containing a symbol name matching this regex (literal substring if it is not valid regex) — the BFS still walks the full graph, this only narrows which finished chains are reported, so a chain passing through a matching symbol on its way to an unrelated root still surfaces; when it matches none of the chains that do exist, the output names how many were filtered out rather than reading as genuinely caller-less. A chain the walk abandoned because it ran out of `--depth` ends in a `(depth-limit)` marker, so a truncated chain is never mistaken for one that reached a real entry point. |
|
|
59
59
|
| `token-goat impact <symbol>` | Walk the call-reference graph forward (breadth-first) and list every function that depends on a symbol, with hop depth; module-scope callers are surfaced as `(module scope) <file>` entries. Run before a refactor to size up the blast radius without starting a build. Accepts `file::symbol` to disambiguate WHICH same-named definition the walk starts from — the file only narrows which definition, callers can still be found in any file. `--exclude-tests` prunes callers (including module-scope entries) whose call site is a test file BEFORE they're enqueued for further traversal, so nothing walks through a test node either (opt-in — omitted, output is unchanged); when every caller was a test, the "no callers found" error names how many were hidden instead of reading as genuinely unreferenced. A bare name that isn't indexed at all reports `Symbol not found: <name>` (with a `Did you mean:` suggestion when a near-name candidate is indexed) instead of the misleading "no callers found", which is reserved for a real, indexed symbol that genuinely has zero impact. `--grep <pattern>` only shows impacted entries whose symbol name (or `(module scope) <file>` key) matches this regex (literal substring if it is not valid regex), the same filter `call-chain --grep`/`dead --grep` apply to their own results — applied BEFORE the `--top` slice, so it selects from the whole impacted set rather than an already-capped page; when it matches none of the impacted entries that do exist, the output names how many were filtered out instead of reading as genuinely impact-free. |
|
|
60
60
|
| `token-goat context-for <task>` | Takes a natural-language task description, runs semantic search across the indexed codebase, and emits a prioritized list of `token-goat read` commands trimmed to a token budget. Fetches only the relevant slices instead of loading entire files. `--budget N` sets the token ceiling; `--top N` limits the file count; `--json` for structured output. Every emitted command carries the `file::symbol@LINE` anchor, so a suggestion still runs when the same symbol name has more than one definition in its file; `--json` entries carry the matching `line` field. |
|
|
61
|
+
| `token-goat answer "<question>"` | Deterministic question router: classifies a plain-English question to one of six intents (where a symbol is, who calls it, what tests cover it, what a file exports, what a file imports, what breaks if a symbol changes), resolves the subject against the index, and delegates in-process to the command that already answers it. No model call, no inference, no fallback to a fuzzy match. Because the subject is looked up before anything runs, the router picks the right ARGUMENT FORM too: `answer "what tests cover foldPath"` resolves `foldPath` to `src/path_containment.ts` and runs `test-for` against the file, where passing a symbol to `test-for` directly fails. A bare basename resolves when exactly one file in the project carries it, and lists the candidates when several do. Every answer is preceded by `via: token-goat <command> <argument>` so the delegate can be verified or re-run with its own flags. Anything else refuses: `cannot answer deterministically: <why>; try: <command>`, exit 1. Questions of judgement, intent or runtime behaviour (`why does X...`, `is X correct`, `should I...`) refuse even when they name a real symbol. |
|
|
61
62
|
| `token-goat ask "<question>"` *(experimental)* | Retrieves relevant slices via full-text (BM25) search over the symbol index — not semantic/embedding search — and lists them as pointer-citations plus `token-goat read` commands. Set `TOKEN_GOAT_ASK_BACKEND=claude` or `TOKEN_GOAT_ASK_BACKEND=codex` to synthesize a short answer via that CLI (whatever model it defaults to; token-goat does not force Haiku or any particular tier); with the env var unset, or the named CLI missing from PATH, `ask` degrades to printing the retrieved pointers with no network call. `--top N` caps the number of FTS hits (default 8); `--json` for structured output. Answers are not cached — each call re-retrieves and re-synthesizes from scratch. Every emitted command carries the `file::symbol@LINE` anchor, so a suggestion still runs when the same symbol name has more than one definition in its file; `--json` entries carry the matching `line` field. |
|
|
62
63
|
| `token-goat changed [<ref>]` | List files (or `--symbol` for symbols) changed since a git ref, without reading the full diff. `<ref>` and `--since <ref>` are equivalent (default `HEAD~5`); `--since` wins if both are given. `--json` for structured output. `--grep <pattern>` only lists changed files whose path matches this regex (literal substring if it is not valid regex) — applied to the file list even in `--symbol` mode, before any downstream slicing; when it matches none of the files that did change, the output names the active filter instead of reading like nothing changed. `--exclude-tests` hides changed files that live in a test file (opt-in — omitted, output is unchanged), completing the flag family already on `refs`/`callers`/`dead`/`call-chain`/`impact`/`semantic`/`symbol`. `--grep` can only ever *select* a path, so there was no reliable way to ask for the non-test half of a diff: the negative-lookahead regex that expresses "not a test" silently degrades to a literal substring match whenever the regex-compile fallback fires. Test files are a large share of a typical diff — measured against this repo, 35–54% of changed files across the last 5, 10 and 20 commits. Like `--grep` it filters the file path, so it applies in `--symbol` mode too, and it prunes before the per-file index lookup rather than after, so a test file is never queried at all. Composable with `--grep` (a file must satisfy both; when both are active and `--grep` is what emptied the list, the `--grep` notice takes priority, and when `--grep` left only test files so that `--exclude-tests` emptied it, the message names both filters rather than claiming no non-test file changed). When it hides every changed file there was, the output names how many were hidden and exits 0, rather than a bare "No files changed." that would read as a clean diff. Every zero-row path emits the shared `{items, truncated, totalCount}` envelope under `--json`. |
|
|
63
64
|
| `token-goat diff "file::symbol" [range]` | Show only the git diff hunk(s) that fall within one symbol's line range, e.g. `token-goat diff "file.ts::myFn" HEAD~3..HEAD`, instead of the whole file's diff. Also accepts `read`'s `symbol@LINE` anchor to pick out an otherwise-ambiguous candidate. |
|
|
@@ -81,7 +82,7 @@ token-goat pdf-extract manual.pdf --pages 12-15 --layout --head 120
|
|
|
81
82
|
| `token-goat coverage-gaps` | Find callables in non-test source files that never appear in a test file's reference records. Useful for spotting untested surface area before a refactor or release. `--top N` caps output; `--json` for structured output. |
|
|
82
83
|
| `token-goat recent [N]` | Show the N most recently edited/accessed files with their symbols. |
|
|
83
84
|
| `token-goat grep "<pattern>" [paths...]` | Built-in fallback regex search over files (no `rg` shell-out, no caching) — session-aware dedup for raw `rg`/`grep` Bash calls is a separate hook, not this command. Accepts zero or more paths: omit to walk cwd, or pass several to search them together with hits merged in argument order under one `--max-lines` cap. `-C, --context <n>` shows `n` lines before and after each match. `--symbol` annotates each hit with its enclosing indexed symbol — ` [name (kind)]` appended in text mode, a `symbol: {name, kind, lineStart, lineEnd} | null` field per item under `--json` — `null`/no tag when the hit falls outside any indexed symbol (e.g. module-level code). |
|
|
84
|
-
| `token-goat semantic "<query>"` | Find code by meaning, not by filename: embedding-vector similarity search over indexed file chunks and full-text search (BM25) over symbol names/bodies both run on every query and are fused by Reciprocal Rank Fusion (`score = sum of 1/(60 + rank)` per list), so an exact keyword match can outrank a weak vector hit instead of being shadowed by the vector branch. Covers extracted text from PDF/DOCX/PPTX/XLSX files alongside source code, so a query can surface a spec PDF, design doc, deck, or spreadsheet, not just code. Results are re-ranked with a path-priority multiplier so live source wins ties/near-ties against stale or archival prose (`archive/`, `archived/`, `old/`, `deprecated/`, `plans/`, `drafts/`, `CHANGELOG*`, `*.bak`, `*.orig`) and, more mildly, general docs (`docs/**`, `*.md`) — a nudge, not a hard filter, so a genuinely much better archival match still surfaces. Configure with `token-goat config set semantic.archive_weight <0-1>` / `token-goat config set semantic.docs_weight <0-1>` (both default `<1`; set to `1` to disable that penalty entirely, e.g. for a project with a genuinely live `plans/` directory). `--preflight` runs an advance diagnostic verification of embedding runtime availability, model weight cache presence, and project coverage without executing a query; add `--warm` to load/cache the ONNX model into memory. `--limit <n>` caps result count; `--json` for structured output: `{source, items, truncated, totalCount}`, where `source` is `"hybrid"` when both the embedding and BM25 branches contributed at least one raw hit, `"embeddings"` when only the embedding branch did (e.g. no vector index exists yet: optional embedding deps unavailable, or `indexing.embeddings_enabled` is off), or `"fts"` when only BM25 did, and every item carries the same keys (`filePath`, `name`, `kind`, `startLine`, `endLine`, `distance`, `preview`), with `null` for whichever of `name`/`kind`/`distance` don't apply to that item's source, and `filePath` rendered root-relative when a project root resolves, absolute when none does, matching plain-text output. On an embeddings hit, `name`/`kind` are resolved to the innermost indexed symbol whose line range contains the hit's start line (`null`/`null` when the hit falls outside any symbol, e.g. a top-of-file imports chunk); text output appends the same as a `— inside <name> (<kind>)` suffix. `--grep <pattern>` only shows hits whose FILE PATH matches this regex (literal substring if it is not valid regex), tested against the path exactly as rendered (so an anchored `--grep "^src/"` matches what you see, not the stored absolute path) — the high-value case is dropping test/vendored noise from a project-wide semantic hit list. Applied before the `--limit` slice in both the embeddings and full-text-fallback branches, so it selects from the whole hit set, not an already-capped page; when it matches nothing among hits that do exist, the output names the active filter (and `--json` sets `grepFilteredToEmpty: true`) instead of reading like the search found nothing. `--exclude-tests` hides hits whose file is a test file (opt-in — omitted, output is unchanged), covering the case `--grep` structurally cannot: `--grep` can only ever *select* a path, and the negative-lookahead pattern that would express "not a test" silently degrades to a literal substring match whenever the regex-compile fallback fires. Applied before the `--limit` slice in both branches and composable with `--grep` (a hit must satisfy both); when it hides every hit there was, the output names how many were hidden (and `--json` sets `excludeTestsFilteredToEmpty: true`) and exits 0, instead of the exit-1 "no matches" a genuinely empty search returns. With both filters set and both emptying the view, the `--grep` notice takes priority. |
|
|
85
|
+
| `token-goat semantic "<query>"` | Find code by meaning, not by filename: embedding-vector similarity search over indexed file chunks and full-text search (BM25) over symbol names/bodies both run on every query and are fused by Reciprocal Rank Fusion (`score = sum of 1/(60 + rank)` per list), so an exact keyword match can outrank a weak vector hit instead of being shadowed by the vector branch. Covers extracted text from PDF/DOCX/PPTX/XLSX files alongside source code, so a query can surface a spec PDF, design doc, deck, or spreadsheet, not just code. Results are re-ranked with a path-priority multiplier so live source wins ties/near-ties against stale or archival prose (`archive/`, `archived/`, `old/`, `deprecated/`, `plans/`, `drafts/`, `CHANGELOG*`, `*.bak`, `*.orig`) and, more mildly, general docs (`docs/**`, `*.md`) — a nudge, not a hard filter, so a genuinely much better archival match still surfaces. Configure with `token-goat config set semantic.archive_weight <0-1>` / `token-goat config set semantic.docs_weight <0-1>` (both default `<1`; set to `1` to disable that penalty entirely, e.g. for a project with a genuinely live `plans/` directory). `token-goat config set semantic.max_distance <0.05-2>` is the relevance floor: a vector hit whose raw distance exceeds it is dropped before fusion, so a corpus whose own distances have been measured can stop the vector half answering a question it has nothing for. It defaults to `1.2`, the retrieval bound's own value, so out of the box it filters nothing — measured genuine matches run from 0.635 on a large index to 0.934 on a two-file one, overlapping the off-corpus band, so no fixed number separates them across corpus sizes and the default is deliberately inert. When the floor empties the vector half, the command says so and names the closest distance it rejected, so the number to change is visible rather than inferred. The key is refused from a project's own `.token-goat.toml` (see `docs/security.md`), since a checked-in file setting it near the minimum would withhold that repository's code from the vector half while BM25 went on answering. `--preflight` runs an advance diagnostic verification of embedding runtime availability, model weight cache presence, and project coverage without executing a query; add `--warm` to load/cache the ONNX model into memory. `--limit <n>` caps result count; `--json` for structured output: `{source, items, truncated, totalCount}`, where `source` is `"hybrid"` when both the embedding and BM25 branches contributed at least one raw hit, `"embeddings"` when only the embedding branch did (e.g. no vector index exists yet: optional embedding deps unavailable, or `indexing.embeddings_enabled` is off), or `"fts"` when only BM25 did, and every item carries the same keys (`filePath`, `name`, `kind`, `startLine`, `endLine`, `distance`, `preview`, `rank`, `rrf`, `retrieval`), with `null` for whichever of `name`/`kind`/`distance` don't apply to that item's source. `rank` is the item's 1-based position, `rrf` the fused score it was sorted on, and `retrieval` is `"dense"`, `"lexical"` or `"both"` — together they are what makes the ordering reproducible, since `distance` is only the vector leg's own score and a lexical-only item has none at all, so an item at distance 0.850 legitimately sits above one at 0.776 when the keyword pass voted for the first as well, and `filePath` rendered root-relative when a project root resolves, absolute when none does, matching plain-text output. On an embeddings hit, `name`/`kind` are resolved to the innermost indexed symbol whose line range contains the hit's start line (`null`/`null` when the hit falls outside any symbol, e.g. a top-of-file imports chunk); text output appends the same as a `— inside <name> (<kind>)` suffix. `--grep <pattern>` only shows hits whose FILE PATH matches this regex (literal substring if it is not valid regex), tested against the path exactly as rendered (so an anchored `--grep "^src/"` matches what you see, not the stored absolute path) — the high-value case is dropping test/vendored noise from a project-wide semantic hit list. Applied before the `--limit` slice in both the embeddings and full-text-fallback branches, so it selects from the whole hit set, not an already-capped page; when it matches nothing among hits that do exist, the output names the active filter (and `--json` sets `grepFilteredToEmpty: true`) instead of reading like the search found nothing. `--exclude-tests` hides hits whose file is a test file (opt-in — omitted, output is unchanged), covering the case `--grep` structurally cannot: `--grep` can only ever *select* a path, and the negative-lookahead pattern that would express "not a test" silently degrades to a literal substring match whenever the regex-compile fallback fires. Applied before the `--limit` slice in both branches and composable with `--grep` (a hit must satisfy both); when it hides every hit there was, the output names how many were hidden (and `--json` sets `excludeTestsFilteredToEmpty: true`) and exits 0, instead of the exit-1 "no matches" a genuinely empty search returns. With both filters set and both emptying the view, the `--grep` notice takes priority. |
|
|
85
86
|
| `token-goat map` | Get a compact orientation of the repo. Add `--compact` to fit a fixed 2000-token budget. `--json` emits the project map as JSON instead of text. |
|
|
86
87
|
| `token-goat deps "file"` | One-level import listing for a single file: resolves relative imports to project files (`internal`, root-relative paths) and groups everything else as `external`. `--json` for structured output. `--grep <pattern>` only shows dependencies whose MODULE SPECIFIER (the resolved internal path or the external package name) matches this regex (literal substring if it is not valid regex), applied before output is built; when it matches nothing among real dependencies, the output names the active filter instead of reading like the file has no imports at all. Complemented by `token-goat arch` for the project-wide graph. |
|
|
87
88
|
| `token-goat arch` | Project-wide import graph summary: hub modules (most imported), entry points (nothing imports them), and circular chains. `--modules` adds a grouping of the files that mostly import each other, naming each group by its most connected file, saying whether the group is one directory or spread across several, and listing which groups reach into which. That section also prints the grouping's modularity and calls it out when it is too weak to mean anything, since the algorithm returns groups for any graph, including one with no real structure. `--json` carries each group's full member list. Complements `token-goat deps <file>` for per-file depth. |
|
|
@@ -98,7 +99,7 @@ token-goat pdf-extract manual.pdf --pages 12-15 --layout --head 120
|
|
|
98
99
|
| `token-goat waste [--project <path>] [--transcript <path>] [--top <n>] [--json] [--copilot]` | Session spend-ledger: parses the current project's Claude Code session transcript and reports token cost by tool, by file, the top N most expensive individual tool calls, files read once and never referenced again, Bash commands run repeatedly without hitting token-goat's own bash-output cache, and the assistant's own text-output cost (generated tokens plus a cache-unaware re-send upper bound). See [Session waste ledger](#session-waste-ledger) below. |
|
|
99
100
|
| `token-goat mcp-audit [--project <path>] [--json]` | MCP server schema cost report: scans .mcp.json for installed MCP servers, estimates per-server token costs from cached tool calls, correlates schema complexity against real call frequency. Outputs as markdown table or JSON. |
|
|
100
101
|
| `token-goat recall ["<query>"] [--type bash\|web\|mcp] [--limit <n>] [--json]` | Full-text search across every cached bash-output, web-output, and mcp-output entry at once — one command instead of remembering which cache type holds a prior result. Ranked by relevance (BM25 via SQLite FTS5). With **no query**, lists every cached entry newest-first instead of searching, so you can browse when the ids have scrolled out of context and you have no term to search for. `--type` narrows to one cache type; `--limit` caps results (default 10). Each hit shows its cache type, id, the exact recall command (`bash-output <id>` / `web-output <id>` / `mcp-output <id>`), and a content snippet. See [Cross-cache recall](#cross-cache-recall) below. |
|
|
101
|
-
| `token-goat hint-stats [--json] [--reset] [--mark-effective <cat>] [--mark-ineffective <cat>]` | Per-category efficacy report for token-goat's discretionary hint hooks: how often each hint category was emitted, how often the agent actually followed its specific suggestion within the next few tool calls, whether the category is currently auto-suppressed, and the bytes each category spent (injected into context) plus an all-time saved/spent
|
|
102
|
+
| `token-goat hint-stats [--json] [--reset] [--mark-effective <cat>] [--mark-ineffective <cat>]` | Per-category efficacy report for token-goat's discretionary hint hooks: how often each hint category was emitted, how often the agent actually followed its specific suggestion within the next few tool calls, whether the category is currently auto-suppressed, and the bytes each category spent (injected into context) plus an all-time saved-bytes/spent-bytes summary line (the two cover disjoint populations and are deliberately never netted against each other). `--reset` clears all tracked data; `--mark-effective`/`--mark-ineffective <category>` record a manual vote as a supplement to the automatic signal. See [Hint efficacy tracking](#hint-efficacy-tracking) below. |
|
|
102
103
|
| `token-goat history` | Show current session access history: bash commands and URLs fetched. |
|
|
103
104
|
| `token-goat session-outline` | Turn-by-turn structure (role, preview, tool calls, approx size) of a Claude Code session JSONL transcript, instead of a raw Read; defaults to the current project's most recent session. |
|
|
104
105
|
| `token-goat session-slice <turns>` | Full content of one turn range from a Claude Code session JSONL transcript (see `session-outline` for turn numbers), instead of a raw Read. |
|
|
@@ -151,6 +152,7 @@ token-goat pdf-extract manual.pdf --pages 12-15 --layout --head 120
|
|
|
151
152
|
| `token-goat reclaim-index` | Shrink an oversized symbol index (`VACUUM` + WAL checkpoint). `--rebuild` also drops every derived row — files/symbols/refs/chunks — so the next `token-goat index` re-derives them under current parser rules, which is what actually reclaims space held by rows a since-fixed extractor wrote too large. Refuses to run while the worker daemon is writing to the index unless `--force`. `token-goat doctor` points here when `global.db` grows past 1 GB and at least a tenth of it is free pages (it points to `project prune` instead for scratch files indexed under the OS temp dir), and separately when it finds a stored symbol body above the parser's own size cap — a leftover from a since-fixed extractor bug, which only `--rebuild` can clear (a plain `VACUUM` reclaims freed pages but never deletes row content). |
|
|
152
153
|
| `token-goat prune-cache` | Manually trigger LRU eviction across all cache directories (images, bash, web, skills). |
|
|
153
154
|
| `token-goat session-summary` | Compact one-liner about current session state — designed for orchestrators and multi-agent loops. |
|
|
155
|
+
| `token-goat audit` | Run a session audit on recent agent activity, tracking surgical reads, token burn, cache hits, tool calls, and missed savings across Claude Code and Copilot CLI sessions. `--copilot` scopes to Copilot sessions; `--top <n>` limits item breakdowns; `--json` outputs structured audit data. |
|
|
154
156
|
| `token-goat cache-audit` | Audit your Claude Code config for patterns that bust the prompt cache. |
|
|
155
157
|
| `token-goat pack <patterns>` | Collect files matching glob patterns into a single LLM-ready output — Markdown (default), XML, or plain text — with a manifest table of per-file line and token counts. `--line-numbers` prefixes each line; `--instruction-file` appends a task prompt; `--output` writes to a file; `--no-ignore` bypasses `.tokengoatignore`. `--strip-comments` removes language-appropriate comments before packing (shebangs preserved; `#` inside string literals is a known limitation). `--scan-secrets` checks for credentials and exits 2 with per-file warnings if any are found. `--budget N` exits 3 when the estimated token count exceeds N — lets a shell script treat an oversized context as a hard error. Reads file paths from stdin when no patterns are given. |
|
|
156
158
|
| `token-goat budget <patterns>` | Estimate the token cost of a file set without reading them into context. Prints results sorted by cost descending; `--context <N>` shows each file as a share of an N-thousand-token window. `--json` for machine-readable output. Run before `pack` to decide what to include. |
|
|
@@ -177,7 +179,7 @@ token-goat pdf-extract manual.pdf --pages 12-15 --layout --head 120
|
|
|
177
179
|
| `token-goat project prune [--dry-run]` | Remove blocked/excluded roots that no longer exist on disk, and index rows for files under the OS temp dir. `--dry-run` previews removals without touching the config file. Useful after deleting or moving projects. |
|
|
178
180
|
| `token-goat install` | Wire up hooks (and, with the harness flags below, other AI tool integrations). No `--dry-run` or `--verify` flag — run `token-goat doctor` after install to audit the result. |
|
|
179
181
|
| `token-goat upgrade` | Check for updates or upgrade token-goat to the latest version via npm. Pass `--check` to report status without installing. `--json` emits machine-readable version status. |
|
|
180
|
-
| `token-goat doctor` | Confirm everything is wired correctly. Surfaces install state, cold-import timing, cache hit rates, compaction-budget telemetry, opt-in flag status, and canonical-root sanity. A **Parser freshness** check reports how much of the index for this project was written by a different build of the extractor. The stamp only refreshes when something touches a file, so after an upgrade a project goes on answering `symbol`, `read`, `outline` and `skeleton` from the previous extractor with nothing saying so. It warns past a quarter and names the fix: a plain `token-goat index` in that project, which is enough on its own, since a parser mismatch reparses without `--force`. A **Tool names** check reports any tool name a harness sent that reached no handler wanting it, and calls out the ones that differ from a handled name only by capitalisation or punctuation — the signature of a bridge that forgot to rename something, which is otherwise invisible. A **Security** section reports the posture in one place: whether offline mode is on, whether injection scanning is on, whether the Google Drive integration is enabled, whether fetching runs against an allow list or a deny list, whether MCP reads are confined to the project root, and whether the data directory is readable by other local users. It only warns when a protection that ships on has been switched off, so a default install stays quiet. It read-only audits `~/.copilot/mcp-config.json` for globally configured Chrome DevTools or Playwright `npx` launchers, recommending project scope or removal when inactive; it never prints server configuration or secrets. On Windows, it also reports duplicate Chrome DevTools/Playwright MCP launchers and orphaned Node processes without terminating anything. Pass `--context` to show the **Context footprint** section: a fill bar with severity (ok / warn / high / URGENT), per-component breakdown (skills catalog, loaded skill bodies, CLAUDE.md+MEMORY.md, conversation estimate), session-to-session growth trend with sessions-to-URGENT projection, and tiered compaction recommendations (Tier 0–4) naming the exact commands to run. Auto-shown when fill > 40 % or any loaded skill > 2 K tokens lacks a compact. `--json` emits the check results (one entry per check, with `ok`/`warn`/`fail` status) as JSON instead of text. |
|
|
182
|
+
| `token-goat doctor` | Confirm everything is wired correctly. Surfaces install state, cold-import timing, cache hit rates, compaction-budget telemetry, opt-in flag status, and canonical-root sanity. A **Parser freshness** check reports how much of the index for this project was written by a different build of the extractor. The stamp only refreshes when something touches a file, so after an upgrade a project goes on answering `symbol`, `read`, `outline` and `skeleton` from the previous extractor with nothing saying so. It warns past a quarter and names the fix: a plain `token-goat index` in that project, which is enough on its own, since a parser mismatch reparses without `--force`. A **Tool names** check reports any tool name a harness sent that reached no handler wanting it, and calls out the ones that differ from a handled name only by capitalisation or punctuation — the signature of a bridge that forgot to rename something, which is otherwise invisible. When `global.db` passes 1 GB, the warning now also breaks the total down by category (symbol bodies, refs, chunk text, embedding vectors, stats detail) with each one's measured byte share and the exact command that shrinks it -- `reclaim-index --rebuild` for the derived-row categories, disabling `indexing.embeddings_enabled` first for vectors so they don't regrow, and nothing manual for stats detail, which ages out on its own. A **Security** section reports the posture in one place: whether offline mode is on, whether injection scanning is on, whether the Google Drive integration is enabled, whether fetching runs against an allow list or a deny list, whether MCP reads are confined to the project root, and whether the data directory is readable by other local users. It only warns when a protection that ships on has been switched off, so a default install stays quiet. It read-only audits `~/.copilot/mcp-config.json` for globally configured Chrome DevTools or Playwright `npx` launchers, recommending project scope or removal when inactive; it never prints server configuration or secrets. On Windows, it also reports duplicate Chrome DevTools/Playwright MCP launchers and orphaned Node processes without terminating anything. Pass `--context` to show the **Context footprint** section: a fill bar with severity (ok / warn / high / URGENT), per-component breakdown (skills catalog, loaded skill bodies, CLAUDE.md+MEMORY.md, conversation estimate), session-to-session growth trend with sessions-to-URGENT projection, and tiered compaction recommendations (Tier 0–4) naming the exact commands to run. Auto-shown when fill > 40 % or any loaded skill > 2 K tokens lacks a compact. `--json` emits the check results (one entry per check, with `ok`/`warn`/`fail` status) as JSON instead of text. |
|
|
181
183
|
| `token-goat capabilities` | List every capability that can send data off this machine or leave data on it, with whether it is currently on, the config key that decides that, and the exact `file::symbol` where the decision is made — so a reviewer can open the code rather than take the list's word for it. `--json` emits the same thing for a pipeline to assert on, which is the point: the answer comes from the binary installed on your machine, not from documentation that may describe a different build. A test in the suite fails the build when a module that can open a network connection is missing from this list, and equally when the list names one that no longer connects anywhere. |
|
|
182
184
|
| `token-goat baseline` | Emit a project map: file count, per-language file counts, the top indexed symbols (by name/kind/location), and the most recently modified files. `--subagent` emits a terser variant (fewer symbols, fewer recent files) for context handed to a freshly spawned subagent; `--json` for the machine-readable form. |
|
|
183
185
|
| `token-goat compact-doc <path>` | Build an extractive compact sidecar for a large reference doc (`.md`/`.markdown`). The compact is stored in the token-goat data dir as a SHA-keyed sidecar; `pre_read` serves it in place of the full file when it exists and is fresh, typically saving 60–95% of the tokens the full read would cost (measured across this repo's own 44 docs: median 67%, range 5–99%, depending on how much of the doc is prose under headings). Use `--force` to rebuild, `--sentences N` to control lines per section (default 2), `--show` to print the result. The sidecar is automatically marked stale when you edit the source file. Config: `[hints] stable_doc_compacts = true` (default on). |
|
|
@@ -307,22 +309,22 @@ worth its keep only if it's actually followed. `token-goat hint-stats` reports,
|
|
|
307
309
|
category: how many times it fired, how many times a later Bash command in the same session
|
|
308
310
|
actually invoked the specific `token-goat` command (or referenced the specific cached-output id)
|
|
309
311
|
the hint pointed at, the resulting efficacy percentage, whether the category is currently
|
|
310
|
-
auto-suppressed, and the `spent` column (bytes of hint text actually injected into context for
|
|
312
|
+
auto-suppressed, and the `spent-bytes` column (bytes of hint text actually injected into context for
|
|
311
313
|
that category — the real cost of emitting it, not just how often it fired):
|
|
312
314
|
|
|
313
315
|
```
|
|
314
316
|
$ token-goat hint-stats
|
|
315
|
-
category emitted acted-on efficacy suppressed manual+ manual- spent
|
|
316
|
-
bash_redirect 42 9 21.4% no 0 0 3150
|
|
317
|
-
bash_recall 18 15 83.3% no 0 0 1080
|
|
318
|
-
read_reread_dedup 11 2 18.2% * no 0 0 660
|
|
319
|
-
read_structural_nav 7 1 14.3% yes 0 1 420
|
|
320
|
-
edit_reread_suggest 3 0 0% * no 0 0 180
|
|
321
|
-
|
|
322
|
-
TOTAL saved=48200 spent=5490
|
|
317
|
+
category emitted undisplayed acted-on efficacy suppressed manual+ manual- spent-bytes
|
|
318
|
+
bash_redirect 42 - 9 21.4% no 0 0 3150
|
|
319
|
+
bash_recall 18 - 15 83.3% no 0 0 1080
|
|
320
|
+
read_reread_dedup 11 - 2 18.2% * no 0 0 660
|
|
321
|
+
read_structural_nav 7 3 1 14.3% yes 0 1 420
|
|
322
|
+
edit_reread_suggest 3 - 0 0% * no 0 0 180
|
|
323
|
+
|
|
324
|
+
TOTAL saved-bytes=48200 (all-time, every hint kind) spent-bytes=5490 (hint_emissions ledger only)
|
|
323
325
|
```
|
|
324
326
|
|
|
325
|
-
`spent` (and the `TOTAL` line's `spent
|
|
327
|
+
`spent-bytes` (and the `TOTAL` line's `spent-bytes`) render `n/a` instead of a fake `0` whenever a
|
|
326
328
|
category — or, for the total, the whole store — has no tracked spend figure at all: either
|
|
327
329
|
nothing has fired yet, or every emission predates this feature and was recorded before spend
|
|
328
330
|
tracking existed. A partially-tracked category shows the real sum plus how many legacy rows it
|
package/docs/install.md
CHANGED
|
@@ -360,7 +360,7 @@ For VS Code using the token-goat MCP server (`token-goat install --vscode`), ena
|
|
|
360
360
|
|
|
361
361
|
One table in `src/language_specs.ts` drives every per-language list token-goat uses, so this is the whole set. **Symbols** means `symbol`, `read "file::Name"`, `skeleton` and `outline` return named declarations from the file. **Structure only** means the file is indexed for headings, keys or sections rather than code symbols.
|
|
362
362
|
|
|
363
|
-
**Symbols:** ABAP (`.abap`), Apex (`.cls`, `.trigger`), Assembly (`.s`, `.asm`, `.nasm`), Bash and compatible shells (`.sh`, `.bash`, `.zsh`, `.ksh`, `.bats`), C (`.c`, `.h`), C# (`.cs`), C++ (`.cpp`, `.cc`, `.cxx`, `.hpp`, `.hxx`), Clojure (`.clj`, `.cljs`, `.cljc`), CMake (`.cmake`, `CMakeLists.txt`), COBOL (`.cbl`, `.cob`, `.cobol`, `.cpy`), Common Lisp (`.lisp`, `.lsp`, `.cl`), Dart (`.dart`), Elixir (`.ex`, `.exs`), Emacs Lisp (`.el`), Erlang (`.erl`, `.hrl`), F# (`.fs`, `.fsi`, `.fsx`), Fortran (`.f`, `.for`, `.f77`, `.f90`, `.f95`, `.f03`, `.f08`), GLSL (`.glsl`, `.vert`, `.frag`, `.comp`, `.geom`, `.tesc`, `.tese`), Go (`.go`), GraphQL (`.graphql`, `.gql`), Groovy (`.groovy`, `.gvy`, `.gradle`, `Jenkinsfile`), Haskell (`.hs`), HLSL (`.hlsl`, `.hlsli`), Java (`.java`), JavaScript (`.js`, `.jsx`, `.mjs`, `.cjs`), JCL (`.jcl`), Kotlin (`.kt`, `.kts`), Lua (`.lua`), MATLAB, Metal (`.metal`), Natural (`.nsp`, `.nsn`, `.nss`, `.nsa`, `.nsl`, `.nsg`, `.nsc`, `.nsh`), Nix (`.nix`), Objective-C (`.mm`, and a `.h` that declares an `@interface` or `@protocol`), OCaml (`.ml`, `.mli`), OpenEdge ABL, Pascal (`.pas`, `.dpr`, `.dpk`, `.lpr`, `.dfm`), Perl (`.pl`, `.pm`), PHP (`.php`), PL/I (`.pli`, `.pl1`), PowerShell (`.ps1`, `.psm1`), Protocol Buffers (`.proto`), Python and Starlark (`.py`, `.pyi`, `.bzl`, `.star`, `BUILD`, `WORKSPACE`, `MODULE.bazel`), R (`.r`), Racket (`.rkt`, `.rktl`), RPG (`.rpgle`, `.sqlrpgle`), Ruby (`.rb`, `.ruby`, `.rake`, `Gemfile`, `Rakefile`, and the other extensionless Ruby DSL files), Rust (`.rs`), Salesforce markup (`.cmp`, `.app`, `.evt`, `.intf`, `.design`, `.auradoc`, `.tokens`, `.page`, `.component`, `.email`), Salesforce metadata, SAS (`.sas`), Scala (`.scala`, `.sc`), Scheme (`.scm`, `.ss`), Solidity (`.sol`), SQL and PL/SQL (`.sql`, `.pks`, `.pkb`, `.pls`, `.plsql`, `.pck`, `.prc`, `.fnc`, `.trg`, `.tps`, `.tpb`), Swift (`.swift`), Terraform (`.tf`, `.tfvars`, `.hcl`), Thrift (`.thrift`), TypeScript (`.ts`, `.tsx`, `.mts`, `.cts`), VHDL (`.vhd`, `.vhdl`), Visual Basic (`.vb`, `.bas`, `.vbs`, `.frm`), WGSL (`.wgsl`), Windows batch (`.bat`, `.cmd`), Zig (`.zig`).
|
|
363
|
+
**Symbols:** ABAP (`.abap`), Apache (`httpd.conf`, `apache2.conf`, `.htaccess`, `.apache`, `.apache2`), Apex (`.cls`, `.trigger`), Assembly (`.s`, `.asm`, `.nasm`), Bash and compatible shells (`.sh`, `.bash`, `.zsh`, `.ksh`, `.bats`), C (`.c`, `.h`), C# (`.cs`), C++ (`.cpp`, `.cc`, `.cxx`, `.hpp`, `.hxx`), Caddyfile (`Caddyfile`, `.caddy`), Clojure (`.clj`, `.cljs`, `.cljc`), CMake (`.cmake`, `CMakeLists.txt`), COBOL (`.cbl`, `.cob`, `.cobol`, `.cpy`), Common Lisp (`.lisp`, `.lsp`, `.cl`), Dart (`.dart`), Elixir (`.ex`, `.exs`), Emacs Lisp (`.el`), Erlang (`.erl`, `.hrl`), F# (`.fs`, `.fsi`, `.fsx`), Fortran (`.f`, `.for`, `.f77`, `.f90`, `.f95`, `.f03`, `.f08`), GLSL (`.glsl`, `.vert`, `.frag`, `.comp`, `.geom`, `.tesc`, `.tese`), Go (`.go`), GraphQL (`.graphql`, `.gql`), Groovy (`.groovy`, `.gvy`, `.gradle`, `Jenkinsfile`), Haskell (`.hs`), HLSL (`.hlsl`, `.hlsli`), Java (`.java`), JavaScript (`.js`, `.jsx`, `.mjs`, `.cjs`), JCL (`.jcl`), Kotlin (`.kt`, `.kts`), Lua (`.lua`), MATLAB, Metal (`.metal`), Natural (`.nsp`, `.nsn`, `.nss`, `.nsa`, `.nsl`, `.nsg`, `.nsc`, `.nsh`), Nginx (`nginx.conf`, `.nginx`), Nix (`.nix`), Objective-C (`.mm`, and a `.h` that declares an `@interface` or `@protocol`), OCaml (`.ml`, `.mli`), OpenEdge ABL, Pascal (`.pas`, `.dpr`, `.dpk`, `.lpr`, `.dfm`), Perl (`.pl`, `.pm`), PHP (`.php`), PL/I (`.pli`, `.pl1`), PowerShell (`.ps1`, `.psm1`), Protocol Buffers (`.proto`), Python and Starlark (`.py`, `.pyi`, `.bzl`, `.star`, `BUILD`, `WORKSPACE`, `MODULE.bazel`), R (`.r`), Racket (`.rkt`, `.rktl`), RPG (`.rpgle`, `.sqlrpgle`), Ruby (`.rb`, `.ruby`, `.rake`, `Gemfile`, `Rakefile`, and the other extensionless Ruby DSL files), Rust (`.rs`), Salesforce markup (`.cmp`, `.app`, `.evt`, `.intf`, `.design`, `.auradoc`, `.tokens`, `.page`, `.component`, `.email`), Salesforce metadata, SAS (`.sas`), Scala (`.scala`, `.sc`), Scheme (`.scm`, `.ss`), Solidity (`.sol`), SQL and PL/SQL (`.sql`, `.pks`, `.pkb`, `.pls`, `.plsql`, `.pck`, `.prc`, `.fnc`, `.trg`, `.tps`, `.tpb`; including DuckDB `CREATE MACRO` and `CREATE SECRET` statements), Swift (`.swift`), Terraform (`.tf`, `.tfvars`, `.hcl`), Thrift (`.thrift`), TypeScript (`.ts`, `.tsx`, `.mts`, `.cts`), VHDL (`.vhd`, `.vhdl`), Visual Basic (`.vb`, `.bas`, `.vbs`, `.frm`), WGSL (`.wgsl`), Windows batch (`.bat`, `.cmd`), Zig (`.zig`).
|
|
364
364
|
|
|
365
365
|
**Structure only:** Astro (`.astro`), CSS and its preprocessors (`.css`, `.scss`, `.sass`, `.less`), Dockerfile, environment files (`.env`, `.envrc`), HTML (`.html`, `.htm`), INI (`.ini`, `.cfg`, `.conf`), JSON including JSON with comments and Avro schemas (`.json`, `.jsonc`, `.avsc`), Jupyter notebooks (`.ipynb`), Makefiles (`.mk`, `Makefile`), Markdown (`.md`, `.markdown`, `.mdx`), Svelte (`.svelte`), TOML (`.toml`), Vue (`.vue`), YAML (`.yaml`, `.yml`), and seven template-engine dialects read as HTML: Liquid (`.liquid`), Jinja2 (`.j2`, `.jinja`, `.jinja2`), Handlebars (`.hbs`, `.handlebars`), ERB (`.erb`), EJS (`.ejs`), Nunjucks (`.njk`), Twig (`.twig`).
|
|
366
366
|
|
package/docs/security.md
CHANGED
|
@@ -20,9 +20,9 @@ Outbound network is reserved to these explicit cases:
|
|
|
20
20
|
|
|
21
21
|
**One switch for all of it.** Set `network.offline = true` (env `TOKEN_GOAT_OFFLINE`) and every one of the paths above refuses instead of connecting, saying so rather than failing quietly. Anything already cached keeps working: a machine that has the embedding model still runs `semantic`, and one that has the language data still reads text out of images. This is one of the settings a per-project config file may not touch, so cloning a repository cannot switch it back off.
|
|
22
22
|
|
|
23
|
-
**A repository cannot reconfigure the security controls.** A project-root `.token-goat.toml` layers on top of your global config, which is what it is for: hint thresholds, indexing settings, compression tuning. But that file arrives with the repository, so whoever wrote the repository wrote it. Seven whole sections are therefore off limits to it, and come from your global config or the environment only: `injection` (prompt-injection fencing), `webfetch` (the fetch allow and deny lists), `gdrive` (the Google Drive integration), `mcp` (root confinement and the allowed-roots list), `network` (offline mode), `redaction` (the secret-redaction rules), and `screenshot` (the headless browser).
|
|
23
|
+
**A repository cannot reconfigure the security controls.** A project-root `.token-goat.toml` layers on top of your global config, which is what it is for: hint thresholds, indexing settings, compression tuning. But that file arrives with the repository, so whoever wrote the repository wrote it. Seven whole sections are therefore off limits to it, and come from your global config or the environment only: `injection` (prompt-injection fencing), `webfetch` (the fetch allow and deny lists), `gdrive` (the Google Drive integration), `mcp` (root confinement and the allowed-roots list), `network` (offline mode), `redaction` (the secret-redaction rules), and `screenshot` (the headless browser). Seventeen individual keys inside otherwise-overridable sections are locked the same way. Six decide what gets indexed at all: `indexing.cross_project_symbols`, `indexing.skip_dirs`, `indexing.skip_files`, `indexing.large_file_skip_kb`, `indexing.large_file_symbol_only_kb` and `indexing.max_chunks_per_file`, which matter because an unindexed file answers `symbol`, `read` and `semantic` in the same words a name that never existed does. Five more are the switches that decide whether a large file arrives folded rather than whole: `hints.fold_code_bodies`, `hints.fold_comment_blocks`, `hints.fold_prose_paragraphs`, `hints.outline_large_documents` and `hints.skeleton_large_sources`. The last four are `image_shrink.max_image_pixels`, the decompression-bomb cap, `worker.blocked_roots`, the folders you have kept out of the index, `compact_assist.summary_budget_chars`, the length target token-goat hands to whoever writes your compaction summary, which a repository must not be able to set low enough to erase the session, and `semantic.max_distance`, the relevance floor for `semantic`, which a repository must not be able to set low enough that its own code stops matching while the command goes on answering from keyword search as though nothing were missing. Two more govern the shared database: `indexing.max_db_size_mb` and `indexing.auto_reclaim_embeddings`, which prevent a repository from manipulating global database disk caps or forcing vector purging across projects. A project file that sets one of them is ignored, and token-goat prints a line naming what it dropped. Everything else stays project-overridable.
|
|
24
24
|
|
|
25
|
-
**The lock covers the config file, not the environment.** These settings still read a `TOKEN_GOAT_*` environment variable, and a repository has ways to set one: a `.envrc` for direnv, a `terminal.integrated.env.*` block in a committed `.vscode/settings.json`, a `containerEnv` entry in a devcontainer. Refusing environment overrides would break the operator who exports a variable in their own shell, which is the legitimate case and the common one, so token-goat reports instead of refusing: `token-goat doctor` prints a `Security config overrides` line naming every locked security setting the environment is currently deciding, and the variable to unset. It covers all twenty locked keys that read an environment variable, and it derives that set from the same two tables that define what a project config may not write, rather than from a list kept alongside them. Settings with a safe side (booleans) are reported when the environment holds them open; settings without one (lists, sizes) are reported whenever the environment supplies a value at all, since there is nothing to compare against. A default install, where nothing is set, prints a single ok line.
|
|
25
|
+
**The lock covers the config file, not the environment.** These settings still read a `TOKEN_GOAT_*` environment variable, and a repository has ways to set one: a `.envrc` for direnv, a `terminal.integrated.env.*` block in a committed `.vscode/settings.json`, a `containerEnv` entry in a devcontainer. Refusing environment overrides would break the operator who exports a variable in their own shell, which is the legitimate case and the common one, so token-goat reports instead of refusing: `token-goat doctor` prints a `Security config overrides` line naming every locked security setting the environment is currently deciding, and the variable to unset. It covers all twenty-two locked keys that read an environment variable, and it derives that set from the same two tables that define what a project config may not write, rather than from a list kept alongside them. Settings with a safe side (booleans) are reported when the environment holds them open; settings without one (lists, sizes) are reported whenever the environment supplies a value at all, since there is nothing to compare against. A default install, where nothing is set, prints a single ok line.
|
|
26
26
|
|
|
27
27
|
**Security reports.** See [SECURITY.md](../SECURITY.md). Email `token-goat@dfkhelper.com`; do not file as a GitHub issue. Reports are acknowledged within 7 days; coordinated disclosure with a 90-day default window.
|
|
28
28
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "token-goat",
|
|
3
|
-
"version": "2.9.
|
|
3
|
+
"version": "2.9.20",
|
|
4
4
|
"description": "Surgical token-reduction companion for Claude Code and other AI coding agents",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "./dist/token-goat.mjs",
|
|
@@ -132,6 +132,7 @@
|
|
|
132
132
|
"unzipper": "^0.12.5",
|
|
133
133
|
"adm-zip": "^0.6.0",
|
|
134
134
|
"fast-uri": "^3.1.6",
|
|
135
|
-
"qs": "^6.16.0"
|
|
135
|
+
"qs": "^6.16.0",
|
|
136
|
+
"tesseract.js-core": "6.1.2"
|
|
136
137
|
}
|
|
137
138
|
}
|