token-goat 2.9.18 → 2.9.19
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/dist/{token-goat-chunk-V3ZWBEHM.mjs → token-goat-chunk-2GTB5RBZ.mjs} +365 -26
- package/dist/{token-goat-chunk-ARNRFR6M.mjs → token-goat-chunk-3W5P673V.mjs} +1 -1
- package/dist/{token-goat-chunk-YQZE5NS6.mjs → token-goat-chunk-62VJ47WN.mjs} +6 -6
- package/dist/{token-goat-chunk-CTOYJ232.mjs → token-goat-chunk-7DGTVBWP.mjs} +45 -8
- package/dist/{token-goat-chunk-5MMJAJF3.mjs → token-goat-chunk-AVRD7SBO.mjs} +60 -8
- package/dist/{token-goat-chunk-7VLWNQDQ.mjs → token-goat-chunk-BTKDZW7F.mjs} +1 -1
- package/dist/{token-goat-chunk-LMFO66YD.mjs → token-goat-chunk-DA43OE7L.mjs} +1 -1
- package/dist/{token-goat-chunk-6LGVCCMY.mjs → token-goat-chunk-DA6TIDLI.mjs} +21 -16
- package/dist/{token-goat-chunk-UENEL6KX.mjs → token-goat-chunk-DHIXE6GV.mjs} +2 -2
- package/dist/{token-goat-chunk-W3E65QOW.mjs → token-goat-chunk-EP7TCWHM.mjs} +304 -46
- package/dist/{token-goat-chunk-HIZXQVNU.mjs → token-goat-chunk-EXLLLIC3.mjs} +4 -4
- package/dist/{token-goat-chunk-H3FONFBO.mjs → token-goat-chunk-F24NE3FO.mjs} +3 -3
- package/dist/{token-goat-chunk-LA3O4JGD.mjs → token-goat-chunk-FJZ5MZK7.mjs} +3 -3
- package/dist/{token-goat-chunk-BFEWOMA3.mjs → token-goat-chunk-FQYMZASI.mjs} +632 -352
- package/dist/{token-goat-chunk-OWSJJY5R.mjs → token-goat-chunk-FX5DHKGF.mjs} +5 -5
- package/dist/{token-goat-chunk-K4DG6EEF.mjs → token-goat-chunk-GYC34EGE.mjs} +7 -8
- package/dist/{token-goat-chunk-WFX2V3OF.mjs → token-goat-chunk-HPMMLV37.mjs} +358 -116
- package/dist/{token-goat-chunk-I67CPTDD.mjs → token-goat-chunk-IBTFGZBX.mjs} +19 -14
- package/dist/{token-goat-chunk-L7VCX33I.mjs → token-goat-chunk-JNIFSVWM.mjs} +36 -10
- package/dist/{token-goat-chunk-EPXNOKIV.mjs → token-goat-chunk-KWM6KBPA.mjs} +2 -2
- package/dist/{token-goat-chunk-BWCRC23L.mjs → token-goat-chunk-L2UOJBV4.mjs} +21 -20
- package/dist/{token-goat-chunk-52IMXJ2L.mjs → token-goat-chunk-LACH3WKS.mjs} +4 -4
- package/dist/token-goat-chunk-MU6I4IDI.mjs +337 -0
- package/dist/{token-goat-chunk-GRI6HWX3.mjs → token-goat-chunk-NJKNAC37.mjs} +1 -1
- package/dist/{token-goat-chunk-ERTXEKB6.mjs → token-goat-chunk-NMTKNYGF.mjs} +2 -0
- package/dist/{token-goat-chunk-34OUJ2IE.mjs → token-goat-chunk-NYFYIRG3.mjs} +12 -12
- package/dist/{token-goat-chunk-4EXFN2AW.mjs → token-goat-chunk-OZRSREKR.mjs} +2 -2
- package/dist/token-goat-chunk-P46RTOAQ.mjs +27 -0
- package/dist/{token-goat-chunk-QML7ITHJ.mjs → token-goat-chunk-PIAESKVU.mjs} +139 -23
- package/dist/{token-goat-chunk-HPOFR6QX.mjs → token-goat-chunk-S7RFIX4Y.mjs} +180 -50
- package/dist/{token-goat-chunk-CA6V36KW.mjs → token-goat-chunk-SOKAPOOC.mjs} +23 -4
- package/dist/token-goat-chunk-VJQBYSXG.mjs +20 -0
- package/dist/{token-goat-chunk-LTYWLKAG.mjs → token-goat-chunk-WKO5TVNA.mjs} +922 -611
- package/dist/{token-goat-chunk-QIDLJGXZ.mjs → token-goat-chunk-ZGQXRYKE.mjs} +29 -5
- package/dist/token-goat-hook.mjs +12 -12
- package/dist/token-goat.core.mjs +21 -20
- package/docs/cli.md +2 -1
- package/docs/security.md +1 -1
- package/package.json +1 -1
- package/dist/token-goat-chunk-2KTZ6J7T.mjs +0 -173
- package/dist/token-goat-chunk-RHHFEXKF.mjs +0 -93
|
@@ -2,14 +2,17 @@ import { createRequire as __cjsRequire } from 'node:module';
|
|
|
2
2
|
const require = __cjsRequire(import.meta.url);
|
|
3
3
|
import {
|
|
4
4
|
getDb
|
|
5
|
-
} from "./token-goat-chunk-
|
|
5
|
+
} from "./token-goat-chunk-JNIFSVWM.mjs";
|
|
6
6
|
import {
|
|
7
7
|
getHarnessName,
|
|
8
8
|
loadConfig
|
|
9
|
-
} from "./token-goat-chunk-
|
|
9
|
+
} from "./token-goat-chunk-SOKAPOOC.mjs";
|
|
10
10
|
import {
|
|
11
11
|
registerReset
|
|
12
12
|
} from "./token-goat-chunk-EEIDFMEM.mjs";
|
|
13
|
+
import {
|
|
14
|
+
Database
|
|
15
|
+
} from "./token-goat-chunk-OSUFN2FV.mjs";
|
|
13
16
|
import {
|
|
14
17
|
VERSION,
|
|
15
18
|
atomicWriteText,
|
|
@@ -17,12 +20,13 @@ import {
|
|
|
17
20
|
dataDir,
|
|
18
21
|
dataDirForHome,
|
|
19
22
|
ensureDirSync,
|
|
23
|
+
extractErrorMessage,
|
|
20
24
|
sanitizeIdForFilename,
|
|
21
25
|
tokenGoatHome
|
|
22
|
-
} from "./token-goat-chunk-
|
|
26
|
+
} from "./token-goat-chunk-ZGQXRYKE.mjs";
|
|
23
27
|
import {
|
|
24
28
|
displaySafeText
|
|
25
|
-
} from "./token-goat-chunk-
|
|
29
|
+
} from "./token-goat-chunk-NMTKNYGF.mjs";
|
|
26
30
|
import {
|
|
27
31
|
growsExponentially,
|
|
28
32
|
hasNestedQuantifier
|
|
@@ -108,6 +112,7 @@ var C = {
|
|
|
108
112
|
|
|
109
113
|
// src/stats.ts
|
|
110
114
|
init_define_import_meta_env();
|
|
115
|
+
import * as fs from "node:fs";
|
|
111
116
|
import * as path from "node:path";
|
|
112
117
|
|
|
113
118
|
// src/render/stats_renderer.ts
|
|
@@ -954,11 +959,7 @@ var KIND_TO_SOURCE = {
|
|
|
954
959
|
compact_doc: SOURCE_READ,
|
|
955
960
|
note_read: SOURCE_READ,
|
|
956
961
|
note_list: SOURCE_READ,
|
|
957
|
-
// note-add is a write (like insert-section/replace, which record no stat at all -- neither
|
|
958
|
-
// has a "full source it replaces" savings concept). It still gets an event-only entry here
|
|
959
|
-
// (no bytesSaved/tokensSaved argument, same as skill_load) purely so `token-goat note-add`
|
|
960
|
-
// usage is visible in `token-goat stats --full` at all -- SOURCE_OTHER, not SOURCE_READ,
|
|
961
|
-
// since it is not a token-savings substitute for a read.
|
|
962
|
+
// note-add is a write (like insert-section/replace, which record no stat at all -- neither has a "full source it replaces" savings concept). It still gets an event-only entry here (no bytesSaved/tokensSaved argument, same as skill_load) purely so `token-goat note-add` usage is visible in `token-goat stats --full` at all -- SOURCE_OTHER, not SOURCE_READ, since it is not a token-savings substitute for a read.
|
|
962
963
|
note_write: SOURCE_OTHER,
|
|
963
964
|
web_fetch: SOURCE_WEB,
|
|
964
965
|
injection_detected: SOURCE_WEB,
|
|
@@ -973,6 +974,8 @@ var KIND_TO_SOURCE = {
|
|
|
973
974
|
dirty_queue_append_failed: SOURCE_OTHER,
|
|
974
975
|
worker_healthcheck_failed: SOURCE_OTHER,
|
|
975
976
|
known_root_record_failed: SOURCE_OTHER,
|
|
977
|
+
// Same fail-soft shape, from hooks_session_start.ts's reconcileNote: a thrown reconcile sweep must not block session start, but it also must not vanish silently, so the catch that swallows it records why instead of returning null with nothing recorded.
|
|
978
|
+
reconcile_note_failed: SOURCE_OTHER,
|
|
976
979
|
// Measurement of what a compaction produced (hooks_compact.ts postCompactHandler): summary size and how many manifest paths survived into it. SOURCE_OTHER and always recorded at (0, 0) -- the summary was written whether or not token-goat was watching, so there is no counterfactual in which those bytes were saved. Filing it anywhere with a savings total would credit token-goat for the whole summary, which is the accounting mistake this registry exists to prevent.
|
|
977
980
|
compact_summary: SOURCE_OTHER,
|
|
978
981
|
// Envelope compaction of an oversized subagent report (hooks_agent_spawn.ts). SOURCE_CONTENT, not SOURCE_HINT: the handler's sibling session_hint entry is advisory (it only appends a recall pointer and genuinely saves nothing), whereas this kind records a real rewrite with real bytes removed, so filing it under the advisory bucket would understate the compaction and repeat the zero-savings desync this registry keeps getting bitten by.
|
|
@@ -986,12 +989,7 @@ var KIND_TO_SOURCE = {
|
|
|
986
989
|
plan_echo_collapse: SOURCE_CONTENT,
|
|
987
990
|
// Lossless re-layout of Grep content-mode output (hooks_grep.ts foldGrepContentHandler). SOURCE_CONTENT, not SOURCE_HINT, for the same reason as agent_report_compact above: its sibling grep_dedup_hint is advisory and saves nothing directly, whereas this is a real rewrite with real bytes removed. Filing it under the advisory bucket would silently add non-hint savings to hint_stats.ts's savedBytes (which reads by_source[SOURCE_HINT] wholesale) and overstate the hint ledger's net benefit.
|
|
988
991
|
"grep:fold": SOURCE_CONTENT,
|
|
989
|
-
// Withholding of already-served stretches from a completed Read (hooks_read.ts
|
|
990
|
-
// elideAlreadyServedLines). SOURCE_CONTENT, not SOURCE_HINT, for the same reason as
|
|
991
|
-
// grep:fold above: its siblings read_count_deny and read_served_deny are decisions about
|
|
992
|
-
// whether a read happens at all, whereas this is a rewrite of a result that did happen,
|
|
993
|
-
// with real bytes removed from it. Filing it under the advisory bucket would add non-hint
|
|
994
|
-
// savings to hint_stats.ts's savedBytes, which reads by_source[SOURCE_HINT] wholesale.
|
|
992
|
+
// Withholding of already-served stretches from a completed Read (hooks_read.ts elideAlreadyServedLines). SOURCE_CONTENT, not SOURCE_HINT, for the same reason as grep:fold above: its siblings read_count_deny and read_served_deny are decisions about whether a read happens at all, whereas this is a rewrite of a result that did happen, with real bytes removed from it. Filing it under the advisory bucket would add non-hint savings to hint_stats.ts's savedBytes, which reads by_source[SOURCE_HINT] wholesale.
|
|
995
993
|
"read:served_elide": SOURCE_CONTENT,
|
|
996
994
|
// Same bucket and same reasoning as read:served_elide directly above: a rewrite of a Read that did happen, with real bytes removed, not an advisory about whether to read at all.
|
|
997
995
|
"read:body_fold": SOURCE_CONTENT,
|
|
@@ -1011,7 +1009,9 @@ var KIND_PREFIX_TO_SOURCE = [
|
|
|
1011
1009
|
["skill_body:", SOURCE_SKILL],
|
|
1012
1010
|
["skill_compact:", SOURCE_SKILL],
|
|
1013
1011
|
["bashoutput:", SOURCE_BASH],
|
|
1014
|
-
["taskoutput:", SOURCE_CONTENT]
|
|
1012
|
+
["taskoutput:", SOURCE_CONTENT],
|
|
1013
|
+
// Hook wall-clock timing (relay.ts's relayInProcess), one row per invocation, always 0 bytes/0 tokens: it measures token-goat's own overhead, not a saving. See hook_latency.ts's hookLatencyBreakdown() for the dedicated read path and pruneHookStats() below for its own (shorter) retention.
|
|
1014
|
+
["hook:", SOURCE_OTHER]
|
|
1015
1015
|
];
|
|
1016
1016
|
var COMMAND_KINDS = {
|
|
1017
1017
|
symbol: /* @__PURE__ */ new Set(["symbol_lookup"]),
|
|
@@ -1132,7 +1132,8 @@ CREATE TABLE IF NOT EXISTS stats (
|
|
|
1132
1132
|
detail TEXT,
|
|
1133
1133
|
harness TEXT,
|
|
1134
1134
|
traceparent TEXT,
|
|
1135
|
-
tg_version TEXT
|
|
1135
|
+
tg_version TEXT,
|
|
1136
|
+
duration_ms INTEGER
|
|
1136
1137
|
);
|
|
1137
1138
|
CREATE INDEX IF NOT EXISTS idx_stats_ts ON stats(ts);
|
|
1138
1139
|
CREATE INDEX IF NOT EXISTS idx_stats_kind ON stats(kind);
|
|
@@ -1190,6 +1191,29 @@ function migrateGlobalSchema(db) {
|
|
|
1190
1191
|
} catch (err) {
|
|
1191
1192
|
if (!(err instanceof Error) || !/duplicate column/i.test(err.message)) throw err;
|
|
1192
1193
|
}
|
|
1194
|
+
try {
|
|
1195
|
+
db.exec("ALTER TABLE stats ADD COLUMN duration_ms INTEGER");
|
|
1196
|
+
} catch (err) {
|
|
1197
|
+
if (!(err instanceof Error) || !/duplicate column/i.test(err.message)) throw err;
|
|
1198
|
+
}
|
|
1199
|
+
}
|
|
1200
|
+
var TEST_ISOLATION_LEAK_WINDOW_START_TS = 1780406895;
|
|
1201
|
+
var TEST_ISOLATION_LEAK_WINDOW_END_TS = 1782119331;
|
|
1202
|
+
function pruneTestIsolationLeakRows(db) {
|
|
1203
|
+
try {
|
|
1204
|
+
db.prepare(
|
|
1205
|
+
`DELETE FROM stats WHERE tg_version IS NULL AND ts BETWEEN ? AND ? AND (detail LIKE '%pytest-of-%' OR detail LIKE '%/fake/%')`
|
|
1206
|
+
).run(TEST_ISOLATION_LEAK_WINDOW_START_TS, TEST_ISOLATION_LEAK_WINDOW_END_TS);
|
|
1207
|
+
} catch {
|
|
1208
|
+
}
|
|
1209
|
+
}
|
|
1210
|
+
function dropRetiredPythonTables(db) {
|
|
1211
|
+
try {
|
|
1212
|
+
db.prepare("DROP TABLE IF EXISTS grep_patterns").run();
|
|
1213
|
+
db.prepare("DROP TABLE IF EXISTS miss_patterns").run();
|
|
1214
|
+
db.prepare("DROP TABLE IF EXISTS wal_bloat").run();
|
|
1215
|
+
} catch {
|
|
1216
|
+
}
|
|
1193
1217
|
}
|
|
1194
1218
|
var _harnessColumnByDb = /* @__PURE__ */ new WeakMap();
|
|
1195
1219
|
function statsHasHarnessColumn(db) {
|
|
@@ -1236,6 +1260,21 @@ function statsHasVersionColumn(db) {
|
|
|
1236
1260
|
_versionColumnByDb.set(db, present);
|
|
1237
1261
|
return present;
|
|
1238
1262
|
}
|
|
1263
|
+
var _durationColumnByDb = /* @__PURE__ */ new WeakMap();
|
|
1264
|
+
function statsHasDurationColumn(db) {
|
|
1265
|
+
const cached = _durationColumnByDb.get(db);
|
|
1266
|
+
if (cached !== void 0) return cached;
|
|
1267
|
+
let present;
|
|
1268
|
+
try {
|
|
1269
|
+
present = db.prepare("PRAGMA table_info(stats)").all().some(
|
|
1270
|
+
(c) => c.name === "duration_ms"
|
|
1271
|
+
);
|
|
1272
|
+
} catch {
|
|
1273
|
+
present = false;
|
|
1274
|
+
}
|
|
1275
|
+
_durationColumnByDb.set(db, present);
|
|
1276
|
+
return present;
|
|
1277
|
+
}
|
|
1239
1278
|
function getGlobalDb(homeDir) {
|
|
1240
1279
|
const basePath = homeDir ? dataDirForHome(homeDir) : dataDir();
|
|
1241
1280
|
const dbPath = path.join(basePath, "global.db");
|
|
@@ -1248,6 +1287,7 @@ function getGlobalDb(homeDir) {
|
|
|
1248
1287
|
return db;
|
|
1249
1288
|
}
|
|
1250
1289
|
var STATS_RETENTION_DAYS = 180;
|
|
1290
|
+
var UNMAPPED_TOOL_RETENTION_DAYS = 30;
|
|
1251
1291
|
var STATS_ROLLUP_INTERVAL_MS = 6 * 60 * 60 * 1e3;
|
|
1252
1292
|
function rollupAndPruneStats(db, retentionDays = STATS_RETENTION_DAYS) {
|
|
1253
1293
|
try {
|
|
@@ -1269,11 +1309,31 @@ function rollupAndPruneStats(db, retentionDays = STATS_RETENTION_DAYS) {
|
|
|
1269
1309
|
tokens_saved = tokens_saved + excluded.tokens_saved`
|
|
1270
1310
|
).run(cutoff);
|
|
1271
1311
|
db.prepare(`DELETE FROM stats WHERE ts < ?`).run(cutoff);
|
|
1312
|
+
try {
|
|
1313
|
+
const unmappedCutoff = Math.floor(Date.now() / 1e3) - UNMAPPED_TOOL_RETENTION_DAYS * 86400;
|
|
1314
|
+
db.prepare(`DELETE FROM unmapped_tools WHERE last_seen < ?`).run(unmappedCutoff);
|
|
1315
|
+
} catch {
|
|
1316
|
+
}
|
|
1272
1317
|
});
|
|
1273
1318
|
run(cutoffTs);
|
|
1274
1319
|
} catch {
|
|
1275
1320
|
}
|
|
1276
1321
|
}
|
|
1322
|
+
function pruneHintEmissions(db, retentionDays = STATS_RETENTION_DAYS) {
|
|
1323
|
+
try {
|
|
1324
|
+
const cutoffMs = Date.now() - retentionDays * 86400 * 1e3;
|
|
1325
|
+
db.prepare(`DELETE FROM hint_emissions WHERE emitted_at < ?`).run(cutoffMs);
|
|
1326
|
+
} catch {
|
|
1327
|
+
}
|
|
1328
|
+
}
|
|
1329
|
+
var HOOK_STATS_RETENTION_DAYS = 7;
|
|
1330
|
+
function pruneHookStats(db, retentionDays = HOOK_STATS_RETENTION_DAYS) {
|
|
1331
|
+
try {
|
|
1332
|
+
const cutoffTs = Math.floor(Date.now() / 1e3) - retentionDays * 86400;
|
|
1333
|
+
db.prepare(`DELETE FROM stats WHERE kind LIKE 'hook:%' AND ts < ?`).run(cutoffTs);
|
|
1334
|
+
} catch {
|
|
1335
|
+
}
|
|
1336
|
+
}
|
|
1277
1337
|
function maybeRunStatsMaintenance(db) {
|
|
1278
1338
|
try {
|
|
1279
1339
|
const now = Date.now();
|
|
@@ -1284,7 +1344,11 @@ function maybeRunStatsMaintenance(db) {
|
|
|
1284
1344
|
`INSERT INTO stats_maintenance (id, last_rollup_ts) VALUES (1, ?)
|
|
1285
1345
|
ON CONFLICT(id) DO UPDATE SET last_rollup_ts = excluded.last_rollup_ts`
|
|
1286
1346
|
).run(now);
|
|
1347
|
+
pruneHookStats(db);
|
|
1287
1348
|
rollupAndPruneStats(db);
|
|
1349
|
+
pruneHintEmissions(db);
|
|
1350
|
+
pruneTestIsolationLeakRows(db);
|
|
1351
|
+
dropRetiredPythonTables(db);
|
|
1288
1352
|
} catch {
|
|
1289
1353
|
}
|
|
1290
1354
|
}
|
|
@@ -1303,9 +1367,33 @@ function noStatsMessage(windowDays, homeDir) {
|
|
|
1303
1367
|
if (total === 0) return "No stats recorded yet.";
|
|
1304
1368
|
return `No stats in the last ${countNoun(windowDays, "day")} (${total} recorded outside this window; use --window-days 0 for all time).`;
|
|
1305
1369
|
}
|
|
1306
|
-
|
|
1370
|
+
var STATS_WRITE_FAILURE_LOG_MIN_INTERVAL_MS = 60 * 1e3;
|
|
1371
|
+
function statsWriteFailureMarkerPath(dir) {
|
|
1372
|
+
return path.join(dir, "stats-write-failed.marker");
|
|
1373
|
+
}
|
|
1374
|
+
function statsWriteFailureLogPath(dir) {
|
|
1375
|
+
return path.join(dir, "stats-write-failed.log");
|
|
1376
|
+
}
|
|
1377
|
+
function recordStatWriteFailure(kind, err, dir = dataDir()) {
|
|
1307
1378
|
try {
|
|
1308
|
-
const
|
|
1379
|
+
const markerPath = statsWriteFailureMarkerPath(dir);
|
|
1380
|
+
try {
|
|
1381
|
+
if (Date.now() - fs.statSync(markerPath).mtimeMs < STATS_WRITE_FAILURE_LOG_MIN_INTERVAL_MS) return;
|
|
1382
|
+
} catch {
|
|
1383
|
+
}
|
|
1384
|
+
ensureDirSync(dir);
|
|
1385
|
+
fs.writeFileSync(markerPath, "");
|
|
1386
|
+
const line = `${(/* @__PURE__ */ new Date()).toISOString()} recordStat write failed for kind=${kind}: ${extractErrorMessage(err)}`;
|
|
1387
|
+
fs.appendFileSync(statsWriteFailureLogPath(dir), displaySafeText(line.replace(/[\n\r]+$/, "")) + "\n");
|
|
1388
|
+
} catch {
|
|
1389
|
+
}
|
|
1390
|
+
}
|
|
1391
|
+
var STATS_WRITE_BUSY_TIMEOUT_MS = 200;
|
|
1392
|
+
function recordStat(kind, bytesSaved = 0, tokensSaved = 0, _testDb, detail, traceparent, durationMs) {
|
|
1393
|
+
let db;
|
|
1394
|
+
try {
|
|
1395
|
+
if (!_testDb) getGlobalDb();
|
|
1396
|
+
db = _testDb ?? new Database(path.join(dataDir(), "global.db"), { timeout: STATS_WRITE_BUSY_TIMEOUT_MS });
|
|
1309
1397
|
const ts = Math.floor(Date.now() / 1e3);
|
|
1310
1398
|
const tp = traceparent ?? process.env["TRACEPARENT"] ?? process.env["traceparent"] ?? null;
|
|
1311
1399
|
const cols = ["ts", "kind", "bytes_saved", "tokens_saved", "detail"];
|
|
@@ -1322,11 +1410,23 @@ function recordStat(kind, bytesSaved = 0, tokensSaved = 0, _testDb, detail, trac
|
|
|
1322
1410
|
cols.push("tg_version");
|
|
1323
1411
|
vals.push(VERSION);
|
|
1324
1412
|
}
|
|
1413
|
+
if (durationMs !== void 0 && statsHasDurationColumn(db)) {
|
|
1414
|
+
cols.push("duration_ms");
|
|
1415
|
+
vals.push(Math.round(durationMs));
|
|
1416
|
+
}
|
|
1325
1417
|
db.prepare(
|
|
1326
1418
|
`INSERT INTO stats (${cols.join(", ")}) VALUES (${cols.map(() => "?").join(", ")})`
|
|
1327
1419
|
).run(...vals);
|
|
1328
1420
|
maybeRunStatsMaintenance(db);
|
|
1329
|
-
} catch {
|
|
1421
|
+
} catch (e) {
|
|
1422
|
+
recordStatWriteFailure(kind, e);
|
|
1423
|
+
} finally {
|
|
1424
|
+
if (db && !_testDb) {
|
|
1425
|
+
try {
|
|
1426
|
+
db.close();
|
|
1427
|
+
} catch {
|
|
1428
|
+
}
|
|
1429
|
+
}
|
|
1330
1430
|
}
|
|
1331
1431
|
}
|
|
1332
1432
|
var MAX_TOOL_NAME_CHARS = 200;
|
|
@@ -1356,6 +1456,18 @@ function readUnmappedTools(dbPath, homeDir) {
|
|
|
1356
1456
|
return [];
|
|
1357
1457
|
}
|
|
1358
1458
|
}
|
|
1459
|
+
function pruneStalePatternCoveredUnmappedTools(db, patterns) {
|
|
1460
|
+
if (patterns.length === 0) return;
|
|
1461
|
+
const regexes = patterns.map((p) => new RegExp(p));
|
|
1462
|
+
try {
|
|
1463
|
+
const rows = db.prepare("SELECT DISTINCT tool_name FROM unmapped_tools").all();
|
|
1464
|
+
const stale = rows.filter((r) => regexes.some((re) => re.test(r.tool_name)));
|
|
1465
|
+
if (stale.length === 0) return;
|
|
1466
|
+
const del = db.prepare("DELETE FROM unmapped_tools WHERE tool_name = ?");
|
|
1467
|
+
for (const r of stale) del.run(r.tool_name);
|
|
1468
|
+
} catch {
|
|
1469
|
+
}
|
|
1470
|
+
}
|
|
1359
1471
|
function summarize(windowDays = 30, testDb, homeDir) {
|
|
1360
1472
|
const t0 = Date.now();
|
|
1361
1473
|
const sinceTs = windowDays > 0 ? Math.floor((Date.now() - windowDays * 24 * 60 * 60 * 1e3) / 1e3) : null;
|
|
@@ -1496,18 +1608,11 @@ function _totalsLines(summary) {
|
|
|
1496
1608
|
`Total events: ${summary.total_events}`,
|
|
1497
1609
|
`Bytes saved: ${fmtBytes(summary.total_bytes_saved)}`,
|
|
1498
1610
|
`Tokens saved: ${summary.total_tokens_saved}`,
|
|
1499
|
-
// Disclosure, not a correction: `total_tokens_saved` above sums rows written under whichever
|
|
1500
|
-
// pricing formula was live when each was recorded, and `tg_version` cannot be read back into
|
|
1501
|
-
// "which formula" for a row from before this disclosure existed (PRICING_VERSION_UNRECORDED is
|
|
1502
|
-
// the overwhelming majority of all-time rows). Excluding those rows from the headline would
|
|
1503
|
-
// discard nearly the whole figure rather than fix it, so the honest move is to keep the sum and
|
|
1504
|
-
// say plainly that it spans more than one era, not to quietly present a mixed total as single-formula.
|
|
1611
|
+
// Disclosure, not a correction: `total_tokens_saved` above sums rows written under whichever pricing formula was live when each was recorded, and `tg_version` cannot be read back into "which formula" for a row from before this disclosure existed (PRICING_VERSION_UNRECORDED is the overwhelming majority of all-time rows). Excluding those rows from the headline would discard nearly the whole figure rather than fix it, so the honest move is to keep the sum and say plainly that it spans more than one era, not to quietly present a mixed total as single-formula.
|
|
1505
1612
|
...hasMixedPricingEras(summary) ? [
|
|
1506
1613
|
`Pricing note: totals mix ${countNoun(Object.keys(summary.by_pricing_version).length, "tg_version era")} (${countNoun(summary.by_pricing_version[PRICING_VERSION_UNRECORDED]?.events ?? 0, "row")} unrecorded); see 'token-goat stats --json' -> by_pricing_version for the breakdown`
|
|
1507
1614
|
] : [],
|
|
1508
|
-
// Printed on its own line, below the token total and never inside it, because it counts
|
|
1509
|
-
// placeholders rather than tokens. Omitted entirely when nothing was redacted, so the line is
|
|
1510
|
-
// information rather than a permanent zero. See COUNT_ONLY_KINDS.
|
|
1615
|
+
// Printed on its own line, below the token total and never inside it, because it counts placeholders rather than tokens. Omitted entirely when nothing was redacted, so the line is information rather than a permanent zero. See COUNT_ONLY_KINDS.
|
|
1511
1616
|
...summary.counts["secret_redacted"] ? [`Secrets hidden: ${summary.counts["secret_redacted"]} (a count, not tokens)`] : [],
|
|
1512
1617
|
`Window: ${summary.window_days} days`
|
|
1513
1618
|
];
|
|
@@ -1515,7 +1620,8 @@ function _totalsLines(summary) {
|
|
|
1515
1620
|
function _useRichStats() {
|
|
1516
1621
|
if (process.env["NO_COLOR"]) return false;
|
|
1517
1622
|
if (process.stdout.isTTY === true) return true;
|
|
1518
|
-
|
|
1623
|
+
const forceColor = process.env["FORCE_COLOR"];
|
|
1624
|
+
return forceColor !== void 0 && forceColor !== "0";
|
|
1519
1625
|
}
|
|
1520
1626
|
function _renderShortTotals(summary) {
|
|
1521
1627
|
const lines = [
|
|
@@ -1678,9 +1784,7 @@ var SECRET_PATTERNS = [
|
|
|
1678
1784
|
// Redacts only the token itself, not the "Authorization: Bearer " prefix -- the lookbehind anchors on the header name and scheme so the surrounding request-log line stays readable, matching how AWS_ACCESS_KEY_ID=... above keeps its own prefix intact. The optional quotes on either side of the colon are what let this see a header carried in JSON rather than in raw wire format. Without them the lookbehind demanded the colon sit directly against the header name and the scheme directly against the space, so a body like {"Authorization": "Bearer <token>"} -- the shape any logged fetch or MCP result arrives in -- matched nothing and the token was cached verbatim. `token` is the second scheme spelling in wide use -- it is what curl and gh examples pass for GitHub and many other APIs -- and an opaque value behind it carries exactly the same authority as one behind `Bearer`. The trailing gap is `{1,8}` rather than a single space for the same reason every other gap in this lookbehind already is: a hand-aligned or reformatted header ("Authorization: Bearer <token>") is ordinary, and demanding exactly one space there made this the one position in the pattern that a second space defeated.
|
|
1679
1785
|
["auth_bearer_token", /(?<=Authorization["']?[ \t]{0,8}:[ \t]{0,8}["']?[ \t]{0,8}(?:Bearer|token)[ \t]{1,8})[A-Za-z0-9\-._~+/]{10,}=*/gi],
|
|
1680
1786
|
["auth_basic_token", /(?<=Authorization["']?[ \t]{0,8}:[ \t]{0,8}["']?[ \t]{0,8}Basic[ \t])[A-Za-z0-9+/]{6,}=*/gi],
|
|
1681
|
-
// JWTs have no distinctive prefix of their own, but the base64url encoding of the smallest realistic header ('{"alg":' or similar) always starts with "eyJ", so that's the practical anchor here -- each of the three dot-separated segments requires a minimum length to avoid matching a short, coincidentally dotted token.
|
|
1682
|
-
//
|
|
1683
|
-
// The trailing `={0,2}` on each segment is what makes a padded token match. Base64url as the JWT spec defines it drops the `=` padding, but producers that reach for a plain base64 encoder emit it anyway, and a `=` in the header or payload segment used to defeat the match outright: not a partial redaction, but none at all, so the entire token was printed. `=` cannot appear anywhere except the end of a segment, since it is not one of the characters the segment body allows, so accepting it here cannot widen the match onto anything else.
|
|
1787
|
+
// JWTs have no distinctive prefix of their own, but the base64url encoding of the smallest realistic header ('{"alg":' or similar) always starts with "eyJ", so that's the practical anchor here -- each of the three dot-separated segments requires a minimum length to avoid matching a short, coincidentally dotted token. The trailing `={0,2}` on each segment is what makes a padded token match. Base64url as the JWT spec defines it drops the `=` padding, but producers that reach for a plain base64 encoder emit it anyway, and a `=` in the header or payload segment used to defeat the match outright: not a partial redaction, but none at all, so the entire token was printed. `=` cannot appear anywhere except the end of a segment, since it is not one of the characters the segment body allows, so accepting it here cannot widen the match onto anything else.
|
|
1684
1788
|
["jwt", /eyJ[A-Za-z0-9_-]{10,}={0,2}\.[A-Za-z0-9_-]{10,}={0,2}\.[A-Za-z0-9_-]{10,}={0,2}/g],
|
|
1685
1789
|
["npm_token", /npm_[A-Za-z0-9]{36}/g],
|
|
1686
1790
|
// rk_live_ (restricted keys) share the sk_live_/sk_test_ secret-key shape and risk level, so one pattern covers all three rather than adding a near-duplicate entry.
|
|
@@ -1692,9 +1796,7 @@ var SECRET_PATTERNS = [
|
|
|
1692
1796
|
["url_credentials", /(?<=:\/\/[^\s:@/]{1,64}:)[^\s:@/]{1,256}(?=@)/g],
|
|
1693
1797
|
// Azure storage account connection strings (`DefaultEndpointsProtocol=https;AccountName=...; AccountKey=<base64>==;EndpointSuffix=core.windows.net`) carry a full read/write key to the account in the AccountKey field, and none of the generic patterns above catch it: generic_secret_assignment's keyword list (password|passwd|secret|api[_-]?key| access[_-]?token|refresh[_-]?token|id[_-]?token) has nothing that matches "AccountKey", and an unanchored base64-shape pattern was deliberately rejected -- this module's header already warns that a bare high-entropy-blob heuristic false-fires on ordinary code, JSON and log output, and an 88-char base64 value is exactly the shape a hash, a compiled asset digest or a generated id can also take. Anchoring on the literal `AccountKey=` field name instead keeps the match specific to this one connection-string field. The value class is base64 proper (letters, digits, `+`, `/`, trailing `=` padding) and none of those characters include `;`, so the match terminates on its own at the `;` that starts the next `Name=` field -- unlike generic_secret_assignment's separator characters (`& ; # , :`), which double as ordinary credential characters and need a lookahead to tell the two roles apart, `;` is never valid base64 and needs no such lookahead here. The lookbehind keeps `AccountKey=` itself in the output, matching auth_bearer_token and presigned_signature above, so `;EndpointSuffix=core.windows.net` after it stays fully readable too. `SharedAccessKey` is the same credential one Azure service over: Service Bus, Event Hubs and Relay spell it that way (`Endpoint=sb://ns.servicebus.windows.net/;SharedAccessKeyName=Root; SharedAccessKey=<base64>`) and it carries the same authority over that namespace that AccountKey does over a storage account. `SharedAccessKeyName` is a plain identifier rather than a secret, and never matches: the separator in the lookbehind sits directly against the key name, so the `Name` in between stops it dead. The separator is spelled the way auth_bearer_token above spells its own, for the same reasons that comment records having learned the hard way. Optional quotes on either side of it, so the JSON and YAML forms a logged MCP result or api response actually arrives in are matched rather than stopped dead at the opening quote. A bounded gap rather than exactly one space, so a hand-aligned or reformatted `AccountKey = ...` in an appsettings file is not the single variant that defeats the whole pattern. Case-insensitive for the same reason presigned_signature is. The leading word boundary is what keeps the widened name from reaching into the middle of a longer identifier such as `myaccountkey=`.
|
|
1694
1798
|
["azure_storage_key", /(?<=\b(?:AccountKey|SharedAccessKey)["']?[ \t]{0,8}[:=][ \t]{0,8}["']?)[A-Za-z0-9+/]{40,}=*/gi],
|
|
1695
|
-
// Generic key=value assignments in .env-file and connection-string/query-string shape. The lookbehind again redacts only the value, and the value's character class deliberately excludes whitespace, '&', ';', '#', quote characters, and '[' ']' ':' -- that exclusion is what stops this from swallowing the rest of the line (a trailing comment or the next key=value pair) or the remainder of a query string past the matched parameter, which is exactly the kind of over-eager match this module's own design note above warns broad heuristics produce. The '[' ']' ':' exclusion also matters because this pattern runs last: an earlier pattern's own "OPENAI_API_KEY=[REDACTED:openai_project_key]" replacement text contains "API_KEY=" too, and without excluding those characters this pattern would re-match and double-redact its own placeholder. The length has a lower bound only (no upper bound): capping it at 64 used to leave the tail of any longer secret unredacted in plain text, which is worse than no redaction because it looks handled. A single negated-class quantifier like this cannot backtrack catastrophically -- there is no nested or overlapping quantifier for the engine to explore multiple ways of matching, so removing the upper bound does not introduce a ReDoS risk. Quotes are permitted around the separator, but never inside the value class. A quoted value is the ordinary way secrets are written -- .env files, JSON, YAML, TOML all quote by default -- and the lookbehind used to stop dead at the opening quote, so `API_KEY="..."` passed through in full while the bare `API_KEY=...` was caught. The closing quote of the key name blocked it from the other side too, which is what kept every JSON body unredacted. Keeping quotes out of the value class is still what stops the match running past the closing quote. The keyword may be a prefix of a longer key name rather than the whole of it, so the trailing identifier class below is load-bearing: without it the lookbehind required the keyword to sit immediately before the separator, and AWS_SECRET_ACCESS_KEY=, SECRET_KEY=, and DB_PASSWORD_HASH= all passed through in full. That class matches identifier characters only, so prose that merely mentions a keyword still never reaches a separator and stays unredacted. api[_-]?key covers the apikey and api-key spellings too. `& ; # , :` play two incompatible roles. They separate one field from the next (a query string, a cookie header, an inline env list), and they are also perfectly ordinary credential characters. Rejecting them outright got the first role right and the second badly wrong: the match stopped at the first one and left everything after it in plain text, so `password=corr&horse&battery` redacted four characters and printed the rest, and `DB_PASSWORD=Aa1:xyz` matched nothing at all because the run before the `:` was under the four-character floor. A tail left sitting in the open is the outcome this module's header calls worse than no redaction, because it reads as handled.
|
|
1696
|
-
//
|
|
1697
|
-
// So the separator role is decided by what follows rather than assumed: one of these characters ends the value only when the next thing along is another `name=` / `name:` pair, which is what an actual field separator is always followed by. `,OTHER=public` and `; other=1` still end it; the `&` in the middle of a passphrase does not. Whitespace, quotes and brackets are unchanged -- they end a value unconditionally, which is also what keeps this pattern from re-matching the `[REDACTED:...]` placeholder it just wrote.
|
|
1799
|
+
// Generic key=value assignments in .env-file and connection-string/query-string shape. The lookbehind again redacts only the value, and the value's character class deliberately excludes whitespace, '&', ';', '#', quote characters, and '[' ']' ':' -- that exclusion is what stops this from swallowing the rest of the line (a trailing comment or the next key=value pair) or the remainder of a query string past the matched parameter, which is exactly the kind of over-eager match this module's own design note above warns broad heuristics produce. The '[' ']' ':' exclusion also matters because this pattern runs last: an earlier pattern's own "OPENAI_API_KEY=[REDACTED:openai_project_key]" replacement text contains "API_KEY=" too, and without excluding those characters this pattern would re-match and double-redact its own placeholder. The length has a lower bound only (no upper bound): capping it at 64 used to leave the tail of any longer secret unredacted in plain text, which is worse than no redaction because it looks handled. A single negated-class quantifier like this cannot backtrack catastrophically -- there is no nested or overlapping quantifier for the engine to explore multiple ways of matching, so removing the upper bound does not introduce a ReDoS risk. Quotes are permitted around the separator, but never inside the value class. A quoted value is the ordinary way secrets are written -- .env files, JSON, YAML, TOML all quote by default -- and the lookbehind used to stop dead at the opening quote, so `API_KEY="..."` passed through in full while the bare `API_KEY=...` was caught. The closing quote of the key name blocked it from the other side too, which is what kept every JSON body unredacted. Keeping quotes out of the value class is still what stops the match running past the closing quote. The keyword may be a prefix of a longer key name rather than the whole of it, so the trailing identifier class below is load-bearing: without it the lookbehind required the keyword to sit immediately before the separator, and AWS_SECRET_ACCESS_KEY=, SECRET_KEY=, and DB_PASSWORD_HASH= all passed through in full. That class matches identifier characters only, so prose that merely mentions a keyword still never reaches a separator and stays unredacted. api[_-]?key covers the apikey and api-key spellings too. `& ; # , :` play two incompatible roles. They separate one field from the next (a query string, a cookie header, an inline env list), and they are also perfectly ordinary credential characters. Rejecting them outright got the first role right and the second badly wrong: the match stopped at the first one and left everything after it in plain text, so `password=corr&horse&battery` redacted four characters and printed the rest, and `DB_PASSWORD=Aa1:xyz` matched nothing at all because the run before the `:` was under the four-character floor. A tail left sitting in the open is the outcome this module's header calls worse than no redaction, because it reads as handled. So the separator role is decided by what follows rather than assumed: one of these characters ends the value only when the next thing along is another `name=` / `name:` pair, which is what an actual field separator is always followed by. `,OTHER=public` and `; other=1` still end it; the `&` in the middle of a passphrase does not. Whitespace, quotes and brackets are unchanged -- they end a value unconditionally, which is also what keeps this pattern from re-matching the `[REDACTED:...]` placeholder it just wrote.
|
|
1698
1800
|
["generic_secret_assignment", /(?<=(?:password|passwd|secret|api[_-]?key|access[_-]?token|refresh[_-]?token|id[_-]?token)[a-z0-9_-]{0,64}["']?[ \t]{0,8}[:=][ \t]{0,8}["']?)(?:\\[^\n]|[^\s\\&;#,:'"[\]{}]|[&;#,:](?![ \t]*[A-Za-z_][A-Za-z0-9_.-]*[ \t]*[:=])){4,}/gi]
|
|
1699
1801
|
];
|
|
1700
1802
|
function countRedactionPlaceholders(text) {
|
|
@@ -1777,6 +1879,28 @@ function characterClasses(s) {
|
|
|
1777
1879
|
function looksLikeCredential(s) {
|
|
1778
1880
|
return characterClasses(s) >= 3 && entropyBitsPerChar(s) >= 3.5;
|
|
1779
1881
|
}
|
|
1882
|
+
var RECALL_TUNED_KIND = "generic_secret_assignment";
|
|
1883
|
+
function matchesAnywhere(pattern, text) {
|
|
1884
|
+
pattern.lastIndex = 0;
|
|
1885
|
+
const hit = pattern.test(text);
|
|
1886
|
+
pattern.lastIndex = 0;
|
|
1887
|
+
return hit;
|
|
1888
|
+
}
|
|
1889
|
+
function hasPreciseSecret(text, config = loadConfig()) {
|
|
1890
|
+
for (const [kind, pattern] of SECRET_PATTERNS) {
|
|
1891
|
+
if (kind === RECALL_TUNED_KIND) continue;
|
|
1892
|
+
if (matchesAnywhere(pattern, text)) return true;
|
|
1893
|
+
}
|
|
1894
|
+
for (const custom of compileCustomPatterns(config.redaction.custom_patterns).patterns) {
|
|
1895
|
+
if (matchesAnywhere(custom, text)) return true;
|
|
1896
|
+
}
|
|
1897
|
+
if (config.redaction.strict) {
|
|
1898
|
+
for (const candidate of text.match(STRICT_CANDIDATE) ?? []) {
|
|
1899
|
+
if (looksLikeCredential(candidate)) return true;
|
|
1900
|
+
}
|
|
1901
|
+
}
|
|
1902
|
+
return false;
|
|
1903
|
+
}
|
|
1780
1904
|
function redactSecrets(text, config = loadConfig()) {
|
|
1781
1905
|
let count = 0;
|
|
1782
1906
|
let out = text;
|
|
@@ -1804,7 +1928,7 @@ function redactSecrets(text, config = loadConfig()) {
|
|
|
1804
1928
|
|
|
1805
1929
|
// src/disk_cache.ts
|
|
1806
1930
|
init_define_import_meta_env();
|
|
1807
|
-
import * as
|
|
1931
|
+
import * as fs2 from "node:fs";
|
|
1808
1932
|
import * as path2 from "node:path";
|
|
1809
1933
|
var DEFAULT_MAX_COUNT = 200;
|
|
1810
1934
|
var DEFAULT_MAX_AGE_MS = 24 * 3600 * 1e3;
|
|
@@ -1831,7 +1955,7 @@ function isBlobStale(subdir, id) {
|
|
|
1831
1955
|
const p = blobPath(subdir, id);
|
|
1832
1956
|
if (p === null) return false;
|
|
1833
1957
|
try {
|
|
1834
|
-
const stat =
|
|
1958
|
+
const stat = fs2.statSync(p);
|
|
1835
1959
|
return Date.now() - stat.mtimeMs > DEFAULT_MAX_AGE_MS;
|
|
1836
1960
|
} catch {
|
|
1837
1961
|
return false;
|
|
@@ -1873,7 +1997,7 @@ function storeBlob(subdir, id, value, opts = {}) {
|
|
|
1873
1997
|
if (Number.isFinite(maxBytesPerItem) && Buffer.byteLength(json, "utf-8") > maxBytesPerItem) return false;
|
|
1874
1998
|
try {
|
|
1875
1999
|
const dir = path2.dirname(p);
|
|
1876
|
-
if (!
|
|
2000
|
+
if (!fs2.existsSync(dir)) ensureDirSync(dir);
|
|
1877
2001
|
atomicWriteText(p, json);
|
|
1878
2002
|
} catch {
|
|
1879
2003
|
return false;
|
|
@@ -1891,8 +2015,8 @@ function loadBlob(subdir, id) {
|
|
|
1891
2015
|
const p = blobPath(subdir, id);
|
|
1892
2016
|
if (!p) return null;
|
|
1893
2017
|
try {
|
|
1894
|
-
if (!
|
|
1895
|
-
return JSON.parse(
|
|
2018
|
+
if (!fs2.existsSync(p)) return null;
|
|
2019
|
+
return JSON.parse(fs2.readFileSync(p, "utf8"));
|
|
1896
2020
|
} catch {
|
|
1897
2021
|
return null;
|
|
1898
2022
|
}
|
|
@@ -1901,13 +2025,13 @@ function listBlobs(subdir) {
|
|
|
1901
2025
|
const dir = blobDir(subdir);
|
|
1902
2026
|
const out = [];
|
|
1903
2027
|
try {
|
|
1904
|
-
if (!
|
|
1905
|
-
for (const file of
|
|
2028
|
+
if (!fs2.existsSync(dir)) return out;
|
|
2029
|
+
for (const file of fs2.readdirSync(dir)) {
|
|
1906
2030
|
if (!file.endsWith(".json")) continue;
|
|
1907
2031
|
const id = file.slice(0, -5);
|
|
1908
2032
|
let mtime = 0;
|
|
1909
2033
|
try {
|
|
1910
|
-
mtime =
|
|
2034
|
+
mtime = fs2.statSync(path2.join(dir, file)).mtimeMs;
|
|
1911
2035
|
} catch {
|
|
1912
2036
|
}
|
|
1913
2037
|
const value = loadBlob(subdir, id);
|
|
@@ -1924,15 +2048,15 @@ function pruneBlobs(subdir, maxCount = DEFAULT_MAX_COUNT, maxAgeMs = DEFAULT_MAX
|
|
|
1924
2048
|
function pruneBlobDir(dir, maxCount = DEFAULT_MAX_COUNT, maxAgeMs = DEFAULT_MAX_AGE_MS, maxBytes = Number.POSITIVE_INFINITY, protectedPath) {
|
|
1925
2049
|
let removed = 0;
|
|
1926
2050
|
try {
|
|
1927
|
-
if (!
|
|
2051
|
+
if (!fs2.existsSync(dir)) return 0;
|
|
1928
2052
|
const cutoff = Date.now() - maxAgeMs;
|
|
1929
2053
|
let kept = [];
|
|
1930
2054
|
let protectedEntry;
|
|
1931
|
-
for (const file of
|
|
2055
|
+
for (const file of fs2.readdirSync(dir)) {
|
|
1932
2056
|
const full = path2.join(dir, file);
|
|
1933
2057
|
let stat;
|
|
1934
2058
|
try {
|
|
1935
|
-
stat =
|
|
2059
|
+
stat = fs2.statSync(full);
|
|
1936
2060
|
} catch {
|
|
1937
2061
|
continue;
|
|
1938
2062
|
}
|
|
@@ -1943,7 +2067,7 @@ function pruneBlobDir(dir, maxCount = DEFAULT_MAX_COUNT, maxAgeMs = DEFAULT_MAX_
|
|
|
1943
2067
|
}
|
|
1944
2068
|
if (stat.mtimeMs < cutoff) {
|
|
1945
2069
|
try {
|
|
1946
|
-
|
|
2070
|
+
fs2.unlinkSync(full);
|
|
1947
2071
|
removed++;
|
|
1948
2072
|
} catch {
|
|
1949
2073
|
continue;
|
|
@@ -1959,7 +2083,7 @@ function pruneBlobDir(dir, maxCount = DEFAULT_MAX_COUNT, maxAgeMs = DEFAULT_MAX_
|
|
|
1959
2083
|
kept = kept.slice(kept.length - countBudget);
|
|
1960
2084
|
for (const [full] of excess) {
|
|
1961
2085
|
try {
|
|
1962
|
-
|
|
2086
|
+
fs2.unlinkSync(full);
|
|
1963
2087
|
removed++;
|
|
1964
2088
|
} catch {
|
|
1965
2089
|
continue;
|
|
@@ -1974,7 +2098,7 @@ function pruneBlobDir(dir, maxCount = DEFAULT_MAX_COUNT, maxAgeMs = DEFAULT_MAX_
|
|
|
1974
2098
|
if (!oldest) break;
|
|
1975
2099
|
const [full, , size] = oldest;
|
|
1976
2100
|
try {
|
|
1977
|
-
|
|
2101
|
+
fs2.unlinkSync(full);
|
|
1978
2102
|
removed++;
|
|
1979
2103
|
total -= size;
|
|
1980
2104
|
} catch {
|
|
@@ -2026,15 +2150,21 @@ export {
|
|
|
2026
2150
|
SOURCE_HINT,
|
|
2027
2151
|
savedTokensFromBytes,
|
|
2028
2152
|
formatLocalTimestamp,
|
|
2153
|
+
statsHasHarnessColumn,
|
|
2154
|
+
statsHasDurationColumn,
|
|
2155
|
+
getGlobalDb,
|
|
2156
|
+
HOOK_STATS_RETENTION_DAYS,
|
|
2029
2157
|
recordStat,
|
|
2030
2158
|
recordUnmappedTool,
|
|
2031
2159
|
readUnmappedTools,
|
|
2160
|
+
pruneStalePatternCoveredUnmappedTools,
|
|
2032
2161
|
summarize,
|
|
2033
2162
|
_useRichStats,
|
|
2034
2163
|
renderShortStats,
|
|
2035
2164
|
renderStats2 as renderStats,
|
|
2036
2165
|
countRedactionPlaceholders,
|
|
2037
2166
|
compileCustomPatterns,
|
|
2167
|
+
hasPreciseSecret,
|
|
2038
2168
|
redactSecrets,
|
|
2039
2169
|
DEFAULT_MAX_COUNT,
|
|
2040
2170
|
DEFAULT_MAX_AGE_MS,
|
|
@@ -13,10 +13,10 @@ import {
|
|
|
13
13
|
foldPath,
|
|
14
14
|
projectConfigPath,
|
|
15
15
|
runGit
|
|
16
|
-
} from "./token-goat-chunk-
|
|
16
|
+
} from "./token-goat-chunk-ZGQXRYKE.mjs";
|
|
17
17
|
import {
|
|
18
18
|
normalizeDarwinSystemAlias
|
|
19
|
-
} from "./token-goat-chunk-
|
|
19
|
+
} from "./token-goat-chunk-NMTKNYGF.mjs";
|
|
20
20
|
import {
|
|
21
21
|
init_define_import_meta_env
|
|
22
22
|
} from "./token-goat-chunk-A37V4PBF.mjs";
|
|
@@ -1202,6 +1202,8 @@ var PROJECT_LOCKED_KEYS = [
|
|
|
1202
1202
|
"indexing.skip_files",
|
|
1203
1203
|
"indexing.large_file_skip_kb",
|
|
1204
1204
|
"indexing.large_file_symbol_only_kb",
|
|
1205
|
+
// Same reasoning as the two thresholds above, on the chunk axis: a checked-in `.token-goat.toml` dropping this to 1 would take every file in the repository out of `semantic` for a reviewing agent, and `semantic` would answer "no matches" in the words it uses for a genuine absence.
|
|
1206
|
+
"indexing.max_chunks_per_file",
|
|
1205
1207
|
"indexing.cross_project_symbols",
|
|
1206
1208
|
"worker.blocked_roots"
|
|
1207
1209
|
];
|
|
@@ -1514,7 +1516,8 @@ var CONFIG_DEFAULTS = {
|
|
|
1514
1516
|
session_start_reminder: true
|
|
1515
1517
|
},
|
|
1516
1518
|
hooks: {
|
|
1517
|
-
watchdog_ms: 700
|
|
1519
|
+
watchdog_ms: 700,
|
|
1520
|
+
latency_budget_ms: 1500
|
|
1518
1521
|
},
|
|
1519
1522
|
webfetch: {
|
|
1520
1523
|
allow: [],
|
|
@@ -1542,6 +1545,8 @@ var CONFIG_DEFAULTS = {
|
|
|
1542
1545
|
indexing: {
|
|
1543
1546
|
large_file_symbol_only_kb: 500,
|
|
1544
1547
|
large_file_skip_kb: 2048,
|
|
1548
|
+
// 600 comes from a census of one real machine-wide index: 243,603 chunks over 62 projects and roughly 4,400 files. The largest hand-written source file anywhere in it produced 387 chunks (a 4,000-line test file); the next tier up starts at 513 and is entirely generated data snapshots, minified vendor assets and one HTML draft. 600 sits in that gap with about 1.5x headroom over the largest real file, and above it 32 files held 54,674 chunks -- 22% of the whole index from 0.7% of its files.
|
|
1549
|
+
max_chunks_per_file: 600,
|
|
1545
1550
|
skip_dirs: [],
|
|
1546
1551
|
skip_files: ["coverage.json", "coverage-final.json"],
|
|
1547
1552
|
embeddings_enabled: true,
|
|
@@ -1572,6 +1577,7 @@ var CONFIG_DEFAULTS = {
|
|
|
1572
1577
|
},
|
|
1573
1578
|
hint_stats: {
|
|
1574
1579
|
suppress_threshold_pct: 15,
|
|
1580
|
+
defiance_threshold_pct: 85,
|
|
1575
1581
|
min_sample_size: 5
|
|
1576
1582
|
},
|
|
1577
1583
|
semantic: {
|
|
@@ -1710,6 +1716,7 @@ var NUMERIC_FIELD_BOUNDS = {
|
|
|
1710
1716
|
"hints.cross_session_read_dedup_ttl_secs": { min: 1, max: 86400 },
|
|
1711
1717
|
"hints.mcp_dedup_ttl_secs": { min: 1, max: 3600 },
|
|
1712
1718
|
"hooks.watchdog_ms": { min: 100, max: 3e4 },
|
|
1719
|
+
"hooks.latency_budget_ms": { min: 1, max: 6e5 },
|
|
1713
1720
|
"webfetch.max_file_count": { min: 0, max: 1e7 },
|
|
1714
1721
|
"webfetch.max_bytes": { min: 0, max: 100 * 1024 * 1024 * 1024 },
|
|
1715
1722
|
"webfetch.compress_min_bytes": { min: 1024, max: 10 * 1024 * 1024 },
|
|
@@ -1717,8 +1724,10 @@ var NUMERIC_FIELD_BOUNDS = {
|
|
|
1717
1724
|
"worker.embed_threads": { min: 1, max: 16 },
|
|
1718
1725
|
"indexing.large_file_symbol_only_kb": { min: 1, max: 1048576, clampTo: "indexing.large_file_skip_kb" },
|
|
1719
1726
|
"indexing.large_file_skip_kb": { min: 1, max: 1048576 },
|
|
1727
|
+
"indexing.max_chunks_per_file": { min: 1, max: 1e6 },
|
|
1720
1728
|
"context.model_window_tokens": { min: 1e4, max: 1e7 },
|
|
1721
1729
|
"hint_stats.suppress_threshold_pct": { min: 0, max: 100 },
|
|
1730
|
+
"hint_stats.defiance_threshold_pct": { min: 0, max: 100 },
|
|
1722
1731
|
"hint_stats.min_sample_size": { min: 1, max: 1e4 },
|
|
1723
1732
|
"semantic.archive_weight": { min: 0.05, max: 1 },
|
|
1724
1733
|
"semantic.docs_weight": { min: 0.05, max: 1 }
|
|
@@ -2182,6 +2191,8 @@ function _buildConfig(raw, projectRaw = {}) {
|
|
|
2182
2191
|
const hk = getDefaultConfig("hooks");
|
|
2183
2192
|
hk.watchdog_ms = validatedInt(hk_raw["watchdog_ms"], hk.watchdog_ms, ...boundsOf("hooks.watchdog_ms"));
|
|
2184
2193
|
hk.watchdog_ms = envInt("TOKEN_GOAT_HOOK_WATCHDOG_MS", hk.watchdog_ms, ...boundsOf("hooks.watchdog_ms"));
|
|
2194
|
+
hk.latency_budget_ms = validatedInt(hk_raw["latency_budget_ms"], hk.latency_budget_ms, ...boundsOf("hooks.latency_budget_ms"));
|
|
2195
|
+
hk.latency_budget_ms = envInt("TOKEN_GOAT_HOOK_LATENCY_BUDGET_MS", hk.latency_budget_ms, ...boundsOf("hooks.latency_budget_ms"));
|
|
2185
2196
|
const wf_raw = section(raw, "webfetch");
|
|
2186
2197
|
const wf = getDefaultConfig("webfetch");
|
|
2187
2198
|
wf.allow = validatedStrList(wf_raw["allow"], wf.allow);
|
|
@@ -2209,6 +2220,7 @@ function _buildConfig(raw, projectRaw = {}) {
|
|
|
2209
2220
|
ix.large_file_symbol_only_kb = validatedInt(ix_raw["large_file_symbol_only_kb"], ix.large_file_symbol_only_kb, ...boundsOf("indexing.large_file_symbol_only_kb"));
|
|
2210
2221
|
ix.large_file_skip_kb = validatedInt(ix_raw["large_file_skip_kb"], ix.large_file_skip_kb, ...boundsOf("indexing.large_file_skip_kb"));
|
|
2211
2222
|
ix.large_file_symbol_only_kb = Math.min(ix.large_file_symbol_only_kb, ix.large_file_skip_kb);
|
|
2223
|
+
ix.max_chunks_per_file = validatedInt(ix_raw["max_chunks_per_file"], ix.max_chunks_per_file, ...boundsOf("indexing.max_chunks_per_file"));
|
|
2212
2224
|
ix.skip_dirs = validatedStrList(ix_raw["skip_dirs"], ix.skip_dirs);
|
|
2213
2225
|
ix.skip_files = validatedStrList(ix_raw["skip_files"], ix.skip_files);
|
|
2214
2226
|
ix.embeddings_enabled = validatedBool(ix_raw["embeddings_enabled"], ix.embeddings_enabled);
|
|
@@ -2250,6 +2262,8 @@ function _buildConfig(raw, projectRaw = {}) {
|
|
|
2250
2262
|
const hs_raw = section(raw, "hint_stats");
|
|
2251
2263
|
const hs = getDefaultConfig("hint_stats");
|
|
2252
2264
|
hs.suppress_threshold_pct = validatedInt(hs_raw["suppress_threshold_pct"], hs.suppress_threshold_pct, ...boundsOf("hint_stats.suppress_threshold_pct"));
|
|
2265
|
+
hs.defiance_threshold_pct = validatedInt(hs_raw["defiance_threshold_pct"], 100 - hs.suppress_threshold_pct, ...boundsOf("hint_stats.defiance_threshold_pct"));
|
|
2266
|
+
hs.defiance_threshold_pct = envInt("TOKEN_GOAT_HINT_DEFIANCE_THRESHOLD_PCT", hs.defiance_threshold_pct, ...boundsOf("hint_stats.defiance_threshold_pct"));
|
|
2253
2267
|
hs.min_sample_size = validatedInt(hs_raw["min_sample_size"], hs.min_sample_size, ...boundsOf("hint_stats.min_sample_size"));
|
|
2254
2268
|
const sem_raw = section(raw, "semantic");
|
|
2255
2269
|
const sem = getDefaultConfig("semantic");
|
|
@@ -2345,6 +2359,7 @@ var CONFIG_KEY_ENV_OVERRIDES = {
|
|
|
2345
2359
|
"hints.pre_skill_advisory": ["TOKEN_GOAT_PRE_SKILL_ADVISORY"],
|
|
2346
2360
|
"hints.quiet_hours": ["TOKEN_GOAT_QUIET_HOURS"],
|
|
2347
2361
|
"hooks.watchdog_ms": ["TOKEN_GOAT_HOOK_WATCHDOG_MS"],
|
|
2362
|
+
"hooks.latency_budget_ms": ["TOKEN_GOAT_HOOK_LATENCY_BUDGET_MS"],
|
|
2348
2363
|
"webfetch.max_file_count": ["TOKEN_GOAT_WEB_CACHE_MAX_FILES"],
|
|
2349
2364
|
"webfetch.max_bytes": ["TOKEN_GOAT_WEB_CACHE_MAX_BYTES"],
|
|
2350
2365
|
"webfetch.compress_bodies": ["TOKEN_GOAT_WEB_COMPRESS"],
|
|
@@ -2359,6 +2374,7 @@ var CONFIG_KEY_ENV_OVERRIDES = {
|
|
|
2359
2374
|
"redaction.custom_patterns": ["TOKEN_GOAT_REDACTION_CUSTOM_PATTERNS"],
|
|
2360
2375
|
"redaction.strict": ["TOKEN_GOAT_REDACTION_STRICT"],
|
|
2361
2376
|
"network.offline": ["TOKEN_GOAT_OFFLINE"],
|
|
2377
|
+
"hint_stats.defiance_threshold_pct": ["TOKEN_GOAT_HINT_DEFIANCE_THRESHOLD_PCT"],
|
|
2362
2378
|
"mcp.confine_reads_to_project_root": ["TOKEN_GOAT_MCP_CONFINE_READS"],
|
|
2363
2379
|
"mcp.allowed_roots": ["TOKEN_GOAT_MCP_ALLOWED_ROOTS"],
|
|
2364
2380
|
"webfetch.allow": ["TOKEN_GOAT_WEBFETCH_ALLOW"],
|
|
@@ -2498,7 +2514,8 @@ function saveConfig(config) {
|
|
|
2498
2514
|
session_start_reminder: config.hints.session_start_reminder
|
|
2499
2515
|
},
|
|
2500
2516
|
hooks: {
|
|
2501
|
-
watchdog_ms: config.hooks.watchdog_ms
|
|
2517
|
+
watchdog_ms: config.hooks.watchdog_ms,
|
|
2518
|
+
latency_budget_ms: config.hooks.latency_budget_ms
|
|
2502
2519
|
},
|
|
2503
2520
|
webfetch: {
|
|
2504
2521
|
allow: config.webfetch.allow,
|
|
@@ -2517,6 +2534,7 @@ function saveConfig(config) {
|
|
|
2517
2534
|
indexing: {
|
|
2518
2535
|
large_file_symbol_only_kb: config.indexing.large_file_symbol_only_kb,
|
|
2519
2536
|
large_file_skip_kb: config.indexing.large_file_skip_kb,
|
|
2537
|
+
max_chunks_per_file: config.indexing.max_chunks_per_file,
|
|
2520
2538
|
skip_dirs: config.indexing.skip_dirs,
|
|
2521
2539
|
skip_files: config.indexing.skip_files,
|
|
2522
2540
|
embeddings_enabled: config.indexing.embeddings_enabled,
|
|
@@ -2546,6 +2564,7 @@ function saveConfig(config) {
|
|
|
2546
2564
|
},
|
|
2547
2565
|
hint_stats: {
|
|
2548
2566
|
suppress_threshold_pct: config.hint_stats.suppress_threshold_pct,
|
|
2567
|
+
defiance_threshold_pct: config.hint_stats.defiance_threshold_pct,
|
|
2549
2568
|
min_sample_size: config.hint_stats.min_sample_size
|
|
2550
2569
|
},
|
|
2551
2570
|
semantic: {
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
import { createRequire as __cjsRequire } from 'node:module';
|
|
2
|
+
const require = __cjsRequire(import.meta.url);
|
|
3
|
+
import {
|
|
4
|
+
checkUpdateStatus,
|
|
5
|
+
cmdUpgrade,
|
|
6
|
+
compareSemver,
|
|
7
|
+
fetchLatestVersion
|
|
8
|
+
} from "./token-goat-chunk-GYC34EGE.mjs";
|
|
9
|
+
import "./token-goat-chunk-SOKAPOOC.mjs";
|
|
10
|
+
import "./token-goat-chunk-EEIDFMEM.mjs";
|
|
11
|
+
import "./token-goat-chunk-ZGQXRYKE.mjs";
|
|
12
|
+
import "./token-goat-chunk-NMTKNYGF.mjs";
|
|
13
|
+
import "./token-goat-chunk-GMOUBOX4.mjs";
|
|
14
|
+
import "./token-goat-chunk-A37V4PBF.mjs";
|
|
15
|
+
export {
|
|
16
|
+
checkUpdateStatus,
|
|
17
|
+
cmdUpgrade,
|
|
18
|
+
compareSemver,
|
|
19
|
+
fetchLatestVersion
|
|
20
|
+
};
|