token-goat 2.9.12 → 2.9.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/README.md +20 -1
  2. package/dist/token-goat-chunk-2ESBO4IN.mjs +209 -0
  3. package/dist/token-goat-chunk-3BTK54F3.mjs +1733 -0
  4. package/dist/{token-goat-chunk-6NIGPPN6.mjs → token-goat-chunk-3XQPEJMV.mjs} +1 -1
  5. package/dist/token-goat-chunk-4SDX3QP3.mjs +122 -0
  6. package/dist/token-goat-chunk-7OGKZ7AP.mjs +142 -0
  7. package/dist/token-goat-chunk-AH6QILZM.mjs +26 -0
  8. package/dist/token-goat-chunk-AMYCQJX4.mjs +1548 -0
  9. package/dist/{token-goat-chunk-OTW7LC4Y.mjs → token-goat-chunk-ASVVF4JV.mjs} +13 -10
  10. package/dist/{token-goat-chunk-GCXX67HM.mjs → token-goat-chunk-DAMXYVIW.mjs} +12477 -12156
  11. package/dist/token-goat-chunk-DBNY4RLN.mjs +308 -0
  12. package/dist/token-goat-chunk-EG3663UT.mjs +226 -0
  13. package/dist/token-goat-chunk-EZNVAIR3.mjs +21 -0
  14. package/dist/token-goat-chunk-GMOUBOX4.mjs +386 -0
  15. package/dist/{token-goat-chunk-6F5TLJC7.mjs → token-goat-chunk-HF6H7RNK.mjs} +1 -1
  16. package/dist/{token-goat-chunk-GM6QZWCF.mjs → token-goat-chunk-IT6O3PNN.mjs} +2660 -6877
  17. package/dist/token-goat-chunk-NEI4NC54.mjs +424 -0
  18. package/dist/token-goat-chunk-O5WAMISC.mjs +167 -0
  19. package/dist/{token-goat-chunk-LHLQFGWQ.mjs → token-goat-chunk-OMNQUUIT.mjs} +9 -27
  20. package/dist/{token-goat-chunk-SF2CKCFQ.mjs → token-goat-chunk-PGGDW7DZ.mjs} +30 -13
  21. package/dist/token-goat-chunk-POBYR64E.mjs +1632 -0
  22. package/dist/token-goat-chunk-PRJVGIC5.mjs +58 -0
  23. package/dist/{token-goat-chunk-4CK445AW.mjs → token-goat-chunk-RUDOKYPJ.mjs} +10074 -9765
  24. package/dist/token-goat-chunk-SAQ5PG4L.mjs +2634 -0
  25. package/dist/token-goat-chunk-T2IWWTHB.mjs +3155 -0
  26. package/dist/{token-goat-chunk-TOGYS5A7.mjs → token-goat-chunk-VZYD4OZB.mjs} +1469 -5752
  27. package/dist/token-goat-chunk-XEH6KBWW.mjs +23 -0
  28. package/dist/token-goat-chunk-XQF5J25J.mjs +33 -0
  29. package/dist/token-goat-chunk-XTQAOTSO.mjs +89 -0
  30. package/dist/{token-goat-chunk-THLHC6QJ.mjs → token-goat-chunk-XVZ4MNQC.mjs} +614 -603
  31. package/dist/token-goat-chunk-Y4AFKTHK.mjs +22 -0
  32. package/dist/token-goat-chunk-YHGTGG6K.mjs +396 -0
  33. package/dist/{token-goat-chunk-SBYRP3X4.mjs → token-goat-chunk-YTQHZJXW.mjs} +61 -27
  34. package/dist/token-goat-chunk-Z6UXPYJA.mjs +62 -0
  35. package/dist/token-goat-chunk-ZD4EM4LR.mjs +30 -0
  36. package/dist/token-goat-chunk-ZFM4PWXL.mjs +585 -0
  37. package/dist/{token-goat-chunk-A6QTLAWO.mjs → token-goat-chunk-ZYNNQ36L.mjs} +128 -328
  38. package/dist/token-goat-hook.mjs +12 -6
  39. package/dist/token-goat.core.mjs +22 -7
  40. package/docs/cli.md +12 -8
  41. package/docs/security.md +1 -1
  42. package/package.json +4 -2
  43. package/dist/token-goat-chunk-HKFOH6JH.mjs +0 -28
  44. package/dist/token-goat-chunk-LOCOX2ML.mjs +0 -3564
  45. package/dist/token-goat-chunk-VRWX6QYW.mjs +0 -24
@@ -0,0 +1,3155 @@
1
+ import { createRequire as __cjsRequire } from 'node:module';
2
+ const require = __cjsRequire(import.meta.url);
3
+ import {
4
+ getHarnessName,
5
+ loadConfig
6
+ } from "./token-goat-chunk-SAQ5PG4L.mjs";
7
+ import {
8
+ registerReset
9
+ } from "./token-goat-chunk-EEIDFMEM.mjs";
10
+ import {
11
+ _detectOpenQuote,
12
+ _lineClosesQuote,
13
+ isAblSource,
14
+ isLatexClassFile,
15
+ isMatlabSource,
16
+ isObjcHeader,
17
+ isObjcSource,
18
+ isPascalSource,
19
+ isPerlSource,
20
+ isPrologSource
21
+ } from "./token-goat-chunk-POBYR64E.mjs";
22
+ import {
23
+ SYMBOL_BODY_CHAR_CAP,
24
+ VERSION,
25
+ atomicWriteText,
26
+ countNoun,
27
+ dataDir,
28
+ dataDirForHome,
29
+ displaySafeText,
30
+ ensureDirSync,
31
+ foldCase,
32
+ foldPath,
33
+ safeJoin,
34
+ sanitizeIdForFilename,
35
+ sleepSync,
36
+ tokenGoatHome
37
+ } from "./token-goat-chunk-AMYCQJX4.mjs";
38
+ import {
39
+ growsExponentially,
40
+ hasNestedQuantifier
41
+ } from "./token-goat-chunk-GMOUBOX4.mjs";
42
+ import {
43
+ init_define_import_meta_env
44
+ } from "./token-goat-chunk-A37V4PBF.mjs";
45
+
46
+ // src/parser_types.ts
47
+ init_define_import_meta_env();
48
+ import * as fs from "node:fs";
49
+ import * as path from "node:path";
50
+
51
+ // src/language_specs.ts
52
+ init_define_import_meta_env();
53
+ var CODE = { symbolBearing: true, sourceHints: true, grepSource: true, diffable: true };
54
+ var DATA = { symbolBearing: false, sourceHints: false, grepSource: false, diffable: false };
55
+ var LANGUAGE_SPECS = [
56
+ { id: "typescript", extraction: "tree-sitter", extensions: [".ts", ".tsx", ".mts", ".cts"], ...CODE, fence: "typescript", fenceByExtension: { ".tsx": "tsx" } },
57
+ { id: "javascript", extraction: "tree-sitter", extensions: [".js", ".jsx", ".mjs", ".cjs"], ...CODE, fence: "javascript", fenceByExtension: { ".jsx": "jsx" } },
58
+ // Bazel and Starlark (`.bzl`, `.star`, BUILD/WORKSPACE/MODULE.bazel) are a Python dialect: `def` and top-level calls such as `load(...)` parse with the Python grammar (https://github.com/bazelbuild/starlark/blob/master/spec.md).
59
+ { id: "python", extraction: "tree-sitter", extensions: [".py", ".pyi", ".bzl", ".star"], basenames: ["build.bazel", "workspace.bazel", "module.bazel"], exactBasenames: ["BUILD", "WORKSPACE"], ...CODE, fence: "python" },
60
+ { id: "go", extraction: "tree-sitter", extensions: [".go"], ...CODE, fence: "go" },
61
+ { id: "rust", extraction: "tree-sitter", extensions: [".rs"], ...CODE, fence: "rust" },
62
+ // Rake task files, Gemfile, Rakefile and the other extensionless Ruby DSL files are plain Ruby syntax.
63
+ { id: "ruby", extraction: "tree-sitter", extensions: [".rb", ".ruby", ".rake"], basenames: ["gemfile", "rakefile", "vagrantfile", "guardfile", "podfile", "capfile", "fastfile", "brewfile"], ...CODE, fence: "ruby" },
64
+ { id: "java", extraction: "tree-sitter", extensions: [".java"], ...CODE, fence: "java" },
65
+ // `.h` is C unless it declares an Objective-C `@interface` or `@protocol` (refineLanguageByContent); a C++ header parses with the C grammar.
66
+ { id: "c", extraction: "tree-sitter", extensions: [".c", ".h"], label: "C", ...CODE, fence: "c" },
67
+ { id: "cpp", extraction: "tree-sitter", extensions: [".cpp", ".cc", ".cxx", ".hpp", ".hxx"], label: "C++", ...CODE, fence: "cpp" },
68
+ // zsh, ksh and bats scripts share the POSIX `name() {` and `function name {` function forms the bash adapter reads.
69
+ { id: "bash", extraction: "regex", extensions: [".sh", ".bash", ".zsh", ".ksh", ".bats"], ...DATA, symbolBearing: true, fence: "bash" },
70
+ // MDX headings are plain ATX; `.rst` needs an underline heading parser this adapter lacks, so it stays unmapped.
71
+ { id: "markdown", extraction: "regex", extensions: [".md", ".markdown", ".mdx"], ...DATA, fence: "markdown" },
72
+ { id: "toml", extraction: "regex", extensions: [".toml"], basenames: ["cargo.toml", "pyproject.toml"], label: "TOML", ...DATA, diffable: true, fence: "toml" },
73
+ // JSON with comments (`.jsonc`) and Avro schemas (`.avsc`) read with the JSON adapter, which skips `//` and `/* */` comments.
74
+ { id: "json", extraction: "regex", extensions: [".json", ".jsonc", ".avsc"], basenames: ["package.json", "tsconfig.json"], label: "JSON", ...DATA, diffable: true, fence: "json" },
75
+ { id: "yaml", extraction: "regex", extensions: [".yaml", ".yml"], label: "YAML", ...DATA, diffable: true, fence: "yaml" },
76
+ { id: "css", extraction: "regex", extensions: [".css", ".scss", ".sass", ".less"], label: "CSS", ...DATA, grepSource: true, diffable: true, fence: "css", fenceByExtension: { ".scss": "scss", ".sass": "sass", ".less": "less" } },
77
+ { id: "dockerfile", extraction: "regex", extensions: [], basenames: ["dockerfile"], ...DATA, fence: "dockerfile" },
78
+ { id: "csharp", extraction: "regex", extensions: [".cs"], label: "C#", ...CODE, fence: "csharp" },
79
+ { id: "php", extraction: "regex", extensions: [".php"], label: "PHP", ...CODE, fence: "php" },
80
+ { id: "html", extraction: "regex", extensions: [".html", ".htm"], label: "HTML", ...DATA, fence: "html" },
81
+ { id: "liquid", extraction: "regex", extensions: [".liquid"], ...DATA, fence: "liquid" },
82
+ // Six template dialects, each masking its own delimiters out then handing off to the HTML
83
+ // extractor (src/languages/templates_idx.ts) -- markup formats, so DATA defaults like html/liquid.
84
+ { id: "jinja2", extraction: "regex", extensions: [".j2", ".jinja", ".jinja2"], label: "Jinja2", ...DATA, fence: "jinja" },
85
+ { id: "handlebars", extraction: "regex", extensions: [".hbs", ".handlebars"], label: "Handlebars", ...DATA, fence: "handlebars" },
86
+ { id: "erb", extraction: "regex", extensions: [".erb"], label: "ERB", ...DATA, fence: "erb" },
87
+ { id: "ejs", extraction: "regex", extensions: [".ejs"], label: "EJS", ...DATA, fence: "ejs" },
88
+ { id: "nunjucks", extraction: "regex", extensions: [".njk"], label: "Nunjucks", ...DATA, fence: "html" },
89
+ { id: "twig", extraction: "regex", extensions: [".twig"], label: "Twig", ...DATA, fence: "twig" },
90
+ { id: "kotlin", extraction: "regex", extensions: [".kt", ".kts"], ...CODE, fence: "kotlin" },
91
+ { id: "swift", extraction: "regex", extensions: [".swift"], ...CODE, fence: "swift" },
92
+ { id: "scala", extraction: "regex", extensions: [".scala", ".sc"], ...CODE, fence: "scala" },
93
+ { id: "lua", extraction: "regex", extensions: [".lua"], ...CODE, fence: "lua" },
94
+ { id: "elixir", extraction: "regex", extensions: [".ex", ".exs"], ...CODE, fence: "elixir" },
95
+ { id: "dart", extraction: "regex", extensions: [".dart"], ...CODE, fence: "dart" },
96
+ { id: "zig", extraction: "regex", extensions: [".zig"], ...CODE, fence: "zig" },
97
+ { id: "r", extraction: "regex", extensions: [".r"], ...CODE, fence: "r" },
98
+ { id: "graphql", extraction: "regex", extensions: [".graphql", ".gql"], label: "GraphQL", ...DATA, symbolBearing: true, fence: "graphql" },
99
+ // Oracle PL/SQL sources: package spec and body, standalone procedure/function, trigger, and object type spec and body.
100
+ { id: "sql", extraction: "regex", extensions: [".sql", ".pks", ".pkb", ".pls", ".plsql", ".pck", ".prc", ".fnc", ".trg", ".tps", ".tpb"], label: "SQL", ...DATA, symbolBearing: true, diffable: true, fence: "sql" },
101
+ { id: "ini", extraction: "regex", extensions: [".ini", ".cfg", ".conf"], label: "INI", ...DATA, fence: "ini" },
102
+ // `.mk` fragments (config.mk, rules.mk) share a bare Makefile's syntax.
103
+ { id: "makefile", extraction: "regex", extensions: [".mk"], basenames: ["makefile", "gnumakefile", "bsdmakefile"], ...DATA, fence: "makefile", basenameImportsExtension: ".mk" },
104
+ { id: "proto", extraction: "regex", extensions: [".proto"], label: "Protocol Buffers", ...DATA, symbolBearing: true, fence: "protobuf" },
105
+ { id: "terraform", extraction: "regex", extensions: [".tf", ".tfvars", ".hcl"], ...DATA, symbolBearing: true, fence: "hcl" },
106
+ // `.env` and `.env.<suffix>` are matched by parser_types.ts's DOTENV_VARIANT_RE before this table.
107
+ { id: "env_file", extraction: "regex", extensions: [".env"], basenames: [".envrc"], label: "env file", ...DATA },
108
+ { id: "powershell", extraction: "regex", extensions: [".ps1", ".psm1"], label: "PowerShell", ...CODE, fence: "powershell" },
109
+ // VB.NET, VB6/VBA standard modules, VBScript and VB6 forms. A VB6 class module shares `.cls` with Apex (and LaTeX) and is told apart by content in refineLanguageByContent.
110
+ { id: "vb", extraction: "regex", extensions: [".vb", ".bas", ".vbs", ".frm"], label: "Visual Basic", ...CODE, fence: "vb", fenceByExtension: { ".vb": "vbnet", ".vbs": "vbscript" } },
111
+ {
112
+ id: "cobol",
113
+ extraction: "own-result",
114
+ extensions: [".cbl", ".cob", ".cobol", ".cpy"],
115
+ label: "COBOL",
116
+ ...CODE,
117
+ fence: "cobol",
118
+ partialRefsReason: "PERFORM, GO TO and CALL 'literal' are recorded as references, but a paragraph also runs by falling through from the one above it and a program can be called through a data item holding its name, so `dead` skips COBOL"
119
+ },
120
+ // Natural object sources as NaturalONE and SYSOBJH export them. Maps (.nsm) and DDMs (.nsd) are layouts, not code, and stay unmapped.
121
+ {
122
+ id: "natural",
123
+ extraction: "own-result",
124
+ extensions: [".nsp", ".nsn", ".nss", ".nsa", ".nsl", ".nsg", ".nsc", ".nsh"],
125
+ label: "Natural",
126
+ ...CODE,
127
+ fence: "natural",
128
+ partialRefsReason: "PERFORM, CALLNAT 'literal' and FETCH 'literal' are recorded as references, but an object can also be called through a variable holding its name, so `dead` skips Natural"
129
+ },
130
+ { id: "abap", extraction: "regex", extensions: [".abap"], label: "ABAP", ...CODE, fence: "abap" },
131
+ { id: "sas", extraction: "regex", extensions: [".sas"], label: "SAS", ...CODE, fence: "sas" },
132
+ { id: "pli", extraction: "regex", extensions: [".pli", ".pl1"], label: "PL/I", ...CODE, fence: "pli" },
133
+ // `.rpg` stays unmapped: RPG II and RPG III sources use it, and their fixed layout predates the ILE RPG forms this adapter reads.
134
+ { id: "rpg", extraction: "regex", extensions: [".rpgle", ".sqlrpgle"], label: "RPG", ...CODE, fence: "rpgle" },
135
+ { id: "jcl", extraction: "regex", extensions: [".jcl"], label: "JCL", ...CODE, fence: "jcl" },
136
+ // `.mm` is always Objective-C++. A `.m` (MATLAB uses it too) is Objective-C only on an `#import`, `@interface`, `@implementation` or `@protocol` line, and a `.h` only on `@interface` or `@protocol`: refineLanguageByContent in parser_types.ts decides.
137
+ { id: "objc", extraction: "regex", extensions: [".mm"], label: "Objective-C", ...CODE, fence: "objectivec" },
138
+ // Gradle build scripts and Jenkinsfiles are Groovy.
139
+ { id: "groovy", extraction: "regex", extensions: [".groovy", ".gvy", ".gradle"], basenames: ["jenkinsfile"], label: "Groovy", ...CODE, fence: "groovy", basenameImportsExtension: ".groovy" },
140
+ // A Prolog `.pl` stays unknown (refineLanguageByContent), and a `.t` is Perl only on a Perl marker line.
141
+ { id: "perl", extraction: "regex", extensions: [".pl", ".pm"], label: "Perl", ...CODE, fence: "perl" },
142
+ { id: "solidity", extraction: "regex", extensions: [".sol"], label: "Solidity", ...CODE, fence: "solidity" },
143
+ { id: "thrift", extraction: "regex", extensions: [".thrift"], label: "Thrift", ...DATA, symbolBearing: true, fence: "thrift" },
144
+ { id: "glsl", extraction: "regex", extensions: [".glsl", ".vert", ".frag", ".comp", ".geom", ".tesc", ".tese"], label: "GLSL", ...CODE, fence: "glsl" },
145
+ // `.fx` stays unmapped: other languages use it too.
146
+ { id: "hlsl", extraction: "regex", extensions: [".hlsl", ".hlsli"], label: "HLSL", ...CODE, fence: "hlsl" },
147
+ { id: "wgsl", extraction: "regex", extensions: [".wgsl"], label: "WGSL", ...CODE, fence: "wgsl" },
148
+ { id: "metal", extraction: "regex", extensions: [".metal"], label: "Metal", ...CODE, fence: "metal" },
149
+ // `.f`, `.for` and `.f77` are read as fixed form unless code starts in column 1. `.fpp` stays unmapped: it is used for both forms.
150
+ { id: "fortran", extraction: "regex", extensions: [".f", ".for", ".f77", ".f90", ".f95", ".f03", ".f08"], label: "Fortran", ...CODE, fence: "fortran" },
151
+ // `.inc` stays unmapped (many languages use it), and a `.pp` (Puppet uses it too) is Pascal only on a unit, program or library header: refineLanguageByContent in parser_types.ts decides. Text-form `.dfm` forms list their components.
152
+ { id: "pascal", extraction: "regex", extensions: [".pas", ".dpr", ".dpk", ".lpr", ".dfm"], label: "Pascal", ...CODE, fence: "pascal" },
153
+ // MATLAB has no extension of its own: a `.m` that is not Objective-C is MATLAB only on a `function` or `classdef` header line, which refineLanguageByContent in parser_types.ts checks, so a Mathematica or Mercury `.m` stays unknown.
154
+ { id: "matlab", extraction: "regex", extensions: [], label: "MATLAB", ...CODE, fence: "matlab" },
155
+ { id: "cmake", extraction: "regex", extensions: [".cmake"], basenames: ["cmakelists.txt"], label: "CMake", ...CODE, fence: "cmake", basenameImportsExtension: ".cmake" },
156
+ // One adapter for the three assembly dialects that share these extensions: GNU as (`.s`, and `.S` through the lowercase lookup), NASM (`.asm`, `.nasm`) and IBM High Level Assembler (`.asm`), which the adapter tells apart by content. `.inc` stays unmapped: many languages use it.
157
+ { id: "asm", extraction: "regex", extensions: [".s", ".asm", ".nasm"], label: "Assembly", ...CODE, fence: "asm" },
158
+ { id: "batch", extraction: "regex", extensions: [".bat", ".cmd"], label: "Windows batch", ...CODE, fence: "batch" },
159
+ { id: "erlang", extraction: "regex", extensions: [".erl", ".hrl"], label: "Erlang", ...CODE, fence: "erlang" },
160
+ { id: "vhdl", extraction: "regex", extensions: [".vhd", ".vhdl"], label: "VHDL", ...CODE, fence: "vhdl" },
161
+ // Five Lisp-family dialects, each its own row: none shares an extractor (see common_lisp.ts's
162
+ // module doc for why their lexical rules stay separate rather than a single Lisp masker).
163
+ { id: "common_lisp", extraction: "regex", extensions: [".lisp", ".lsp", ".cl"], label: "Common Lisp", ...CODE, fence: "lisp" },
164
+ { id: "scheme", extraction: "regex", extensions: [".scm", ".ss"], label: "Scheme", ...CODE, fence: "scheme" },
165
+ { id: "racket", extraction: "regex", extensions: [".rkt", ".rktl"], label: "Racket", ...CODE, fence: "racket" },
166
+ { id: "clojure", extraction: "regex", extensions: [".clj", ".cljs", ".cljc"], label: "Clojure", ...CODE, fence: "clojure" },
167
+ { id: "emacs_lisp", extraction: "regex", extensions: [".el"], label: "Emacs Lisp", ...CODE, fence: "lisp" },
168
+ // `.lhs` (literate Haskell) is deliberately not claimed here -- see haskell.ts's module doc.
169
+ { id: "haskell", extraction: "regex", extensions: [".hs"], label: "Haskell", ...CODE, fence: "haskell" },
170
+ // `.mli` interface files share `.ml`'s lexical rules (comments, strings, quoted strings) and are read with the same extractor -- see ocaml.ts's module doc.
171
+ { id: "ocaml", extraction: "regex", extensions: [".ml", ".mli"], label: "OCaml", ...CODE, fence: "ocaml" },
172
+ // `.fsi` signature files and `.fsx` scripts share `.fs`'s lexical rules and are read with the same extractor -- see fsharp.ts's module doc.
173
+ { id: "fsharp", extraction: "regex", extensions: [".fs", ".fsi", ".fsx"], label: "F#", ...CODE, fence: "fsharp" },
174
+ { id: "nix", extraction: "regex", extensions: [".nix"], label: "Nix", ...CODE, fence: "nix" },
175
+ // OpenEdge ABL has no extension of its own: a `.p` or `.w` (Pascal and CWEB use them too) or a `.cls` (Apex, VB6, LaTeX) is ABL only when its head carries an ABL marker, which refineLanguageByContent in parser_types.ts checks. The path-only hooks see a `.p` or `.w` as unknown and a `.cls` as Apex.
176
+ { id: "abl", extraction: "regex", extensions: [], label: "OpenEdge ABL", ...CODE, fence: "abl" },
177
+ { id: "apex", extraction: "regex", extensions: [".cls", ".trigger"], label: "Apex", ...CODE, fence: "apex" },
178
+ // Matched by the `-meta.xml` suffix in detectLanguage, not by an extension.
179
+ { id: "salesforce_metadata", extraction: "regex", extensions: [], label: "Salesforce metadata", ...DATA, symbolBearing: true, sourceHints: true, grepSource: true, fence: "xml" },
180
+ { id: "salesforce_markup", extraction: "own-result", extensions: [".cmp", ".app", ".evt", ".intf", ".design", ".auradoc", ".tokens", ".page", ".component", ".email"], label: "Salesforce markup", ...DATA, symbolBearing: true, sourceHints: true, grepSource: true, fence: "xml" },
181
+ { id: "vue", extraction: "own-result", extensions: [".vue"], ...DATA, fence: "vue" },
182
+ { id: "svelte", extraction: "own-result", extensions: [".svelte"], ...DATA, fence: "svelte" },
183
+ { id: "astro", extraction: "own-result", extensions: [".astro"], ...DATA, fence: "astro" },
184
+ // Notebooks index through their code cells as Python, in parser.ts's ipynb branch.
185
+ { id: "ipynb", extraction: "own-result", extensions: [".ipynb"], label: "Jupyter notebook", ...DATA, fence: "json" }
186
+ ];
187
+ var SPEC_BY_ID = new Map(LANGUAGE_SPECS.map((s) => [s.id, s]));
188
+ function rows() {
189
+ return LANGUAGE_SPECS;
190
+ }
191
+ var EXTENSION_LANGUAGE = new Map(
192
+ LANGUAGE_SPECS.flatMap((s) => s.extensions.map((e) => [e, s.id]))
193
+ );
194
+ var FILENAME_LANGUAGE = new Map(
195
+ rows().flatMap((s) => (s.basenames ?? []).map((b) => [b, s.id]))
196
+ );
197
+ var EXACT_FILENAME_LANGUAGE = new Map(
198
+ rows().flatMap((s) => (s.exactBasenames ?? []).map((b) => [b, s.id]))
199
+ );
200
+ var TREE_SITTER_LANGUAGES = LANGUAGE_SPECS.filter((s) => s.extraction === "tree-sitter").map((s) => s.id);
201
+ function languageHasFlag(language, flag) {
202
+ return SPEC_BY_ID.get(language)?.[flag] === true;
203
+ }
204
+ function languageLabel(language) {
205
+ if (language === "unknown") return "this file type";
206
+ return SPEC_BY_ID.get(language)?.label ?? language;
207
+ }
208
+ function fenceFor(language, ext) {
209
+ const spec = SPEC_BY_ID.get(language);
210
+ if (spec === void 0) return "";
211
+ return spec.fenceByExtension?.[ext] ?? spec.fence ?? "";
212
+ }
213
+ function basenameImportsExtension(language) {
214
+ return SPEC_BY_ID.get(language)?.basenameImportsExtension;
215
+ }
216
+ function partialRefsReason(language) {
217
+ return SPEC_BY_ID.get(language)?.partialRefsReason;
218
+ }
219
+
220
+ // src/parser_types.ts
221
+ var DOTENV_VARIANT_RE = /^\.env(\..+)?$/;
222
+ var VB6_HEADER_SCAN_LINES = 40;
223
+ function isVb6ClassModule(content) {
224
+ if (content.includes("\0")) return false;
225
+ const lines = (content.charCodeAt(0) === 65279 ? content.slice(1) : content).split(/\r?\n/, VB6_HEADER_SCAN_LINES);
226
+ const first = lines.find((l) => l.trim() !== "");
227
+ if (first !== void 0 && /^VERSION\s+1\.0\s+CLASS\b/i.test(first.trim())) return true;
228
+ return lines.some((l) => /^Attribute\s+VB_Name\s*=\s*"/i.test(l.trim()));
229
+ }
230
+ function refineLanguageByContent(filePath, language, content) {
231
+ const sniff = CONTENT_SNIFFS.get(path.extname(filePath).toLowerCase());
232
+ if (sniff === void 0 || sniff.from !== language) return language;
233
+ return sniff.refine(content) ?? language;
234
+ }
235
+ var LANGUAGE_SNIFF_BYTES = 8192;
236
+ function sniffHead(content) {
237
+ const head = content.slice(0, LANGUAGE_SNIFF_BYTES);
238
+ return Buffer.byteLength(head, "utf8") <= LANGUAGE_SNIFF_BYTES ? head : Buffer.from(head, "utf8").subarray(0, LANGUAGE_SNIFF_BYTES).toString("utf8");
239
+ }
240
+ var ablOrUnknown = (c) => isAblSource(c) ? "abl" : void 0;
241
+ var clsRefine = (c) => {
242
+ if (isVb6ClassModule(c)) return "vb";
243
+ const abl = ablOrUnknown(c);
244
+ if (abl !== void 0) return abl;
245
+ if (isLatexClassFile(c)) return "unknown";
246
+ return void 0;
247
+ };
248
+ var CONTENT_SNIFFS = /* @__PURE__ */ new Map([
249
+ [".cls", { from: "apex", refine: clsRefine }],
250
+ [".p", { from: "unknown", refine: ablOrUnknown }],
251
+ [".w", { from: "unknown", refine: ablOrUnknown }],
252
+ [".m", { from: "unknown", refine: (c) => isObjcSource(sniffHead(c)) ? "objc" : isMatlabSource(sniffHead(c)) ? "matlab" : void 0 }],
253
+ [".pp", { from: "unknown", refine: (c) => isPascalSource(sniffHead(c)) ? "pascal" : void 0 }],
254
+ [".h", { from: "c", refine: (c) => isObjcHeader(sniffHead(c)) ? "objc" : void 0 }],
255
+ [".pl", { from: "perl", refine: (c) => isPrologSource(sniffHead(c)) ? "unknown" : void 0 }],
256
+ [".t", { from: "unknown", refine: (c) => isPerlSource(sniffHead(c)) ? "perl" : void 0 }]
257
+ ]);
258
+ function needsContentSniff(filePath, language) {
259
+ return CONTENT_SNIFFS.get(path.extname(filePath).toLowerCase())?.from === language;
260
+ }
261
+ function detectLanguageOfFile(filePath) {
262
+ const language = detectLanguage(filePath);
263
+ if (!needsContentSniff(filePath, language)) return language;
264
+ try {
265
+ const fd = fs.openSync(filePath, "r");
266
+ try {
267
+ const buf = Buffer.alloc(LANGUAGE_SNIFF_BYTES);
268
+ const n = fs.readSync(fd, buf, 0, buf.length, 0);
269
+ return refineLanguageByContent(filePath, language, buf.subarray(0, n).toString("utf8"));
270
+ } finally {
271
+ fs.closeSync(fd);
272
+ }
273
+ } catch {
274
+ return language;
275
+ }
276
+ }
277
+ function detectLanguage(filePath) {
278
+ const exactBase = path.basename(filePath);
279
+ const base = exactBase.toLowerCase();
280
+ if (DOTENV_VARIANT_RE.test(base)) return "env_file";
281
+ const byName = EXACT_FILENAME_LANGUAGE.get(exactBase) ?? FILENAME_LANGUAGE.get(base);
282
+ if (byName !== void 0) return byName;
283
+ if (base.endsWith("-meta.xml")) {
284
+ return "salesforce_metadata";
285
+ }
286
+ const ext = path.extname(base).toLowerCase();
287
+ return EXTENSION_LANGUAGE.get(ext) ?? "unknown";
288
+ }
289
+ var UNSUPPORTED_LANGUAGE_EXTENSIONS = /* @__PURE__ */ new Map([
290
+ [".rpg", "RPG II or RPG III"],
291
+ [".nsm", "Natural map"],
292
+ [".nsd", "Natural DDM"]
293
+ ]);
294
+ function nonTreeSitterLanguageCount() {
295
+ return LANGUAGE_SPECS.filter((s) => s.extraction !== "tree-sitter" && s.id !== "ipynb").length;
296
+ }
297
+ function unsupportedLanguageName(filePath) {
298
+ const ext = path.extname(filePath).toLowerCase();
299
+ return UNSUPPORTED_LANGUAGE_EXTENSIONS.get(ext);
300
+ }
301
+
302
+ // src/dotenv_redact.ts
303
+ init_define_import_meta_env();
304
+ var DOTENV_VALUE_PLACEHOLDER = "[REDACTED:dotenv_value]";
305
+ var ASSIGNMENT_RE = /^(\s*(?:export\s+)?[A-Za-z_][\w.-]*\s*(?:\+?=|:(?!\/\/)))/;
306
+ var SAFE_LINE_RE = /^\s*(?:[#;].*)?$/;
307
+ function isDotenvPath(filePath) {
308
+ return detectLanguage(filePath) === "env_file";
309
+ }
310
+ function redactDotenvValues(text) {
311
+ const lines = text.split("\n");
312
+ const out = [];
313
+ let openQuote = null;
314
+ for (const raw of lines) {
315
+ const hasCr = raw.endsWith("\r");
316
+ const line = hasCr ? raw.slice(0, -1) : raw;
317
+ const eol = hasCr ? "\r" : "";
318
+ const emit = (s) => {
319
+ out.push(`${s}${eol}`);
320
+ };
321
+ if (openQuote !== null) {
322
+ if (_lineClosesQuote(line, openQuote)) openQuote = null;
323
+ emit(DOTENV_VALUE_PLACEHOLDER);
324
+ continue;
325
+ }
326
+ if (SAFE_LINE_RE.test(line)) {
327
+ emit(line);
328
+ continue;
329
+ }
330
+ const m = ASSIGNMENT_RE.exec(line);
331
+ if (m === null || m[1] === void 0) {
332
+ emit(DOTENV_VALUE_PLACEHOLDER);
333
+ continue;
334
+ }
335
+ const prefix = m[1];
336
+ openQuote = _detectOpenQuote(line.slice(prefix.length));
337
+ emit(`${prefix}${DOTENV_VALUE_PLACEHOLDER}`);
338
+ }
339
+ return out.join("\n");
340
+ }
341
+ function redactIfDotenv(filePath, text) {
342
+ return isDotenvPath(filePath) ? redactDotenvValues(text) : text;
343
+ }
344
+
345
+ // src/db.ts
346
+ init_define_import_meta_env();
347
+ import * as fs3 from "node:fs";
348
+ import { createRequire as createRequire2 } from "node:module";
349
+ import * as path2 from "node:path";
350
+
351
+ // src/sqlite_driver.ts
352
+ init_define_import_meta_env();
353
+ import * as fs2 from "node:fs";
354
+ import { createRequire } from "node:module";
355
+ var _require = createRequire(import.meta.url);
356
+ function suppressSqliteExperimentalWarning() {
357
+ const original = process.emit;
358
+ let armed = true;
359
+ const restore = () => {
360
+ if (!armed) return;
361
+ armed = false;
362
+ process.emit = original;
363
+ };
364
+ process.emit = function patched(name, ...rest) {
365
+ const data = rest[0];
366
+ if (armed && name === "warning" && data instanceof Error && data.name === "ExperimentalWarning" && /sqlite/i.test(data.message)) {
367
+ restore();
368
+ return false;
369
+ }
370
+ return original.call(this, name, ...rest);
371
+ };
372
+ setImmediate(restore);
373
+ return restore;
374
+ }
375
+ var restoreWarnings = suppressSqliteExperimentalWarning();
376
+ var nodeSqlite;
377
+ try {
378
+ nodeSqlite = _require("node:sqlite");
379
+ } catch (e) {
380
+ restoreWarnings();
381
+ throw e;
382
+ }
383
+ var { DatabaseSync } = nodeSqlite;
384
+ var SQLITE_PRIMARY_CODES = [
385
+ "SQLITE_OK",
386
+ "SQLITE_ERROR",
387
+ "SQLITE_INTERNAL",
388
+ "SQLITE_PERM",
389
+ "SQLITE_ABORT",
390
+ "SQLITE_BUSY",
391
+ "SQLITE_LOCKED",
392
+ "SQLITE_NOMEM",
393
+ "SQLITE_READONLY",
394
+ "SQLITE_INTERRUPT",
395
+ "SQLITE_IOERR",
396
+ "SQLITE_CORRUPT",
397
+ "SQLITE_NOTFOUND",
398
+ "SQLITE_FULL",
399
+ "SQLITE_CANTOPEN",
400
+ "SQLITE_PROTOCOL",
401
+ "SQLITE_EMPTY",
402
+ "SQLITE_SCHEMA",
403
+ "SQLITE_TOOBIG",
404
+ "SQLITE_CONSTRAINT",
405
+ "SQLITE_MISMATCH",
406
+ "SQLITE_MISUSE",
407
+ "SQLITE_NOLFS",
408
+ "SQLITE_AUTH",
409
+ "SQLITE_FORMAT",
410
+ "SQLITE_RANGE",
411
+ "SQLITE_NOTADB",
412
+ "SQLITE_NOTICE",
413
+ "SQLITE_WARNING"
414
+ ];
415
+ var SQLITE_EXTENDED_SUFFIXES = {
416
+ SQLITE_OK: ["LOAD_PERMANENTLY", "SYMLINK"],
417
+ SQLITE_ERROR: ["MISSING_COLLSEQ", "RETRY", "SNAPSHOT"],
418
+ SQLITE_ABORT: [null, "ROLLBACK"],
419
+ SQLITE_BUSY: ["RECOVERY", "SNAPSHOT", "TIMEOUT"],
420
+ SQLITE_LOCKED: ["SHAREDCACHE", "VTAB"],
421
+ SQLITE_READONLY: ["RECOVERY", "CANTLOCK", "ROLLBACK", "DBMOVED", "CANTINIT", "DIRECTORY"],
422
+ SQLITE_IOERR: [
423
+ "READ",
424
+ "SHORT_READ",
425
+ "WRITE",
426
+ "FSYNC",
427
+ "DIR_FSYNC",
428
+ "TRUNCATE",
429
+ "FSTAT",
430
+ "UNLOCK",
431
+ "RDLOCK",
432
+ "DELETE",
433
+ "BLOCKED",
434
+ "NOMEM",
435
+ "ACCESS",
436
+ "CHECKRESERVEDLOCK",
437
+ "LOCK",
438
+ "CLOSE",
439
+ "DIR_CLOSE",
440
+ "SHMOPEN",
441
+ "SHMSIZE",
442
+ "SHMLOCK",
443
+ "SHMMAP",
444
+ "SEEK",
445
+ "DELETE_NOENT",
446
+ "MMAP",
447
+ "GETTEMPPATH",
448
+ "CONVPATH",
449
+ "VNODE",
450
+ "AUTH",
451
+ "BEGIN_ATOMIC",
452
+ "COMMIT_ATOMIC",
453
+ "ROLLBACK_ATOMIC",
454
+ "DATA",
455
+ "CORRUPTFS",
456
+ "IN_PAGE"
457
+ ],
458
+ SQLITE_CORRUPT: ["VTAB", "SEQUENCE", "INDEX"],
459
+ SQLITE_CANTOPEN: ["NOTEMPDIR", "ISDIR", "FULLPATH", "CONVPATH", "DIRTYWAL", "SYMLINK"],
460
+ SQLITE_CONSTRAINT: [
461
+ "CHECK",
462
+ "COMMITHOOK",
463
+ "FOREIGNKEY",
464
+ "FUNCTION",
465
+ "NOTNULL",
466
+ "PRIMARYKEY",
467
+ "TRIGGER",
468
+ "UNIQUE",
469
+ "VTAB",
470
+ "ROWID",
471
+ "PINNED",
472
+ "DATATYPE"
473
+ ],
474
+ SQLITE_AUTH: ["USER"],
475
+ SQLITE_NOTICE: ["RECOVER_WAL", "RECOVER_ROLLBACK", "RBU"],
476
+ SQLITE_WARNING: ["AUTOINDEX"]
477
+ };
478
+ function sqliteResultCodeName(errcode) {
479
+ if (!Number.isInteger(errcode) || errcode < 0) return "ERR_SQLITE_ERROR";
480
+ if (errcode === 100) return "SQLITE_ROW";
481
+ if (errcode === 101) return "SQLITE_DONE";
482
+ const primary = SQLITE_PRIMARY_CODES[errcode & 255];
483
+ if (primary === void 0) return "ERR_SQLITE_ERROR";
484
+ const subcode = errcode >> 8;
485
+ if (subcode === 0) return primary;
486
+ const suffix = SQLITE_EXTENDED_SUFFIXES[primary]?.[subcode - 1];
487
+ return suffix === void 0 || suffix === null ? primary : `${primary}_${suffix}`;
488
+ }
489
+ function attempt(fn) {
490
+ try {
491
+ return fn();
492
+ } catch (e) {
493
+ const err = e;
494
+ if (err.code === "ERR_SQLITE_ERROR" && typeof err.errcode === "number") {
495
+ err.code = sqliteResultCodeName(err.errcode);
496
+ }
497
+ throw e;
498
+ }
499
+ }
500
+ var Statement = class {
501
+ #stmt;
502
+ #pluck = false;
503
+ constructor(stmt) {
504
+ this.#stmt = stmt;
505
+ }
506
+ get source() {
507
+ return this.#stmt.sourceSQL;
508
+ }
509
+ /**
510
+ * better-sqlite3's `reader` flag: does this statement return rows?
511
+ *
512
+ * `node:sqlite` has no equivalent, so it is derived from the prepared statement's own column
513
+ * count -- SQLite gives a row-producing statement its result columns at prepare time and gives a
514
+ * non-producing one none. That is a derivation, and this is the third defence-in-depth layer in
515
+ * `sqlite_query.ts`'s read-only guard, so it is not taken on faith: the driver tests run both
516
+ * libraries side by side over SELECT, a CTE, VALUES, EXPLAIN, an empty-result SELECT, INSERT,
517
+ * UPDATE, DELETE, CREATE, a reading PRAGMA and an assigning PRAGMA, and require every verdict to
518
+ * agree. If a future SQLite statement form ever breaks the equivalence, that test fails rather
519
+ * than the guard quietly weakening.
520
+ */
521
+ get reader() {
522
+ return attempt(() => this.#stmt.columns()).length > 0;
523
+ }
524
+ // A plucked row is "the first column", which for an object row means the first *inserted* key. V8 preserves insertion order for string keys, and node:sqlite builds the row by walking the result columns left to right, so Object.values()[0] is the leftmost column -- not the column named in the SQL text, the same rule better-sqlite3 applies.
525
+ #shape(row) {
526
+ if (!this.#pluck || row === void 0 || row === null) return row;
527
+ const values = Object.values(row);
528
+ return values.length === 0 ? void 0 : values[0];
529
+ }
530
+ get(...params) {
531
+ return this.#shape(attempt(() => this.#stmt.get(...params)));
532
+ }
533
+ all(...params) {
534
+ const rows2 = attempt(() => this.#stmt.all(...params));
535
+ return this.#pluck ? rows2.map((r) => this.#shape(r)) : rows2;
536
+ }
537
+ run(...params) {
538
+ return attempt(() => this.#stmt.run(...params));
539
+ }
540
+ // Wrapped rather than returned directly so pluck applies lazily, one row at a time: the whole point of iterate() here is that sqlite_query.ts caps the row count without buffering the rest, and mapping the iterator through .all() first would defeat that.
541
+ *iterate(...params) {
542
+ const rows2 = attempt(() => this.#stmt.iterate(...params))[Symbol.iterator]();
543
+ for (; ; ) {
544
+ const next = attempt(() => rows2.next());
545
+ if (next.done === true) return;
546
+ yield this.#shape(next.value);
547
+ }
548
+ }
549
+ pluck(toggle = true) {
550
+ this.#pluck = toggle;
551
+ return this;
552
+ }
553
+ safeIntegers(toggle = true) {
554
+ this.#stmt.setReadBigInts(toggle);
555
+ return this;
556
+ }
557
+ columns() {
558
+ return attempt(() => this.#stmt.columns());
559
+ }
560
+ };
561
+ var Database = class {
562
+ #db;
563
+ #path;
564
+ #readonly;
565
+ #savepoints = 0;
566
+ constructor(dbPath, options = {}) {
567
+ const wantsExisting = options.readonly === true || options.fileMustExist === true;
568
+ if (wantsExisting && dbPath !== ":memory:" && !fs2.existsSync(dbPath)) {
569
+ throw new Error("unable to open database file");
570
+ }
571
+ this.#db = attempt(() => new DatabaseSync(dbPath, {
572
+ readOnly: options.readonly === true,
573
+ // sqlite-vec is loaded through db.loadExtension by initConnection, which node:sqlite refuses unless the connection opted in at construction. Harmless when no extension is ever loaded.
574
+ allowExtension: true,
575
+ // better-sqlite3 opens every connection with busy_timeout at 5000ms; node:sqlite opens at 0, so a connection that named no timeout would silently go from five seconds of patience to none. db.ts overrides this to 15000 in initConnection, but sqlite_query.ts opens a user's arbitrary database readonly and takes whatever the default is -- which would have turned ordinary contention with another writer into an immediate "database is locked".
576
+ timeout: options.timeout ?? 5e3
577
+ }));
578
+ this.#path = dbPath;
579
+ this.#readonly = options.readonly === true;
580
+ }
581
+ get open() {
582
+ return this.#db.isOpen;
583
+ }
584
+ get inTransaction() {
585
+ return this.#db.isTransaction;
586
+ }
587
+ get readonly() {
588
+ return this.#readonly;
589
+ }
590
+ get name() {
591
+ return this.#path;
592
+ }
593
+ prepare(sql) {
594
+ return new Statement(attempt(() => this.#db.prepare(sql)));
595
+ }
596
+ exec(sql) {
597
+ attempt(() => this.#db.exec(sql));
598
+ }
599
+ pragma(source, options = {}) {
600
+ const rows2 = attempt(() => this.#db.prepare(`PRAGMA ${source}`).all());
601
+ if (options.simple !== true) return rows2;
602
+ const first = rows2[0];
603
+ if (first === void 0) return void 0;
604
+ const values = Object.values(first);
605
+ return values.length === 0 ? void 0 : values[0];
606
+ }
607
+ function(name, options, fn) {
608
+ attempt(() => this.#db.function(name, options, fn));
609
+ }
610
+ loadExtension(extensionPath) {
611
+ attempt(() => this.#db.loadExtension(extensionPath));
612
+ }
613
+ close() {
614
+ attempt(() => this.#db.close());
615
+ }
616
+ /**
617
+ * Wrap `fn` so it runs inside a transaction, committing on return and rolling back on throw.
618
+ *
619
+ * Nesting uses SAVEPOINT, which is what makes it safe for a transactional helper to call another
620
+ * one: an inner `BEGIN` would throw ("cannot start a transaction within a transaction"), an inner
621
+ * SAVEPOINT composes. Whether we are nested is read from SQLite via `isTransaction` rather than
622
+ * tracked in a counter here, so a transaction some other code path opened still nests correctly.
623
+ *
624
+ * The rollback is best-effort and never replaces the caller's error: if the ROLLBACK itself fails
625
+ * -- the connection died, the transaction was already unwound -- the original failure is still
626
+ * what propagates, because that is the one that explains what went wrong.
627
+ */
628
+ transaction(fn) {
629
+ const build = (beginSql) => (...args) => {
630
+ if (this.#db.isTransaction) {
631
+ const name = `tg_sp_${this.#savepoints++}`;
632
+ attempt(() => this.#db.exec(`SAVEPOINT ${name}`));
633
+ try {
634
+ const result = fn(...args);
635
+ attempt(() => this.#db.exec(`RELEASE ${name}`));
636
+ return result;
637
+ } catch (e) {
638
+ try {
639
+ attempt(() => this.#db.exec(`ROLLBACK TO ${name}`));
640
+ attempt(() => this.#db.exec(`RELEASE ${name}`));
641
+ } catch {
642
+ }
643
+ throw e;
644
+ }
645
+ }
646
+ attempt(() => this.#db.exec(beginSql));
647
+ try {
648
+ const result = fn(...args);
649
+ attempt(() => this.#db.exec("COMMIT"));
650
+ return result;
651
+ } catch (e) {
652
+ try {
653
+ attempt(() => this.#db.exec("ROLLBACK"));
654
+ } catch {
655
+ }
656
+ throw e;
657
+ }
658
+ };
659
+ const wrapped = build("BEGIN");
660
+ wrapped.default = wrapped;
661
+ wrapped.deferred = build("BEGIN");
662
+ wrapped.immediate = build("BEGIN IMMEDIATE");
663
+ wrapped.exclusive = build("BEGIN EXCLUSIVE");
664
+ return wrapped;
665
+ }
666
+ };
667
+
668
+ // src/db.ts
669
+ var _require2 = createRequire2(import.meta.url);
670
+ var _connections = /* @__PURE__ */ new Map();
671
+ var SCHEMA_SQL = `
672
+ CREATE TABLE IF NOT EXISTS files (
673
+ path TEXT PRIMARY KEY,
674
+ sha TEXT,
675
+ mtime REAL,
676
+ language TEXT,
677
+ indexed_at REAL,
678
+ embed_sha TEXT,
679
+ retry_count INTEGER NOT NULL DEFAULT 0,
680
+ parser_sha TEXT
681
+ );
682
+ -- Expression index on TG_LOWER(path) -- see pathEqClause (sql_path.ts) and TG_LOWER's
683
+ -- registration above. TG_LOWER is registered { deterministic: true }, which is required for
684
+ -- SQLite to index an expression at all; without it CREATE INDEX on a function call throws
685
+ -- "non-deterministic functions prohibited in index expressions". Because pathEqClause emits
686
+ -- this exact 'TG_LOWER(path) = ?' text for every case-insensitive-filesystem query, the planner
687
+ -- matches it against this index and uses SEARCH instead of a full table SCAN, without requiring
688
+ -- any writer to populate a separate folded column (verified via EXPLAIN QUERY PLAN in
689
+ -- db.test.ts / sql_path.test.ts). CREATE INDEX IF NOT EXISTS is purely additive and safe to run
690
+ -- against an already-populated table on every connection open, unlike an ALTER TABLE column add
691
+ -- -- no MIGRATIONS entry or SCHEMA_VERSION bump is needed for this index.
692
+ CREATE INDEX IF NOT EXISTS idx_files_path_folded ON files(TG_LOWER(path));
693
+
694
+ CREATE TABLE IF NOT EXISTS symbols (
695
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
696
+ file_path TEXT,
697
+ name TEXT,
698
+ kind TEXT,
699
+ line_start INTEGER,
700
+ line_end INTEGER,
701
+ body TEXT,
702
+ docstring TEXT,
703
+ parent TEXT NOT NULL DEFAULT ''
704
+ );
705
+ CREATE INDEX IF NOT EXISTS idx_symbols_name ON symbols(name);
706
+ CREATE INDEX IF NOT EXISTS idx_symbols_file ON symbols(file_path);
707
+ CREATE INDEX IF NOT EXISTS idx_symbols_name_kind ON symbols(name, kind);
708
+ CREATE INDEX IF NOT EXISTS idx_symbols_file_folded ON symbols(TG_LOWER(file_path));
709
+ CREATE INDEX IF NOT EXISTS idx_symbols_file_name_folded ON symbols(TG_LOWER(file_path), name);
710
+ -- Partial index backing checkSymbolBodySize (cli_doctor.ts), which every SessionStart hook runs.
711
+ -- Its predicate cannot be served by any index above, so the check had to read the whole symbols
712
+ -- table -- 226 MB / 231324 rows here, 229 ms per session start, and the early-exit LIMIT 1 never
713
+ -- fires on a healthy index because there is nothing to find. Indexing the *violating* rows only
714
+ -- makes the check a lookup into a b-tree that is empty on a healthy index: measured 229 ms -> 0.0
715
+ -- ms, 4 KB on disk, and no measurable insert cost (-0.2%, within noise, over 40000 real rows),
716
+ -- because SQLite evaluates the predicate and skips the b-tree write for every row under the cap.
717
+ -- SQLite uses a partial index only where the query's WHERE implies the index's, so the probe in
718
+ -- cli_doctor.ts spells its comparison the same way and against the same constant. That makes the
719
+ -- threshold part of the stored schema -- see SYMBOL_BODY_CHAR_CAP in constants.ts for what
720
+ -- changing it requires. A query with a lower threshold correctly gets a full scan instead, so no
721
+ -- other reader can be served stale rows by this index.
722
+ CREATE INDEX IF NOT EXISTS idx_symbols_oversized_body ON symbols(id) WHERE LENGTH(body) > ${SYMBOL_BODY_CHAR_CAP};
723
+
724
+ CREATE TABLE IF NOT EXISTS refs (
725
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
726
+ file_path TEXT,
727
+ name TEXT,
728
+ line INTEGER,
729
+ col INTEGER,
730
+ context TEXT
731
+ );
732
+ CREATE INDEX IF NOT EXISTS idx_refs_name ON refs(name);
733
+ CREATE INDEX IF NOT EXISTS idx_refs_file ON refs(file_path);
734
+ CREATE INDEX IF NOT EXISTS idx_refs_file_folded ON refs(TG_LOWER(file_path));
735
+ CREATE INDEX IF NOT EXISTS idx_refs_file_name_folded ON refs(TG_LOWER(file_path), name);
736
+
737
+ CREATE TABLE IF NOT EXISTS chunks (
738
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
739
+ file_path TEXT,
740
+ start_line INTEGER,
741
+ end_line INTEGER,
742
+ text TEXT,
743
+ kind TEXT
744
+ );
745
+ CREATE INDEX IF NOT EXISTS idx_chunks_file ON chunks(file_path);
746
+ CREATE INDEX IF NOT EXISTS idx_chunks_file_folded ON chunks(TG_LOWER(file_path));
747
+ CREATE INDEX IF NOT EXISTS idx_chunks_file_kind_folded ON chunks(TG_LOWER(file_path), kind);
748
+
749
+ -- Tracks every project root a hook has ever seen an edit for, so the worker's periodic sweep
750
+ -- (sweepKnownRoots in index_prune.ts) knows which roots to auto-prune without scanning the
751
+ -- entire shared files table for distinct top-level directories on every cycle. Purely additive
752
+ -- (no SCHEMA_VERSION bump needed): last_seen_ms is refreshed on every observed edit,
753
+ -- first_missing_ms is set the first sweep that finds the root unreachable and cleared the
754
+ -- moment it's seen reachable again -- see sweepKnownRoots' grace-period logic.
755
+ CREATE TABLE IF NOT EXISTS known_roots (
756
+ root TEXT PRIMARY KEY,
757
+ last_seen_ms REAL NOT NULL,
758
+ first_missing_ms REAL
759
+ );
760
+
761
+ -- Resume point for a budget-truncated reconcile sweep (reconcile.ts), one row per project root. A
762
+ -- project too large to finish a sweep inside DEFAULT_RECONCILE_BUDGET_MS would otherwise scan the
763
+ -- same deterministic (git ls-files) prefix every session forever, leaving every file after the
764
+ -- budget cutoff permanently unchecked. last_scanned_path is the last tracked file the sweep
765
+ -- finished examining before its budget ran out; the next sweep rotates its scan order to resume
766
+ -- right after that file, wrapping back to the start, so repeated truncated sweeps eventually cover
767
+ -- the whole project. Cleared (row deleted) the moment a sweep completes a full lap without running
768
+ -- out of budget. Purely additive (no SCHEMA_VERSION bump needed): a missing row just means "start
769
+ -- from the beginning", the same as a fresh database.
770
+ CREATE TABLE IF NOT EXISTS reconcile_cursor (
771
+ root TEXT PRIMARY KEY,
772
+ last_scanned_path TEXT NOT NULL,
773
+ updated_at REAL NOT NULL
774
+ );
775
+
776
+ -- Cross-cache full-text search index for 'token-goat recall' (recall_index.ts). One row
777
+ -- per bash-output/web-output/mcp-output blob-store entry (see disk_cache.ts), refreshed
778
+ -- in place (ON CONFLICT DO UPDATE) whenever storeBashOutput/storeWebOutput/storeMcpOutput
779
+ -- write that entry, so recall never needs a separate rebuild step. row_id is a plain
780
+ -- surrogate integer key -- entry_id is the real blob-store id (bash/mcp ids are hex,
781
+ -- web ids are the cache's own scheme) and is not unique on its own since bash-output and
782
+ -- mcp-output ids share one namespace (BASH_OUTPUT_SUBDIR) while web-output ids are a
783
+ -- separate namespace; cache_type disambiguates.
784
+ CREATE TABLE IF NOT EXISTS cache_recall (
785
+ row_id INTEGER PRIMARY KEY AUTOINCREMENT,
786
+ cache_type TEXT NOT NULL,
787
+ entry_id TEXT NOT NULL,
788
+ label TEXT,
789
+ content TEXT,
790
+ stored_at REAL,
791
+ UNIQUE(cache_type, entry_id)
792
+ );
793
+ CREATE INDEX IF NOT EXISTS idx_cache_recall_type ON cache_recall(cache_type);
794
+
795
+ -- Per-emission ledger for 'token-goat hint-stats' (hint_stats.ts). One row per hint
796
+ -- emission event (a hook returning a 'context' HookOutput classified as a discretionary
797
+ -- efficiency nudge, as opposed to a mandatory informational injection -- see hint_stats.ts's
798
+ -- doc comment for the exact category list and what is deliberately excluded). correlator is a
799
+ -- best-effort file-path/output-id substring extracted from the hint's own text, used to check
800
+ -- whether a later Bash tool call in the same session actually followed the hint's specific
801
+ -- pointer (see resolvePendingHintsForEvent) -- NULL when no such pointer could be extracted,
802
+ -- in which case the row is inserted already resolved with acted_on=0 (counted as emitted, never
803
+ -- eligible for auto-detected credit). calls_remaining is the countdown of subsequent tool-use
804
+ -- events still eligible to resolve this row before it is considered timed out.
805
+ CREATE TABLE IF NOT EXISTS hint_emissions (
806
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
807
+ category TEXT NOT NULL,
808
+ session_id TEXT NOT NULL,
809
+ harness TEXT NOT NULL,
810
+ correlator TEXT,
811
+ emitted_at REAL NOT NULL,
812
+ resolved INTEGER NOT NULL DEFAULT 0,
813
+ acted_on INTEGER NOT NULL DEFAULT 0,
814
+ calls_remaining INTEGER NOT NULL DEFAULT 0,
815
+ bytes_emitted INTEGER
816
+ );
817
+ CREATE INDEX IF NOT EXISTS idx_hint_emissions_category ON hint_emissions(category);
818
+ CREATE INDEX IF NOT EXISTS idx_hint_emissions_session_pending ON hint_emissions(session_id, resolved);
819
+
820
+ -- Manual efficacy votes for a hint category (token-goat hint-stats --mark-effective/--mark-ineffective),
821
+ -- kept separate from hint_emissions' automatic acted_on signal so the two are never silently
822
+ -- blended -- see hint_stats.ts's doc comment on why some categories only support this manual signal.
823
+ CREATE TABLE IF NOT EXISTS hint_manual_marks (
824
+ category TEXT PRIMARY KEY,
825
+ effective_count INTEGER NOT NULL DEFAULT 0,
826
+ ineffective_count INTEGER NOT NULL DEFAULT 0
827
+ );
828
+
829
+ -- Durable counter backing hint_stats.ts's backoff-threshold probe-recovery schedule: how many
830
+ -- CONSECUTIVE suppressed occasions have elapsed for (category, harness) since a hint in this
831
+ -- category was last actually shown (either organically, because shouldSuppress no longer holds,
832
+ -- or via a prior probe). shouldSuppress itself stays a pure function of hint_emissions -- this
833
+ -- table exists only because a suppressed occasion is deliberately never written to
834
+ -- hint_emissions (see that table's own comment), so without a separate durable counter here
835
+ -- there would be no way to know "how many suppressed occasions have we seen" across the
836
+ -- short-lived hook CLI processes that call applyHintTracking. Keyed by (category, harness), not
837
+ -- category alone, to match shouldSuppress/categoryStats' own per-harness scoping -- unlike
838
+ -- hint_manual_marks (a human-entered vote, deliberately not harness-split).
839
+ CREATE TABLE IF NOT EXISTS hint_suppression_probes (
840
+ category TEXT NOT NULL,
841
+ harness TEXT NOT NULL,
842
+ streak INTEGER NOT NULL DEFAULT 0,
843
+ PRIMARY KEY (category, harness)
844
+ );
845
+
846
+ -- Free-text architecture/rationale notes (the "why" layer -- see notes.ts), attached either to
847
+ -- a whole file (symbol = '') or to one specific indexed symbol within it (symbol = that
848
+ -- symbol's name). '' rather than NULL for the whole-file case because SQLite's UNIQUE treats
849
+ -- NULLs as pairwise-distinct (never conflicting with each other), which would let note-add
850
+ -- accumulate unlimited duplicate whole-file notes for the same file instead of upserting one;
851
+ -- '' is a real, comparable value so UNIQUE(file_path, symbol) enforces "at most one note per
852
+ -- attachment point" for both cases identically. 'fingerprint' is a SHA-256 digest (see
853
+ -- fingerprintContent in fingerprint.ts) captured at write time of exactly what the note
854
+ -- describes -- the resolved symbol's current body text for a symbol-scoped note, or a stable
855
+ -- digest of the file's current top-level symbol manifest (name:kind:line-range per symbol,
856
+ -- sorted) for a file-scoped note -- so 'token-goat note-list --stale-only' can recompute the
857
+ -- same fingerprint against the live index later and flag a mismatch (see notes.ts's
858
+ -- isNoteStale). Staleness detection is purely advisory: nothing here ever auto-rewrites or
859
+ -- deletes a note's content, only flags that the code it describes has moved since it was
860
+ -- written -- a human/agent re-review decides what to do with a stale note.
861
+ CREATE TABLE IF NOT EXISTS notes (
862
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
863
+ file_path TEXT NOT NULL,
864
+ symbol TEXT NOT NULL DEFAULT '',
865
+ content TEXT NOT NULL,
866
+ fingerprint TEXT NOT NULL,
867
+ created_at REAL NOT NULL,
868
+ updated_at REAL NOT NULL,
869
+ UNIQUE(file_path, symbol)
870
+ );
871
+ CREATE INDEX IF NOT EXISTS idx_notes_file_folded ON notes(TG_LOWER(file_path));
872
+
873
+ -- Baseline for skill_version_drift.ts's one-shot nudge: the token-goat CLI version (and its
874
+ -- flat command-name set, JSON-encoded) active the moment the token-goat skill's body was
875
+ -- last (re)loaded into this session -- see hooks_skill.ts's postSkillHandler. A session that
876
+ -- keeps running after the CLI is upgraded has no other way to learn that new surgical-read
877
+ -- commands now exist (the skill only re-announces itself on an explicit reload), so
878
+ -- checkSkillVersionDrift compares this snapshot against the live command set on each user turn
879
+ -- and fires the nudge exactly once (notified_at) per (re)load. session_id is the primary key
880
+ -- because only one skill (token-goat) is ever tracked here.
881
+ CREATE TABLE IF NOT EXISTS skill_version_snapshots (
882
+ session_id TEXT PRIMARY KEY,
883
+ skill_name TEXT NOT NULL,
884
+ loaded_version TEXT NOT NULL,
885
+ loaded_commands_json TEXT NOT NULL,
886
+ notified_at REAL
887
+ );
888
+
889
+ -- Which embedding stack produced the vectors currently in chunk_vectors -- the model, its
890
+ -- pinned revision, and the inference runtime (see embeddingProvenance in embeddings.ts). The
891
+ -- vector table itself is vec0(rowid, embedding) and has nowhere to record this, so without
892
+ -- this row a database that was embedded by one stack and then added to by another holds two
893
+ -- incomparable sets of vectors under one index, with nothing able to tell them apart. That is
894
+ -- not hypothetical: global.db is machine-wide across every project on the machine (see
895
+ -- constants.ts), so upgrading the runtime, or changing the model or its pinned revision, mixes
896
+ -- old and new vectors for as long as the old files go untouched. Measured drift between two
897
+ -- runtime versions of the same quantized model is 0.9925-0.9978 cosine on the final vector --
898
+ -- small, but enough to reorder near-ties, and invisible to every existing check.
899
+ --
900
+ -- Single-row by construction (the CHECK pins the key), because there is exactly one vector
901
+ -- table per database. An EMPTY table on a database that already holds chunks means the vectors
902
+ -- predate this stamp and their provenance is unknowable -- see ensureEmbeddingProvenance, which
903
+ -- treats that exactly like a mismatch. That is what makes this work without a migration step.
904
+ CREATE TABLE IF NOT EXISTS embedding_provenance (
905
+ id INTEGER PRIMARY KEY CHECK (id = 1),
906
+ provenance TEXT NOT NULL
907
+ );
908
+ `;
909
+ var FTS_TOKENIZER = "unicode61 remove_diacritics 2";
910
+ var FTS_SQL = `
911
+ CREATE VIRTUAL TABLE IF NOT EXISTS symbols_fts USING fts5(
912
+ name,
913
+ body,
914
+ docstring,
915
+ content='symbols',
916
+ content_rowid='id',
917
+ tokenize='${FTS_TOKENIZER}'
918
+ );
919
+ CREATE TRIGGER IF NOT EXISTS symbols_ai AFTER INSERT ON symbols BEGIN
920
+ INSERT INTO symbols_fts(rowid, name, body, docstring)
921
+ VALUES (new.id, new.name, new.body, new.docstring);
922
+ END;
923
+ CREATE TRIGGER IF NOT EXISTS symbols_ad AFTER DELETE ON symbols BEGIN
924
+ INSERT INTO symbols_fts(symbols_fts, rowid, name, body, docstring)
925
+ VALUES ('delete', old.id, old.name, old.body, old.docstring);
926
+ END;
927
+ CREATE TRIGGER IF NOT EXISTS symbols_au AFTER UPDATE ON symbols BEGIN
928
+ INSERT INTO symbols_fts(symbols_fts, rowid, name, body, docstring)
929
+ VALUES ('delete', old.id, old.name, old.body, old.docstring);
930
+ INSERT INTO symbols_fts(rowid, name, body, docstring)
931
+ VALUES (new.id, new.name, new.body, new.docstring);
932
+ END;
933
+
934
+ -- Content-linked FTS5 mirror of cache_recall (recall_index.ts), same shape as symbols_fts
935
+ -- above. An INSERT ... ON CONFLICT DO UPDATE against cache_recall fires the AFTER UPDATE
936
+ -- trigger (not AFTER INSERT) on the conflicting row, same as any other SQLite upsert, so the
937
+ -- delete+reinsert pattern below keeps the fts index correct on a re-indexed (overwritten)
938
+ -- entry, not just a brand-new one.
939
+ CREATE VIRTUAL TABLE IF NOT EXISTS cache_recall_fts USING fts5(
940
+ label,
941
+ content,
942
+ content='cache_recall',
943
+ content_rowid='row_id',
944
+ tokenize='${FTS_TOKENIZER}'
945
+ );
946
+ CREATE TRIGGER IF NOT EXISTS cache_recall_ai AFTER INSERT ON cache_recall BEGIN
947
+ INSERT INTO cache_recall_fts(rowid, label, content)
948
+ VALUES (new.row_id, new.label, new.content);
949
+ END;
950
+ CREATE TRIGGER IF NOT EXISTS cache_recall_ad AFTER DELETE ON cache_recall BEGIN
951
+ INSERT INTO cache_recall_fts(cache_recall_fts, rowid, label, content)
952
+ VALUES ('delete', old.row_id, old.label, old.content);
953
+ END;
954
+ CREATE TRIGGER IF NOT EXISTS cache_recall_au AFTER UPDATE ON cache_recall BEGIN
955
+ INSERT INTO cache_recall_fts(cache_recall_fts, rowid, label, content)
956
+ VALUES ('delete', old.row_id, old.label, old.content);
957
+ INSERT INTO cache_recall_fts(rowid, label, content)
958
+ VALUES (new.row_id, new.label, new.content);
959
+ END;
960
+ `;
961
+ var SCHEMA_VERSION = 14;
962
+ function alterTableIdempotent(conn, sql) {
963
+ try {
964
+ conn.exec(sql);
965
+ } catch (err) {
966
+ if (!(err instanceof Error) || !/duplicate column/i.test(err.message)) throw err;
967
+ }
968
+ }
969
+ function purgeDotenvEmbeddings(conn) {
970
+ let paths;
971
+ try {
972
+ paths = conn.prepare("SELECT DISTINCT file_path FROM chunks").all().map((r) => r.file_path).filter(isDotenvPath);
973
+ } catch {
974
+ return;
975
+ }
976
+ if (paths.length === 0) return;
977
+ for (const p of paths) {
978
+ try {
979
+ conn.prepare("DELETE FROM chunk_vectors WHERE rowid IN (SELECT id FROM chunks WHERE file_path = ?)").run(p);
980
+ } catch {
981
+ }
982
+ conn.prepare("DELETE FROM chunks WHERE file_path = ?").run(p);
983
+ try {
984
+ conn.prepare("UPDATE files SET embed_sha = NULL WHERE path = ?").run(p);
985
+ } catch {
986
+ }
987
+ }
988
+ }
989
+ function rebuildFtsAtCurrentTokenizer(conn) {
990
+ try {
991
+ const declarations = conn.prepare("SELECT sql FROM sqlite_master WHERE name IN ('symbols_fts','cache_recall_fts')").all();
992
+ const current = declarations.map((d) => /tokenize\s*=\s*'([^']*)'/.exec(d.sql ?? "")?.[1] ?? "");
993
+ if (current.length === 2 && current.every((t) => t === FTS_TOKENIZER)) return;
994
+ conn.transaction(() => {
995
+ conn.exec("DROP TABLE IF EXISTS symbols_fts; DROP TABLE IF EXISTS cache_recall_fts;");
996
+ conn.exec(FTS_SQL);
997
+ conn.exec("INSERT INTO symbols_fts(symbols_fts) VALUES('rebuild');");
998
+ conn.exec("INSERT INTO cache_recall_fts(cache_recall_fts) VALUES('rebuild');");
999
+ }).immediate();
1000
+ } catch {
1001
+ }
1002
+ }
1003
+ var MIGRATIONS = {
1004
+ // v1 -> v2: adds files.embed_sha, tracked separately from files.sha so embedding freshness can be gated independently of parse freshness (see makeIndexer in worker.ts). A pre-existing v1 database's `files` table predates the column, so it needs an explicit ALTER TABLE here; a brand-new database already has the column from SCHEMA_SQL's CREATE TABLE above, so the ALTER TABLE would fail with "duplicate column name" there -- swallow exactly that error and rethrow anything else, so a genuine ALTER TABLE failure is never silently lost.
1005
+ 1: (conn) => alterTableIdempotent(conn, "ALTER TABLE files ADD COLUMN embed_sha TEXT"),
1006
+ // v2 -> v3: adds files.retry_count, a durable per-path counter for consecutive transient-read-failure requeues (see MAX_TRANSIENT_RETRIES / requeueDirtyPath / clearRetryCount in worker.ts). Previously this counter lived only in an in-memory Map inside worker.ts, which meant the retry-count reset -- run at the time in the short-lived hook CLI process -- could never actually reach the long-lived detached daemon process's own copy of that Map: they are different Node processes with no shared memory, so the reset was a silent no-op in the real deployed topology. Persisting the counter in `files` makes it visible to both processes via the one thing they do share: the index DB. Same swallow-duplicate-column pattern as v1 -> v2 above.
1007
+ 2: (conn) => alterTableIdempotent(conn, "ALTER TABLE files ADD COLUMN retry_count INTEGER NOT NULL DEFAULT 0"),
1008
+ // v8 -> v9: adds symbols.parent (see SCHEMA_VERSION comment above for why). A pre-existing v8 database's `symbols` table predates the column, so it needs an explicit ALTER TABLE here; a brand-new database already has the column from SCHEMA_SQL's CREATE TABLE above, so the ALTER TABLE would fail with "duplicate column name" there -- swallow exactly that error and rethrow anything else, same pattern as v1 -> v2 / v2 -> v3 above.
1009
+ 8: (conn) => alterTableIdempotent(conn, "ALTER TABLE symbols ADD COLUMN parent TEXT NOT NULL DEFAULT ''"),
1010
+ // v9 -> v10: adds hint_emissions.bytes_emitted (see SCHEMA_VERSION comment above for why). A pre-existing v9 database's `hint_emissions` table predates the column, so it needs an explicit ALTER TABLE here; a brand-new database already has the column from SCHEMA_SQL's CREATE TABLE above, so the ALTER TABLE would fail with "duplicate column name" there -- swallow exactly that error and rethrow anything else, same pattern as v1 -> v2 / v2 -> v3 / v8 -> v9 above.
1011
+ 9: (conn) => alterTableIdempotent(conn, "ALTER TABLE hint_emissions ADD COLUMN bytes_emitted INTEGER"),
1012
+ // v10 -> v11: purge chunks (and their vectors) for dotenv files. Until this version, a tracked `.env` was chunked and embedded verbatim on the git path, so `semantic` returned its values -- see dotenv_redact.ts. Redacting from now on is not enough on its own: the embed-freshness gate (isEmbedFresh in parser.ts) skips a file whose bytes have not changed, so an already-indexed .env would have kept serving its pre-fix chunks indefinitely. Deleting the rows here both removes the stored secrets and, by clearing embed_sha, makes the next drain re-embed the file through the redacting path.
1013
+ 10: purgeDotenvEmbeddings,
1014
+ // v12 -> v13: adds files.parser_sha, the digest of the extraction logic that produced this file's rows, tracked separately from files.sha for the same reason embed_sha is -- content freshness and parse freshness are different questions, and the content sha alone could only ever answer the first. A pre-existing v12 database's `files` table predates the column, so it needs an explicit ALTER TABLE here; a brand-new database already has it from SCHEMA_SQL's CREATE TABLE above, so the ALTER TABLE would fail with "duplicate column name" there -- swallow exactly that error and rethrow anything else, same pattern as v1 -> v2 / v2 -> v3 / v8 -> v9 / v9 -> v10 above. Deliberately left NULL for every existing row rather than backfilled with the current fingerprint: NULL is the truthful answer (nobody recorded which parser wrote those rows), and it is also the answer that makes the freshness gates reparse them once, which is exactly what a database indexed by an older parser needs. v13 -> v14: changes both FTS5 tables' tokenizer to `unicode61 remove_diacritics 2`, so a search for `Noi` or `Viet` finds `Hà Nội` and `Việt Nam` -- combining marks that `remove_diacritics 1`, FTS5's default, leaves in place. This is the first schema change that `CREATE VIRTUAL TABLE IF NOT EXISTS` cannot express at all rather than merely cannot express on a populated table: against an existing virtual table that statement is a silent no-op, so without MIGRATIONS[13] the new tokenizer would reach only databases created after this release. The step drops both tables, re-runs FTS_SQL to re-create them at the current declaration, and rebuilds each from its content table.
1015
+ 12: (conn) => alterTableIdempotent(conn, "ALTER TABLE files ADD COLUMN parser_sha TEXT"),
1016
+ // v13 -> v14: re-creates both FTS5 tables at the tokenizer FTS_SQL currently declares (see the SCHEMA_VERSION comment above for why no `IF NOT EXISTS` form can do this).
1017
+ 13: rebuildFtsAtCurrentTokenizer
1018
+ };
1019
+ function runMigrations(conn, fromVersion, toVersion) {
1020
+ for (let v = fromVersion; v < toVersion; v++) {
1021
+ MIGRATIONS[v]?.(conn);
1022
+ }
1023
+ }
1024
+ var WAL_SWITCH_DEADLINE_MS = 15e3;
1025
+ function enableWalWithRetry(conn, budgetMs = WAL_SWITCH_DEADLINE_MS) {
1026
+ const deadline = Date.now() + budgetMs;
1027
+ let lastError;
1028
+ for (; ; ) {
1029
+ try {
1030
+ const mode = conn.pragma("journal_mode = WAL", { simple: true });
1031
+ if (String(mode).toLowerCase() === "wal") return;
1032
+ lastError = new Error(`got: ${String(mode)}`);
1033
+ } catch (e) {
1034
+ lastError = e;
1035
+ }
1036
+ try {
1037
+ if (String(conn.pragma("journal_mode", { simple: true })).toLowerCase() === "wal") return;
1038
+ } catch {
1039
+ }
1040
+ if (Date.now() >= deadline) {
1041
+ throw new Error(`db: failed to enable WAL mode (${lastError instanceof Error ? lastError.message : String(lastError)})`);
1042
+ }
1043
+ sleepSync(25);
1044
+ }
1045
+ }
1046
+ function initConnection(conn) {
1047
+ conn.pragma("busy_timeout = 15000");
1048
+ enableWalWithRetry(conn);
1049
+ conn.pragma("synchronous = NORMAL");
1050
+ conn.pragma("cache_size = -32000");
1051
+ conn.pragma("temp_store = MEMORY");
1052
+ conn.pragma("mmap_size = 134217728");
1053
+ conn.function(
1054
+ "TG_LOWER",
1055
+ { deterministic: true },
1056
+ (value) => value === null ? null : foldCase(String(value))
1057
+ );
1058
+ const storedVersion = Number(conn.pragma("user_version", { simple: true }));
1059
+ if (storedVersion > SCHEMA_VERSION) {
1060
+ throw new Error(
1061
+ `db: index schema version ${storedVersion} is newer than this token-goat build supports (expected ${SCHEMA_VERSION}). Update token-goat, or delete the stale index database and let it rebuild.`
1062
+ );
1063
+ }
1064
+ conn.exec(SCHEMA_SQL);
1065
+ try {
1066
+ conn.exec(FTS_SQL);
1067
+ } catch {
1068
+ }
1069
+ try {
1070
+ const sqliteVec = _require2("sqlite-vec");
1071
+ sqliteVec.load(conn);
1072
+ conn.exec(
1073
+ `CREATE VIRTUAL TABLE IF NOT EXISTS chunk_vectors USING vec0(
1074
+ embedding float[384]
1075
+ );`
1076
+ );
1077
+ } catch {
1078
+ }
1079
+ if (storedVersion < SCHEMA_VERSION) {
1080
+ runMigrations(conn, storedVersion, SCHEMA_VERSION);
1081
+ conn.pragma(`user_version = ${SCHEMA_VERSION}`);
1082
+ }
1083
+ }
1084
+ function resolveDbPath(dbPath) {
1085
+ if (path2.isAbsolute(dbPath)) return dbPath;
1086
+ if (dbPath.includes("/") || dbPath.includes("\\")) return path2.resolve(dbPath);
1087
+ return safeJoin(dataDir(), dbPath);
1088
+ }
1089
+ function connectionKey(dbPath) {
1090
+ const resolved = resolveDbPath(dbPath);
1091
+ return { resolved, key: foldPath(resolved) };
1092
+ }
1093
+ function getDb(dbPath) {
1094
+ const { resolved, key } = connectionKey(dbPath);
1095
+ const existing = _connections.get(key);
1096
+ if (existing !== void 0) return existing;
1097
+ const dir = path2.dirname(resolved);
1098
+ try {
1099
+ ensureDirSync(dir);
1100
+ } catch (e) {
1101
+ if (e.code !== "EEXIST" || !fs3.existsSync(dir)) throw e;
1102
+ }
1103
+ const conn = new Database(resolved);
1104
+ try {
1105
+ initConnection(conn);
1106
+ } catch (e) {
1107
+ try {
1108
+ conn.close();
1109
+ } catch {
1110
+ }
1111
+ throw e;
1112
+ }
1113
+ _connections.set(key, conn);
1114
+ return conn;
1115
+ }
1116
+ function closeAllDbs() {
1117
+ for (const conn of _connections.values()) {
1118
+ try {
1119
+ conn.close();
1120
+ } catch {
1121
+ }
1122
+ }
1123
+ _connections.clear();
1124
+ }
1125
+ registerReset(closeAllDbs);
1126
+
1127
+ // src/render/ansi.ts
1128
+ init_define_import_meta_env();
1129
+ function _colorStream(isatty) {
1130
+ if (process.env["NO_COLOR"]) return false;
1131
+ return isatty;
1132
+ }
1133
+ function colorStdout() {
1134
+ return _colorStream(process.stdout.isTTY === true);
1135
+ }
1136
+ var _E = "\x1B";
1137
+ var RESET = `${_E}[0m`;
1138
+ var _ANSI_ESCAPE_RE = /\x1B\[[0-?]*[ -/]*[@-~]|\x1B\].*?(?:\x07|\x1B\\|$)|\x1B[PX^_].*?\x1B\\|\x1B[@-_]/gs;
1139
+ var _PUA_RE = /[\u{E000}-\u{F8FF}\u{F0000}-\u{FFFFD}]/gu;
1140
+ function stripAnsiEscapes(s) {
1141
+ if (!s.includes("\x1B")) {
1142
+ return s;
1143
+ }
1144
+ return s.replace(_ANSI_ESCAPE_RE, "");
1145
+ }
1146
+ function stripAnsi(s) {
1147
+ if (!s.includes("\x1B")) {
1148
+ return s;
1149
+ }
1150
+ return stripAnsiEscapes(s).replace(_PUA_RE, "");
1151
+ }
1152
+ function fmtBytes(n) {
1153
+ let value = n;
1154
+ const units = ["B", "KB", "MB", "GB", "TB"];
1155
+ for (const unit of units) {
1156
+ if (Math.abs(value) < 1024) {
1157
+ return unit === "B" ? `${Math.trunc(value)}${unit}` : `${value.toFixed(1)}${unit}`;
1158
+ }
1159
+ value = value / 1024;
1160
+ }
1161
+ return `${value.toFixed(1)}PB`;
1162
+ }
1163
+ function fg(r, g, b) {
1164
+ return `${_E}[38;2;${r};${g};${b}m`;
1165
+ }
1166
+ function vlen(s) {
1167
+ return stripAnsi(s).length;
1168
+ }
1169
+ function padR(s, w) {
1170
+ return s + " ".repeat(Math.max(0, w - vlen(s)));
1171
+ }
1172
+ function padL(s, w) {
1173
+ return " ".repeat(Math.max(0, w - vlen(s))) + s;
1174
+ }
1175
+ function lerpRgb(a, b, t) {
1176
+ return [
1177
+ Math.round(a[0] + (b[0] - a[0]) * t),
1178
+ Math.round(a[1] + (b[1] - a[1]) * t),
1179
+ Math.round(a[2] + (b[2] - a[2]) * t)
1180
+ ];
1181
+ }
1182
+ var C = {
1183
+ TEXT_PRIMARY: [201, 209, 217],
1184
+ TEXT_BRIGHT: [240, 246, 252],
1185
+ TEXT_MUTED: [125, 133, 144],
1186
+ TEXT_DIM: [72, 79, 88],
1187
+ BG_TILE: [22, 27, 34],
1188
+ TRACK: [28, 35, 41],
1189
+ GREEN1: [31, 77, 44],
1190
+ GREEN2: [46, 160, 67],
1191
+ GREEN3: [63, 185, 80],
1192
+ GREEN4: [86, 211, 100],
1193
+ GREEN5: [126, 231, 135],
1194
+ BLUE: [88, 166, 255],
1195
+ PURPLE: [188, 140, 255],
1196
+ TEAL: [138, 212, 255],
1197
+ ORANGE: [235, 165, 80],
1198
+ YELLOW: [240, 215, 80],
1199
+ RED: [200, 60, 60]
1200
+ };
1201
+
1202
+ // src/stats.ts
1203
+ init_define_import_meta_env();
1204
+ import * as path3 from "node:path";
1205
+
1206
+ // src/render/stats_renderer.ts
1207
+ init_define_import_meta_env();
1208
+ var _STATS_MESSAGES_FALLBACK = {
1209
+ bytesModeOnlyNote: "tracks bytes, not vision tokens",
1210
+ sessionHintSplitNote: "session_hint shows realized savings; session_hint_overhead shows injected hint cost",
1211
+ insights: {
1212
+ biggestSaver: "Biggest saver ",
1213
+ mostActive: "Most active ",
1214
+ tokenLeader: "Token leader "
1215
+ }
1216
+ };
1217
+ var _STATS_MESSAGES = _STATS_MESSAGES_FALLBACK;
1218
+ var _TERM_W = process.stdout.columns || 100;
1219
+ var _CONTENT_W = Math.min(Math.max(_TERM_W, 80), 140);
1220
+ var _M = " ";
1221
+ var _COL_NAME = 18;
1222
+ var _COL_DATA = 10;
1223
+ var _COL_TOKENS = 12;
1224
+ var _COL_SHARE = 6;
1225
+ var _COL_EVENTS = 6;
1226
+ var _COLS_FIXED = _COL_NAME + 1 + 2 + _COL_DATA + 2 + _COL_TOKENS + 2 + _COL_SHARE + 2 + _COL_EVENTS;
1227
+ var _BAR_W = Math.max(16, _CONTENT_W - _M.length * 2 - _COLS_FIXED);
1228
+ var _RULE = _M + fg(...C.TEXT_DIM) + "\u2500".repeat(_CONTENT_W - _M.length * 2) + RESET;
1229
+ var _BYTE_TIERS = [
1230
+ { threshold: 1e15, divisor: 1e15, unit: "PB", color: C.PURPLE },
1231
+ { threshold: 1e12, divisor: 1e12, unit: "TB", color: C.BLUE },
1232
+ { threshold: 1e9, divisor: 1e9, unit: "GB", color: C.TEAL },
1233
+ { threshold: 1e6, divisor: 1e6, unit: "MB", color: C.GREEN4 },
1234
+ { threshold: 1e3, divisor: 1e3, unit: "KB", color: C.TEXT_MUTED },
1235
+ { threshold: 0, divisor: 1, unit: "B", color: C.TEXT_DIM }
1236
+ ];
1237
+ var _TOKEN_TIERS = [
1238
+ { threshold: 1e12, divisor: 1e12, unit: "Tt", color: C.GREEN5 },
1239
+ { threshold: 1e9, divisor: 1e9, unit: "Gt", color: C.TEAL },
1240
+ { threshold: 1e6, divisor: 1e6, unit: "Mt", color: C.PURPLE },
1241
+ { threshold: 1e3, divisor: 1e3, unit: "kt", color: C.BLUE },
1242
+ { threshold: 0, divisor: 1, unit: "t", color: C.TEXT_DIM }
1243
+ ];
1244
+ function _fmtMagnitude(n, tiers, zeroLabel) {
1245
+ if (zeroLabel !== void 0 && n === 0) {
1246
+ return `${fg(...C.TEXT_DIM)}${zeroLabel}${RESET}`;
1247
+ }
1248
+ if (n < 0) {
1249
+ const a = -n;
1250
+ const color = C.TEXT_DIM;
1251
+ for (const tier of tiers) {
1252
+ if (a >= tier.threshold && tier.threshold > 0) {
1253
+ return `${fg(...color)}-${(a / tier.divisor).toLocaleString("en", { maximumFractionDigits: 1 })} ${tier.unit}${RESET}`;
1254
+ }
1255
+ }
1256
+ const lastTier2 = tiers[tiers.length - 1];
1257
+ if (lastTier2) {
1258
+ return `${fg(...color)}-${a} ${lastTier2.unit}${RESET}`;
1259
+ }
1260
+ return `${fg(...color)}-${a}${RESET}`;
1261
+ }
1262
+ for (const tier of tiers) {
1263
+ if (n >= tier.threshold && tier.threshold > 0) {
1264
+ return `${fg(...tier.color)}${(n / tier.divisor).toLocaleString("en", { maximumFractionDigits: 1 })} ${tier.unit}${RESET}`;
1265
+ }
1266
+ }
1267
+ const lastTier = tiers[tiers.length - 1];
1268
+ if (lastTier) {
1269
+ return `${fg(...lastTier.color)}${n} ${lastTier.unit}${RESET}`;
1270
+ }
1271
+ return `${n}${RESET}`;
1272
+ }
1273
+ function _fmtBytes(n) {
1274
+ return _fmtMagnitude(n, _BYTE_TIERS);
1275
+ }
1276
+ function _fmtTokens(n) {
1277
+ return _fmtMagnitude(n, _TOKEN_TIERS, "0 t");
1278
+ }
1279
+ function _fmtPct(fraction) {
1280
+ return `${(fraction * 100).toFixed(1)}%`;
1281
+ }
1282
+ function _fmtDelta(delta) {
1283
+ if (!delta && delta !== 0) {
1284
+ return "";
1285
+ }
1286
+ const up = (delta ?? 0) >= 0;
1287
+ const color = up ? C.GREEN5 : C.RED;
1288
+ const arrow = up ? "\u2191" : "\u2193";
1289
+ return ` ${fg(...color)}${arrow} ${Math.round(Math.abs(delta ?? 0))}%${RESET}`;
1290
+ }
1291
+ var _EIGHTHS = ["\u258F", "\u258E", "\u258D", "\u258C", "\u258B", "\u258A", "\u2589"];
1292
+ var _BLOCK = "\u2588";
1293
+ var _TRACK = "\u2591";
1294
+ var _GRADIENT = [C.GREEN1, C.GREEN2, C.GREEN3, C.GREEN4, C.GREEN5];
1295
+ function _distribute(total, n) {
1296
+ if (total <= 0 || n <= 0) {
1297
+ return Array(Math.max(0, n)).fill(0);
1298
+ }
1299
+ const base = Math.floor(total / n);
1300
+ const rem = total % n;
1301
+ return Array.from({ length: n }, (_, i) => base + (i >= n - rem ? 1 : 0));
1302
+ }
1303
+ function _renderBar(fraction, width = _BAR_W) {
1304
+ const f = Math.max(0, Math.min(1, fraction));
1305
+ const raw = f * width;
1306
+ let nFull = Math.floor(raw);
1307
+ const eighths = Math.round((raw - nFull) * 8);
1308
+ if (eighths >= 8) {
1309
+ nFull += 1;
1310
+ }
1311
+ const hasPartial = eighths > 0 && eighths < 8;
1312
+ const nTrack = Math.max(0, width - nFull - (hasPartial ? 1 : 0));
1313
+ const counts = _distribute(nFull, _GRADIENT.length);
1314
+ let bar = counts.map((count, i) => count > 0 ? `${fg(_GRADIENT[i][0], _GRADIENT[i][1], _GRADIENT[i][2])}${_BLOCK.repeat(count)}` : "").join("");
1315
+ if (hasPartial) {
1316
+ const lastGrad = _GRADIENT[_GRADIENT.length - 1];
1317
+ bar += `${fg(lastGrad[0], lastGrad[1], lastGrad[2])}${_EIGHTHS[eighths - 1]}`;
1318
+ }
1319
+ if (nTrack > 0) {
1320
+ bar += `${fg(...C.TRACK)}${_TRACK.repeat(nTrack)}`;
1321
+ }
1322
+ return bar + RESET;
1323
+ }
1324
+ var _SPARK = "\u2581\u2582\u2583\u2584\u2585\u2586\u2587\u2588";
1325
+ function _resample(vals, length) {
1326
+ if (vals.length === 0) {
1327
+ return Array(length).fill(0);
1328
+ }
1329
+ if (vals.length === length) {
1330
+ return [...vals];
1331
+ }
1332
+ const result = [];
1333
+ for (let i = 0; i < length; i++) {
1334
+ const src = i / (length - 1 || 1) * (vals.length - 1);
1335
+ const lo = Math.floor(src);
1336
+ const hi = Math.min(vals.length - 1, lo + 1);
1337
+ const t = src - lo;
1338
+ const loVal = vals[lo] ?? 0;
1339
+ const hiVal = vals[hi] ?? 0;
1340
+ result.push(loVal * (1 - t) + hiVal * t);
1341
+ }
1342
+ return result;
1343
+ }
1344
+ function _renderSparkline(values, width = 8) {
1345
+ const pts = _resample(values, width);
1346
+ const hi = pts.length > 0 ? Math.max(...pts) : 1;
1347
+ const lo = pts.length > 0 ? Math.min(...pts) : 0;
1348
+ const span = hi - lo || 1;
1349
+ const chars = [];
1350
+ for (let i = 0; i < pts.length; i++) {
1351
+ const v = pts[i];
1352
+ if (v === void 0) continue;
1353
+ const idx = Math.min(7, Math.floor((v - lo) / span * 8));
1354
+ const color = lerpRgb(C.GREEN1, C.GREEN5, i / (width - 1 || 1));
1355
+ chars.push(`${fg(color[0], color[1], color[2])}${_SPARK[idx]}`);
1356
+ }
1357
+ return chars.join("") + RESET;
1358
+ }
1359
+ function _tokenOrByteShare(itemTokens, itemBytes, totalTokens, totalBytes) {
1360
+ if (totalTokens > 0) {
1361
+ return itemTokens / totalTokens;
1362
+ }
1363
+ if (totalBytes > 0) {
1364
+ return itemBytes / totalBytes;
1365
+ }
1366
+ return 0;
1367
+ }
1368
+ function _barFraction(itemBytes, grossBytes) {
1369
+ return itemBytes > 0 ? itemBytes / grossBytes : 0;
1370
+ }
1371
+ function _computeShareDenominators(items) {
1372
+ let grossBytesSum = 0;
1373
+ let shareByteSum = 0;
1374
+ let shareTokensSum = 0;
1375
+ for (const item of items) {
1376
+ if (item.bytes > 0) {
1377
+ grossBytesSum += item.bytes;
1378
+ }
1379
+ shareByteSum += Math.abs(item.bytes);
1380
+ shareTokensSum += Math.abs(item.tokens);
1381
+ }
1382
+ return {
1383
+ grossBytes: Math.max(grossBytesSum, 1),
1384
+ shareBytesDenom: Math.max(shareByteSum, 1),
1385
+ shareTokensDenom: shareTokensSum
1386
+ };
1387
+ }
1388
+ function _absShare(itemBytes, itemTokens, shareBytesDenom, shareTokensDenom) {
1389
+ if (shareTokensDenom === 0) {
1390
+ return itemBytes / shareBytesDenom;
1391
+ }
1392
+ return itemTokens / shareTokensDenom;
1393
+ }
1394
+ function _sectionHeader(title, subtitle = "") {
1395
+ const sub = subtitle ? ` ${fg(...C.TEXT_MUTED)}${subtitle}${RESET}` : "";
1396
+ return [
1397
+ "",
1398
+ `${_M}${fg(...C.TEXT_BRIGHT)}${title}${RESET}${sub}`,
1399
+ _RULE
1400
+ ];
1401
+ }
1402
+ function _tableHeader(firstColLabel) {
1403
+ return [
1404
+ _M,
1405
+ padR(`${fg(...C.TEXT_DIM)}${firstColLabel}${RESET}`, _COL_NAME),
1406
+ " ",
1407
+ padR(`${fg(...C.TEXT_DIM)}savings${RESET}`, _BAR_W),
1408
+ " ",
1409
+ padL(`${fg(...C.TEXT_DIM)}data saved${RESET}`, _COL_DATA),
1410
+ " ",
1411
+ padL(`${fg(...C.TEXT_DIM)}tokens saved${RESET}`, _COL_TOKENS),
1412
+ " ",
1413
+ padL(`${fg(...C.TEXT_DIM)}share${RESET}`, _COL_SHARE),
1414
+ " ",
1415
+ padL(`${fg(...C.TEXT_DIM)}events${RESET}`, _COL_EVENTS)
1416
+ ].join("");
1417
+ }
1418
+ function _tableRow({
1419
+ name,
1420
+ fraction,
1421
+ bytes,
1422
+ tokens,
1423
+ events,
1424
+ share,
1425
+ bytesModeOnly = false,
1426
+ namePrefix = "",
1427
+ nameColor = C.TEXT_PRIMARY
1428
+ }) {
1429
+ const prefixW = vlen(namePrefix);
1430
+ const maxName = _COL_NAME - prefixW;
1431
+ const truncated = name.length > maxName ? name.slice(0, maxName - 1) + "\u2026" : name;
1432
+ const nameStr = padR(`${namePrefix}${fg(...nameColor)}${truncated}${RESET}`, _COL_NAME);
1433
+ const dataStr = padL(_fmtBytes(bytes), _COL_DATA);
1434
+ const tokStr = bytesModeOnly ? padL(`${fg(...C.TEXT_DIM)}\u2014${RESET}`, _COL_TOKENS) : padL(_fmtTokens(tokens), _COL_TOKENS);
1435
+ const sharePct = share * 100;
1436
+ let shareColor;
1437
+ if (sharePct < 0) {
1438
+ shareColor = C.RED;
1439
+ } else if (sharePct >= 50) {
1440
+ shareColor = C.GREEN5;
1441
+ } else if (sharePct >= 10) {
1442
+ shareColor = C.TEXT_PRIMARY;
1443
+ } else {
1444
+ shareColor = C.TEXT_MUTED;
1445
+ }
1446
+ const shareStr = padL(`${fg(...shareColor)}${_fmtPct(share)}${RESET}`, _COL_SHARE);
1447
+ const evStr = padL(`${fg(...C.TEXT_PRIMARY)}${events.toLocaleString()}${RESET}`, _COL_EVENTS);
1448
+ return [_M, nameStr, " ", _renderBar(fraction), " ", dataStr, " ", tokStr, " ", shareStr, " ", evStr].join("");
1449
+ }
1450
+ function _renderKpiSection(stats) {
1451
+ const totals = stats.totals;
1452
+ const colW = Math.floor((_CONTENT_W - _M.length * 2) / 3);
1453
+ function card(label, value, delta, spark2) {
1454
+ return [
1455
+ padR(`${fg(C.TEXT_MUTED[0], C.TEXT_MUTED[1], C.TEXT_MUTED[2])}${label}${RESET}`, colW),
1456
+ padR(`${fg(C.TEXT_BRIGHT[0], C.TEXT_BRIGHT[1], C.TEXT_BRIGHT[2])}${value}${RESET}${delta}`, colW),
1457
+ spark2 !== null ? padR(spark2, colW) : padR("", colW)
1458
+ ];
1459
+ }
1460
+ const spark = totals.sparklines;
1461
+ const c1 = card(
1462
+ "events",
1463
+ `${totals.events.toLocaleString()}`,
1464
+ _fmtDelta(totals.events_delta ?? null),
1465
+ spark ? _renderSparkline(spark.events) : null
1466
+ );
1467
+ const c2 = card(
1468
+ "data saved",
1469
+ _fmtBytes(totals.bytes),
1470
+ _fmtDelta(totals.bytes_delta ?? null),
1471
+ spark ? _renderSparkline(spark.bytes) : null
1472
+ );
1473
+ const c3 = card(
1474
+ "tokens saved",
1475
+ _fmtTokens(totals.tokens),
1476
+ _fmtDelta(totals.tokens_delta ?? null),
1477
+ spark ? _renderSparkline(spark.tokens) : null
1478
+ );
1479
+ const border = fg(C.TEXT_DIM[0], C.TEXT_DIM[1], C.TEXT_DIM[2]);
1480
+ const frameBar = "\u2500".repeat(colW * 3 + 2);
1481
+ function framed(content) {
1482
+ return `${_M}${border}\u2502${RESET} ${content} ${border}\u2502${RESET}`;
1483
+ }
1484
+ const lines = [
1485
+ "",
1486
+ `${_M}${border}\u256D${frameBar}\u256E${RESET}`,
1487
+ framed(c1[0] + c2[0] + c3[0]),
1488
+ framed(c1[1] + c2[1] + c3[1])
1489
+ ];
1490
+ if (spark) {
1491
+ lines.push(framed(c1[2] + c2[2] + c3[2]));
1492
+ }
1493
+ lines.push(`${_M}${border}\u2570${frameBar}\u256F${RESET}`);
1494
+ return lines;
1495
+ }
1496
+ var _KIND_GROUPS = [
1497
+ {
1498
+ label: "Read savings",
1499
+ members: /* @__PURE__ */ new Set([
1500
+ "read_replacement",
1501
+ "section_replacement",
1502
+ "symbol_read",
1503
+ "section_read",
1504
+ "stub_view",
1505
+ "outline",
1506
+ "exports",
1507
+ "imports",
1508
+ "changed_lookup",
1509
+ "dep_docs",
1510
+ // Every other SOURCE_READ kind in stats.ts's KIND_TO_SOURCE: the surgical-read commands over documents, structured data and session/PR state. They were registered and produced but grouped nowhere, so `stats --full` printed the whole family under 'Other', away from the read-savings siblings they are measured against. image_meta/image_text sit here rather than under 'Images' because stats.ts files them as SOURCE_READ: they save read bytes, they do not shrink an image.
1511
+ "brief_view",
1512
+ "conflicts",
1513
+ "coverage_report_gaps",
1514
+ "csv_query",
1515
+ "csv_profile",
1516
+ "compact_doc",
1517
+ "docx_outline",
1518
+ "docx_tables",
1519
+ "docx_text",
1520
+ "gdrive_sections",
1521
+ "image_meta",
1522
+ "image_text",
1523
+ "json_query",
1524
+ "json_outline",
1525
+ "note_read",
1526
+ "note_list",
1527
+ "openapi_op",
1528
+ "openapi_outline",
1529
+ "pdf_extract",
1530
+ "pdf_locate",
1531
+ "pdf_outline",
1532
+ "pdf_meta",
1533
+ "pptx_outline",
1534
+ "pptx_slide",
1535
+ "pptx_notes",
1536
+ "pptx_text",
1537
+ "pr_slice",
1538
+ "session_outline",
1539
+ "session_slice",
1540
+ "sqlite_query",
1541
+ "sqlite_schema",
1542
+ "sqlite_tables",
1543
+ "transcript",
1544
+ "transcript_outline",
1545
+ "video_chapters",
1546
+ "xlsx_sheets",
1547
+ "xlsx_head",
1548
+ "xlsx_range",
1549
+ "xlsx_query",
1550
+ "xlsx_columns",
1551
+ "xml_query",
1552
+ "xml_outline",
1553
+ "html_query",
1554
+ "html_outline",
1555
+ "yaml_query",
1556
+ "yaml_outline",
1557
+ "zip_list",
1558
+ "zip_read"
1559
+ ])
1560
+ },
1561
+ { label: "Lookups", members: /* @__PURE__ */ new Set(["symbol_lookup", "semantic_search", "map_lookup"]) },
1562
+ {
1563
+ label: "Images",
1564
+ members: /* @__PURE__ */ new Set([
1565
+ "image_shrink",
1566
+ "gdrive_image",
1567
+ "webfetch_image",
1568
+ "image_shrink_skipped",
1569
+ "image_shrink_cache_hit",
1570
+ "image_ocr"
1571
+ ])
1572
+ },
1573
+ {
1574
+ label: "Hints",
1575
+ members: /* @__PURE__ */ new Set([
1576
+ "session_hint",
1577
+ "session_hint_overhead",
1578
+ "session_hint_suppressed",
1579
+ "read_count_deny",
1580
+ "read_served_deny",
1581
+ "subagent_markdown_first_read_deny",
1582
+ "grep_dedup_hint",
1583
+ "glob_dedup_hint",
1584
+ "diff_hint",
1585
+ "predictive_prefetch_hit",
1586
+ "structured_file_hint",
1587
+ "write_rewrite_hint",
1588
+ "websearch_dedup_hint",
1589
+ "large_file_hint_followed",
1590
+ "large_file_hint_ignored",
1591
+ "evidence_cache_hit"
1592
+ ])
1593
+ },
1594
+ // Empty for the same reason as MCP below: every live Bash kind arrives through _kindGroupLabel's `bash_compress:` prefix branch, not through a literal name. The fifteen literal names this set used to carry (bash_output_cached, bash_dedup_hint, env_probe_cache_hit and the rest) came over with the Python port and were never recorded or registered anywhere in this tree, so they grouped rows that could not exist.
1595
+ { label: "Bash", members: /* @__PURE__ */ new Set() },
1596
+ {
1597
+ label: "Web",
1598
+ members: /* @__PURE__ */ new Set([
1599
+ "web_fetch",
1600
+ "injection_detected"
1601
+ ])
1602
+ },
1603
+ // Membership comes from _kindGroupLabel's `mcp:` prefix branch, not from this set, which is why
1604
+ // it is empty. The entry still has to exist: _renderByKindSection iterates _KIND_GROUPS' labels
1605
+ // (plus 'Other') to decide what to print, so a label _kindGroupLabel returns but that is missing
1606
+ // here does not fall back to 'Other' -- its rows disappear from the table entirely.
1607
+ { label: "MCP", members: /* @__PURE__ */ new Set() },
1608
+ {
1609
+ label: "Compact / Skills",
1610
+ members: /* @__PURE__ */ new Set([
1611
+ "skill_load",
1612
+ "skill_oversized_first_load",
1613
+ "skill_compact_inlined",
1614
+ "skill_heading_tree_inlined"
1615
+ ])
1616
+ },
1617
+ // SOURCE_CONTENT: real rewrites of tool output that remove real bytes (agent report compaction, Grep fold, browser tab dedup, bash/content compression and the handoff pair). The by-source table has shown a 'content' row since the source was added, but the by-kind table had no member set for it, so every one of these kinds printed under 'Other'. The taskoutput: prefix branch in _kindGroupLabel routes here too.
1618
+ {
1619
+ label: "Content",
1620
+ members: /* @__PURE__ */ new Set([
1621
+ "content_compress",
1622
+ "content_retrieve",
1623
+ "agent_report_compact",
1624
+ "agent_report_compact_declined",
1625
+ "browser_tab_dedup",
1626
+ "grep:fold",
1627
+ "read:served_elide",
1628
+ "read:body_fold",
1629
+ "read:markdown_outline",
1630
+ "read:source_skeleton",
1631
+ "handoff_create",
1632
+ "handoff_resolve",
1633
+ "plan_echo_collapse"
1634
+ ])
1635
+ }
1636
+ ];
1637
+ function _kindGroupLabel(kind) {
1638
+ if (kind.startsWith("bash_compress:") || kind.startsWith("bashoutput:")) {
1639
+ return "Bash";
1640
+ }
1641
+ if (kind.startsWith("webfetch:") || kind.startsWith("gdrive:")) {
1642
+ return "Web";
1643
+ }
1644
+ if (kind.startsWith("mcp:")) {
1645
+ return "MCP";
1646
+ }
1647
+ if (kind.startsWith("skill_body:") || kind.startsWith("skill_compact:")) {
1648
+ return "Compact / Skills";
1649
+ }
1650
+ if (kind.startsWith("taskoutput:")) {
1651
+ return "Content";
1652
+ }
1653
+ for (const group of _KIND_GROUPS) {
1654
+ if (group.members.has(kind)) {
1655
+ return group.label;
1656
+ }
1657
+ }
1658
+ return "Other";
1659
+ }
1660
+ function _groupSeparator(label) {
1661
+ return `${_M} ${fg(...C.TEXT_DIM)}${label}${RESET}`;
1662
+ }
1663
+ function _renderByKindSection(stats) {
1664
+ if (stats.by_kind.length === 0) {
1665
+ return [];
1666
+ }
1667
+ const lines = [..._sectionHeader("By kind"), _tableHeader("name")];
1668
+ const { grossBytes, shareBytesDenom, shareTokensDenom } = _computeShareDenominators(stats.by_kind);
1669
+ const kindNames = new Set(stats.by_kind.map((k) => k.kind));
1670
+ const bytesModeKinds = stats.by_kind.filter((k) => k.bytes_mode_only).map((k) => k.kind);
1671
+ function share(k) {
1672
+ if (k.bytes_mode_only) {
1673
+ return k.bytes / shareBytesDenom;
1674
+ }
1675
+ return _absShare(k.bytes, k.tokens, shareBytesDenom, shareTokensDenom);
1676
+ }
1677
+ const byGroup = /* @__PURE__ */ new Map();
1678
+ for (const k of stats.by_kind) {
1679
+ const grp = _kindGroupLabel(k.kind);
1680
+ if (!byGroup.has(grp)) {
1681
+ byGroup.set(grp, []);
1682
+ }
1683
+ byGroup.get(grp).push(k);
1684
+ }
1685
+ for (const grpKinds of byGroup.values()) {
1686
+ grpKinds.sort((a, b) => share(b) - share(a));
1687
+ }
1688
+ const groupLabels = [..._KIND_GROUPS.map((g) => g.label), "Other"];
1689
+ let firstGroup = true;
1690
+ for (const label of groupLabels) {
1691
+ const groupKinds = byGroup.get(label);
1692
+ if (!groupKinds || groupKinds.length === 0) {
1693
+ continue;
1694
+ }
1695
+ if (!firstGroup) {
1696
+ lines.push("");
1697
+ }
1698
+ firstGroup = false;
1699
+ lines.push(_groupSeparator(label));
1700
+ for (const k of groupKinds) {
1701
+ const s = share(k);
1702
+ lines.push(
1703
+ _tableRow({
1704
+ name: k.kind,
1705
+ fraction: _barFraction(k.bytes, grossBytes),
1706
+ bytes: k.bytes,
1707
+ tokens: k.tokens,
1708
+ events: k.events,
1709
+ share: s,
1710
+ bytesModeOnly: k.bytes_mode_only ?? false
1711
+ })
1712
+ );
1713
+ }
1714
+ }
1715
+ if (bytesModeKinds.length > 0) {
1716
+ const names = bytesModeKinds.join(", ");
1717
+ lines.push(`${_M}${fg(...C.TEXT_DIM)}i ${names} ${_STATS_MESSAGES.bytesModeOnlyNote}${RESET}`);
1718
+ }
1719
+ if (kindNames.has("session_hint") && kindNames.has("session_hint_overhead")) {
1720
+ lines.push(`${_M}${fg(...C.TEXT_DIM)}i ${_STATS_MESSAGES.sessionHintSplitNote}${RESET}`);
1721
+ }
1722
+ return lines;
1723
+ }
1724
+ var _SOURCE_COLORS = {
1725
+ image: C.PURPLE,
1726
+ hint: C.BLUE,
1727
+ read: C.GREEN4,
1728
+ compact: C.TEAL,
1729
+ bash: C.ORANGE,
1730
+ web: C.YELLOW,
1731
+ other: C.TEXT_MUTED
1732
+ };
1733
+ function _sourceColor(source) {
1734
+ const color = _SOURCE_COLORS[source];
1735
+ return color || C.TEXT_MUTED;
1736
+ }
1737
+ function _renderBySourceSection(stats) {
1738
+ if (!stats.by_source || stats.by_source.length === 0) {
1739
+ return [];
1740
+ }
1741
+ const lines = [..._sectionHeader("By source"), _tableHeader("source")];
1742
+ const { grossBytes, shareBytesDenom, shareTokensDenom } = _computeShareDenominators(stats.by_source);
1743
+ function share(s) {
1744
+ return _absShare(s.bytes, s.tokens, shareBytesDenom, shareTokensDenom);
1745
+ }
1746
+ for (const s of [...stats.by_source].sort((a, b) => share(b) - share(a))) {
1747
+ const s_val = share(s);
1748
+ const color = _sourceColor(s.source);
1749
+ lines.push(
1750
+ _tableRow({
1751
+ name: s.source,
1752
+ fraction: _barFraction(s.bytes, grossBytes),
1753
+ bytes: s.bytes,
1754
+ tokens: s.tokens,
1755
+ events: s.events,
1756
+ share: s_val,
1757
+ namePrefix: `${fg(...color)}\u25CF${RESET} `,
1758
+ nameColor: C.TEXT_PRIMARY
1759
+ })
1760
+ );
1761
+ }
1762
+ return lines;
1763
+ }
1764
+ function _renderByCommandSection(stats) {
1765
+ if (!stats.by_command || stats.by_command.length === 0) {
1766
+ return [];
1767
+ }
1768
+ const lines = [..._sectionHeader("By command"), _tableHeader("command")];
1769
+ const { grossBytes, shareBytesDenom, shareTokensDenom } = _computeShareDenominators(stats.by_command);
1770
+ function share(c) {
1771
+ return _absShare(c.bytes, c.tokens, shareBytesDenom, shareTokensDenom);
1772
+ }
1773
+ for (const c of [...stats.by_command].sort((a, b) => share(b) - share(a))) {
1774
+ const s_val = share(c);
1775
+ lines.push(
1776
+ _tableRow({
1777
+ name: c.command,
1778
+ fraction: _barFraction(c.bytes, grossBytes),
1779
+ bytes: c.bytes,
1780
+ tokens: c.tokens,
1781
+ events: c.events,
1782
+ share: s_val,
1783
+ nameColor: C.TEXT_PRIMARY
1784
+ })
1785
+ );
1786
+ }
1787
+ return lines;
1788
+ }
1789
+ function _renderByHarnessSection(stats) {
1790
+ if (!stats.by_harness || stats.by_harness.length < 2) {
1791
+ return [];
1792
+ }
1793
+ const lines = [..._sectionHeader("By harness"), _tableHeader("harness")];
1794
+ const { grossBytes, shareBytesDenom, shareTokensDenom } = _computeShareDenominators(stats.by_harness);
1795
+ function share(h) {
1796
+ return _absShare(h.bytes, h.tokens, shareBytesDenom, shareTokensDenom);
1797
+ }
1798
+ for (const h of [...stats.by_harness].sort((a, b) => share(b) - share(a))) {
1799
+ lines.push(
1800
+ _tableRow({
1801
+ name: h.harness,
1802
+ fraction: _barFraction(h.bytes, grossBytes),
1803
+ bytes: h.bytes,
1804
+ tokens: h.tokens,
1805
+ events: h.events,
1806
+ share: share(h),
1807
+ nameColor: C.TEXT_PRIMARY
1808
+ })
1809
+ );
1810
+ }
1811
+ return lines;
1812
+ }
1813
+ function _renderByDaySection(stats) {
1814
+ if (stats.by_day.length === 0) {
1815
+ return [];
1816
+ }
1817
+ const lines = [..._sectionHeader("By day"), _tableHeader("date")];
1818
+ function share(d) {
1819
+ return _tokenOrByteShare(d.tokens, d.bytes, stats.totals.tokens, stats.totals.bytes);
1820
+ }
1821
+ for (const d of [...stats.by_day].sort((a, b) => b.date < a.date ? -1 : b.date > a.date ? 1 : 0)) {
1822
+ const s = share(d);
1823
+ lines.push(
1824
+ _tableRow({
1825
+ name: d.date,
1826
+ fraction: s,
1827
+ bytes: d.bytes,
1828
+ tokens: d.tokens,
1829
+ events: d.events,
1830
+ share: s
1831
+ })
1832
+ );
1833
+ }
1834
+ return lines;
1835
+ }
1836
+ var _PROJECT_COLORS = [C.PURPLE, C.TEAL, C.BLUE, C.GREEN4, C.TEXT_MUTED];
1837
+ function _hashColor(hashStr) {
1838
+ let n = 0;
1839
+ for (const c of hashStr) {
1840
+ n += c.charCodeAt(0);
1841
+ }
1842
+ return _PROJECT_COLORS[n % _PROJECT_COLORS.length] ?? C.TEXT_MUTED;
1843
+ }
1844
+ function _renderByProjectSection(stats) {
1845
+ if (stats.by_project.length === 0) {
1846
+ return [];
1847
+ }
1848
+ const lines = [..._sectionHeader(`By project (top ${stats.by_project.length})`), _tableHeader("project")];
1849
+ function share(p) {
1850
+ return _tokenOrByteShare(p.tokens, p.bytes, stats.totals.tokens, stats.totals.bytes);
1851
+ }
1852
+ for (const p of [...stats.by_project].sort((a, b) => share(b) - share(a))) {
1853
+ const s = share(p);
1854
+ const color = _hashColor(p.hash);
1855
+ lines.push(
1856
+ _tableRow({
1857
+ name: p.project,
1858
+ fraction: s,
1859
+ bytes: p.bytes,
1860
+ tokens: p.tokens,
1861
+ events: p.events,
1862
+ share: s,
1863
+ namePrefix: `${fg(...color)}\u25CF${RESET} `,
1864
+ nameColor: C.TEXT_PRIMARY
1865
+ })
1866
+ );
1867
+ lines.push(`${_M} ${fg(...C.TEXT_DIM)}\u2514\u2500 ${p.hash} ${displaySafeText(stripAnsi(p.path))}${RESET}`);
1868
+ }
1869
+ return lines;
1870
+ }
1871
+ function _renderInsightsSection(stats) {
1872
+ const lines = [..._sectionHeader("Insights")];
1873
+ const bullet = `${fg(...C.GREEN3)}\u25B8${RESET}`;
1874
+ function dim(s) {
1875
+ return `${fg(...C.TEXT_MUTED)}${s}${RESET}`;
1876
+ }
1877
+ const savingKinds = stats.by_kind.filter((k) => k.bytes > 0);
1878
+ const topKind = savingKinds.reduce((max, k) => k.bytes > (max?.bytes || -Infinity) ? k : max, savingKinds[0]);
1879
+ if (topKind) {
1880
+ const share = stats.totals.bytes > 0 ? topKind.bytes / stats.totals.bytes : 0;
1881
+ lines.push(
1882
+ `${_M}${bullet} ${dim(_STATS_MESSAGES.insights.biggestSaver)}${fg(...C.TEXT_PRIMARY)}${topKind.kind}${RESET}${dim(" \u2014 ")}${fg(...C.GREEN5)}${_fmtPct(share)}${RESET}${dim(` of saved data across ${topKind.events.toLocaleString()} events`)}`
1883
+ );
1884
+ }
1885
+ const topDay = stats.by_day.reduce((max, d) => d.events > (max?.events || -Infinity) ? d : max, stats.by_day[0]);
1886
+ if (topDay) {
1887
+ lines.push(
1888
+ `${_M}${bullet} ${dim(_STATS_MESSAGES.insights.mostActive)}${fg(...C.TEXT_PRIMARY)}${topDay.date}${RESET}${dim(" \u2014 ")}${topDay.events.toLocaleString()} events, ${_fmtBytes(topDay.bytes)}${dim(" saved")}`
1889
+ );
1890
+ }
1891
+ const tokenKinds = stats.by_kind.filter((k) => !k.bytes_mode_only && k.tokens > 0);
1892
+ const topToken = tokenKinds.reduce((max, k) => k.tokens > (max?.tokens || -Infinity) ? k : max, tokenKinds[0]);
1893
+ if (topToken) {
1894
+ lines.push(
1895
+ `${_M}${bullet} ${dim(_STATS_MESSAGES.insights.tokenLeader)}${fg(...C.TEXT_PRIMARY)}${topToken.kind}${RESET}${dim(" \u2014 ")}${_fmtTokens(topToken.tokens)}${dim(` saved in ${topToken.events.toLocaleString()} events`)}`
1896
+ );
1897
+ }
1898
+ if ((stats.by_command?.length ?? 0) === 0) {
1899
+ const hintSource = stats.by_source?.find((s) => s.source === "hint");
1900
+ if (hintSource && hintSource.events > 0) {
1901
+ lines.push(
1902
+ `${_M}${fg(...C.YELLOW)}\u25B8${RESET} ${dim("0 direct commands ")}${fg(...C.TEXT_PRIMARY)}${hintSource.events.toLocaleString()}${RESET}${dim(" hint(s) fired but not acted on \u2014 run symbol/read/section/semantic/outline/skeleton directly to capture these savings")}`
1903
+ );
1904
+ }
1905
+ }
1906
+ return lines;
1907
+ }
1908
+ function _renderHeader(stats) {
1909
+ let line = `${_M}${fg(...C.TEXT_BRIGHT)}token-goat${RESET}`;
1910
+ if (stats.version) {
1911
+ line += ` ${fg(...C.TEXT_MUTED)}v${stats.version}${RESET}`;
1912
+ }
1913
+ if (stats.window_label) {
1914
+ line += ` ${fg(...C.TEXT_DIM)}\xB7 ${stats.window_label}${RESET}`;
1915
+ }
1916
+ return [line];
1917
+ }
1918
+ function _renderShortHint() {
1919
+ return [
1920
+ "",
1921
+ `${_M}${fg(...C.TEXT_MUTED)}Run 'token-goat stats --full' for the full breakdown (by source, by command, by day).${RESET}`
1922
+ ];
1923
+ }
1924
+ function renderStats(stats, opts) {
1925
+ if (opts?.short) {
1926
+ const sections2 = [_renderHeader(stats), _renderKpiSection(stats), _renderShortHint(), [""]];
1927
+ return sections2.flatMap((s) => s).join("\n");
1928
+ }
1929
+ const sections = [
1930
+ _renderHeader(stats),
1931
+ _renderKpiSection(stats),
1932
+ _renderByKindSection(stats),
1933
+ _renderBySourceSection(stats),
1934
+ _renderByCommandSection(stats),
1935
+ _renderByHarnessSection(stats),
1936
+ _renderByDaySection(stats),
1937
+ _renderByProjectSection(stats),
1938
+ _renderInsightsSection(stats),
1939
+ [""]
1940
+ ];
1941
+ return sections.flatMap((s) => s).join("\n");
1942
+ }
1943
+
1944
+ // src/stats.ts
1945
+ var HARNESS_UNRECORDED = "unrecorded (pre-2.8.1)";
1946
+ var PRICING_VERSION_UNRECORDED = "unrecorded (pre-tg_version column)";
1947
+ function hasMixedPricingEras(summary) {
1948
+ return Object.keys(summary.by_pricing_version).length > 1;
1949
+ }
1950
+ var SOURCE_IMAGE = "image";
1951
+ var SOURCE_HINT = "hint";
1952
+ var SOURCE_READ = "read";
1953
+ var SOURCE_BASH = "bash";
1954
+ var SOURCE_WEB = "web";
1955
+ var SOURCE_MCP = "mcp";
1956
+ var SOURCE_SKILL = "skill";
1957
+ var SOURCE_CONTENT = "content";
1958
+ var SOURCE_OTHER = "other";
1959
+ var _BYTES_MODE_ONLY_KINDS = /* @__PURE__ */ new Set(["webfetch_image", "gdrive_image"]);
1960
+ var COUNT_ONLY_KINDS = /* @__PURE__ */ new Set(["secret_redacted"]);
1961
+ function savedTokensFromBytes(bytes) {
1962
+ return Math.round(Math.max(0, bytes) / 4);
1963
+ }
1964
+ var KIND_TO_SOURCE = {
1965
+ image_shrink: SOURCE_IMAGE,
1966
+ image_shrink_cache_hit: SOURCE_IMAGE,
1967
+ image_shrink_skipped: SOURCE_IMAGE,
1968
+ image_ocr: SOURCE_IMAGE,
1969
+ webfetch_image: SOURCE_IMAGE,
1970
+ gdrive_image: SOURCE_IMAGE,
1971
+ session_hint: SOURCE_HINT,
1972
+ session_hint_suppressed: SOURCE_HINT,
1973
+ diff_hint: SOURCE_HINT,
1974
+ evidence_cache_hit: SOURCE_HINT,
1975
+ structured_file_hint: SOURCE_HINT,
1976
+ predictive_prefetch_hit: SOURCE_HINT,
1977
+ grep_dedup_hint: SOURCE_HINT,
1978
+ glob_dedup_hint: SOURCE_HINT,
1979
+ write_rewrite_hint: SOURCE_HINT,
1980
+ websearch_dedup_hint: SOURCE_HINT,
1981
+ large_file_hint_followed: SOURCE_HINT,
1982
+ large_file_hint_ignored: SOURCE_HINT,
1983
+ read_count_deny: SOURCE_HINT,
1984
+ read_served_deny: SOURCE_HINT,
1985
+ // hooks_read.ts's subagent first-read markdown deny (hints.subagent_markdown_first_read_deny, off by default). Always recorded at 0 bytes / 0 tokens: no first-read deny of this shape exists in the transcript corpus, so its abandoned/substituted/shell-read/retried rates are unknown and can only be borrowed from the re-read heading-tree census. Booking withheld bytes against borrowed rates would claim a saving this path cannot back up. The kind exists to make the intervention countable in session-audit, not to claim a win.
1986
+ subagent_markdown_first_read_deny: SOURCE_HINT,
1987
+ read_replacement: SOURCE_READ,
1988
+ section_replacement: SOURCE_READ,
1989
+ symbol_read: SOURCE_READ,
1990
+ section_read: SOURCE_READ,
1991
+ stub_view: SOURCE_READ,
1992
+ symbol_lookup: SOURCE_READ,
1993
+ semantic_search: SOURCE_READ,
1994
+ map_lookup: SOURCE_READ,
1995
+ changed_lookup: SOURCE_READ,
1996
+ outline: SOURCE_READ,
1997
+ exports: SOURCE_READ,
1998
+ imports: SOURCE_READ,
1999
+ dep_docs: SOURCE_READ,
2000
+ csv_query: SOURCE_READ,
2001
+ csv_profile: SOURCE_READ,
2002
+ pdf_extract: SOURCE_READ,
2003
+ pdf_locate: SOURCE_READ,
2004
+ pdf_outline: SOURCE_READ,
2005
+ pdf_meta: SOURCE_READ,
2006
+ xlsx_sheets: SOURCE_READ,
2007
+ xlsx_head: SOURCE_READ,
2008
+ xlsx_range: SOURCE_READ,
2009
+ xlsx_query: SOURCE_READ,
2010
+ xlsx_columns: SOURCE_READ,
2011
+ pptx_outline: SOURCE_READ,
2012
+ pptx_slide: SOURCE_READ,
2013
+ pptx_notes: SOURCE_READ,
2014
+ pptx_text: SOURCE_READ,
2015
+ docx_outline: SOURCE_READ,
2016
+ docx_tables: SOURCE_READ,
2017
+ docx_text: SOURCE_READ,
2018
+ transcript_outline: SOURCE_READ,
2019
+ transcript: SOURCE_READ,
2020
+ video_chapters: SOURCE_READ,
2021
+ image_meta: SOURCE_READ,
2022
+ image_text: SOURCE_READ,
2023
+ coverage_report_gaps: SOURCE_READ,
2024
+ json_query: SOURCE_READ,
2025
+ json_outline: SOURCE_READ,
2026
+ yaml_query: SOURCE_READ,
2027
+ yaml_outline: SOURCE_READ,
2028
+ xml_query: SOURCE_READ,
2029
+ xml_outline: SOURCE_READ,
2030
+ html_query: SOURCE_READ,
2031
+ html_outline: SOURCE_READ,
2032
+ openapi_op: SOURCE_READ,
2033
+ openapi_outline: SOURCE_READ,
2034
+ zip_list: SOURCE_READ,
2035
+ zip_read: SOURCE_READ,
2036
+ sqlite_query: SOURCE_READ,
2037
+ sqlite_schema: SOURCE_READ,
2038
+ sqlite_tables: SOURCE_READ,
2039
+ conflicts: SOURCE_READ,
2040
+ brief_view: SOURCE_READ,
2041
+ session_outline: SOURCE_READ,
2042
+ session_slice: SOURCE_READ,
2043
+ gdrive_sections: SOURCE_READ,
2044
+ pr_slice: SOURCE_READ,
2045
+ compact_doc: SOURCE_READ,
2046
+ note_read: SOURCE_READ,
2047
+ note_list: SOURCE_READ,
2048
+ // note-add is a write (like insert-section/replace, which record no stat at all -- neither
2049
+ // has a "full source it replaces" savings concept). It still gets an event-only entry here
2050
+ // (no bytesSaved/tokensSaved argument, same as skill_load) purely so `token-goat note-add`
2051
+ // usage is visible in `token-goat stats --full` at all -- SOURCE_OTHER, not SOURCE_READ,
2052
+ // since it is not a token-savings substitute for a read.
2053
+ note_write: SOURCE_OTHER,
2054
+ web_fetch: SOURCE_WEB,
2055
+ injection_detected: SOURCE_WEB,
2056
+ skill_load: SOURCE_SKILL,
2057
+ skill_oversized_first_load: SOURCE_SKILL,
2058
+ // Cold first load of an oversized skill where preSkillHandler inlined the compact slice in its reply instead of pointing at `skill-body --compact`. Unlike its skill_oversized_first_load sibling (event-only, 0 bytes -- the pointer deny saves nothing by itself, the follow-up command does) this one records real savings: the full body never landed, the slice did, so bytesSaved is body minus slice.
2059
+ skill_compact_inlined: SOURCE_SKILL,
2060
+ // Cold first load of an oversized skill with no compact marker at all, where preSkillHandler inlined a heading tree in its reply instead of letting the whole body fall through. Same shape as skill_compact_inlined: real savings, bytesSaved is body minus the rendered tree.
2061
+ skill_heading_tree_inlined: SOURCE_SKILL,
2062
+ secret_redacted: SOURCE_OTHER,
2063
+ // Fail-soft diagnostic counters from hooks_edit.ts: they record that a side task threw, never a byte saving, so "other" is the right home. Listed explicitly rather than left to kindToSource()'s fallback so the registration guard can tell a deliberate placement from an unregistered kind.
2064
+ dirty_queue_append_failed: SOURCE_OTHER,
2065
+ worker_healthcheck_failed: SOURCE_OTHER,
2066
+ known_root_record_failed: SOURCE_OTHER,
2067
+ // Measurement of what a compaction produced (hooks_compact.ts postCompactHandler): summary size and how many manifest paths survived into it. SOURCE_OTHER and always recorded at (0, 0) -- the summary was written whether or not token-goat was watching, so there is no counterfactual in which those bytes were saved. Filing it anywhere with a savings total would credit token-goat for the whole summary, which is the accounting mistake this registry exists to prevent.
2068
+ compact_summary: SOURCE_OTHER,
2069
+ // Envelope compaction of an oversized subagent report (hooks_agent_spawn.ts). SOURCE_CONTENT, not SOURCE_HINT: the handler's sibling session_hint entry is advisory (it only appends a recall pointer and genuinely saves nothing), whereas this kind records a real rewrite with real bytes removed, so filing it under the advisory bucket would understate the compaction and repeat the zero-savings desync this registry keeps getting bitten by.
2070
+ agent_report_compact: SOURCE_CONTENT,
2071
+ // Decline counterpart to agent_report_compact: the fence-collapse net-benefit gate ran and found at least one over-long fence, but declined to rewrite because net savings did not clear the notice cost. Always recorded at (0, 0) -- see the recordStat call site -- so it never contributes to any savings total; it exists purely to make gate hit-rate and near-misses visible instead of the decline being invisible.
2072
+ agent_report_compact_declined: SOURCE_CONTENT,
2073
+ content_compress: SOURCE_CONTENT,
2074
+ // Verbatim-repeat collapse of a browser tool's "Tab Context:" text block (hooks_browser_image.ts postBrowserImageHandler). SOURCE_CONTENT for the same reason as agent_report_compact above: it is a real rewrite with real bytes removed, not an advisory nudge. Deliberately not SOURCE_IMAGE -- it shares a handler with image_shrink but collapses text, and folding text bytes into the image ledger is the two-units-under-one-label mistake this file's image_shrink entry was just fixed for.
2075
+ browser_tab_dedup: SOURCE_CONTENT,
2076
+ // Collapse of the plan echo in an approved ExitPlanMode result (hooks_exitplanmode.ts). SOURCE_CONTENT for the same reason as agent_report_compact: real bytes removed from a tool result, not an advisory nudge. The handler shipped for releases emitting this rewrite and recording nothing at all, so the mechanism was invisible in `stats` and its net benefit could not be checked against the gate that admits it.
2077
+ plan_echo_collapse: SOURCE_CONTENT,
2078
+ // Lossless re-layout of Grep content-mode output (hooks_grep.ts foldGrepContentHandler). SOURCE_CONTENT, not SOURCE_HINT, for the same reason as agent_report_compact above: its sibling grep_dedup_hint is advisory and saves nothing directly, whereas this is a real rewrite with real bytes removed. Filing it under the advisory bucket would silently add non-hint savings to hint_stats.ts's savedBytes (which reads by_source[SOURCE_HINT] wholesale) and overstate the hint ledger's net benefit.
2079
+ "grep:fold": SOURCE_CONTENT,
2080
+ // Withholding of already-served stretches from a completed Read (hooks_read.ts
2081
+ // elideAlreadyServedLines). SOURCE_CONTENT, not SOURCE_HINT, for the same reason as
2082
+ // grep:fold above: its siblings read_count_deny and read_served_deny are decisions about
2083
+ // whether a read happens at all, whereas this is a rewrite of a result that did happen,
2084
+ // with real bytes removed from it. Filing it under the advisory bucket would add non-hint
2085
+ // savings to hint_stats.ts's savedBytes, which reads by_source[SOURCE_HINT] wholesale.
2086
+ "read:served_elide": SOURCE_CONTENT,
2087
+ // Same bucket and same reasoning as read:served_elide directly above: a rewrite of a Read that did happen, with real bytes removed, not an advisory about whether to read at all.
2088
+ "read:body_fold": SOURCE_CONTENT,
2089
+ // Same bucket and same reasoning as read:body_fold directly above: a coarser sibling rewrite of a large untargeted markdown Read (hooks_read.ts foldMarkdownOutline) that replaces the body with a heading tree plus preamble, with real bytes removed, not an advisory about whether to read at all.
2090
+ "read:markdown_outline": SOURCE_CONTENT,
2091
+ // Same bucket and same reasoning as read:markdown_outline directly above, on source instead of prose: the structural-skeleton replacement of a large untargeted source Read (hooks_read.ts foldSourceSkeleton), with real bytes removed, not an advisory about whether to read at all.
2092
+ "read:source_skeleton": SOURCE_CONTENT,
2093
+ content_retrieve: SOURCE_CONTENT,
2094
+ handoff_create: SOURCE_CONTENT,
2095
+ handoff_resolve: SOURCE_CONTENT
2096
+ };
2097
+ var KIND_PREFIX_TO_SOURCE = [
2098
+ ["bash_compress:", SOURCE_BASH],
2099
+ ["webfetch:", SOURCE_WEB],
2100
+ ["gdrive:", SOURCE_WEB],
2101
+ ["mcp:", SOURCE_MCP],
2102
+ ["skill_body:", SOURCE_SKILL],
2103
+ ["skill_compact:", SOURCE_SKILL],
2104
+ ["bashoutput:", SOURCE_BASH],
2105
+ ["taskoutput:", SOURCE_CONTENT]
2106
+ ];
2107
+ var COMMAND_KINDS = {
2108
+ symbol: /* @__PURE__ */ new Set(["symbol_lookup"]),
2109
+ read: /* @__PURE__ */ new Set(["read_replacement"]),
2110
+ section: /* @__PURE__ */ new Set(["section_replacement", "section_read"]),
2111
+ semantic: /* @__PURE__ */ new Set(["semantic_search"]),
2112
+ outline: /* @__PURE__ */ new Set(["outline"]),
2113
+ exports: /* @__PURE__ */ new Set(["exports"]),
2114
+ imports: /* @__PURE__ */ new Set(["imports"]),
2115
+ skeleton: /* @__PURE__ */ new Set(["stub_view"]),
2116
+ refs: /* @__PURE__ */ new Set(["symbol_read"]),
2117
+ map: /* @__PURE__ */ new Set(["map_lookup"]),
2118
+ changed: /* @__PURE__ */ new Set(["changed_lookup"]),
2119
+ "dep-docs": /* @__PURE__ */ new Set(["dep_docs"]),
2120
+ "csv-query": /* @__PURE__ */ new Set(["csv_query"]),
2121
+ "csv-profile": /* @__PURE__ */ new Set(["csv_profile"]),
2122
+ "pdf-extract": /* @__PURE__ */ new Set(["pdf_extract"]),
2123
+ "pdf-locate": /* @__PURE__ */ new Set(["pdf_locate"]),
2124
+ "pdf-outline": /* @__PURE__ */ new Set(["pdf_outline"]),
2125
+ "pdf-meta": /* @__PURE__ */ new Set(["pdf_meta"]),
2126
+ "xlsx-sheets": /* @__PURE__ */ new Set(["xlsx_sheets"]),
2127
+ "xlsx-head": /* @__PURE__ */ new Set(["xlsx_head"]),
2128
+ "xlsx-range": /* @__PURE__ */ new Set(["xlsx_range"]),
2129
+ "xlsx-query": /* @__PURE__ */ new Set(["xlsx_query"]),
2130
+ "xlsx-columns": /* @__PURE__ */ new Set(["xlsx_columns"]),
2131
+ "pptx-outline": /* @__PURE__ */ new Set(["pptx_outline"]),
2132
+ "pptx-slide": /* @__PURE__ */ new Set(["pptx_slide"]),
2133
+ "pptx-notes": /* @__PURE__ */ new Set(["pptx_notes"]),
2134
+ "pptx-text": /* @__PURE__ */ new Set(["pptx_text"]),
2135
+ "docx-outline": /* @__PURE__ */ new Set(["docx_outline"]),
2136
+ "docx-tables": /* @__PURE__ */ new Set(["docx_tables"]),
2137
+ "docx-text": /* @__PURE__ */ new Set(["docx_text"]),
2138
+ "transcript-outline": /* @__PURE__ */ new Set(["transcript_outline"]),
2139
+ transcript: /* @__PURE__ */ new Set(["transcript"]),
2140
+ "video-chapters": /* @__PURE__ */ new Set(["video_chapters"]),
2141
+ "image-meta": /* @__PURE__ */ new Set(["image_meta"]),
2142
+ "image-text": /* @__PURE__ */ new Set(["image_text"]),
2143
+ "coverage-report-gaps": /* @__PURE__ */ new Set(["coverage_report_gaps"]),
2144
+ "json-query": /* @__PURE__ */ new Set(["json_query"]),
2145
+ "json-outline": /* @__PURE__ */ new Set(["json_outline"]),
2146
+ "yaml-query": /* @__PURE__ */ new Set(["yaml_query"]),
2147
+ "yaml-outline": /* @__PURE__ */ new Set(["yaml_outline"]),
2148
+ "xml-query": /* @__PURE__ */ new Set(["xml_query"]),
2149
+ "xml-outline": /* @__PURE__ */ new Set(["xml_outline"]),
2150
+ "html-query": /* @__PURE__ */ new Set(["html_query"]),
2151
+ "html-outline": /* @__PURE__ */ new Set(["html_outline"]),
2152
+ "openapi-op": /* @__PURE__ */ new Set(["openapi_op"]),
2153
+ "openapi-outline": /* @__PURE__ */ new Set(["openapi_outline"]),
2154
+ "zip-list": /* @__PURE__ */ new Set(["zip_list"]),
2155
+ "zip-read": /* @__PURE__ */ new Set(["zip_read"]),
2156
+ "sqlite-query": /* @__PURE__ */ new Set(["sqlite_query"]),
2157
+ "sqlite-schema": /* @__PURE__ */ new Set(["sqlite_schema"]),
2158
+ "sqlite-tables": /* @__PURE__ */ new Set(["sqlite_tables"]),
2159
+ conflicts: /* @__PURE__ */ new Set(["conflicts"]),
2160
+ brief: /* @__PURE__ */ new Set(["brief_view"]),
2161
+ "session-outline": /* @__PURE__ */ new Set(["session_outline"]),
2162
+ "session-slice": /* @__PURE__ */ new Set(["session_slice"]),
2163
+ "gdrive-sections": /* @__PURE__ */ new Set(["gdrive_sections"]),
2164
+ "pr-slice": /* @__PURE__ */ new Set(["pr_slice"]),
2165
+ "compact-doc": /* @__PURE__ */ new Set(["compact_doc"]),
2166
+ "note-add": /* @__PURE__ */ new Set(["note_write"]),
2167
+ "note-get": /* @__PURE__ */ new Set(["note_read"]),
2168
+ "note-list": /* @__PURE__ */ new Set(["note_list"]),
2169
+ "compress-text": /* @__PURE__ */ new Set(["content_compress"]),
2170
+ retrieve: /* @__PURE__ */ new Set(["content_retrieve"]),
2171
+ "handoff-create": /* @__PURE__ */ new Set(["handoff_create"]),
2172
+ "handoff-resolve": /* @__PURE__ */ new Set(["handoff_resolve"]),
2173
+ npm: /* @__PURE__ */ new Set([
2174
+ "bash_compress:npm_install",
2175
+ "bash_compress:npm_ci",
2176
+ "bash_compress:npm_audit",
2177
+ "bash_compress:npm_ls",
2178
+ "bash_compress:npm_outdated",
2179
+ "bash_compress:npx"
2180
+ ])
2181
+ };
2182
+ var OVERHEAD_SUFFIX = "_overhead";
2183
+ function kindToSource(kind) {
2184
+ const src = KIND_TO_SOURCE[kind];
2185
+ if (src !== void 0) return src;
2186
+ if (kind.endsWith(OVERHEAD_SUFFIX)) {
2187
+ const base = kind.slice(0, -OVERHEAD_SUFFIX.length);
2188
+ const baseSrc = KIND_TO_SOURCE[base];
2189
+ if (baseSrc !== void 0) return baseSrc;
2190
+ }
2191
+ for (const [prefix, prefixSrc] of KIND_PREFIX_TO_SOURCE) {
2192
+ if (kind.startsWith(prefix)) return prefixSrc;
2193
+ }
2194
+ return SOURCE_OTHER;
2195
+ }
2196
+ function toLocalDateKey(d) {
2197
+ const year = d.getFullYear();
2198
+ const month = String(d.getMonth() + 1).padStart(2, "0");
2199
+ const day = String(d.getDate()).padStart(2, "0");
2200
+ return `${year}-${month}-${day}`;
2201
+ }
2202
+ function formatLocalTimestamp(d) {
2203
+ const hours = String(d.getHours()).padStart(2, "0");
2204
+ const minutes = String(d.getMinutes()).padStart(2, "0");
2205
+ const seconds = String(d.getSeconds()).padStart(2, "0");
2206
+ return `${toLocalDateKey(d)}T${hours}:${minutes}:${seconds}`;
2207
+ }
2208
+ function zeroBucket() {
2209
+ return { events: 0, bytes_saved: 0, tokens_saved: 0 };
2210
+ }
2211
+ function incBucket(bucket, bytesSaved, tokensSaved) {
2212
+ bucket.events += 1;
2213
+ bucket.bytes_saved += bytesSaved;
2214
+ bucket.tokens_saved += tokensSaved;
2215
+ }
2216
+ var GLOBAL_SCHEMA_SQL = `
2217
+ CREATE TABLE IF NOT EXISTS stats (
2218
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
2219
+ ts INTEGER NOT NULL,
2220
+ kind TEXT NOT NULL,
2221
+ tokens_saved INTEGER NOT NULL DEFAULT 0,
2222
+ bytes_saved INTEGER NOT NULL DEFAULT 0,
2223
+ detail TEXT,
2224
+ harness TEXT,
2225
+ traceparent TEXT,
2226
+ tg_version TEXT
2227
+ );
2228
+ CREATE INDEX IF NOT EXISTS idx_stats_ts ON stats(ts);
2229
+ CREATE INDEX IF NOT EXISTS idx_stats_kind ON stats(kind);
2230
+ -- Day-granularity rollup of stats rows older than STATS_RETENTION_DAYS -- see
2231
+ -- rollupAndPruneStats's doc comment for why raw rows are pruned instead of kept forever, and
2232
+ -- summarize()'s rollup merge for how a --window-days report spanning pruned days still gets a
2233
+ -- correct total. Deliberately day+kind+harness+tg_version, not day+kind alone: those are exactly
2234
+ -- the dimensions summarize() breaks totals out by, and collapsing any of them here would make an
2235
+ -- old --window-days report under-report a breakdown a recent one still shows in full.
2236
+ CREATE TABLE IF NOT EXISTS stats_daily_rollup (
2237
+ day TEXT NOT NULL,
2238
+ kind TEXT NOT NULL,
2239
+ harness TEXT NOT NULL DEFAULT '',
2240
+ tg_version TEXT NOT NULL DEFAULT '',
2241
+ events INTEGER NOT NULL DEFAULT 0,
2242
+ bytes_saved INTEGER NOT NULL DEFAULT 0,
2243
+ tokens_saved INTEGER NOT NULL DEFAULT 0,
2244
+ PRIMARY KEY (day, kind, harness, tg_version)
2245
+ );
2246
+ CREATE INDEX IF NOT EXISTS idx_stats_daily_rollup_day ON stats_daily_rollup(day);
2247
+ -- Single-row (id=1, same pattern as embedding_provenance in db.ts) throttle so the rollup+prune
2248
+ -- pass -- an aggregate scan over the whole stats table -- runs at most once per
2249
+ -- STATS_ROLLUP_INTERVAL_MS regardless of how often recordStat() itself is called (many times a
2250
+ -- minute during active use).
2251
+ CREATE TABLE IF NOT EXISTS stats_maintenance (
2252
+ id INTEGER PRIMARY KEY CHECK (id = 1),
2253
+ last_rollup_ts INTEGER NOT NULL DEFAULT 0
2254
+ );
2255
+ CREATE TABLE IF NOT EXISTS unmapped_tools (
2256
+ harness TEXT NOT NULL,
2257
+ tool_name TEXT NOT NULL,
2258
+ event_name TEXT NOT NULL,
2259
+ near_miss TEXT,
2260
+ first_seen INTEGER NOT NULL,
2261
+ last_seen INTEGER NOT NULL,
2262
+ hits INTEGER NOT NULL DEFAULT 0,
2263
+ PRIMARY KEY (harness, tool_name, event_name)
2264
+ );
2265
+ `;
2266
+ var _globalSchemaApplied = /* @__PURE__ */ new Set();
2267
+ registerReset(() => _globalSchemaApplied.clear());
2268
+ function migrateGlobalSchema(db) {
2269
+ try {
2270
+ db.exec("ALTER TABLE stats ADD COLUMN harness TEXT");
2271
+ } catch (err) {
2272
+ if (!(err instanceof Error) || !/duplicate column/i.test(err.message)) throw err;
2273
+ }
2274
+ try {
2275
+ db.exec("ALTER TABLE stats ADD COLUMN traceparent TEXT");
2276
+ } catch (err) {
2277
+ if (!(err instanceof Error) || !/duplicate column/i.test(err.message)) throw err;
2278
+ }
2279
+ try {
2280
+ db.exec("ALTER TABLE stats ADD COLUMN tg_version TEXT");
2281
+ } catch (err) {
2282
+ if (!(err instanceof Error) || !/duplicate column/i.test(err.message)) throw err;
2283
+ }
2284
+ }
2285
+ var _harnessColumnByDb = /* @__PURE__ */ new WeakMap();
2286
+ function statsHasHarnessColumn(db) {
2287
+ const cached = _harnessColumnByDb.get(db);
2288
+ if (cached !== void 0) return cached;
2289
+ let present;
2290
+ try {
2291
+ present = db.prepare("PRAGMA table_info(stats)").all().some(
2292
+ (c) => c.name === "harness"
2293
+ );
2294
+ } catch {
2295
+ present = false;
2296
+ }
2297
+ _harnessColumnByDb.set(db, present);
2298
+ return present;
2299
+ }
2300
+ var _traceparentColumnByDb = /* @__PURE__ */ new WeakMap();
2301
+ function statsHasTraceparentColumn(db) {
2302
+ const cached = _traceparentColumnByDb.get(db);
2303
+ if (cached !== void 0) return cached;
2304
+ let present;
2305
+ try {
2306
+ present = db.prepare("PRAGMA table_info(stats)").all().some(
2307
+ (c) => c.name === "traceparent"
2308
+ );
2309
+ } catch {
2310
+ present = false;
2311
+ }
2312
+ _traceparentColumnByDb.set(db, present);
2313
+ return present;
2314
+ }
2315
+ var _versionColumnByDb = /* @__PURE__ */ new WeakMap();
2316
+ function statsHasVersionColumn(db) {
2317
+ const cached = _versionColumnByDb.get(db);
2318
+ if (cached !== void 0) return cached;
2319
+ let present;
2320
+ try {
2321
+ present = db.prepare("PRAGMA table_info(stats)").all().some(
2322
+ (c) => c.name === "tg_version"
2323
+ );
2324
+ } catch {
2325
+ present = false;
2326
+ }
2327
+ _versionColumnByDb.set(db, present);
2328
+ return present;
2329
+ }
2330
+ function getGlobalDb(homeDir) {
2331
+ const basePath = homeDir ? dataDirForHome(homeDir) : dataDir();
2332
+ const dbPath = path3.join(basePath, "global.db");
2333
+ const db = getDb(dbPath);
2334
+ if (!_globalSchemaApplied.has(dbPath)) {
2335
+ db.exec(GLOBAL_SCHEMA_SQL);
2336
+ migrateGlobalSchema(db);
2337
+ _globalSchemaApplied.add(dbPath);
2338
+ }
2339
+ return db;
2340
+ }
2341
+ var STATS_RETENTION_DAYS = 180;
2342
+ var STATS_ROLLUP_INTERVAL_MS = 6 * 60 * 60 * 1e3;
2343
+ function rollupAndPruneStats(db, retentionDays = STATS_RETENTION_DAYS) {
2344
+ try {
2345
+ const cutoffTs = Math.floor(Date.now() / 1e3) - retentionDays * 86400;
2346
+ const run = db.transaction((cutoff) => {
2347
+ db.prepare(
2348
+ `INSERT INTO stats_daily_rollup (day, kind, harness, tg_version, events, bytes_saved, tokens_saved)
2349
+ SELECT strftime('%Y-%m-%d', ts, 'unixepoch', 'localtime') AS day,
2350
+ kind,
2351
+ COALESCE(harness, '') AS harness,
2352
+ COALESCE(tg_version, '') AS tg_version,
2353
+ COUNT(*), SUM(bytes_saved), SUM(tokens_saved)
2354
+ FROM stats
2355
+ WHERE ts < ?
2356
+ GROUP BY day, kind, harness, tg_version
2357
+ ON CONFLICT(day, kind, harness, tg_version) DO UPDATE SET
2358
+ events = events + excluded.events,
2359
+ bytes_saved = bytes_saved + excluded.bytes_saved,
2360
+ tokens_saved = tokens_saved + excluded.tokens_saved`
2361
+ ).run(cutoff);
2362
+ db.prepare(`DELETE FROM stats WHERE ts < ?`).run(cutoff);
2363
+ });
2364
+ run(cutoffTs);
2365
+ } catch {
2366
+ }
2367
+ }
2368
+ function maybeRunStatsMaintenance(db) {
2369
+ try {
2370
+ const now = Date.now();
2371
+ const row = db.prepare(`SELECT last_rollup_ts FROM stats_maintenance WHERE id = 1`).get();
2372
+ const last = row?.last_rollup_ts ?? 0;
2373
+ if (now - last < STATS_ROLLUP_INTERVAL_MS) return;
2374
+ db.prepare(
2375
+ `INSERT INTO stats_maintenance (id, last_rollup_ts) VALUES (1, ?)
2376
+ ON CONFLICT(id) DO UPDATE SET last_rollup_ts = excluded.last_rollup_ts`
2377
+ ).run(now);
2378
+ rollupAndPruneStats(db);
2379
+ } catch {
2380
+ }
2381
+ }
2382
+ function noStatsMessage(windowDays, homeDir) {
2383
+ if (windowDays <= 0) return "No stats recorded yet.";
2384
+ const db = getGlobalDb(homeDir);
2385
+ let total = db.prepare("SELECT COUNT(*) as c FROM stats").get().c;
2386
+ try {
2387
+ const rollupTable = db.prepare(`SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = 'stats_daily_rollup'`).get();
2388
+ if (rollupTable !== void 0) {
2389
+ const rolledUp = db.prepare("SELECT COALESCE(SUM(events), 0) as c FROM stats_daily_rollup").get().c;
2390
+ total += rolledUp;
2391
+ }
2392
+ } catch {
2393
+ }
2394
+ if (total === 0) return "No stats recorded yet.";
2395
+ return `No stats in the last ${countNoun(windowDays, "day")} (${total} recorded outside this window; use --window-days 0 for all time).`;
2396
+ }
2397
+ function recordStat(kind, bytesSaved = 0, tokensSaved = 0, _testDb, detail, traceparent) {
2398
+ try {
2399
+ const db = _testDb ?? getGlobalDb();
2400
+ const ts = Math.floor(Date.now() / 1e3);
2401
+ const tp = traceparent ?? process.env["TRACEPARENT"] ?? process.env["traceparent"] ?? null;
2402
+ const cols = ["ts", "kind", "bytes_saved", "tokens_saved", "detail"];
2403
+ const vals = [ts, kind, bytesSaved, tokensSaved, detail ?? null];
2404
+ if (statsHasHarnessColumn(db)) {
2405
+ cols.push("harness");
2406
+ vals.push(getHarnessName());
2407
+ }
2408
+ if (statsHasTraceparentColumn(db)) {
2409
+ cols.push("traceparent");
2410
+ vals.push(tp);
2411
+ }
2412
+ if (statsHasVersionColumn(db)) {
2413
+ cols.push("tg_version");
2414
+ vals.push(VERSION);
2415
+ }
2416
+ db.prepare(
2417
+ `INSERT INTO stats (${cols.join(", ")}) VALUES (${cols.map(() => "?").join(", ")})`
2418
+ ).run(...vals);
2419
+ maybeRunStatsMaintenance(db);
2420
+ } catch {
2421
+ }
2422
+ }
2423
+ var MAX_TOOL_NAME_CHARS = 200;
2424
+ function recordUnmappedTool(toolName, eventName, nearMiss, _testDb) {
2425
+ try {
2426
+ if (!toolName) return;
2427
+ const db = _testDb ?? getGlobalDb();
2428
+ const now = Math.floor(Date.now() / 1e3);
2429
+ db.prepare(
2430
+ `INSERT INTO unmapped_tools (harness, tool_name, event_name, near_miss, first_seen, last_seen, hits)
2431
+ VALUES (?, ?, ?, ?, ?, ?, 1)
2432
+ ON CONFLICT(harness, tool_name, event_name) DO UPDATE SET
2433
+ hits = hits + 1,
2434
+ last_seen = excluded.last_seen,
2435
+ near_miss = excluded.near_miss`
2436
+ ).run(getHarnessName(), toolName.slice(0, MAX_TOOL_NAME_CHARS), eventName, nearMiss, now, now);
2437
+ } catch {
2438
+ }
2439
+ }
2440
+ function readUnmappedTools(dbPath, homeDir) {
2441
+ try {
2442
+ const db = dbPath ? getDb(dbPath) : getGlobalDb(homeDir);
2443
+ return db.prepare(
2444
+ "SELECT harness, tool_name, event_name, near_miss, hits, last_seen FROM unmapped_tools ORDER BY hits DESC, tool_name ASC"
2445
+ ).all();
2446
+ } catch {
2447
+ return [];
2448
+ }
2449
+ }
2450
+ function summarize(windowDays = 30, testDb, homeDir) {
2451
+ const t0 = Date.now();
2452
+ const sinceTs = windowDays > 0 ? Math.floor((Date.now() - windowDays * 24 * 60 * 60 * 1e3) / 1e3) : null;
2453
+ const byKind = {};
2454
+ const byDay = {};
2455
+ const byHarness = {};
2456
+ const byPricingVersion = {};
2457
+ let totalEvents = 0;
2458
+ let totalBytes = 0;
2459
+ let totalTokens = 0;
2460
+ const db = testDb ?? getGlobalDb(homeDir);
2461
+ const hasHarness = statsHasHarnessColumn(db);
2462
+ const hasVersion = statsHasVersionColumn(db);
2463
+ const cols = [
2464
+ "ts",
2465
+ "kind",
2466
+ "bytes_saved",
2467
+ "tokens_saved",
2468
+ ...hasHarness ? ["harness"] : [],
2469
+ ...hasVersion ? ["tg_version"] : []
2470
+ ].join(", ");
2471
+ const query = sinceTs !== null ? `SELECT ${cols} FROM stats WHERE ts >= ? ORDER BY ts DESC` : `SELECT ${cols} FROM stats ORDER BY ts DESC`;
2472
+ const stmt = db.prepare(query);
2473
+ const rows2 = sinceTs !== null ? stmt.all(sinceTs) : stmt.all();
2474
+ const tsToDateCache = {};
2475
+ const counts = {};
2476
+ for (const row of rows2) {
2477
+ const bytesSaved = row.bytes_saved ?? 0;
2478
+ const recorded = row.tokens_saved ?? 0;
2479
+ const kind = row.kind;
2480
+ const isCount = COUNT_ONLY_KINDS.has(kind);
2481
+ if (isCount) counts[kind] = (counts[kind] ?? 0) + recorded;
2482
+ const tokensSaved = isCount ? 0 : recorded;
2483
+ const tsRaw = row.ts;
2484
+ if (tsRaw === void 0) continue;
2485
+ const ts = tsRaw;
2486
+ totalEvents += 1;
2487
+ totalBytes += bytesSaved;
2488
+ totalTokens += tokensSaved;
2489
+ if (!byKind[kind]) {
2490
+ byKind[kind] = zeroBucket();
2491
+ }
2492
+ incBucket(byKind[kind], bytesSaved, tokensSaved);
2493
+ const dateKey = tsToDateCache[ts] || toLocalDateKey(new Date(ts * 1e3));
2494
+ tsToDateCache[ts] = dateKey;
2495
+ if (!byDay[dateKey]) {
2496
+ byDay[dateKey] = zeroBucket();
2497
+ }
2498
+ incBucket(byDay[dateKey], bytesSaved, tokensSaved);
2499
+ const harness = row.harness || HARNESS_UNRECORDED;
2500
+ if (!byHarness[harness]) {
2501
+ byHarness[harness] = zeroBucket();
2502
+ }
2503
+ incBucket(byHarness[harness], bytesSaved, tokensSaved);
2504
+ const pricingVersion = row.tg_version || PRICING_VERSION_UNRECORDED;
2505
+ if (!byPricingVersion[pricingVersion]) {
2506
+ byPricingVersion[pricingVersion] = zeroBucket();
2507
+ }
2508
+ incBucket(byPricingVersion[pricingVersion], bytesSaved, tokensSaved);
2509
+ }
2510
+ try {
2511
+ const rollupTable = db.prepare(`SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = 'stats_daily_rollup'`).get();
2512
+ if (rollupTable !== void 0) {
2513
+ const sinceDayKey = sinceTs !== null ? toLocalDateKey(new Date(sinceTs * 1e3)) : null;
2514
+ const rollupRows = sinceDayKey !== null ? db.prepare(`SELECT day, kind, harness, tg_version, events, bytes_saved, tokens_saved FROM stats_daily_rollup WHERE day >= ?`).all(sinceDayKey) : db.prepare(`SELECT day, kind, harness, tg_version, events, bytes_saved, tokens_saved FROM stats_daily_rollup`).all();
2515
+ const addBucket = (bucket, events, bytesSaved, tokensSaved) => {
2516
+ bucket.events += events;
2517
+ bucket.bytes_saved += bytesSaved;
2518
+ bucket.tokens_saved += tokensSaved;
2519
+ };
2520
+ for (const r of rollupRows) {
2521
+ const isCount = COUNT_ONLY_KINDS.has(r.kind);
2522
+ if (isCount) counts[r.kind] = (counts[r.kind] ?? 0) + r.tokens_saved;
2523
+ const tokensSaved = isCount ? 0 : r.tokens_saved;
2524
+ const bytesSaved = r.bytes_saved;
2525
+ totalEvents += r.events;
2526
+ totalBytes += bytesSaved;
2527
+ totalTokens += tokensSaved;
2528
+ if (!byKind[r.kind]) byKind[r.kind] = zeroBucket();
2529
+ addBucket(byKind[r.kind], r.events, bytesSaved, tokensSaved);
2530
+ if (!byDay[r.day]) byDay[r.day] = zeroBucket();
2531
+ addBucket(byDay[r.day], r.events, bytesSaved, tokensSaved);
2532
+ const harness = r.harness || HARNESS_UNRECORDED;
2533
+ if (!byHarness[harness]) byHarness[harness] = zeroBucket();
2534
+ addBucket(byHarness[harness], r.events, bytesSaved, tokensSaved);
2535
+ const pricingVersion = r.tg_version || PRICING_VERSION_UNRECORDED;
2536
+ if (!byPricingVersion[pricingVersion]) byPricingVersion[pricingVersion] = zeroBucket();
2537
+ addBucket(byPricingVersion[pricingVersion], r.events, bytesSaved, tokensSaved);
2538
+ }
2539
+ }
2540
+ } catch {
2541
+ }
2542
+ const bySourceDict = {};
2543
+ for (const [kind, bucket] of Object.entries(byKind)) {
2544
+ const source = kindToSource(kind);
2545
+ if (!bySourceDict[source]) {
2546
+ bySourceDict[source] = zeroBucket();
2547
+ }
2548
+ bySourceDict[source].events += bucket.events;
2549
+ bySourceDict[source].bytes_saved += bucket.bytes_saved;
2550
+ bySourceDict[source].tokens_saved += bucket.tokens_saved;
2551
+ }
2552
+ const byCommandDict = {};
2553
+ for (const [cmd, kinds] of Object.entries(COMMAND_KINDS)) {
2554
+ byCommandDict[cmd] = zeroBucket();
2555
+ for (const kind of kinds) {
2556
+ if (byKind[kind]) {
2557
+ byCommandDict[cmd].events += byKind[kind].events;
2558
+ byCommandDict[cmd].bytes_saved += byKind[kind].bytes_saved;
2559
+ byCommandDict[cmd].tokens_saved += byKind[kind].tokens_saved;
2560
+ }
2561
+ }
2562
+ }
2563
+ const byDayList = Object.entries(byDay).map(([date, bucket]) => ({ ...bucket, date })).sort((a, b) => new Date(b.date).getTime() - new Date(a.date).getTime());
2564
+ const byProjectList = [];
2565
+ const t1 = Date.now();
2566
+ if (t1 - t0 > 1e3) {
2567
+ console.warn(`summarize took ${t1 - t0}ms (window=${windowDays}d, total=${totalEvents} rows)`);
2568
+ }
2569
+ return {
2570
+ total_events: totalEvents,
2571
+ total_bytes_saved: totalBytes,
2572
+ total_tokens_saved: totalTokens,
2573
+ by_kind: byKind,
2574
+ by_day: byDayList,
2575
+ by_project: byProjectList,
2576
+ by_source: bySourceDict,
2577
+ by_harness: byHarness,
2578
+ by_pricing_version: byPricingVersion,
2579
+ counts,
2580
+ by_command: Object.entries(byCommandDict).map(([command, bucket]) => ({ ...bucket, command })).filter((r) => r.events > 0),
2581
+ window_days: windowDays
2582
+ };
2583
+ }
2584
+ function _totalsLines(summary) {
2585
+ return [
2586
+ "# token-goat stats",
2587
+ `Total events: ${summary.total_events}`,
2588
+ `Bytes saved: ${fmtBytes(summary.total_bytes_saved)}`,
2589
+ `Tokens saved: ${summary.total_tokens_saved}`,
2590
+ // Disclosure, not a correction: `total_tokens_saved` above sums rows written under whichever
2591
+ // pricing formula was live when each was recorded, and `tg_version` cannot be read back into
2592
+ // "which formula" for a row from before this disclosure existed (PRICING_VERSION_UNRECORDED is
2593
+ // the overwhelming majority of all-time rows). Excluding those rows from the headline would
2594
+ // discard nearly the whole figure rather than fix it, so the honest move is to keep the sum and
2595
+ // say plainly that it spans more than one era, not to quietly present a mixed total as single-formula.
2596
+ ...hasMixedPricingEras(summary) ? [
2597
+ `Pricing note: totals mix ${countNoun(Object.keys(summary.by_pricing_version).length, "tg_version era")} (${countNoun(summary.by_pricing_version[PRICING_VERSION_UNRECORDED]?.events ?? 0, "row")} unrecorded); see 'token-goat stats --json' -> by_pricing_version for the breakdown`
2598
+ ] : [],
2599
+ // Printed on its own line, below the token total and never inside it, because it counts
2600
+ // placeholders rather than tokens. Omitted entirely when nothing was redacted, so the line is
2601
+ // information rather than a permanent zero. See COUNT_ONLY_KINDS.
2602
+ ...summary.counts["secret_redacted"] ? [`Secrets hidden: ${summary.counts["secret_redacted"]} (a count, not tokens)`] : [],
2603
+ `Window: ${summary.window_days} days`
2604
+ ];
2605
+ }
2606
+ function _useRichStats() {
2607
+ if (process.env["NO_COLOR"]) return false;
2608
+ if (process.stdout.isTTY === true) return true;
2609
+ return process.stdout.isTTY === void 0 && !process.env["CI"];
2610
+ }
2611
+ function _renderShortTotals(summary) {
2612
+ const lines = [
2613
+ ..._totalsLines(summary),
2614
+ "",
2615
+ "Run 'token-goat stats --full' for the full breakdown (by source, by command, by day)."
2616
+ ];
2617
+ console.log(lines.join("\n"));
2618
+ }
2619
+ function _plainTextStats(summary) {
2620
+ const lines = _totalsLines(summary);
2621
+ if (Object.keys(summary.by_source).length > 0) {
2622
+ lines.push("", "## By Source");
2623
+ const sources = Object.entries(summary.by_source).filter(([, b]) => b.events > 0).sort((a, b) => b[1].tokens_saved - a[1].tokens_saved);
2624
+ for (const [source, bucket] of sources) {
2625
+ lines.push(
2626
+ ` ${source.padEnd(8)} ${bucket.events.toString().padStart(6)} events ${fmtBytes(bucket.bytes_saved).padStart(8)} ${bucket.tokens_saved.toString().padStart(8)} tokens`
2627
+ );
2628
+ }
2629
+ }
2630
+ const harnesses = Object.entries(summary.by_harness).filter(([, b]) => b.events > 0).sort((a, b) => b[1].tokens_saved - a[1].tokens_saved);
2631
+ if (harnesses.length > 1) {
2632
+ lines.push("", "## By Harness");
2633
+ for (const [harness, bucket] of harnesses) {
2634
+ lines.push(
2635
+ ` ${harness.padEnd(22)} ${bucket.events.toString().padStart(6)} events ${fmtBytes(bucket.bytes_saved).padStart(8)} ${bucket.tokens_saved.toString().padStart(8)} tokens`
2636
+ );
2637
+ }
2638
+ }
2639
+ if (summary.by_command.length > 0) {
2640
+ lines.push("", "## By Command");
2641
+ for (const row of summary.by_command) {
2642
+ lines.push(
2643
+ ` ${row.command.padEnd(12)} ${row.events.toString().padStart(6)} events ${fmtBytes(row.bytes_saved).padStart(8)} ${row.tokens_saved.toString().padStart(8)} tokens`
2644
+ );
2645
+ }
2646
+ } else {
2647
+ const hintBucket = summary.by_source[SOURCE_HINT];
2648
+ if (hintBucket && hintBucket.events > 0) {
2649
+ lines.push(
2650
+ "",
2651
+ "## By Command",
2652
+ ` 0 direct command invocations this window -- ${hintBucket.events} hint(s) fired but not acted on.`,
2653
+ " Run token-goat symbol/read/section/semantic/outline/skeleton directly to capture these savings."
2654
+ );
2655
+ }
2656
+ }
2657
+ if (summary.by_day.length > 0) {
2658
+ lines.push("", "## Last 7 Days");
2659
+ for (const row of summary.by_day.slice(0, 7)) {
2660
+ lines.push(
2661
+ ` ${row.date} ${row.events.toString().padStart(6)} events ${fmtBytes(row.bytes_saved).padStart(8)} ${row.tokens_saved.toString().padStart(8)} tokens`
2662
+ );
2663
+ }
2664
+ }
2665
+ console.log(lines.join("\n"));
2666
+ }
2667
+ function _buildStatsData(summary, windowDays) {
2668
+ const now = /* @__PURE__ */ new Date();
2669
+ const periodStart = windowDays > 0 ? new Date(now.getTime() - windowDays * 24 * 60 * 60 * 1e3) : /* @__PURE__ */ new Date(0);
2670
+ const sparkDays = [...summary.by_day].reverse().slice(-30);
2671
+ const sparklines = sparkDays.length > 1 ? {
2672
+ events: sparkDays.map((d) => d.events),
2673
+ bytes: sparkDays.map((d) => d.bytes_saved),
2674
+ tokens: sparkDays.map((d) => d.tokens_saved)
2675
+ } : null;
2676
+ return {
2677
+ period_start: periodStart,
2678
+ period_end: now,
2679
+ version: VERSION,
2680
+ window_label: windowDays > 0 ? `last ${countNoun(windowDays, "day")}` : "all time",
2681
+ totals: {
2682
+ events: summary.total_events,
2683
+ bytes: summary.total_bytes_saved,
2684
+ tokens: summary.total_tokens_saved,
2685
+ sparklines
2686
+ },
2687
+ by_kind: Object.entries(summary.by_kind).map(([kind, bucket]) => ({
2688
+ kind,
2689
+ bytes: bucket.bytes_saved,
2690
+ tokens: bucket.tokens_saved,
2691
+ events: bucket.events,
2692
+ bytes_mode_only: _BYTES_MODE_ONLY_KINDS.has(kind)
2693
+ })).sort((a, b) => b.bytes - a.bytes),
2694
+ by_day: summary.by_day.map((d) => ({
2695
+ date: d.date,
2696
+ bytes: d.bytes_saved,
2697
+ tokens: d.tokens_saved,
2698
+ events: d.events
2699
+ })),
2700
+ by_project: [],
2701
+ by_source: Object.entries(summary.by_source).filter(([, b]) => b.events > 0).map(([source, bucket]) => ({
2702
+ source,
2703
+ bytes: bucket.bytes_saved,
2704
+ tokens: bucket.tokens_saved,
2705
+ events: bucket.events
2706
+ })).sort((a, b) => b.bytes - a.bytes),
2707
+ by_command: summary.by_command.map((c) => ({
2708
+ command: c.command,
2709
+ bytes: c.bytes_saved,
2710
+ tokens: c.tokens_saved,
2711
+ events: c.events
2712
+ })),
2713
+ by_harness: Object.entries(summary.by_harness).filter(([, b]) => b.events > 0).map(([harness, bucket]) => ({
2714
+ harness,
2715
+ bytes: bucket.bytes_saved,
2716
+ tokens: bucket.tokens_saved,
2717
+ events: bucket.events
2718
+ })).sort((a, b) => b.bytes - a.bytes)
2719
+ };
2720
+ }
2721
+ function renderShortStats(opts) {
2722
+ const windowDays = opts?.windowDays ?? 30;
2723
+ const summary = summarize(windowDays, void 0, opts?.homeDir);
2724
+ if (summary.total_events === 0) {
2725
+ console.log(noStatsMessage(windowDays, opts?.homeDir));
2726
+ return;
2727
+ }
2728
+ const useTty = process.env["NO_COLOR"] ? false : opts?.force === true ? true : _useRichStats();
2729
+ if (!useTty) {
2730
+ _renderShortTotals(summary);
2731
+ return;
2732
+ }
2733
+ const statsData = _buildStatsData(summary, windowDays);
2734
+ process.stdout.write(renderStats(statsData, { short: true }) + "\n");
2735
+ }
2736
+ function renderStats2(opts) {
2737
+ const windowDays = opts?.windowDays ?? 30;
2738
+ const summary = summarize(windowDays, void 0, opts?.homeDir);
2739
+ if (summary.total_events === 0) {
2740
+ console.log(noStatsMessage(windowDays, opts?.homeDir));
2741
+ return;
2742
+ }
2743
+ const useTty = _useRichStats();
2744
+ if (!useTty) {
2745
+ _plainTextStats(summary);
2746
+ return;
2747
+ }
2748
+ const statsData = _buildStatsData(summary, windowDays);
2749
+ process.stdout.write(renderStats(statsData) + "\n");
2750
+ }
2751
+
2752
+ // src/secret_redact.ts
2753
+ init_define_import_meta_env();
2754
+ var SECRET_PATTERNS = [
2755
+ // Anthropic keys share OpenAI's "sk-" prefix but are more specific ("sk-ant-"), so they're matched first — the generic OpenAI pattern's negative lookahead below is defense in depth, not the sole guard.
2756
+ ["anthropic_api_key", /sk-ant-[A-Za-z0-9_-]{20,}/g],
2757
+ // Modern (post-2024, now the default) OpenAI project keys: "sk-proj-" followed by a long base64url-ish body that legitimately contains '-' and '_' -- the generic pattern below deliberately excludes those characters (hyphenated prose would false-fire), so this needs its own entry with the more specific prefix matched first, like sk-ant- above.
2758
+ ["openai_project_key", /sk-proj-[A-Za-z0-9_-]{20,}/g],
2759
+ ["openai_api_key", /sk-(?!ant-|proj-)[A-Za-z0-9]{20,}/g],
2760
+ ["aws_access_key", /AKIA[0-9A-Z]{16}/g],
2761
+ // Fine-grained PATs ("github_pat_...") are matched before the classic gh[oprsu]_ pattern so the full token is always consumed as one match -- a fine-grained token's own body can contain '_' and could otherwise partially match the classic pattern.
2762
+ ["github_token", /github_pat_[A-Za-z0-9_]{22,}/g],
2763
+ ["github_token", /gh[oprsu]_[A-Za-z0-9]{36,}/g],
2764
+ // xapp- is Slack's app-level token (Socket Mode), a bearer credential in its own right that the xox[baprs]- prefix does not cover. It gets its own entry rather than joining the alternation above because it needs the full segmented shape (xapp-<ver>-<app id>-<digits>- <hex>) to be safe: a bare /xapp-[A-Za-z0-9-]+/ redacted ordinary identifiers like "xapp-config" and the css class "xapp-container", mangling normal source. The xox* prefixes are distinctive enough on their own; "xapp-" is not.
2765
+ ["slack_token", /xox[baprs]-[A-Za-z0-9-]+/g],
2766
+ ["slack_token", /xapp-\d-[A-Za-z0-9]{6,}-\d{8,}-[A-Za-z0-9]{16,}/g],
2767
+ // Matches the full block (BEGIN marker through its matching END marker), not just the header -- the actual secret material is the base64 body between them, so redacting only the header line would leave the key bytes themselves fully readable in the cached blob. The body is lazy and additionally refuses to cross a following BEGIN marker. Both halves matter. The laziness bounds a successful match at the very next END marker; the negative lookahead bounds a *failing* one, because a BEGIN with no END of its own would otherwise scan to end-of-input, and a blob full of such markers made the pass quadratic -- 6 ms, 20 ms and 78 ms for 2000, 4000 and 8000 of them, four times the work for twice the input, the same shape as the lookbehind incident described above. Private key blocks do not nest, so refusing to cross a BEGIN costs nothing in correctness. The algorithm and ` BLOCK` groups are backreferenced in the END marker rather than repeated, so a BEGIN only ever pairs with its own END spelling. Written as two independent optional groups, `BEGIN PGP PRIVATE KEY` would happily close on a distant `END RSA PRIVATE KEY BLOCK` and redact everything in between. A non-participating group backreferences as the empty string in JS, which is exactly what the unprefixed `BEGIN PRIVATE KEY` form needs. The algorithm list covers every armored private-key header openssl, ssh-keygen and gpg actually emit, not just the ones a first draft happened to think of. ENCRYPTED is the PKCS#8 passphrase-protected form (`openssl genpkey -aes256`, `ssh-keygen -m PKCS8`) and is the most common shape of all; DSA is legacy but still written verbatim; PGP carries the ` BLOCK` suffix, which is why that suffix is optional here. All three used to fall through to disk in full. An encrypted key is still key material: the passphrase can be attacked offline once the bytes are cached, so it is redacted like any other. PuTTY `.ppk` files are deliberately not matched. They have no END marker -- the private section is a `Private-Lines: N` count followed by exactly N base64 lines -- so bounding a match would take a stateful parse rather than a regex, on a path that runs over every command output. A regex guess at where the body ends is exactly the over-eager match this module's header warns against.
2768
+ ["private_key_block", /-----BEGIN (RSA |DSA |EC |OPENSSH |ENCRYPTED |PGP )?PRIVATE KEY( BLOCK)?-----(?:(?!-----BEGIN )[\s\S])*?-----END \1PRIVATE KEY\2-----/g],
2769
+ // Redacts only the token itself, not the "Authorization: Bearer " prefix -- the lookbehind anchors on the header name and scheme so the surrounding request-log line stays readable, matching how AWS_ACCESS_KEY_ID=... above keeps its own prefix intact. The optional quotes on either side of the colon are what let this see a header carried in JSON rather than in raw wire format. Without them the lookbehind demanded the colon sit directly against the header name and the scheme directly against the space, so a body like {"Authorization": "Bearer <token>"} -- the shape any logged fetch or MCP result arrives in -- matched nothing and the token was cached verbatim. `token` is the second scheme spelling in wide use -- it is what curl and gh examples pass for GitHub and many other APIs -- and an opaque value behind it carries exactly the same authority as one behind `Bearer`. The trailing gap is `{1,8}` rather than a single space for the same reason every other gap in this lookbehind already is: a hand-aligned or reformatted header ("Authorization: Bearer <token>") is ordinary, and demanding exactly one space there made this the one position in the pattern that a second space defeated.
2770
+ ["auth_bearer_token", /(?<=Authorization["']?[ \t]{0,8}:[ \t]{0,8}["']?[ \t]{0,8}(?:Bearer|token)[ \t]{1,8})[A-Za-z0-9\-._~+/]{10,}=*/gi],
2771
+ ["auth_basic_token", /(?<=Authorization["']?[ \t]{0,8}:[ \t]{0,8}["']?[ \t]{0,8}Basic[ \t])[A-Za-z0-9+/]{6,}=*/gi],
2772
+ // JWTs have no distinctive prefix of their own, but the base64url encoding of the smallest realistic header ('{"alg":' or similar) always starts with "eyJ", so that's the practical anchor here -- each of the three dot-separated segments requires a minimum length to avoid matching a short, coincidentally dotted token.
2773
+ //
2774
+ // The trailing `={0,2}` on each segment is what makes a padded token match. Base64url as the JWT spec defines it drops the `=` padding, but producers that reach for a plain base64 encoder emit it anyway, and a `=` in the header or payload segment used to defeat the match outright: not a partial redaction, but none at all, so the entire token was printed. `=` cannot appear anywhere except the end of a segment, since it is not one of the characters the segment body allows, so accepting it here cannot widen the match onto anything else.
2775
+ ["jwt", /eyJ[A-Za-z0-9_-]{10,}={0,2}\.[A-Za-z0-9_-]{10,}={0,2}\.[A-Za-z0-9_-]{10,}={0,2}/g],
2776
+ ["npm_token", /npm_[A-Za-z0-9]{36}/g],
2777
+ // rk_live_ (restricted keys) share the sk_live_/sk_test_ secret-key shape and risk level, so one pattern covers all three rather than adding a near-duplicate entry.
2778
+ ["stripe_key", /(?:sk_live_|sk_test_|rk_live_)[A-Za-z0-9]{20,}/g],
2779
+ ["google_api_key", /AIza[A-Za-z0-9_-]{35}/g],
2780
+ // Presigned-url signatures: AWS SigV4 (X-Amz-Signature), Google Cloud Storage (X-Goog-Signature), and Azure blob SAS (sig). These are bearer credentials in query-string clothing -- anyone holding the whole url can read or write the object until it expires, and none of the prefix-anchored patterns above match them because the signature is a bare hex or base64 blob with no distinctive prefix of its own. The '[?&]' anchor and the 16-char floor are what make the short, generic 'sig' name safe to key on: a prose or code mention of "sig" never sits directly after a query separator followed by that much opaque token.
2781
+ ["presigned_signature", /(?<=[?&](?:X-Amz-Signature|X-Goog-Signature|sig)=)[A-Za-z0-9%+/=_-]{16,}/gi],
2782
+ // A password inside a connection url. `postgres://user:hunter2@db.internal` carries the credential in the authority section, where there is no `key=value` separator for the generic pattern below to anchor on, so a DATABASE_URL echoed by a failing migration or a psql error went through untouched. The anchors are what keep this narrow: a scheme's `://`, a userinfo segment, the colon, and a following `@`. `http://host:8080/path` has no `@` and is left alone; so is any url without credentials.
2783
+ ["url_credentials", /(?<=:\/\/[^\s:@/]{1,64}:)[^\s:@/]{1,256}(?=@)/g],
2784
+ // Azure storage account connection strings (`DefaultEndpointsProtocol=https;AccountName=...; AccountKey=<base64>==;EndpointSuffix=core.windows.net`) carry a full read/write key to the account in the AccountKey field, and none of the generic patterns above catch it: generic_secret_assignment's keyword list (password|passwd|secret|api[_-]?key| access[_-]?token|refresh[_-]?token|id[_-]?token) has nothing that matches "AccountKey", and an unanchored base64-shape pattern was deliberately rejected -- this module's header already warns that a bare high-entropy-blob heuristic false-fires on ordinary code, JSON and log output, and an 88-char base64 value is exactly the shape a hash, a compiled asset digest or a generated id can also take. Anchoring on the literal `AccountKey=` field name instead keeps the match specific to this one connection-string field. The value class is base64 proper (letters, digits, `+`, `/`, trailing `=` padding) and none of those characters include `;`, so the match terminates on its own at the `;` that starts the next `Name=` field -- unlike generic_secret_assignment's separator characters (`& ; # , :`), which double as ordinary credential characters and need a lookahead to tell the two roles apart, `;` is never valid base64 and needs no such lookahead here. The lookbehind keeps `AccountKey=` itself in the output, matching auth_bearer_token and presigned_signature above, so `;EndpointSuffix=core.windows.net` after it stays fully readable too. `SharedAccessKey` is the same credential one Azure service over: Service Bus, Event Hubs and Relay spell it that way (`Endpoint=sb://ns.servicebus.windows.net/;SharedAccessKeyName=Root; SharedAccessKey=<base64>`) and it carries the same authority over that namespace that AccountKey does over a storage account. `SharedAccessKeyName` is a plain identifier rather than a secret, and never matches: the separator in the lookbehind sits directly against the key name, so the `Name` in between stops it dead. The separator is spelled the way auth_bearer_token above spells its own, for the same reasons that comment records having learned the hard way. Optional quotes on either side of it, so the JSON and YAML forms a logged MCP result or api response actually arrives in are matched rather than stopped dead at the opening quote. A bounded gap rather than exactly one space, so a hand-aligned or reformatted `AccountKey = ...` in an appsettings file is not the single variant that defeats the whole pattern. Case-insensitive for the same reason presigned_signature is. The leading word boundary is what keeps the widened name from reaching into the middle of a longer identifier such as `myaccountkey=`.
2785
+ ["azure_storage_key", /(?<=\b(?:AccountKey|SharedAccessKey)["']?[ \t]{0,8}[:=][ \t]{0,8}["']?)[A-Za-z0-9+/]{40,}=*/gi],
2786
+ // Generic key=value assignments in .env-file and connection-string/query-string shape. The lookbehind again redacts only the value, and the value's character class deliberately excludes whitespace, '&', ';', '#', quote characters, and '[' ']' ':' -- that exclusion is what stops this from swallowing the rest of the line (a trailing comment or the next key=value pair) or the remainder of a query string past the matched parameter, which is exactly the kind of over-eager match this module's own design note above warns broad heuristics produce. The '[' ']' ':' exclusion also matters because this pattern runs last: an earlier pattern's own "OPENAI_API_KEY=[REDACTED:openai_project_key]" replacement text contains "API_KEY=" too, and without excluding those characters this pattern would re-match and double-redact its own placeholder. The length has a lower bound only (no upper bound): capping it at 64 used to leave the tail of any longer secret unredacted in plain text, which is worse than no redaction because it looks handled. A single negated-class quantifier like this cannot backtrack catastrophically -- there is no nested or overlapping quantifier for the engine to explore multiple ways of matching, so removing the upper bound does not introduce a ReDoS risk. Quotes are permitted around the separator, but never inside the value class. A quoted value is the ordinary way secrets are written -- .env files, JSON, YAML, TOML all quote by default -- and the lookbehind used to stop dead at the opening quote, so `API_KEY="..."` passed through in full while the bare `API_KEY=...` was caught. The closing quote of the key name blocked it from the other side too, which is what kept every JSON body unredacted. Keeping quotes out of the value class is still what stops the match running past the closing quote. The keyword may be a prefix of a longer key name rather than the whole of it, so the trailing identifier class below is load-bearing: without it the lookbehind required the keyword to sit immediately before the separator, and AWS_SECRET_ACCESS_KEY=, SECRET_KEY=, and DB_PASSWORD_HASH= all passed through in full. That class matches identifier characters only, so prose that merely mentions a keyword still never reaches a separator and stays unredacted. api[_-]?key covers the apikey and api-key spellings too. `& ; # , :` play two incompatible roles. They separate one field from the next (a query string, a cookie header, an inline env list), and they are also perfectly ordinary credential characters. Rejecting them outright got the first role right and the second badly wrong: the match stopped at the first one and left everything after it in plain text, so `password=corr&horse&battery` redacted four characters and printed the rest, and `DB_PASSWORD=Aa1:xyz` matched nothing at all because the run before the `:` was under the four-character floor. A tail left sitting in the open is the outcome this module's header calls worse than no redaction, because it reads as handled.
2787
+ //
2788
+ // So the separator role is decided by what follows rather than assumed: one of these characters ends the value only when the next thing along is another `name=` / `name:` pair, which is what an actual field separator is always followed by. `,OTHER=public` and `; other=1` still end it; the `&` in the middle of a passphrase does not. Whitespace, quotes and brackets are unchanged -- they end a value unconditionally, which is also what keeps this pattern from re-matching the `[REDACTED:...]` placeholder it just wrote.
2789
+ ["generic_secret_assignment", /(?<=(?:password|passwd|secret|api[_-]?key|access[_-]?token|refresh[_-]?token|id[_-]?token)[a-z0-9_-]{0,64}["']?[ \t]{0,8}[:=][ \t]{0,8}["']?)(?:\\[^\n]|[^\s\\&;#,:'"[\]{}]|[&;#,:](?![ \t]*[A-Za-z_][A-Za-z0-9_.-]*[ \t]*[:=])){4,}/gi]
2790
+ ];
2791
+ function countRedactionPlaceholders(text) {
2792
+ return text.match(/\[REDACTED:[a-z0-9_]+\]/g)?.length ?? 0;
2793
+ }
2794
+ var MAX_CUSTOM_PATTERNS = 64;
2795
+ var MAX_CUSTOM_PATTERN_LENGTH = 512;
2796
+ var customCache = null;
2797
+ function compileCustomPatterns(sources) {
2798
+ const key = JSON.stringify(sources);
2799
+ if (customCache !== null && customCache.key === key) {
2800
+ return { patterns: customCache.patterns, problems: customCache.problems };
2801
+ }
2802
+ const patterns = [];
2803
+ const problems = [];
2804
+ const written = sources.map((raw) => raw.trim()).filter((source) => source.length > 0);
2805
+ for (const source of written.slice(0, MAX_CUSTOM_PATTERNS)) {
2806
+ if (source.length > MAX_CUSTOM_PATTERN_LENGTH) {
2807
+ problems.push({ pattern: source.slice(0, 60), reason: `longer than ${MAX_CUSTOM_PATTERN_LENGTH} characters` });
2808
+ continue;
2809
+ }
2810
+ if (hasNestedQuantifier(source)) {
2811
+ problems.push({
2812
+ pattern: source,
2813
+ reason: "repeats a group that already repeats, which can take exponential time to match"
2814
+ });
2815
+ continue;
2816
+ }
2817
+ let compiled;
2818
+ try {
2819
+ compiled = new RegExp(source, "g");
2820
+ } catch (e) {
2821
+ problems.push({ pattern: source, reason: e instanceof Error ? e.message : "is not a valid regular expression" });
2822
+ continue;
2823
+ }
2824
+ if (new RegExp(source).test("")) {
2825
+ problems.push({
2826
+ pattern: source,
2827
+ reason: "matches the empty string, so it would replace every position in the text"
2828
+ });
2829
+ continue;
2830
+ }
2831
+ if (growsExponentially(compiled)) {
2832
+ problems.push({
2833
+ pattern: source,
2834
+ reason: "its running time doubles as the text grows, which can stall the process on a short input"
2835
+ });
2836
+ continue;
2837
+ }
2838
+ patterns.push(compiled);
2839
+ }
2840
+ if (written.length > MAX_CUSTOM_PATTERNS) {
2841
+ problems.push({
2842
+ pattern: `(${written.length} entries)`,
2843
+ reason: `only the first ${MAX_CUSTOM_PATTERNS} custom patterns are used`
2844
+ });
2845
+ }
2846
+ customCache = { key, patterns, problems };
2847
+ return { patterns, problems };
2848
+ }
2849
+ var STRICT_CANDIDATE = /[A-Za-z0-9+/_-]{24,}={0,2}/g;
2850
+ function entropyBitsPerChar(s) {
2851
+ const counts = /* @__PURE__ */ new Map();
2852
+ for (const ch of s) counts.set(ch, (counts.get(ch) ?? 0) + 1);
2853
+ let bits = 0;
2854
+ for (const n of counts.values()) {
2855
+ const p = n / s.length;
2856
+ bits -= p * Math.log2(p);
2857
+ }
2858
+ return bits;
2859
+ }
2860
+ function characterClasses(s) {
2861
+ let n = 0;
2862
+ if (/[a-z]/.test(s)) n++;
2863
+ if (/[A-Z]/.test(s)) n++;
2864
+ if (/[0-9]/.test(s)) n++;
2865
+ if (/[+/=_-]/.test(s)) n++;
2866
+ return n;
2867
+ }
2868
+ function looksLikeCredential(s) {
2869
+ return characterClasses(s) >= 3 && entropyBitsPerChar(s) >= 3.5;
2870
+ }
2871
+ function redactSecrets(text, config = loadConfig()) {
2872
+ let count = 0;
2873
+ let out = text;
2874
+ for (const [kind, pattern] of SECRET_PATTERNS) {
2875
+ out = out.replace(pattern, () => {
2876
+ count++;
2877
+ return `[REDACTED:${kind}]`;
2878
+ });
2879
+ }
2880
+ for (const pattern of compileCustomPatterns(config.redaction.custom_patterns).patterns) {
2881
+ out = out.replace(pattern, () => {
2882
+ count++;
2883
+ return "[REDACTED:custom]";
2884
+ });
2885
+ }
2886
+ if (config.redaction.strict) {
2887
+ out = out.replace(STRICT_CANDIDATE, (candidate) => {
2888
+ if (!looksLikeCredential(candidate)) return candidate;
2889
+ count++;
2890
+ return "[REDACTED:high_entropy]";
2891
+ });
2892
+ }
2893
+ return { text: out, count };
2894
+ }
2895
+
2896
+ // src/disk_cache.ts
2897
+ init_define_import_meta_env();
2898
+ import * as fs4 from "node:fs";
2899
+ import * as path4 from "node:path";
2900
+ var DEFAULT_MAX_COUNT = 200;
2901
+ var DEFAULT_MAX_AGE_MS = 24 * 3600 * 1e3;
2902
+ function sanitizeId(id) {
2903
+ return sanitizeIdForFilename(id, 64);
2904
+ }
2905
+ function blobDir(subdir) {
2906
+ return path4.join(tokenGoatHome(), subdir);
2907
+ }
2908
+ function blobPath(subdir, id) {
2909
+ const safe = sanitizeId(id);
2910
+ if (!safe) return null;
2911
+ const dir = blobDir(subdir);
2912
+ const candidate = path4.join(dir, `${safe}.json`);
2913
+ try {
2914
+ const rel = path4.relative(dir, candidate);
2915
+ if (rel.startsWith("..")) return null;
2916
+ } catch {
2917
+ return null;
2918
+ }
2919
+ return candidate;
2920
+ }
2921
+ function isBlobStale(subdir, id) {
2922
+ const p = blobPath(subdir, id);
2923
+ if (p === null) return false;
2924
+ try {
2925
+ const stat = fs4.statSync(p);
2926
+ return Date.now() - stat.mtimeMs > DEFAULT_MAX_AGE_MS;
2927
+ } catch {
2928
+ return false;
2929
+ }
2930
+ }
2931
+ function subdirCacheDefaults(subdir) {
2932
+ try {
2933
+ if (subdir === "bash_outputs") {
2934
+ const bc = loadConfig().bash_compress;
2935
+ return { maxCount: bc.cache_max_file_count, maxBytes: bc.cache_max_bytes, maxBytesPerItem: bc.cache_max_bytes_per_output };
2936
+ }
2937
+ if (subdir === "web_outputs") {
2938
+ const wf = loadConfig().webfetch;
2939
+ return { maxCount: wf.max_file_count, maxBytes: wf.max_bytes, maxBytesPerItem: Number.POSITIVE_INFINITY };
2940
+ }
2941
+ } catch {
2942
+ }
2943
+ return { maxCount: DEFAULT_MAX_COUNT, maxBytes: Number.POSITIVE_INFINITY, maxBytesPerItem: Number.POSITIVE_INFINITY };
2944
+ }
2945
+ function storeBlob(subdir, id, value, opts = {}) {
2946
+ const p = blobPath(subdir, id);
2947
+ if (!p) return false;
2948
+ const defaults = subdirCacheDefaults(subdir);
2949
+ const maxBytesPerItem = opts.maxBytesPerItem ?? defaults.maxBytesPerItem;
2950
+ let rawJson;
2951
+ try {
2952
+ rawJson = JSON.stringify(value);
2953
+ } catch {
2954
+ return false;
2955
+ }
2956
+ let json;
2957
+ try {
2958
+ const result = redactSecrets(rawJson);
2959
+ json = result.text;
2960
+ if (result.count > 0) recordStat("secret_redacted", 0, result.count, void 0, subdir);
2961
+ } catch {
2962
+ return false;
2963
+ }
2964
+ if (Number.isFinite(maxBytesPerItem) && Buffer.byteLength(json, "utf-8") > maxBytesPerItem) return false;
2965
+ try {
2966
+ const dir = path4.dirname(p);
2967
+ if (!fs4.existsSync(dir)) ensureDirSync(dir);
2968
+ atomicWriteText(p, json);
2969
+ } catch {
2970
+ return false;
2971
+ }
2972
+ pruneBlobs(
2973
+ subdir,
2974
+ opts.maxCount ?? defaults.maxCount,
2975
+ opts.maxAgeMs ?? DEFAULT_MAX_AGE_MS,
2976
+ opts.maxBytes ?? defaults.maxBytes,
2977
+ p
2978
+ );
2979
+ return true;
2980
+ }
2981
+ function loadBlob(subdir, id) {
2982
+ const p = blobPath(subdir, id);
2983
+ if (!p) return null;
2984
+ try {
2985
+ if (!fs4.existsSync(p)) return null;
2986
+ return JSON.parse(fs4.readFileSync(p, "utf8"));
2987
+ } catch {
2988
+ return null;
2989
+ }
2990
+ }
2991
+ function listBlobs(subdir) {
2992
+ const dir = blobDir(subdir);
2993
+ const out = [];
2994
+ try {
2995
+ if (!fs4.existsSync(dir)) return out;
2996
+ for (const file of fs4.readdirSync(dir)) {
2997
+ if (!file.endsWith(".json")) continue;
2998
+ const id = file.slice(0, -5);
2999
+ let mtime = 0;
3000
+ try {
3001
+ mtime = fs4.statSync(path4.join(dir, file)).mtimeMs;
3002
+ } catch {
3003
+ }
3004
+ const value = loadBlob(subdir, id);
3005
+ if (value !== null) out.push({ id, mtime, value });
3006
+ }
3007
+ } catch {
3008
+ return out;
3009
+ }
3010
+ return out;
3011
+ }
3012
+ function pruneBlobs(subdir, maxCount = DEFAULT_MAX_COUNT, maxAgeMs = DEFAULT_MAX_AGE_MS, maxBytes = Number.POSITIVE_INFINITY, protectedPath) {
3013
+ return pruneBlobDir(blobDir(subdir), maxCount, maxAgeMs, maxBytes, protectedPath);
3014
+ }
3015
+ function pruneBlobDir(dir, maxCount = DEFAULT_MAX_COUNT, maxAgeMs = DEFAULT_MAX_AGE_MS, maxBytes = Number.POSITIVE_INFINITY, protectedPath) {
3016
+ let removed = 0;
3017
+ try {
3018
+ if (!fs4.existsSync(dir)) return 0;
3019
+ const cutoff = Date.now() - maxAgeMs;
3020
+ let kept = [];
3021
+ let protectedEntry;
3022
+ for (const file of fs4.readdirSync(dir)) {
3023
+ const full = path4.join(dir, file);
3024
+ let stat;
3025
+ try {
3026
+ stat = fs4.statSync(full);
3027
+ } catch {
3028
+ continue;
3029
+ }
3030
+ if (!stat.isFile()) continue;
3031
+ if (protectedPath !== void 0 && full === protectedPath) {
3032
+ protectedEntry = [full, stat.mtimeMs, stat.size];
3033
+ continue;
3034
+ }
3035
+ if (stat.mtimeMs < cutoff) {
3036
+ try {
3037
+ fs4.unlinkSync(full);
3038
+ removed++;
3039
+ } catch {
3040
+ continue;
3041
+ }
3042
+ } else if (file.endsWith(".json")) {
3043
+ kept.push([full, stat.mtimeMs, stat.size]);
3044
+ }
3045
+ }
3046
+ const countBudget = protectedEntry ? Math.max(0, maxCount - 1) : maxCount;
3047
+ if (kept.length > countBudget) {
3048
+ kept.sort((a, b) => a[1] - b[1]);
3049
+ const excess = kept.slice(0, kept.length - countBudget);
3050
+ kept = kept.slice(kept.length - countBudget);
3051
+ for (const [full] of excess) {
3052
+ try {
3053
+ fs4.unlinkSync(full);
3054
+ removed++;
3055
+ } catch {
3056
+ continue;
3057
+ }
3058
+ }
3059
+ }
3060
+ if (Number.isFinite(maxBytes)) {
3061
+ kept.sort((a, b) => a[1] - b[1]);
3062
+ let total = kept.reduce((sum, [, , size]) => sum + size, 0) + (protectedEntry ? protectedEntry[2] : 0);
3063
+ while (total > maxBytes && kept.length > 0) {
3064
+ const oldest = kept.shift();
3065
+ if (!oldest) break;
3066
+ const [full, , size] = oldest;
3067
+ try {
3068
+ fs4.unlinkSync(full);
3069
+ removed++;
3070
+ total -= size;
3071
+ } catch {
3072
+ break;
3073
+ }
3074
+ }
3075
+ }
3076
+ } catch {
3077
+ return removed;
3078
+ }
3079
+ return removed;
3080
+ }
3081
+ var SWEEPABLE_CACHE_SUBDIRS = [
3082
+ { subdir: "bash_outputs", countCapped: true },
3083
+ { subdir: "web_outputs", countCapped: true },
3084
+ { subdir: "mcp_outputs", countCapped: true },
3085
+ { subdir: "sessions", countCapped: false }
3086
+ ];
3087
+ function sweepCacheRoots(extraRoots = []) {
3088
+ let removed = 0;
3089
+ const roots = [tokenGoatHome(), ...extraRoots];
3090
+ const seen = /* @__PURE__ */ new Set();
3091
+ for (const root of roots) {
3092
+ if (!root || seen.has(root)) continue;
3093
+ seen.add(root);
3094
+ for (const { subdir, countCapped } of SWEEPABLE_CACHE_SUBDIRS) {
3095
+ try {
3096
+ const defaults = subdirCacheDefaults(subdir);
3097
+ removed += pruneBlobDir(
3098
+ path4.join(root, subdir),
3099
+ countCapped ? defaults.maxCount : Number.POSITIVE_INFINITY,
3100
+ DEFAULT_MAX_AGE_MS,
3101
+ countCapped ? defaults.maxBytes : Number.POSITIVE_INFINITY
3102
+ );
3103
+ } catch {
3104
+ }
3105
+ }
3106
+ }
3107
+ return removed;
3108
+ }
3109
+
3110
+ export {
3111
+ Database,
3112
+ FILENAME_LANGUAGE,
3113
+ TREE_SITTER_LANGUAGES,
3114
+ languageHasFlag,
3115
+ languageLabel,
3116
+ fenceFor,
3117
+ basenameImportsExtension,
3118
+ partialRefsReason,
3119
+ refineLanguageByContent,
3120
+ detectLanguageOfFile,
3121
+ detectLanguage,
3122
+ nonTreeSitterLanguageCount,
3123
+ unsupportedLanguageName,
3124
+ isDotenvPath,
3125
+ redactIfDotenv,
3126
+ getDb,
3127
+ colorStdout,
3128
+ RESET,
3129
+ stripAnsiEscapes,
3130
+ stripAnsi,
3131
+ fg,
3132
+ C,
3133
+ SOURCE_HINT,
3134
+ savedTokensFromBytes,
3135
+ formatLocalTimestamp,
3136
+ recordStat,
3137
+ recordUnmappedTool,
3138
+ readUnmappedTools,
3139
+ summarize,
3140
+ _useRichStats,
3141
+ renderShortStats,
3142
+ renderStats2 as renderStats,
3143
+ countRedactionPlaceholders,
3144
+ compileCustomPatterns,
3145
+ redactSecrets,
3146
+ DEFAULT_MAX_COUNT,
3147
+ DEFAULT_MAX_AGE_MS,
3148
+ blobPath,
3149
+ isBlobStale,
3150
+ storeBlob,
3151
+ loadBlob,
3152
+ listBlobs,
3153
+ pruneBlobs,
3154
+ sweepCacheRoots
3155
+ };