token-goat 2.9.26 → 2.9.28
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -88
- package/SECURITY.md +14 -0
- package/dist/{token-goat-chunk-CQHY4SL7.mjs → token-goat-chunk-4JJBFT7V.mjs} +3 -3
- package/dist/token-goat-chunk-4JMWPWCS.mjs +96 -0
- package/dist/token-goat-chunk-4RN64ZND.mjs +954 -0
- package/dist/{token-goat-chunk-5M3NRJD2.mjs → token-goat-chunk-4STDIGYC.mjs} +2 -2
- package/dist/{token-goat-chunk-BH4ZVRB2.mjs → token-goat-chunk-5NNYIUGV.mjs} +2 -2
- package/dist/{token-goat-chunk-LHUP4DXZ.mjs → token-goat-chunk-66LODAUT.mjs} +1 -1
- package/dist/{token-goat-chunk-M6CVNHTW.mjs → token-goat-chunk-6CCQDAYY.mjs} +69 -566
- package/dist/{token-goat-chunk-62IHOU6L.mjs → token-goat-chunk-6DQNNFJ6.mjs} +5 -3
- package/dist/{token-goat-chunk-H2CPLEMP.mjs → token-goat-chunk-6SGLIPQB.mjs} +40 -20
- package/dist/{token-goat-chunk-MBDJR6HU.mjs → token-goat-chunk-6Z5ULNE5.mjs} +3570 -3970
- package/dist/token-goat-chunk-7MAFDRVR.mjs +47 -0
- package/dist/{token-goat-chunk-POOI26O6.mjs → token-goat-chunk-7R2XYKC5.mjs} +6 -5
- package/dist/{token-goat-chunk-E6B2UFDF.mjs → token-goat-chunk-BCN2YQC6.mjs} +10 -8
- package/dist/token-goat-chunk-BJBRGPBX.mjs +244 -0
- package/dist/{token-goat-chunk-TG5ZPD6B.mjs → token-goat-chunk-CBMY2FGI.mjs} +479 -544
- package/dist/{token-goat-chunk-LRIZ7K3F.mjs → token-goat-chunk-CC2PYJM7.mjs} +2 -1
- package/dist/{token-goat-chunk-7EPYB34H.mjs → token-goat-chunk-D5ICEPSB.mjs} +14 -14
- package/dist/token-goat-chunk-D7GHQKQD.mjs +30 -0
- package/dist/{token-goat-chunk-JALB3KWJ.mjs → token-goat-chunk-EL7M26WP.mjs} +7623 -8462
- package/dist/{token-goat-chunk-ZZT6PVB3.mjs → token-goat-chunk-F7QZRIWM.mjs} +69 -20
- package/dist/{token-goat-chunk-FLEYOCAS.mjs → token-goat-chunk-FSTWAOW6.mjs} +3 -3
- package/dist/token-goat-chunk-GGLEE35R.mjs +564 -0
- package/dist/{token-goat-chunk-IKZLSRII.mjs → token-goat-chunk-HUDC5GS5.mjs} +2 -2
- package/dist/{token-goat-chunk-43JFN26X.mjs → token-goat-chunk-HXNAC4ZF.mjs} +22 -964
- package/dist/token-goat-chunk-I7CB3X4K.mjs +1379 -0
- package/dist/{token-goat-chunk-GKZ2US5B.mjs → token-goat-chunk-IFORDBGC.mjs} +32 -27
- package/dist/{token-goat-chunk-CXBOFPF4.mjs → token-goat-chunk-IHKZ3QNA.mjs} +26 -15
- package/dist/token-goat-chunk-IZXKVT4A.mjs +711 -0
- package/dist/{token-goat-chunk-PAV6NNY7.mjs → token-goat-chunk-JXK2HQXS.mjs} +22 -8
- package/dist/{token-goat-chunk-6E2IDLHF.mjs → token-goat-chunk-L45ZQRT7.mjs} +1 -1
- package/dist/{token-goat-chunk-MDV5VWF4.mjs → token-goat-chunk-L7SDVGBT.mjs} +196 -14
- package/dist/token-goat-chunk-LEG3JNJ6.mjs +139 -0
- package/dist/{token-goat-chunk-7SXPJOSZ.mjs → token-goat-chunk-M56BWPBN.mjs} +1 -1
- package/dist/{token-goat-chunk-G6EYO5CD.mjs → token-goat-chunk-MAMQPC43.mjs} +2 -87
- package/dist/{token-goat-chunk-VVROQBQJ.mjs → token-goat-chunk-MMLMALWK.mjs} +5 -3
- package/dist/{token-goat-chunk-D2X7XN5M.mjs → token-goat-chunk-MVSWBNMD.mjs} +7 -231
- package/dist/{token-goat-chunk-C2PG4K5D.mjs → token-goat-chunk-NILVK7XM.mjs} +4 -4
- package/dist/{token-goat-chunk-BAWKGODL.mjs → token-goat-chunk-Q4S3UQGO.mjs} +1 -1
- package/dist/{token-goat-chunk-MWID3YOY.mjs → token-goat-chunk-QM76S6YS.mjs} +3 -3
- package/dist/{token-goat-chunk-4GEDH2WU.mjs → token-goat-chunk-RJH64KQH.mjs} +81 -70
- package/dist/{token-goat-chunk-STYRLQIW.mjs → token-goat-chunk-RSS3TLXU.mjs} +2 -2
- package/dist/{token-goat-chunk-5BQ7W7D7.mjs → token-goat-chunk-SJEKA7Y2.mjs} +21 -16
- package/dist/{token-goat-chunk-RIB6XY2V.mjs → token-goat-chunk-UBUKHVYW.mjs} +1 -1
- package/dist/{token-goat-chunk-7Y6GWTII.mjs → token-goat-chunk-VE27HPIF.mjs} +14 -10
- package/dist/{token-goat-chunk-RH27ZSPF.mjs → token-goat-chunk-VWZKZPSR.mjs} +100 -23
- package/dist/{token-goat-chunk-L45IWWMC.mjs → token-goat-chunk-XI2VFAV6.mjs} +6 -6
- package/dist/{token-goat-chunk-WNBSCSAI.mjs → token-goat-chunk-XKY4VKJ4.mjs} +1 -1
- package/dist/{token-goat-chunk-L43RKNHA.mjs → token-goat-chunk-YWSP6IU7.mjs} +1 -1
- package/dist/token-goat-hook.mjs +23 -17
- package/dist/token-goat.core.mjs +32 -27
- package/docs/cli.md +32 -21
- package/docs/install.md +2 -0
- package/package.json +1 -1
- package/dist/token-goat-chunk-ADMR3QVF.mjs +0 -41
- package/dist/token-goat-chunk-MZTO4FXH.mjs +0 -26
package/docs/cli.md
CHANGED
|
@@ -41,15 +41,15 @@ token-goat pdf-extract manual.pdf --pages 12-15 --layout --head 120
|
|
|
41
41
|
| `token-goat skill-section "<name>::<heading>"` | Extract a named section from an installed skill without reading the full skill file. |
|
|
42
42
|
| `token-goat skeleton "file"` | Show all signatures in a file without bodies — typically 70–90% fewer tokens than a full read. `--force-refresh` reparses from disk first, bypassing a stale index. `--stats` adds a per-symbol reference count and doc-coverage flag, computed live from the index. `--grep <pattern>` narrows to symbols whose name matches a regex (a literal substring when the pattern is not valid regex), which is how you skim one area of a large file without dumping its whole symbol list; `--min-lines <n>` drops symbols shorter than N lines. Both compose, and if a filter removes everything the output says so and names the filter, rather than looking like a file with no symbols. Accepts a comma-separated file list (`"a,b,c"`) to cover several files in one call, one clearly-headed block per file; extra space-separated file arguments are reported in a note naming that comma form instead of being silently dropped. With `--json`, a comma-separated list returns one merged document (rows carry their own `filePath`), not one document per file. |
|
|
43
43
|
| `token-goat outline "file"` | List top-level symbols with line ranges and docstring hints — one-glance file map. Doc hints are clipped to about a sentence with a visible ellipsis; the full doc comment is one `read "file::symbol"` away (`--json` carries it whole). `--force-refresh` reparses from disk first, bypassing a stale index. `--stats` adds a per-symbol reference count and doc-coverage flag, computed live from the index. `--grep <pattern>` narrows to symbols whose name matches a regex (a literal substring when the pattern is not valid regex), which is how you skim one area of a large file without dumping its whole symbol list; `--min-lines <n>` drops symbols shorter than N lines. Both compose, and if a filter removes everything the output says so and names the filter, rather than looking like a file with no symbols. Accepts a comma-separated file list (`"a,b,c"`) to cover several files in one call, one clearly-headed block per file; extra space-separated file arguments are reported in a note naming that comma form instead of being silently dropped. With `--json`, a comma-separated list returns one merged document (rows carry their own `filePath`), not one document per file. |
|
|
44
|
-
| `token-goat yaml-outline <file>` | Structural summary of a YAML document (array shape / object key types) instead of a raw Read. Multi-document streams (`---`-separated) outline as an array of documents. |
|
|
44
|
+
| `token-goat yaml-outline <file>` | Structural summary of a YAML document (array shape / object key types) instead of a raw Read. Multi-document streams (`---`-separated) outline as an array of documents. `--filter TEXT` keeps only the top-level keys containing the text and reports how many matched. |
|
|
45
45
|
| `token-goat yaml-query <file> <path>` | Extract one value or a projected/filtered subset from a YAML document by dot-path instead of a raw Read (same grammar as `json-query`: `[n]` index, `[*]` wildcard, `[field=value]` filter — e.g. `items[status=active].name`). `--head <n>` caps a projected/filtered result. |
|
|
46
46
|
| `token-goat xml-outline <file>` | Structural summary of an XML document (element tag hierarchy, attribute keys, child counts) instead of a raw Read. Supports `--depth <n>` (alias for `--max-depth`) and `--json`. |
|
|
47
47
|
| `token-goat xml-query <file> [path]` | Extract one value, element text/XML, or a projected/filtered subset from an XML document by dot-path or `--xpath <expr>` instead of a raw Read (dot-path grammar matches `json-query`: element tags, `@attr`, `[n]`, `[*]`, `[attr=value]`). Supports `--xpath <expression>` with namespace and attribute predicate support, `--with-lines` to output exact source line ranges (`(lines start-end)`), `--decode-embedded-xml` to pretty-print and bound entity-encoded nested AML/XML payloads, `--head <n>`, and `--json`. |
|
|
48
48
|
| `token-goat html-outline <file>` | Structural outline of an HTML document (DOM hierarchy, tags, IDs, classes, element counts, depth, landmarks, tables, forms) instead of reading thousands of lines of markup. Supports `--json`. |
|
|
49
49
|
| `token-goat html-query <file> <selector>` | Extract matching HTML elements or text using standard CSS selectors (tags, `#id`, `.class`, attribute operators `[attr]`, `[attr=val]`, `[attr*=val]`, `[attr^=val]`, `[attr$=val]`, child `>` and descendant combinators) without a headless browser or heavy DOM dependencies. Supports `--text` to strip tags, `--head <n>`, and `--json`. |
|
|
50
50
|
| `token-goat html-lint <file>` | Fast, zero-dependency HTML5 structural linter checking for unclosed tags, void element violations, duplicate IDs, missing viewport/charset, missing alt attributes, and inline script bloat. Supports `--json`. |
|
|
51
|
-
| `token-goat json-outline <file
|
|
52
|
-
| `token-goat json-query <file> <path>` | Extract one value or a projected/filtered subset from a JSON document by dot-path instead of a raw Read: dot-separated keys with optional bracket segments — `[n]` index, `[*]` wildcard (projects every element/value), `[field=value]` filter. Examples: `data.items[3].name`, `items[*].id`, `items[status=active]`. |
|
|
51
|
+
| `token-goat json-outline <file> [--filter TEXT]` | Structural summary of a JSON document (array shape / object key types) instead of a raw Read. `--filter TEXT` keeps only the top-level keys containing the text, for a registry with hundreds of them: `json-outline registry.json --filter 11862` prints ten keys and `(10 of 93 keys contain "11862")`. |
|
|
52
|
+
| `token-goat json-query <file> <path>` | Extract one value or a projected/filtered subset from a JSON document by dot-path instead of a raw Read: dot-separated keys with optional bracket segments — `[n]` index, `[*]` wildcard (projects every element/value), `[field=value]` filter. A quoted bracket segment is one literal key, the only way to name a key holding a dot or a space: `"['a.b'].c"` reads `c` under the key `a.b`, where `a.b.c` walks three levels. Examples: `data.items[3].name`, `items[*].id`, `items[status=active]`. |
|
|
53
53
|
| `token-goat brief "file::symbol"` | Bundle a symbol's body, resolved callers (grouped by enclosing function), and its containing doc section into one round-trip instead of three separate `read`/`callers`/`section` calls. `--limit <n>` caps the callers shown per symbol (default 20; the true caller count is reported even when truncated). Comma-separated `"file::a,b"` fetches several symbols' bundles from one file in a single call, mirroring `read`'s `file::a,b` multi-symbol grammar. Cross-file `"a.ts::x,b.ts::y"` bundles symbols from several files in one call, mirroring `read`'s cross-file grammar — a bare segment inherits the file to its left, and once more than one file is involved each bundle is keyed by the full `file::symbol` so two files contributing the same symbol name stay distinct. Also accepts `read`'s `symbol@LINE` anchor to pick out an otherwise-ambiguous candidate. `-C, --context <n>` adds N lines of real call-site source around each entry of the caller block. `--json`'s `symbol.filePath` and `callers[].file` render root-relative when a project root resolves, absolute when none does — matching the plain-text block above. `--exclude-tests` hides callers whose call site is in a test file, matching `refs`/`callers`; the caller count and the elided tail both count the filtered set, so they never disagree with the rows shown, and when the filter empties the block it says so instead of reporting a bare zero that would read as "nothing calls this". `--json` adds `hiddenByExcludeTests` only when the filter actually hid something. `--grep <pattern>` narrows the caller block to callers whose enclosing symbol name matches this regex (literal substring if it is not valid regex), the same filter `refs --grep`/`call-chain --grep` apply to their own results — useful for a high-fanout symbol whose default 20-caller window is otherwise mostly noise; composes with `--exclude-tests`, and reports `hiddenByGrep` under `--json` only when it hid something. |
|
|
54
54
|
| `token-goat scope "file:line"` | Show symbols in scope at a given line — avoids reading the whole file to understand locals. |
|
|
55
55
|
| `token-goat exports "file"` | List public (exported) symbols with types, docstring hints, and line ranges (`(lineStart-lineEnd)` in text mode, `lineStart`/`lineEnd` fields under `--json`). Names caught only by the source-text scan (no corresponding index row — e.g. certain re-export forms) report no location: omitted from text mode, `null` under `--json`. Accepts a comma-separated file list (`"a,b,c"`) to cover several files in one call, one clearly-headed block per file; extra space-separated file arguments are reported in a note naming that comma form instead of being silently dropped. `--grep <pattern>` only shows exported symbols whose NAME matches this regex (literal substring if it is not valid regex), applied before output is built; when it matches nothing among real exports, the output names the active filter instead of reading like the file has no exports at all. |
|
|
@@ -82,7 +82,7 @@ token-goat pdf-extract manual.pdf --pages 12-15 --layout --head 120
|
|
|
82
82
|
| `token-goat coverage-gaps` | Find callables in non-test source files that never appear in a test file's reference records. Useful for spotting untested surface area before a refactor or release. `--top N` caps output; `--json` for structured output. |
|
|
83
83
|
| `token-goat recent [N]` | Show the N most recently edited/accessed files with their symbols. |
|
|
84
84
|
| `token-goat grep "<pattern>" [paths...]` | Built-in fallback regex search over files (no `rg` shell-out, no caching) — session-aware dedup for raw `rg`/`grep` Bash calls is a separate hook, not this command. Accepts zero or more paths: omit to walk cwd, or pass several to search them together with hits merged in argument order under one `--max-lines` cap. `-C, --context <n>` shows `n` lines before and after each match. `--symbol` annotates each hit with its enclosing indexed symbol — ` [name (kind)]` appended in text mode, a `symbol: {name, kind, lineStart, lineEnd} | null` field per item under `--json` — `null`/no tag when the hit falls outside any indexed symbol (e.g. module-level code). |
|
|
85
|
-
| `token-goat semantic "<query>"` | Find code by meaning, not by filename: embedding-vector similarity search over indexed file chunks and full-text search (BM25) over symbol names/bodies both run on every query and are fused by Reciprocal Rank Fusion (`score = sum of 1/(60 + rank)` per list), so an exact keyword match can outrank a weak vector hit instead of being shadowed by the vector branch. Covers extracted text from PDF/DOCX/PPTX/XLSX files alongside source code, so a query can surface a spec PDF, design doc, deck, or spreadsheet, not just code. Results are re-ranked with a path-priority multiplier so live source wins ties/near-ties against stale or archival prose (`archive/`, `archived/`, `old/`, `deprecated/`, `plans/`, `drafts/`, `CHANGELOG*`, `*.bak`, `*.orig`) and, more mildly, general docs (`docs/**`, `*.md`) — a nudge, not a hard filter, so a genuinely much better archival match still surfaces. Configure with `token-goat config set semantic.archive_weight <0-1>` / `token-goat config set semantic.docs_weight <0-1>` (both default `<1`; set to `1` to disable that penalty entirely, e.g. for a project with a genuinely live `plans/` directory). `token-goat config set semantic.max_distance <0.05-2>` is the relevance floor: a vector hit whose raw distance exceeds it is dropped before fusion, so a corpus whose own distances have been measured can stop the vector half answering a question it has nothing for. It defaults to `1.2`, the retrieval bound's own value, so out of the box it filters nothing — measured genuine matches run from 0.635 on a large index to 0.934 on a two-file one, overlapping the off-corpus band, so no fixed number separates them across corpus sizes and the default is deliberately inert. When the floor empties the vector half, the command says so and names the closest distance it rejected, so the number to change is visible rather than inferred. The key is refused from a project's own `.token-goat.toml` (see `docs/security.md`), since a checked-in file setting it near the minimum would withhold that repository's code from the vector half while BM25 went on answering. `--preflight` runs an advance diagnostic verification of embedding runtime availability, model weight cache presence, and project coverage without executing a query; add `--warm` to load/cache the ONNX model into memory. `--limit <n>` caps result count; `--json` for structured output: `{source, items, truncated, totalCount}`, where `source` is `"hybrid"` when both the embedding and BM25 branches contributed at least one raw hit, `"embeddings"` when only the embedding branch did (e.g. no vector index exists yet: optional embedding deps unavailable, or `indexing.embeddings_enabled` is off), or `"fts"` when only BM25 did, and every item carries the same keys (`filePath`, `name`, `kind`, `startLine`, `endLine`, `distance`, `preview`, `rank`, `rrf`, `retrieval`), with `null` for whichever of `name`/`kind`/`distance` don't apply to that item's source. `rank` is the item's 1-based position, `rrf` the fused score it was sorted on, and `retrieval` is `"dense"`, `"lexical"` or `"both"` — together they are what makes the ordering reproducible, since `distance` is only the vector leg's own score and a lexical-only item has none at all, so an item at distance 0.850 legitimately sits above one at 0.776 when the keyword pass voted for the first as well, and `filePath` rendered root-relative when a project root resolves, absolute when none does, matching plain-text output. On an embeddings hit, `name`/`kind` are resolved to the innermost indexed symbol whose line range contains the hit's start line (`null`/`null` when the hit falls outside any symbol, e.g. a top-of-file imports chunk); text output appends the same as a `— inside <name> (<kind>)` suffix. `--grep <pattern>` only shows hits whose FILE PATH matches this regex (literal substring if it is not valid regex), tested against the path exactly as rendered (so an anchored `--grep "^src/"` matches what you see, not the stored absolute path) — the high-value case is dropping test/vendored noise from a project-wide semantic hit list. Applied before the `--limit` slice in both the embeddings and full-text-fallback branches, so it selects from the whole hit set, not an already-capped page; when it matches nothing among hits that do exist, the output names the active filter (and `--json` sets `grepFilteredToEmpty: true`) instead of reading like the search found nothing. `--exclude-tests` hides hits whose file is a test file (opt-in — omitted, output is unchanged), covering the case `--grep` structurally cannot: `--grep` can only ever *select* a path, and the negative-lookahead pattern that would express "not a test" silently degrades to a literal substring match whenever the regex-compile fallback fires. Applied before the `--limit` slice in both branches and composable with `--grep` (a hit must satisfy both); when it hides every hit there was, the output names how many were hidden (and `--json` sets `excludeTestsFilteredToEmpty: true`) and exits 0, instead of the exit-1 "no matches" a genuinely empty search returns. With both filters set and both emptying the view, the `--grep` notice takes priority. |
|
|
85
|
+
| `token-goat semantic "<query>" ["<query2>" ...]` | Find code by meaning, not by filename: embedding-vector similarity search over indexed file chunks and full-text search (BM25) over symbol names/bodies both run on every query and are fused by Reciprocal Rank Fusion (`score = sum of 1/(60 + rank)` per list), so an exact keyword match can outrank a weak vector hit instead of being shadowed by the vector branch. Several queries run in one call: each prints its own headed block in argument order (`--json`: an array of per-query payloads, each carrying its `query`); a single query's output is unchanged. `--preflight` combined with extra queries is rejected, since it never searches. When the closest hit that survives the relevance floor is farther than 0.85, a notice on stderr names the distance and points at a keyword route or a rephrase; `--json` carries it as `lowConfidence: {closestDistance, threshold}`. Covers extracted text from PDF/DOCX/PPTX/XLSX files alongside source code, so a query can surface a spec PDF, design doc, deck, or spreadsheet, not just code. Results are re-ranked with a path-priority multiplier so live source wins ties/near-ties against stale or archival prose (`archive/`, `archived/`, `old/`, `deprecated/`, `plans/`, `drafts/`, `CHANGELOG*`, `*.bak`, `*.orig`) and, more mildly, general docs (`docs/**`, `*.md`) — a nudge, not a hard filter, so a genuinely much better archival match still surfaces. Configure with `token-goat config set semantic.archive_weight <0-1>` / `token-goat config set semantic.docs_weight <0-1>` (both default `<1`; set to `1` to disable that penalty entirely, e.g. for a project with a genuinely live `plans/` directory). `token-goat config set semantic.max_distance <0.05-2>` is the relevance floor: a vector hit whose raw distance exceeds it is dropped before fusion, so a corpus whose own distances have been measured can stop the vector half answering a question it has nothing for. It defaults to `1.2`, the retrieval bound's own value, so out of the box it filters nothing — measured genuine matches run from 0.635 on a large index to 0.934 on a two-file one, overlapping the off-corpus band, so no fixed number separates them across corpus sizes and the default is deliberately inert. When the floor empties the vector half, the command says so and names the closest distance it rejected, so the number to change is visible rather than inferred. The key is refused from a project's own `.token-goat.toml` (see `docs/security.md`), since a checked-in file setting it near the minimum would withhold that repository's code from the vector half while BM25 went on answering. `--preflight` runs an advance diagnostic verification of embedding runtime availability, model weight cache presence, and project coverage without executing a query; add `--warm` to load/cache the ONNX model into memory. `--limit <n>` caps result count; `--json` for structured output: `{source, items, truncated, totalCount}`, where `source` is `"hybrid"` when both the embedding and BM25 branches contributed at least one raw hit, `"embeddings"` when only the embedding branch did (e.g. no vector index exists yet: optional embedding deps unavailable, or `indexing.embeddings_enabled` is off), or `"fts"` when only BM25 did, and every item carries the same keys (`filePath`, `name`, `kind`, `startLine`, `endLine`, `distance`, `preview`, `rank`, `rrf`, `retrieval`), with `null` for whichever of `name`/`kind`/`distance` don't apply to that item's source. `rank` is the item's 1-based position, `rrf` the fused score it was sorted on, and `retrieval` is `"dense"`, `"lexical"` or `"both"` — together they are what makes the ordering reproducible, since `distance` is only the vector leg's own score and a lexical-only item has none at all, so an item at distance 0.850 legitimately sits above one at 0.776 when the keyword pass voted for the first as well, and `filePath` rendered root-relative when a project root resolves, absolute when none does, matching plain-text output. On an embeddings hit, `name`/`kind` are resolved to the innermost indexed symbol whose line range contains the hit's start line (`null`/`null` when the hit falls outside any symbol, e.g. a top-of-file imports chunk); text output appends the same as a `— inside <name> (<kind>)` suffix. `--grep <pattern>` only shows hits whose FILE PATH matches this regex (literal substring if it is not valid regex), tested against the path exactly as rendered (so an anchored `--grep "^src/"` matches what you see, not the stored absolute path) — the high-value case is dropping test/vendored noise from a project-wide semantic hit list. Applied before the `--limit` slice in both the embeddings and full-text-fallback branches, so it selects from the whole hit set, not an already-capped page; when it matches nothing among hits that do exist, the output names the active filter (and `--json` sets `grepFilteredToEmpty: true`) instead of reading like the search found nothing. `--exclude-tests` hides hits whose file is a test file (opt-in — omitted, output is unchanged), covering the case `--grep` structurally cannot: `--grep` can only ever *select* a path, and the negative-lookahead pattern that would express "not a test" silently degrades to a literal substring match whenever the regex-compile fallback fires. Applied before the `--limit` slice in both branches and composable with `--grep` (a hit must satisfy both); when it hides every hit there was, the output names how many were hidden (and `--json` sets `excludeTestsFilteredToEmpty: true`) and exits 0, instead of the exit-1 "no matches" a genuinely empty search returns. With both filters set and both emptying the view, the `--grep` notice takes priority. When a single identifier returns zero matches, the command suggests checking `token-goat symbol "<name>"` and notes if that identifier is already cataloged in the symbol index. |
|
|
86
86
|
| `token-goat map` | Get a compact orientation of the repo. Add `--compact` to fit a fixed 2000-token budget. `--json` emits the project map as JSON instead of text. |
|
|
87
87
|
| `token-goat deps "file"` | One-level import listing for a single file: resolves relative imports to project files (`internal`, root-relative paths) and groups everything else as `external`. `--json` for structured output. `--grep <pattern>` only shows dependencies whose MODULE SPECIFIER (the resolved internal path or the external package name) matches this regex (literal substring if it is not valid regex), applied before output is built; when it matches nothing among real dependencies, the output names the active filter instead of reading like the file has no imports at all. Complemented by `token-goat arch` for the project-wide graph. |
|
|
88
88
|
| `token-goat arch` | Project-wide import graph summary: hub modules (most imported), entry points (nothing imports them), and circular chains. `--modules` adds a grouping of the files that mostly import each other, naming each group by its most connected file, saying whether the group is one directory or spread across several, and listing which groups reach into which. That section also prints the grouping's modularity and calls it out when it is too weak to mean anything, since the algorithm returns groups for any graph, including one with no real structure. `--json` carries each group's full member list. Complements `token-goat deps <file>` for per-file depth. |
|
|
@@ -103,10 +103,10 @@ token-goat pdf-extract manual.pdf --pages 12-15 --layout --head 120
|
|
|
103
103
|
| `token-goat history` | Show current session access history: bash commands and URLs fetched. |
|
|
104
104
|
| `token-goat session-outline` | Turn-by-turn structure (role, preview, tool calls, approx size) of a Claude Code session JSONL transcript, instead of a raw Read; defaults to the current project's most recent session. |
|
|
105
105
|
| `token-goat session-slice <turns>` | Full content of one turn range from a Claude Code session JSONL transcript (see `session-outline` for turn numbers), instead of a raw Read. |
|
|
106
|
-
| `token-goat session-audit [--dir <path>] [--json]` | Corpus-wide token attribution across every local Claude Code session transcript, including nested subagent transcripts (default corpus: `~/.claude/projects`): measured billed usage from each API response's own usage record, estimated content size by source and by tool, a per-attachment-kind census ranked by modeled billed cost (cache write plus compaction-capped cache re-reads over the model-visible fields only), a hook-output census split by origin, a subagent-lane rollup (spawn-prefix size and its modeled billed carriage, plus a per-agent-type breakdown from each lane's meta file), a Read-interception census (diverted reads versus full serves, with each large full serve split into first read versus repeat and repeats classified as deliberate paging or divert-miss candidates), a Bash filter fire-rate census (results carrying a token-goat marker versus the untouched remainder, bucketed by bare command head: the binary name only), and billed cost by session position. Output is aggregate counts only, never transcript content or command lines. |
|
|
106
|
+
| `token-goat session-audit [--dir <path>] [--json] [--tool-errors]` | Corpus-wide token attribution across every local Claude Code session transcript, including nested subagent transcripts (default corpus: `~/.claude/projects`): measured billed usage from each API response's own usage record, estimated content size by source and by tool, a per-attachment-kind census ranked by modeled billed cost (cache write plus compaction-capped cache re-reads over the model-visible fields only), a hook-output census split by origin, a subagent-lane rollup (spawn-prefix size and its modeled billed carriage, plus a per-agent-type breakdown from each lane's meta file), a Read-interception census (diverted reads versus full serves, with each large full serve split into first read versus repeat and repeats classified as deliberate paging or divert-miss candidates), a Bash filter fire-rate census (results carrying a token-goat marker versus the untouched remainder, bucketed by bare command head: the binary name only), and billed cost by session position. `--tool-errors` replaces this with a census of failed tool calls instead: calls, errors and unknown errors per tool and per model, with the most frequent unknown error prefixes, so a call that fails without a recognized reason still points at what to classify next. A deny that delivers the content it withholds, such as a skill's compact slice, is counted separately and named on its own line rather than folded into the error counts. Output is aggregate counts only, never transcript content or command lines. |
|
|
107
107
|
| `token-goat bash-output <id>` | Retrieve a cached Bash output by ID instead of re-running the command. Large outputs return a head(30)+tail(80) view by default; pass `--full` for the entire stored entry with no elision, `--head N`/`--tail N` for a specific slice, or narrow with `--grep PATTERN` (cap `--grep` to the first N hits with `--max-matches N`). Read a file directly with `--file <path>` (e.g. a background task's `tasks/<id>.output`); add `--transcript` to parse that file as a subagent JSONL transcript, keeping only assistant text blocks in order before the slicers apply. Add `--verify-last-write [seconds]` (with optional `--strict`) when reading with `--file` to verify the output file was modified within the last N seconds (default 60s), alerting or rejecting stale/cached reads from silent no-op commands before repeating runs. |
|
|
108
108
|
| `token-goat bash-history` | List cached Bash outputs (newest first) with their IDs, byte sizes, and exit codes. |
|
|
109
|
-
| `token-goat compress --cmd '<command>'` |
|
|
109
|
+
| `token-goat compress [--cmd '<command>' | --cmd-b64 <b64>] [--shell <type>]` | Run a command through token-goat's output compression filters. Accepts an inline command via `--cmd` or base64-encoded string via `--cmd-b64` (preventing shell quoting and escaping issues). `--shell` specifies runner (`bash`, `pwsh`, `powershell`, or `native`, defaulting to bash on POSIX and resolving PowerShell on Windows). `-f passthrough` applies no filter, only the size limits. `--max-tokens <n>` caps what is delivered, and output cut by the cap or a filter ends with a `bash-output <id> --full` line to recall the rest. The Bash hook adds `--cap-hint-b64`, a base64 sentence printed after that line naming the narrower command to run next. |
|
|
110
110
|
| `token-goat web-output <id>` | Retrieve a cached WebFetch response body by ID — same head+tail default and `--full`/`--head`/`--tail`/`--grep`/`--max-matches` slicers as `bash-output`. `--raw` returns the body as actually fetched, before `webfetch.compress_bodies`'s HTML-cleaning pass, for recovering a selector/script tag/embedded JSON that the default cleaned text drops; falls back to the (already-raw) cleaned body when no separate raw copy was stored. |
|
|
111
111
|
| `token-goat web-history` | List cached WebFetch responses (newest first) with their IDs, byte sizes, status codes, and URL previews. |
|
|
112
112
|
| `token-goat mcp-output [id]` | Retrieve a cached MCP tool result by ID (or an on-disk tool spill file via `--file <path>`). Supports `--json-query '<path>'` for narrow dot-path and bracket slicing on large Jira, Confluence, or JSON responses (e.g. `issues[*].key`, `total`), avoiding whole-file re-reads. Also supports `--head <n>`, `--tail <n>`, `--grep <pattern>`, `--section <heading>`, `--json`, and `--full`. |
|
|
@@ -173,7 +173,7 @@ token-goat pdf-extract manual.pdf --pages 12-15 --layout --head 120
|
|
|
173
173
|
| `token-goat lockdeps [path]` | Summarize lock file dependencies as a compact table. Reads poetry.lock, uv.lock, requirements.txt, Pipfile.lock, package-lock.json, Cargo.lock, and yarn.lock. Direct dependencies only — optional and transitive entries excluded. `--json` for structured output. |
|
|
174
174
|
| `token-goat logfold [src]` | Collapse consecutive duplicate log lines. Runs of identical or structurally equivalent lines fold to `[Nx] line`. Normalizes timestamps, UUIDs, IPs, hex IDs, and bare integers (counters, PIDs, ports, byte counts) before comparing so the same event with different values folds correctly. `--tail N` keeps last N lines; `--no-normalize` disables normalization; `--fold-repeats` also folds non-consecutive duplicates anywhere in the input, attributing the total count to the first occurrence (capped at 20,000 distinct keys, past which it falls back to consecutive-only); `--json` for structured output. |
|
|
175
175
|
| `token-goat hot [--limit N]` | Cross-session file frequency table: read and edit counts tallied from all stored sessions, ranked by total activity. Shows which files dominate your token spend across your entire history. `--project <dir>` filters to one project; `--json` for structured output. |
|
|
176
|
-
| `token-goat note set/get/unset/list/clear` | Persistent per-project notes stored as key-value pairs. Token-goat
|
|
176
|
+
| `token-goat note set/get/unset/list/clear` | Persistent per-project notes stored as key-value pairs. Token-goat prints them at every session start and in the compaction manifest, ahead of the files read, so they survive conversation rollover; on Codex, whose session start hook is not wired, the manifest is how they come back. Use to pin decisions, constraints, or reminders that would otherwise vanish after compaction. `note list --json` for machine-readable output; `note clear` removes everything at once. |
|
|
177
177
|
| `token-goat project list` | Show all project roots indexed by token-goat with their file counts. Roots on the blocklist appear tagged `[excluded]`. `--json` for structured output. |
|
|
178
178
|
| `token-goat project exclude <path>` | Add a project root to the blocklist so the worker never indexes it. Writes the resolved absolute path to `[worker] blocked_roots` in `config.toml`; idempotent. It also removes anything already indexed under that path and says how many files went, so excluding a directory means its contents stop being readable through `symbol` rather than merely stopping future indexing. Remove the entry from the config to re-enable indexing, then run `token-goat index` to bring the contents back. |
|
|
179
179
|
| `token-goat project prune [--dry-run]` | Remove blocked/excluded roots that no longer exist on disk, and index rows for files under the OS temp dir. `--dry-run` previews removals without touching the config file. Useful after deleting or moving projects. |
|
|
@@ -268,16 +268,21 @@ Total tokens: 18420
|
|
|
268
268
|
$ token-goat waste --copilot
|
|
269
269
|
|
|
270
270
|
## Per-request fixed overhead (Copilot's own token counts)
|
|
271
|
-
System prompt:
|
|
272
|
-
Tool definitions:
|
|
273
|
-
Conversation:
|
|
271
|
+
System prompt: 7,897 tok
|
|
272
|
+
Tool definitions: 10,993 tok
|
|
273
|
+
Conversation: 577 tok
|
|
274
274
|
|
|
275
275
|
## Tool definitions by MCP server (estimated)
|
|
276
|
-
github-mcp-server: 6 tools, 6 KB, ~2,135 tok
|
|
276
|
+
github-mcp-server: 6 tools, 6 KB, ~2,135 tok, 0 calls this session
|
|
277
277
|
~2,135 tok estimated across 1 server, re-sent every request.
|
|
278
|
-
Copilot counted
|
|
278
|
+
Copilot counted 10,993 tok of tool definitions in total, so this is roughly 19.4% of it.
|
|
279
|
+
Dropping a server saves its line above on every request for the rest of the session.
|
|
280
|
+
Never called this session: `copilot mcp disable github-mcp-server` drops it from future sessions.
|
|
281
|
+
'copilot mcp enable <name>' restores one, and '--disable-mcp-server <name>' drops one for a single run instead.
|
|
279
282
|
```
|
|
280
283
|
|
|
284
|
+
Calls per server come from the server name Copilot records on every MCP tool call. A server with none gets the command that turns it off, which Copilot keeps in its settings until `copilot mcp enable` undoes it. A log that records no tool calls at all says so instead, since it cannot tell a server nobody needed from a session that never ran a tool. `token-goat audit` carries the same verdict into its recommended fix.
|
|
285
|
+
|
|
281
286
|
The fixed overhead is the largest number in a Copilot session and no hook can reach it: Copilot assembles the system prompt and the tool definitions natively, with nothing between assembly and send. Only configuration moves it. The per-server breakdown exists to make that configuration decision possible, since one aggregate says the tool definitions are expensive without saying which tools. It is read from Copilot's own MCP tool cache, counts only the fields a model is actually sent, and is labeled an estimate throughout: it comes from byte length rather than Copilot's tokeniser, and it deliberately does not add up to Copilot's total, because Copilot's own built-in tools are not cached there.
|
|
282
287
|
|
|
283
288
|
The "Assistant output" section is separate from the tool-call ledger above it: `generatedTokens` is what was actually paid, once, to produce the assistant's own text turns. `resendCeilingTokens` is a cache-unaware upper bound on how much re-sending those turns as conversation history on every later request could cost — not real spend, since Claude Code's prompt caching bills a repeated conversation prefix at cache-read rates, a fraction of full input price. Treat it as a ceiling on how bad unbounded verbosity could get, not as a dollar figure.
|
|
@@ -304,7 +309,7 @@ Run `token-goat recall` with **no query** to browse instead of search: every cac
|
|
|
304
309
|
|
|
305
310
|
### Hint efficacy tracking
|
|
306
311
|
|
|
307
|
-
Every hint hook (the re-read/dedup/surgical-read nudges in the Bash, Read, and
|
|
312
|
+
Every hint hook (the re-read/dedup/surgical-read nudges in the Bash, Read, Edit, Grep and Glob hooks, and the call-streak hints) is
|
|
308
313
|
worth its keep only if it's actually followed. `token-goat hint-stats` reports, per hint
|
|
309
314
|
category: how many times it fired, how many times a later Bash command in the same session
|
|
310
315
|
actually invoked the specific `token-goat` command (or referenced the specific cached-output id)
|
|
@@ -314,14 +319,18 @@ that category — the real cost of emitting it, not just how often it fired):
|
|
|
314
319
|
|
|
315
320
|
```
|
|
316
321
|
$ token-goat hint-stats
|
|
317
|
-
category emitted undisplayed acted-on efficacy
|
|
318
|
-
bash_redirect
|
|
319
|
-
bash_recall
|
|
320
|
-
read_reread_dedup
|
|
321
|
-
read_structural_nav
|
|
322
|
-
edit_reread_suggest
|
|
323
|
-
|
|
324
|
-
|
|
322
|
+
category emitted undisplayed acted-on efficacy suppressed manual+ manual- spent-bytes
|
|
323
|
+
bash_redirect 42 - 9 21.4% no 0 0 3150
|
|
324
|
+
bash_recall 18 - 15 83.3% no 0 0 1080
|
|
325
|
+
read_reread_dedup 11 - 2 18.2% * no 0 0 660
|
|
326
|
+
read_structural_nav 7 3 1 14.3% * yes 0 1 420
|
|
327
|
+
edit_reread_suggest 3 - 0 0% * no 0 0 180
|
|
328
|
+
read_batch 6 - 4 66.7% no 0 0 900
|
|
329
|
+
search_brake 2 - 1 50.0% no 0 0 330
|
|
330
|
+
grep_dedup_hint 4 - 0 n/a no 0 0 240
|
|
331
|
+
glob_dedup_hint 1 - 0 n/a no 0 0 60
|
|
332
|
+
|
|
333
|
+
TOTAL saved-bytes=48200 (all-time, every hint kind) spent-bytes=7020 (hint_emissions ledger only)
|
|
325
334
|
```
|
|
326
335
|
|
|
327
336
|
`spent-bytes` (and the `TOTAL` line's `spent-bytes`) render `n/a` instead of a fake `0` whenever a
|
|
@@ -357,6 +366,8 @@ Note that what this feature calls "harness" (Claude Code, Codex, Gemini, ...) is
|
|
|
357
366
|
"which LLM model" — no bridge in this codebase exposes an LLM model identifier to hooks, so
|
|
358
367
|
harness is the closest real signal available.
|
|
359
368
|
|
|
369
|
+
Two categories are scored on a pattern of calls rather than a named command. `read_batch` (three or more reads or searches in a row, each sent a full turn after the previous result) counts as acted on when the first later turn that makes read-only calls makes at least two of them together. `search_brake` (three searches in a row found nothing) counts as acted on when the next search from a later turn is `token-goat answer` or `token-goat semantic`. Until that later call arrives the emission stays pending and counts as not acted on. `grep_dedup_hint` and `glob_dedup_hint` (an identical Grep or Glob already ran this session) are never scored: the note rides on the re-run it describes, so no later call can show whether it was heeded. They appear in `spent-bytes` and as the `~N` beside `emitted`, never in `emitted` itself or the efficacy figure.
|
|
370
|
+
|
|
360
371
|
### Scoring the compressors — `token-goat bench`
|
|
361
372
|
|
|
362
373
|
`token-goat bench` replays a fixed corpus of captured command output through the same function
|
package/docs/install.md
CHANGED
|
@@ -20,6 +20,8 @@ token-goat doctor # confirms the hooks, index, and integrations are hea
|
|
|
20
20
|
|
|
21
21
|
Three commands. Done. Hooks register and start working immediately; no terminal popups, no tray icon, no service to babysit.
|
|
22
22
|
|
|
23
|
+
> **WSL performance tip:** Keep active repositories on WSL's native ext4 filesystem (`~/projects/...`) rather than Windows mounts (`/mnt/c/...`) to avoid 9P cross-OS filesystem translation overhead during initial indexing.
|
|
24
|
+
|
|
23
25
|
### Agents choose the commands
|
|
24
26
|
|
|
25
27
|
People install token-goat. Agents use it. You do not need to memorize its commands or tell the agent which file type it has.
|
package/package.json
CHANGED
|
@@ -1,41 +0,0 @@
|
|
|
1
|
-
import { createRequire as __cjsRequire } from 'node:module';
|
|
2
|
-
const require = __cjsRequire(import.meta.url);
|
|
3
|
-
import {
|
|
4
|
-
buildEvent,
|
|
5
|
-
relay,
|
|
6
|
-
relayInProcess
|
|
7
|
-
} from "./token-goat-chunk-JALB3KWJ.mjs";
|
|
8
|
-
import "./token-goat-chunk-ZZT6PVB3.mjs";
|
|
9
|
-
import {
|
|
10
|
-
MAX_STDIN_BYTES,
|
|
11
|
-
readStdinJson
|
|
12
|
-
} from "./token-goat-chunk-7Y6GWTII.mjs";
|
|
13
|
-
import "./token-goat-chunk-62IHOU6L.mjs";
|
|
14
|
-
import "./token-goat-chunk-MDV5VWF4.mjs";
|
|
15
|
-
import "./token-goat-chunk-MBDJR6HU.mjs";
|
|
16
|
-
import "./token-goat-chunk-4GEDH2WU.mjs";
|
|
17
|
-
import "./token-goat-chunk-VVROQBQJ.mjs";
|
|
18
|
-
import "./token-goat-chunk-M6CVNHTW.mjs";
|
|
19
|
-
import "./token-goat-chunk-G6EYO5CD.mjs";
|
|
20
|
-
import "./token-goat-chunk-43JFN26X.mjs";
|
|
21
|
-
import "./token-goat-chunk-D2X7XN5M.mjs";
|
|
22
|
-
import "./token-goat-chunk-3BTK54F3.mjs";
|
|
23
|
-
import "./token-goat-chunk-7EPYB34H.mjs";
|
|
24
|
-
import "./token-goat-chunk-IKZLSRII.mjs";
|
|
25
|
-
import "./token-goat-chunk-7SXPJOSZ.mjs";
|
|
26
|
-
import "./token-goat-chunk-LRIZ7K3F.mjs";
|
|
27
|
-
import "./token-goat-chunk-EEIDFMEM.mjs";
|
|
28
|
-
import "./token-goat-chunk-OSUFN2FV.mjs";
|
|
29
|
-
import "./token-goat-chunk-6E2IDLHF.mjs";
|
|
30
|
-
import "./token-goat-chunk-NMTKNYGF.mjs";
|
|
31
|
-
import "./token-goat-chunk-Y4AFKTHK.mjs";
|
|
32
|
-
import "./token-goat-chunk-GMOUBOX4.mjs";
|
|
33
|
-
import "./token-goat-chunk-RRNZMM3A.mjs";
|
|
34
|
-
import "./token-goat-chunk-A37V4PBF.mjs";
|
|
35
|
-
export {
|
|
36
|
-
MAX_STDIN_BYTES,
|
|
37
|
-
buildEvent,
|
|
38
|
-
readStdinJson,
|
|
39
|
-
relay,
|
|
40
|
-
relayInProcess
|
|
41
|
-
};
|
|
@@ -1,26 +0,0 @@
|
|
|
1
|
-
import { createRequire as __cjsRequire } from 'node:module';
|
|
2
|
-
const require = __cjsRequire(import.meta.url);
|
|
3
|
-
import {
|
|
4
|
-
MAX_RESUME_CHARS,
|
|
5
|
-
MAX_RESUME_TOKENS,
|
|
6
|
-
buildResumePacket
|
|
7
|
-
} from "./token-goat-chunk-L45IWWMC.mjs";
|
|
8
|
-
import "./token-goat-chunk-MDV5VWF4.mjs";
|
|
9
|
-
import "./token-goat-chunk-M6CVNHTW.mjs";
|
|
10
|
-
import "./token-goat-chunk-G6EYO5CD.mjs";
|
|
11
|
-
import "./token-goat-chunk-43JFN26X.mjs";
|
|
12
|
-
import "./token-goat-chunk-D2X7XN5M.mjs";
|
|
13
|
-
import "./token-goat-chunk-IKZLSRII.mjs";
|
|
14
|
-
import "./token-goat-chunk-7SXPJOSZ.mjs";
|
|
15
|
-
import "./token-goat-chunk-LRIZ7K3F.mjs";
|
|
16
|
-
import "./token-goat-chunk-EEIDFMEM.mjs";
|
|
17
|
-
import "./token-goat-chunk-OSUFN2FV.mjs";
|
|
18
|
-
import "./token-goat-chunk-6E2IDLHF.mjs";
|
|
19
|
-
import "./token-goat-chunk-NMTKNYGF.mjs";
|
|
20
|
-
import "./token-goat-chunk-GMOUBOX4.mjs";
|
|
21
|
-
import "./token-goat-chunk-A37V4PBF.mjs";
|
|
22
|
-
export {
|
|
23
|
-
MAX_RESUME_CHARS,
|
|
24
|
-
MAX_RESUME_TOKENS,
|
|
25
|
-
buildResumePacket
|
|
26
|
-
};
|