slash-tokens 1.4.1 → 1.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -1
- package/dist/cli.js +10 -5
- package/dist/intercept.js +7 -7
- package/dist/models.js +11 -2
- package/dist/preflight.js +2 -2
- package/dist/slash.js +68 -13
- package/package.json +2 -1
package/README.md
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# /slash-tokens
|
|
2
2
|
|
|
3
3
|
[](https://github.com/Wolfe-Jam/slash-tokens/actions/workflows/test.yml)
|
|
4
|
+
[](https://faf.one)
|
|
4
5
|
[](https://www.npmjs.com/package/slash-tokens)
|
|
5
6
|
[](https://www.npmjs.com/package/slash-tokens)
|
|
6
7
|
[](https://bundlephobia.com/package/slash-tokens)
|
|
@@ -60,7 +61,7 @@ Full docs, examples, and model pricing at **[GitHub](https://github.com/Wolfe-Ja
|
|
|
60
61
|
|
|
61
62
|
## License
|
|
62
63
|
|
|
63
|
-
**Code: MIT.** Fork it, ship it, change it, sell it.
|
|
64
|
+
**Code: MIT.** Fork it, ship it, change it, show it, share it, sell it.
|
|
64
65
|
|
|
65
66
|
**Brand: reserved.** The slash-tokens name, ⚡ mark, and red/gold colors stay with the project. If you're building on top of the SDK, ship under your own name and colors — don't represent your app as Slash.
|
|
66
67
|
|
package/dist/cli.js
CHANGED
|
@@ -41,11 +41,16 @@ function writeToMemory(content) {
|
|
|
41
41
|
// src/slash.ts
|
|
42
42
|
var WASM_INPUT_OFFSET2 = 4096;
|
|
43
43
|
var CALIBRATION = {
|
|
44
|
-
"claude-opus": 1.
|
|
45
|
-
"claude-opus-4.7": 1.
|
|
46
|
-
"claude-sonnet": 1.
|
|
47
|
-
"claude-haiku": 1.
|
|
44
|
+
"claude-opus": 1.85,
|
|
45
|
+
"claude-opus-4.7": 1.85,
|
|
46
|
+
"claude-sonnet": 1.85,
|
|
47
|
+
"claude-haiku": 1.4,
|
|
48
|
+
"gemini-3.1-pro": 1.4,
|
|
49
|
+
"gemini-2.5-flash": 1.4,
|
|
50
|
+
"grok-4.20": 1.15,
|
|
51
|
+
"grok-4-1-fast": 1.15
|
|
48
52
|
};
|
|
53
|
+
var DEFAULT_UNKNOWN_MODEL_FACTOR = 1.85;
|
|
49
54
|
function slash(content, model) {
|
|
50
55
|
if (!content)
|
|
51
56
|
return 0;
|
|
@@ -54,7 +59,7 @@ function slash(content, model) {
|
|
|
54
59
|
const raw = instance.exports.estimate_tokens(WASM_INPUT_OFFSET2, len);
|
|
55
60
|
if (!model)
|
|
56
61
|
return raw;
|
|
57
|
-
const factor = CALIBRATION[model] ??
|
|
62
|
+
const factor = CALIBRATION[model] ?? DEFAULT_UNKNOWN_MODEL_FACTOR;
|
|
58
63
|
return factor === 1 ? raw : Math.ceil(raw * factor);
|
|
59
64
|
}
|
|
60
65
|
|
package/dist/intercept.js
CHANGED
|
@@ -5,16 +5,16 @@ import { PROVIDER_MODELS } from './providers.js';
|
|
|
5
5
|
// Reverse lookup: model name → provider model names in the API
|
|
6
6
|
// (what to put back in the request body)
|
|
7
7
|
const MODEL_API_NAMES = {
|
|
8
|
-
'claude-opus': 'claude-opus-
|
|
9
|
-
'claude-sonnet': 'claude-sonnet-
|
|
8
|
+
'claude-opus': 'claude-opus-5',
|
|
9
|
+
'claude-sonnet': 'claude-sonnet-5',
|
|
10
10
|
'claude-haiku': 'claude-haiku-4-5-20251001',
|
|
11
11
|
'gpt-5.4': 'gpt-5.4',
|
|
12
12
|
'gpt-5.4-mini': 'gpt-5.4-mini',
|
|
13
13
|
'gpt-5.4-nano': 'gpt-5.4-nano',
|
|
14
|
-
'grok-4.20': 'grok-4.20',
|
|
15
|
-
'grok-4-1-fast': 'grok-4
|
|
16
|
-
'gemini-3.1-pro': 'gemini-
|
|
17
|
-
'gemini-2.5-flash': 'gemini-
|
|
14
|
+
'grok-4.20': 'grok-4.20-0309-non-reasoning',
|
|
15
|
+
'grok-4-1-fast': 'grok-4.3',
|
|
16
|
+
'gemini-3.1-pro': 'gemini-pro-latest',
|
|
17
|
+
'gemini-2.5-flash': 'gemini-flash-latest',
|
|
18
18
|
};
|
|
19
19
|
// AI API endpoint detection
|
|
20
20
|
const AI_ENDPOINTS = [
|
|
@@ -140,7 +140,7 @@ export function patchFetch() {
|
|
|
140
140
|
const content = extractContent(body);
|
|
141
141
|
const rawModel = match.modelExtractor(body, url);
|
|
142
142
|
const originalModel = normalizeModel(rawModel);
|
|
143
|
-
const tokens = slash(content);
|
|
143
|
+
const tokens = slash(content, originalModel);
|
|
144
144
|
const originalInfo = getModel(originalModel);
|
|
145
145
|
const originalCost = originalInfo ? Math.round(((tokens / 1000000) * originalInfo.input) * 1000000) / 1000000 : 0;
|
|
146
146
|
const fits = originalInfo ? tokens <= originalInfo.context : true;
|
package/dist/models.js
CHANGED
|
@@ -1,4 +1,13 @@
|
|
|
1
1
|
// Pricing as of April 2026 — USD per million tokens
|
|
2
|
+
// xAI pricing re-derived 2026-08-23: grok-4.20 and grok-4-1-fast (the literal
|
|
3
|
+
// API IDs, not just the generic keys below) were both fully retired — not
|
|
4
|
+
// just old snapshots, they 404 on the live API. Current lineup has no cheap
|
|
5
|
+
// tier at all; grok-4.20 (generic) now targets grok-4.20-0309-non-reasoning
|
|
6
|
+
// and grok-4-1-fast (generic) targets grok-4.3, both $1.25/$2.50/1M — see
|
|
7
|
+
// intercept.ts MODEL_API_NAMES. There is currently no xAI model cheaper than
|
|
8
|
+
// $1.25/M input, so routing between these two generic keys yields zero
|
|
9
|
+
// savings (findCheapestRoute requires strictly cheaper — this is honest,
|
|
10
|
+
// not a bug: the old $0.20/M "fast" tier no longer exists).
|
|
2
11
|
export const MODELS = {
|
|
3
12
|
// Anthropic
|
|
4
13
|
'claude-opus': { input: 5.00, output: 25.00, context: 1000000 },
|
|
@@ -6,8 +15,8 @@ export const MODELS = {
|
|
|
6
15
|
'claude-sonnet': { input: 3.00, output: 15.00, context: 1000000 },
|
|
7
16
|
'claude-haiku': { input: 1.00, output: 5.00, context: 200000 },
|
|
8
17
|
// xAI
|
|
9
|
-
'grok-4.20': { input:
|
|
10
|
-
'grok-4-1-fast': { input:
|
|
18
|
+
'grok-4.20': { input: 1.25, output: 2.50, context: 1000000 },
|
|
19
|
+
'grok-4-1-fast': { input: 1.25, output: 2.50, context: 1000000 },
|
|
11
20
|
// Google
|
|
12
21
|
'gemini-3.1-pro': { input: 2.00, output: 12.00, context: 1000000 },
|
|
13
22
|
'gemini-2.5-flash': { input: 0.30, output: 2.50, context: 1000000 },
|
package/dist/preflight.js
CHANGED
|
@@ -34,7 +34,7 @@ function buildAlternative(model, originalCost, tokens, info) {
|
|
|
34
34
|
* entries (e.g. given model='claude-opus', options[0]?.model CAN be 'grok-...').
|
|
35
35
|
*/
|
|
36
36
|
export function preflight(content, model) {
|
|
37
|
-
const tokens = slash(content);
|
|
37
|
+
const tokens = slash(content, model);
|
|
38
38
|
const info = getModel(model);
|
|
39
39
|
if (!info) {
|
|
40
40
|
throw new Error(`Unknown model: "${model}". Available: ${Object.keys(MODELS).join(', ')}`);
|
|
@@ -74,7 +74,7 @@ export function preflight(content, model) {
|
|
|
74
74
|
* both functions return the same model name.
|
|
75
75
|
*/
|
|
76
76
|
export function preflightRoute(content, model) {
|
|
77
|
-
const tokens = slash(content);
|
|
77
|
+
const tokens = slash(content, model);
|
|
78
78
|
const info = getModel(model);
|
|
79
79
|
if (!info)
|
|
80
80
|
return null;
|
package/dist/slash.js
CHANGED
|
@@ -2,28 +2,83 @@ import { getInstance, writeToMemory } from './wasm.js';
|
|
|
2
2
|
const WASM_INPUT_OFFSET = 4096;
|
|
3
3
|
/**
|
|
4
4
|
* Per-model calibration factors.
|
|
5
|
-
* WASM estimator was tuned for Opus 4.6. Newer tokenizers may drift.
|
|
6
5
|
* Factor > 1.0 means "model uses more tokens than WASM predicts."
|
|
7
6
|
* Applied as: estimate = wasm_estimate * factor (rounded up).
|
|
8
7
|
*
|
|
9
8
|
* Default 1.0 = no adjustment. Update after running bench/calibrate.ts.
|
|
10
|
-
*/
|
|
11
|
-
/**
|
|
12
|
-
* Calibration factors derived from bench/calibrate.ts (2026-04-16).
|
|
13
|
-
* WASM under-reports 4.6 by ~12% median. 4.7 tokenizer uses 1.29x more
|
|
14
|
-
* tokens than 4.6 median. Combined correction with safety margin:
|
|
15
9
|
*
|
|
16
|
-
*
|
|
17
|
-
*
|
|
10
|
+
* Re-derived 2026-08-23 from a fresh benchmark against real Anthropic
|
|
11
|
+
* count_tokens ground truth (the 2026-04-16 factors were measured
|
|
12
|
+
* against claude-opus-4-20250514, since retired — that baseline no
|
|
13
|
+
* longer exists to verify against). Factor chosen as
|
|
14
|
+
* 1 / min(observed_ratio) with a small safety margin, so it does not
|
|
15
|
+
* under-report even against the worst sample in this benchmark:
|
|
16
|
+
*
|
|
17
|
+
* opus-4.7: 1.85 (was 1.50 — that value was insufficient: min observed
|
|
18
|
+
* ratio 0.571 needs >=1.75 just to avoid under-report, before
|
|
19
|
+
* any margin. Previously the factor also went UNUSED in
|
|
20
|
+
* production — see intercept.ts/preflight.ts history, fixed
|
|
21
|
+
* 2026-08-23 same day)
|
|
22
|
+
* claude-opus (generic, now routed to claude-opus-5): 1.85 — carried
|
|
23
|
+
* over from opus-4.7's measurement as the closest tested
|
|
24
|
+
* sibling; opus-5 itself has not been directly benchmarked
|
|
25
|
+
* claude-sonnet (routed to claude-sonnet-5): 1.85 — measured directly,
|
|
26
|
+
* nearly identical drift to opus-4.7 (min ratio 0.580)
|
|
27
|
+
* claude-haiku (routed to claude-haiku-4.5): 1.40 — measured directly,
|
|
28
|
+
* drifts less than the two above (min ratio 0.762)
|
|
29
|
+
* gemini-3.1-pro / gemini-2.5-flash: 1.40 — measured 2026-08-23 against
|
|
30
|
+
* Google's free countTokens endpoint via the gemini-pro-latest
|
|
31
|
+
* / gemini-flash-latest aliases (min ratio 0.768, identical
|
|
32
|
+
* for both — pro/flash share one tokenizer this generation).
|
|
33
|
+
* Note: 'gemini-3.1-pro' as a literal model ID does not exist
|
|
34
|
+
* on the live API (404) — same stale-hardcoded-ID class of bug
|
|
35
|
+
* as the retired Claude snapshots, caught the same day. See
|
|
36
|
+
* intercept.ts MODEL_API_NAMES: the wire name is now the
|
|
37
|
+
* '-latest' alias, which sidesteps this whole bug class going
|
|
38
|
+
* forward since Google repoints it, not us.
|
|
39
|
+
* grok-4.20 / grok-4-1-fast: 1.15 — measured 2026-08-23 (min ratio 0.928,
|
|
40
|
+
* the least drift of any provider tested — Grok's tokenizer
|
|
41
|
+
* runs fewer tokens per content than the others, so raw WASM
|
|
42
|
+
* already over-reports on most samples). xAI has no free
|
|
43
|
+
* count-tokens endpoint; this used real chat completions
|
|
44
|
+
* (max_tokens: 1) with a per-model baseline subtracted to
|
|
45
|
+
* remove xAI's fixed system-preamble overhead (~185-193
|
|
46
|
+
* tokens, mostly cached) from the content-only count.
|
|
47
|
+
* Both literal API IDs ('grok-4.20', 'grok-4-1-fast') are
|
|
48
|
+
* fully retired — not just old snapshots, gone entirely —
|
|
49
|
+
* remapped to grok-4.20-0309-non-reasoning / grok-4.3 in
|
|
50
|
+
* intercept.ts MODEL_API_NAMES the same day this was found.
|
|
51
|
+
*
|
|
52
|
+
* Known limitation: derived from a 9-sample benchmark corpus, several
|
|
53
|
+
* of which are slash-tokens' own code/docs (not representative
|
|
54
|
+
* third-party content). Re-run bench/calibrate.ts with a larger,
|
|
55
|
+
* more diverse corpus before treating these as final.
|
|
18
56
|
*
|
|
19
57
|
* Slash must NEVER under-report. Over-reporting is safe (go/no-go only).
|
|
20
58
|
*/
|
|
21
59
|
const CALIBRATION = {
|
|
22
|
-
'claude-opus': 1.
|
|
23
|
-
'claude-opus-4.7': 1.
|
|
24
|
-
'claude-sonnet': 1.
|
|
25
|
-
'claude-haiku': 1.
|
|
60
|
+
'claude-opus': 1.85,
|
|
61
|
+
'claude-opus-4.7': 1.85,
|
|
62
|
+
'claude-sonnet': 1.85,
|
|
63
|
+
'claude-haiku': 1.40,
|
|
64
|
+
'gemini-3.1-pro': 1.40,
|
|
65
|
+
'gemini-2.5-flash': 1.40,
|
|
66
|
+
'grok-4.20': 1.15,
|
|
67
|
+
'grok-4-1-fast': 1.15,
|
|
26
68
|
};
|
|
69
|
+
/**
|
|
70
|
+
* Fallback factor for any model not in CALIBRATION — new model IDs,
|
|
71
|
+
* providers we haven't benchmarked yet, anything unrecognized.
|
|
72
|
+
*
|
|
73
|
+
* Deliberately the highest known-safe factor, not 1.0. "Unknown" must
|
|
74
|
+
* never mean "no correction" — that's the most dangerous case, worse
|
|
75
|
+
* than any calibrated model, because it silently assumes a brand-new
|
|
76
|
+
* tokenizer behaves exactly like the raw WASM heuristic with zero
|
|
77
|
+
* evidence either way. Slash must NEVER under-report; for a model we
|
|
78
|
+
* haven't measured, the only safe assumption is the worst one we've
|
|
79
|
+
* actually observed (see CALIBRATION comment above).
|
|
80
|
+
*/
|
|
81
|
+
const DEFAULT_UNKNOWN_MODEL_FACTOR = 1.85;
|
|
27
82
|
/**
|
|
28
83
|
* Estimate token count for a string.
|
|
29
84
|
* Sub-millisecond. Zero allocations in WASM.
|
|
@@ -37,7 +92,7 @@ export function slash(content, model) {
|
|
|
37
92
|
const raw = instance.exports.estimate_tokens(WASM_INPUT_OFFSET, len);
|
|
38
93
|
if (!model)
|
|
39
94
|
return raw;
|
|
40
|
-
const factor = CALIBRATION[model] ??
|
|
95
|
+
const factor = CALIBRATION[model] ?? DEFAULT_UNKNOWN_MODEL_FACTOR;
|
|
41
96
|
return factor === 1.0 ? raw : Math.ceil(raw * factor);
|
|
42
97
|
}
|
|
43
98
|
/**
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "slash-tokens",
|
|
3
|
-
"version": "1.
|
|
3
|
+
"version": "1.5.0",
|
|
4
4
|
"description": "Token Optimization for Context Engineers. 4.8 KB WASM. Sub-millisecond. Zero dependencies.",
|
|
5
5
|
"main": "dist/index.js",
|
|
6
6
|
"module": "dist/index.js",
|
|
@@ -50,6 +50,7 @@
|
|
|
50
50
|
"homepage": "https://slashtokens.com",
|
|
51
51
|
"devDependencies": {
|
|
52
52
|
"@types/node": "^25.5.0",
|
|
53
|
+
"js-tiktoken": "^1.0.21",
|
|
53
54
|
"typescript": "^6.0.2"
|
|
54
55
|
}
|
|
55
56
|
}
|