slash-tokens 1.5.0 → 1.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli.js +11 -8
- package/dist/slash.js +70 -33
- package/package.json +1 -1
package/dist/cli.js
CHANGED
|
@@ -41,16 +41,19 @@ function writeToMemory(content) {
|
|
|
41
41
|
// src/slash.ts
|
|
42
42
|
var WASM_INPUT_OFFSET2 = 4096;
|
|
43
43
|
var CALIBRATION = {
|
|
44
|
-
"claude-opus":
|
|
45
|
-
"claude-opus-4.7":
|
|
46
|
-
"claude-sonnet":
|
|
47
|
-
"claude-haiku": 1.
|
|
48
|
-
"gemini-3.1-pro": 1.
|
|
49
|
-
"gemini-2.5-flash": 1.
|
|
44
|
+
"claude-opus": 2.05,
|
|
45
|
+
"claude-opus-4.7": 2.05,
|
|
46
|
+
"claude-sonnet": 2.05,
|
|
47
|
+
"claude-haiku": 1.45,
|
|
48
|
+
"gemini-3.1-pro": 1.45,
|
|
49
|
+
"gemini-2.5-flash": 1.45,
|
|
50
50
|
"grok-4.20": 1.15,
|
|
51
|
-
"grok-4-1-fast": 1.15
|
|
51
|
+
"grok-4-1-fast": 1.15,
|
|
52
|
+
"gpt-5.4": 1.15,
|
|
53
|
+
"gpt-5.4-mini": 1.15,
|
|
54
|
+
"gpt-5.4-nano": 1.15
|
|
52
55
|
};
|
|
53
|
-
var DEFAULT_UNKNOWN_MODEL_FACTOR =
|
|
56
|
+
var DEFAULT_UNKNOWN_MODEL_FACTOR = 2.05;
|
|
54
57
|
function slash(content, model) {
|
|
55
58
|
if (!content)
|
|
56
59
|
return 0;
|
package/dist/slash.js
CHANGED
|
@@ -14,57 +14,94 @@ const WASM_INPUT_OFFSET = 4096;
|
|
|
14
14
|
* 1 / min(observed_ratio) with a small safety margin, so it does not
|
|
15
15
|
* under-report even against the worst sample in this benchmark:
|
|
16
16
|
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
*
|
|
21
|
-
*
|
|
22
|
-
*
|
|
17
|
+
* SUPERSEDED same day: the first pass (1.85 / 1.85 / 1.40 below) was
|
|
18
|
+
* measured against a 9-sample corpus with no non-English text and no
|
|
19
|
+
* SQL/bash. Expanding the corpus to 29 samples (still 2026-08-23) found
|
|
20
|
+
* a WORSE case for every Claude model — Spanish prose (ratio 0.517) and
|
|
21
|
+
* SQL/bash syntax (~0.53) drift far more than the original English/JS
|
|
22
|
+
* corpus ever showed. 1.85 was live in production (v1.5.0) as an
|
|
23
|
+
* insufficient factor for real Spanish-language and SQL/bash traffic
|
|
24
|
+
* until this second pass caught it — the exact failure mode this whole
|
|
25
|
+
* exercise exists to prevent. Re-verify again if the corpus grows further.
|
|
26
|
+
*
|
|
27
|
+
* opus-4.7: 2.05 (was 1.85, was 1.50 before that — min observed ratio
|
|
28
|
+
* now 0.517 needs >=1.934 before margin, found on Spanish
|
|
29
|
+
* prose, not on any sample in the original 9-item corpus)
|
|
30
|
+
* claude-opus (generic, now routed to claude-opus-5): 2.05 — carried
|
|
23
31
|
* over from opus-4.7's measurement as the closest tested
|
|
24
32
|
* sibling; opus-5 itself has not been directly benchmarked
|
|
25
|
-
* claude-sonnet (routed to claude-sonnet-5):
|
|
26
|
-
* nearly identical drift to opus-4.7 (min ratio 0.
|
|
27
|
-
*
|
|
28
|
-
*
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
33
|
+
* claude-sonnet (routed to claude-sonnet-5): 2.05 — measured directly,
|
|
34
|
+
* nearly identical drift to opus-4.7 (min ratio 0.525, same
|
|
35
|
+
* Spanish-prose sample)
|
|
36
|
+
* claude-haiku (routed to claude-haiku-4.5): 1.45 (was 1.40) — drifts
|
|
37
|
+
* less than the two above but still worsened under the
|
|
38
|
+
* bigger corpus (min ratio 0.744, was 0.762)
|
|
39
|
+
* gemini-3.1-pro / gemini-2.5-flash: 1.45 (was 1.40) — re-verified
|
|
40
|
+
* 2026-08-23 against the 29-sample corpus via Google's free
|
|
41
|
+
* countTokens endpoint (gemini-pro-latest / gemini-flash-latest
|
|
42
|
+
* aliases; identical ratios for both — pro/flash share one
|
|
43
|
+
* tokenizer this generation). Worst case shifted from the
|
|
44
|
+
* original 9-sample corpus: now graphql-response JSON (ratio
|
|
45
|
+
* 0.749, needs >=1.335), not the prose sample that mattered
|
|
46
|
+
* before. 1.40 was technically still safe here (unlike
|
|
47
|
+
* Claude's factor, which actually failed) but at the thinnest
|
|
48
|
+
* margin of any provider (4.9%) — bumped to match the ~6-8%
|
|
49
|
+
* band used everywhere else rather than leave the one factor
|
|
50
|
+
* most likely to fail next time this corpus grows again.
|
|
33
51
|
* Note: 'gemini-3.1-pro' as a literal model ID does not exist
|
|
34
52
|
* on the live API (404) — same stale-hardcoded-ID class of bug
|
|
35
53
|
* as the retired Claude snapshots, caught the same day. See
|
|
36
54
|
* intercept.ts MODEL_API_NAMES: the wire name is now the
|
|
37
55
|
* '-latest' alias, which sidesteps this whole bug class going
|
|
38
56
|
* forward since Google repoints it, not us.
|
|
39
|
-
*
|
|
40
|
-
*
|
|
41
|
-
*
|
|
42
|
-
*
|
|
43
|
-
*
|
|
44
|
-
*
|
|
45
|
-
*
|
|
46
|
-
*
|
|
57
|
+
* gpt-5.4 / gpt-5.4-mini / gpt-5.4-nano: 1.15 — measured 2026-08-23
|
|
58
|
+
* against js-tiktoken's o200k_base (free, local, exact — same
|
|
59
|
+
* encoding for the whole GPT-5.4 family per OpenAI's own
|
|
60
|
+
* tiktoken docs). Min ratio 0.923 against the expanded
|
|
61
|
+
* 29-sample corpus. This was a real gap, not just an accuracy
|
|
62
|
+
* tweak: GPT had a calibration SCRIPT (bench/calibrate-openai.ts)
|
|
63
|
+
* from the same day as the Claude fixes, but was never actually
|
|
64
|
+
* added to this table — it was silently falling through to
|
|
65
|
+
* DEFAULT_UNKNOWN_MODEL_FACTOR (1.85), a ~60% larger correction
|
|
66
|
+
* than it needed. Still safe either way (1.85 > required
|
|
67
|
+
* minimum), just needlessly inflated for real GPT users.
|
|
68
|
+
* grok-4.20 / grok-4-1-fast: 1.15 — re-verified 2026-08-23 against the
|
|
69
|
+
* 29-sample corpus (20 new samples: more languages, Spanish/
|
|
70
|
+
* Japanese prose, more JSON shapes) and UNCHANGED — same
|
|
71
|
+
* worst case (technical-docs prose, ratio 0.928) as the
|
|
72
|
+
* original 9-sample measurement. The only one of the three
|
|
73
|
+
* non-GPT providers whose factor held without a bump; Claude
|
|
74
|
+
* and Gemini both needed one, on different content types each
|
|
75
|
+
* (Spanish prose, then JSON). xAI has no free count-tokens
|
|
76
|
+
* endpoint; this used real chat completions (max_tokens: 1)
|
|
77
|
+
* with a per-model baseline subtracted to remove xAI's fixed
|
|
78
|
+
* system-preamble overhead (~185-193 tokens, mostly cached)
|
|
79
|
+
* from the content-only count.
|
|
47
80
|
* Both literal API IDs ('grok-4.20', 'grok-4-1-fast') are
|
|
48
81
|
* fully retired — not just old snapshots, gone entirely —
|
|
49
82
|
* remapped to grok-4.20-0309-non-reasoning / grok-4.3 in
|
|
50
83
|
* intercept.ts MODEL_API_NAMES the same day this was found.
|
|
51
84
|
*
|
|
52
|
-
*
|
|
53
|
-
*
|
|
54
|
-
*
|
|
55
|
-
*
|
|
85
|
+
* All four providers (Claude, Gemini, Grok, GPT) are now verified against
|
|
86
|
+
* the same 29-sample corpus (expanded 2026-08-23 from the original 9,
|
|
87
|
+
* which was mostly slash-tokens' own code/docs — not representative
|
|
88
|
+
* third-party content). Re-verify again if the corpus grows further —
|
|
89
|
+
* Claude and Gemini both needed real bumps the first time it grew.
|
|
56
90
|
*
|
|
57
91
|
* Slash must NEVER under-report. Over-reporting is safe (go/no-go only).
|
|
58
92
|
*/
|
|
59
93
|
const CALIBRATION = {
|
|
60
|
-
'claude-opus':
|
|
61
|
-
'claude-opus-4.7':
|
|
62
|
-
'claude-sonnet':
|
|
63
|
-
'claude-haiku': 1.
|
|
64
|
-
'gemini-3.1-pro': 1.
|
|
65
|
-
'gemini-2.5-flash': 1.
|
|
94
|
+
'claude-opus': 2.05,
|
|
95
|
+
'claude-opus-4.7': 2.05,
|
|
96
|
+
'claude-sonnet': 2.05,
|
|
97
|
+
'claude-haiku': 1.45,
|
|
98
|
+
'gemini-3.1-pro': 1.45,
|
|
99
|
+
'gemini-2.5-flash': 1.45,
|
|
66
100
|
'grok-4.20': 1.15,
|
|
67
101
|
'grok-4-1-fast': 1.15,
|
|
102
|
+
'gpt-5.4': 1.15,
|
|
103
|
+
'gpt-5.4-mini': 1.15,
|
|
104
|
+
'gpt-5.4-nano': 1.15,
|
|
68
105
|
};
|
|
69
106
|
/**
|
|
70
107
|
* Fallback factor for any model not in CALIBRATION — new model IDs,
|
|
@@ -78,7 +115,7 @@ const CALIBRATION = {
|
|
|
78
115
|
* haven't measured, the only safe assumption is the worst one we've
|
|
79
116
|
* actually observed (see CALIBRATION comment above).
|
|
80
117
|
*/
|
|
81
|
-
const DEFAULT_UNKNOWN_MODEL_FACTOR =
|
|
118
|
+
const DEFAULT_UNKNOWN_MODEL_FACTOR = 2.05;
|
|
82
119
|
/**
|
|
83
120
|
* Estimate token count for a string.
|
|
84
121
|
* Sub-millisecond. Zero allocations in WASM.
|