slash-tokens 1.5.0 → 1.5.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.js CHANGED
@@ -41,16 +41,19 @@ function writeToMemory(content) {
41
41
  // src/slash.ts
42
42
  var WASM_INPUT_OFFSET2 = 4096;
43
43
  var CALIBRATION = {
44
- "claude-opus": 1.85,
45
- "claude-opus-4.7": 1.85,
46
- "claude-sonnet": 1.85,
47
- "claude-haiku": 1.4,
48
- "gemini-3.1-pro": 1.4,
49
- "gemini-2.5-flash": 1.4,
44
+ "claude-opus": 2.05,
45
+ "claude-opus-4.7": 2.05,
46
+ "claude-sonnet": 2.05,
47
+ "claude-haiku": 1.45,
48
+ "gemini-3.1-pro": 1.45,
49
+ "gemini-2.5-flash": 1.45,
50
50
  "grok-4.20": 1.15,
51
- "grok-4-1-fast": 1.15
51
+ "grok-4-1-fast": 1.15,
52
+ "gpt-5.4": 1.15,
53
+ "gpt-5.4-mini": 1.15,
54
+ "gpt-5.4-nano": 1.15
52
55
  };
53
- var DEFAULT_UNKNOWN_MODEL_FACTOR = 1.85;
56
+ var DEFAULT_UNKNOWN_MODEL_FACTOR = 2.05;
54
57
  function slash(content, model) {
55
58
  if (!content)
56
59
  return 0;
package/dist/slash.js CHANGED
@@ -14,57 +14,94 @@ const WASM_INPUT_OFFSET = 4096;
14
14
  * 1 / min(observed_ratio) with a small safety margin, so it does not
15
15
  * under-report even against the worst sample in this benchmark:
16
16
  *
17
- * opus-4.7: 1.85 (was 1.50 — that value was insufficient: min observed
18
- * ratio 0.571 needs >=1.75 just to avoid under-report, before
19
- * any margin. Previously the factor also went UNUSED in
20
- * production — see intercept.ts/preflight.ts history, fixed
21
- * 2026-08-23 same day)
22
- * claude-opus (generic, now routed to claude-opus-5): 1.85 — carried
17
+ * SUPERSEDED same day: the first pass (1.85 / 1.85 / 1.40 below) was
18
+ * measured against a 9-sample corpus with no non-English text and no
19
+ * SQL/bash. Expanding the corpus to 29 samples (still 2026-08-23) found
20
+ * a WORSE case for every Claude model — Spanish prose (ratio 0.517) and
21
+ * SQL/bash syntax (~0.53) drift far more than the original English/JS
22
+ * corpus ever showed. 1.85 was live in production (v1.5.0) as an
23
+ * insufficient factor for real Spanish-language and SQL/bash traffic
24
+ * until this second pass caught it — the exact failure mode this whole
25
+ * exercise exists to prevent. Re-verify again if the corpus grows further.
26
+ *
27
+ * opus-4.7: 2.05 (was 1.85, was 1.50 before that — min observed ratio
28
+ * now 0.517 needs >=1.934 before margin, found on Spanish
29
+ * prose, not on any sample in the original 9-item corpus)
30
+ * claude-opus (generic, now routed to claude-opus-5): 2.05 — carried
23
31
  * over from opus-4.7's measurement as the closest tested
24
32
  * sibling; opus-5 itself has not been directly benchmarked
25
- * claude-sonnet (routed to claude-sonnet-5): 1.85 — measured directly,
26
- * nearly identical drift to opus-4.7 (min ratio 0.580)
27
- * claude-haiku (routed to claude-haiku-4.5): 1.40 — measured directly,
28
- * drifts less than the two above (min ratio 0.762)
29
- * gemini-3.1-pro / gemini-2.5-flash: 1.40 — measured 2026-08-23 against
30
- * Google's free countTokens endpoint via the gemini-pro-latest
31
- * / gemini-flash-latest aliases (min ratio 0.768, identical
32
- * for both — pro/flash share one tokenizer this generation).
33
+ * claude-sonnet (routed to claude-sonnet-5): 2.05 — measured directly,
34
+ * nearly identical drift to opus-4.7 (min ratio 0.525, same
35
+ * Spanish-prose sample)
36
+ * claude-haiku (routed to claude-haiku-4.5): 1.45 (was 1.40) — drifts
37
+ * less than the two above but still worsened under the
38
+ * bigger corpus (min ratio 0.744, was 0.762)
39
+ * gemini-3.1-pro / gemini-2.5-flash: 1.45 (was 1.40) — re-verified
40
+ * 2026-08-23 against the 29-sample corpus via Google's free
41
+ * countTokens endpoint (gemini-pro-latest / gemini-flash-latest
42
+ * aliases; identical ratios for both — pro/flash share one
43
+ * tokenizer this generation). Worst case shifted from the
44
+ * original 9-sample corpus: now graphql-response JSON (ratio
45
+ * 0.749, needs >=1.335), not the prose sample that mattered
46
+ * before. 1.40 was technically still safe here (unlike
47
+ * Claude's factor, which actually failed) but at the thinnest
48
+ * margin of any provider (4.9%) — bumped to match the ~6-8%
49
+ * band used everywhere else rather than leave the one factor
50
+ * most likely to fail next time this corpus grows again.
33
51
  * Note: 'gemini-3.1-pro' as a literal model ID does not exist
34
52
  * on the live API (404) — same stale-hardcoded-ID class of bug
35
53
  * as the retired Claude snapshots, caught the same day. See
36
54
  * intercept.ts MODEL_API_NAMES: the wire name is now the
37
55
  * '-latest' alias, which sidesteps this whole bug class going
38
56
  * forward since Google repoints it, not us.
39
- * grok-4.20 / grok-4-1-fast: 1.15 — measured 2026-08-23 (min ratio 0.928,
40
- * the least drift of any provider tested — Grok's tokenizer
41
- * runs fewer tokens per content than the others, so raw WASM
42
- * already over-reports on most samples). xAI has no free
43
- * count-tokens endpoint; this used real chat completions
44
- * (max_tokens: 1) with a per-model baseline subtracted to
45
- * remove xAI's fixed system-preamble overhead (~185-193
46
- * tokens, mostly cached) from the content-only count.
57
+ * gpt-5.4 / gpt-5.4-mini / gpt-5.4-nano: 1.15 — measured 2026-08-23
58
+ * against js-tiktoken's o200k_base (free, local, exact — same
59
+ * encoding for the whole GPT-5.4 family per OpenAI's own
60
+ * tiktoken docs). Min ratio 0.923 against the expanded
61
+ * 29-sample corpus. This was a real gap, not just an accuracy
62
+ * tweak: GPT had a calibration SCRIPT (bench/calibrate-openai.ts)
63
+ * from the same day as the Claude fixes, but was never actually
64
+ * added to this table — it was silently falling through to
65
+ * DEFAULT_UNKNOWN_MODEL_FACTOR (1.85), a ~60% larger correction
66
+ * than it needed. Still safe either way (1.85 > required
67
+ * minimum), just needlessly inflated for real GPT users.
68
+ * grok-4.20 / grok-4-1-fast: 1.15 — re-verified 2026-08-23 against the
69
+ * 29-sample corpus (20 new samples: more languages, Spanish/
70
+ * Japanese prose, more JSON shapes) and UNCHANGED — same
71
+ * worst case (technical-docs prose, ratio 0.928) as the
72
+ * original 9-sample measurement. The only one of the three
73
+ * non-GPT providers whose factor held without a bump; Claude
74
+ * and Gemini both needed one, on different content types each
75
+ * (Spanish prose, then JSON). xAI has no free count-tokens
76
+ * endpoint; this used real chat completions (max_tokens: 1)
77
+ * with a per-model baseline subtracted to remove xAI's fixed
78
+ * system-preamble overhead (~185-193 tokens, mostly cached)
79
+ * from the content-only count.
47
80
  * Both literal API IDs ('grok-4.20', 'grok-4-1-fast') are
48
81
  * fully retired — not just old snapshots, gone entirely —
49
82
  * remapped to grok-4.20-0309-non-reasoning / grok-4.3 in
50
83
  * intercept.ts MODEL_API_NAMES the same day this was found.
51
84
  *
52
- * Known limitation: derived from a 9-sample benchmark corpus, several
53
- * of which are slash-tokens' own code/docs (not representative
54
- * third-party content). Re-run bench/calibrate.ts with a larger,
55
- * more diverse corpus before treating these as final.
85
+ * All four providers (Claude, Gemini, Grok, GPT) are now verified against
86
+ * the same 29-sample corpus (expanded 2026-08-23 from the original 9,
87
+ * which was mostly slash-tokens' own code/docs — not representative
88
+ * third-party content). Re-verify again if the corpus grows further —
89
+ * Claude and Gemini both needed real bumps the first time it grew.
56
90
  *
57
91
  * Slash must NEVER under-report. Over-reporting is safe (go/no-go only).
58
92
  */
59
93
  const CALIBRATION = {
60
- 'claude-opus': 1.85,
61
- 'claude-opus-4.7': 1.85,
62
- 'claude-sonnet': 1.85,
63
- 'claude-haiku': 1.40,
64
- 'gemini-3.1-pro': 1.40,
65
- 'gemini-2.5-flash': 1.40,
94
+ 'claude-opus': 2.05,
95
+ 'claude-opus-4.7': 2.05,
96
+ 'claude-sonnet': 2.05,
97
+ 'claude-haiku': 1.45,
98
+ 'gemini-3.1-pro': 1.45,
99
+ 'gemini-2.5-flash': 1.45,
66
100
  'grok-4.20': 1.15,
67
101
  'grok-4-1-fast': 1.15,
102
+ 'gpt-5.4': 1.15,
103
+ 'gpt-5.4-mini': 1.15,
104
+ 'gpt-5.4-nano': 1.15,
68
105
  };
69
106
  /**
70
107
  * Fallback factor for any model not in CALIBRATION — new model IDs,
@@ -78,7 +115,7 @@ const CALIBRATION = {
78
115
  * haven't measured, the only safe assumption is the worst one we've
79
116
  * actually observed (see CALIBRATION comment above).
80
117
  */
81
- const DEFAULT_UNKNOWN_MODEL_FACTOR = 1.85;
118
+ const DEFAULT_UNKNOWN_MODEL_FACTOR = 2.05;
82
119
  /**
83
120
  * Estimate token count for a string.
84
121
  * Sub-millisecond. Zero allocations in WASM.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "slash-tokens",
3
- "version": "1.5.0",
3
+ "version": "1.5.2",
4
4
  "description": "Token Optimization for Context Engineers. 4.8 KB WASM. Sub-millisecond. Zero dependencies.",
5
5
  "main": "dist/index.js",
6
6
  "module": "dist/index.js",