@byokit/usage 0.1.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +20 -0
- package/README.md +257 -11
- package/dist/backoff.d.ts +7 -0
- package/dist/backoff.js +26 -0
- package/dist/calls.d.ts +68 -0
- package/dist/calls.js +131 -0
- package/dist/index.d.ts +6 -0
- package/dist/index.js +216 -57
- package/dist/ledger.d.ts +49 -0
- package/dist/ledger.js +87 -0
- package/dist/providers.d.ts +38 -9
- package/dist/providers.js +127 -14
- package/dist/quota.d.ts +8 -0
- package/dist/quota.js +91 -0
- package/dist/room.d.ts +3 -0
- package/dist/room.js +25 -0
- package/dist/store.d.ts +12 -9
- package/dist/store.js +41 -10
- package/dist/types.d.ts +114 -11
- package/dist/windows.d.ts +5 -2
- package/dist/windows.js +42 -12
- package/dist/words.json +2 -1
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,26 @@
|
|
|
2
2
|
|
|
3
3
|
## Unreleased
|
|
4
4
|
|
|
5
|
+
## 0.3.0 (2026-10-01)
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
- FIX: Subscription hard-limit flags now override positive percentage room, including blocks without quota windows; cached reset times never clear them.
|
|
10
|
+
- FIX: Claude subscription normalized and scoped quota rows now take precedence over legacy aggregates; missing usage remains unknown.
|
|
11
|
+
- Expose observation age and poll outcomes separately; retain last-good figures through failed polls without changing account health.
|
|
12
|
+
- Add cancellable host origin pacing and account-scoped retry policies for rate limits, refresh failures and transient outcomes.
|
|
13
|
+
- Document reset-time units when passing normalized subscription readings to account selection.
|
|
14
|
+
|
|
15
|
+
## 0.2.0 (2026-09-30)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
- Export `fingerprint`, `fileUsageStore`, `memoryBackoffPolicy`, `retryAfterMs` and `backoffDelayMs` for host subscription usage adapters.
|
|
20
|
+
- Add token quota sources for Claude, Codex, Copilot, Grok, MiniMax, Gemini and Kimi, with BYOKit's own user agent and no credential discovery.
|
|
21
|
+
- Add `callLedger`, `normalizeTokens` and `priceCall` for attributed model calls, reported/partial/unknown counts and estimates from app price tables only, on the member ledger store seam.
|
|
22
|
+
- Add a member-scoped token ledger with local-day totals, trailing seven-day caps and an injectable store (in-memory by default).
|
|
23
|
+
- Add `roomOf`; reset timestamps now use epoch milliseconds.
|
|
24
|
+
- Add host Claude reader, last-good store and 429 backoff hooks. Persist only normalized quota fields and fingerprint non-secret identities.
|
|
5
25
|
- Read subscription usage windows per provider and per account, with last-good readings and bounded provider reads.
|
|
6
26
|
|
|
7
27
|
## 0.1.0
|
package/README.md
CHANGED
|
@@ -1,25 +1,271 @@
|
|
|
1
1
|
# @byokit/usage
|
|
2
2
|
|
|
3
|
-
Read subscription
|
|
3
|
+
Read subscription quota windows per provider and per account on Node 22.18 or later.
|
|
4
|
+
The app owns sign-in, token renewal, account labels and selection. The kit reads room
|
|
5
|
+
left, estimates no cost and never rotates an account.
|
|
4
6
|
|
|
5
7
|
```ts
|
|
6
|
-
import { usage } from '@byokit/usage';
|
|
8
|
+
import { usage, roomOf } from '@byokit/usage';
|
|
7
9
|
const reader = usage({ stateDir: '/app/state/usage' });
|
|
8
|
-
|
|
10
|
+
// Pass the token and account id from your app-owned sign-in store.
|
|
11
|
+
declare const token: string;
|
|
12
|
+
declare const account: { id: string };
|
|
13
|
+
const source = { provider: 'codex' as const, access: token, accountId: account.id };
|
|
9
14
|
const reading = await reader.read(source);
|
|
10
|
-
|
|
15
|
+
const room = roomOf(reading, Date.now());
|
|
11
16
|
```
|
|
12
17
|
|
|
13
|
-
|
|
18
|
+
Sources:
|
|
14
19
|
|
|
15
|
-
|
|
20
|
+
- `{ provider: 'codex', access, accountId }` reads ChatGPT's `wham/usage`.
|
|
21
|
+
- `{ provider: 'claude', access, accountUuid?, accountId? }` reads Claude plan usage.
|
|
22
|
+
- `{ provider: 'copilot' | 'grok' | 'minimax' | 'kimi', access, accountId? }` reads
|
|
23
|
+
the corresponding subscription quota. MiniMax's `access` is the plan key.
|
|
24
|
+
- `{ provider: 'gemini', access, project?, accountId? }` reads Code Assist quota.
|
|
25
|
+
Without a project, the kit discovers it through `loadCodeAssist` first.
|
|
26
|
+
- `{ provider: 'opencode' | 'zai', key, accountId? }` reads plan-key quota.
|
|
27
|
+
- `{ provider: 'codex', bin, home, env? }` retains the explicit local app-server source.
|
|
28
|
+
- `{ provider: 'claude', credentialsFile, configFile?, statuslineFile? }` is a
|
|
29
|
+
read-only adapter for files the app explicitly supplies. It reads `claudeAiOauth`
|
|
30
|
+
and optionally `oauthAccount.accountUuid`. A statusline snapshot uses its body `fetched_at` (epoch milliseconds or ISO date),
|
|
31
|
+
never file mtime. Known snapshots younger than five minutes precede the endpoint;
|
|
32
|
+
undated/future snapshots remain visible with unknown/future age. An expired token is never sent.
|
|
33
|
+
- `{ provider: 'claude', accountUuid, read, origin?, connected? }` delegates to the app's
|
|
34
|
+
reader. `read({ nowMs, signal })` returns `{ raw?, code?, retryAfterMs?, at?, limited? }` using
|
|
35
|
+
the same Claude payload dialect. The host supplies the actual observation `at`;
|
|
36
|
+
omitted means unknown age, including engine-cached figures. `limited: true` can
|
|
37
|
+
report an authoritative block without fabricating a window. `origin` is the
|
|
38
|
+
host reader's optional HTTP origin for pacing. It has a ten-second deadline, with the signal
|
|
39
|
+
aborted at expiry. The optional synchronous `connected()` hook controls whether
|
|
40
|
+
last-good readings remain visible; exceptions count as disconnected.
|
|
16
41
|
|
|
17
|
-
`read(source, { nowMs? })`
|
|
42
|
+
`read(source, { nowMs?, signal? })` returns `{ provider, windows, at?, limited?, poll?, code? }`.
|
|
43
|
+
`at` is source observation time; it is absent when unavailable. `poll` contains
|
|
44
|
+
last attempt time, `outcome` (`ok` or a safe failure code) and optional `retryAt`.
|
|
45
|
+
All timestamps are epoch milliseconds. A failed poll keeps the figures and their
|
|
46
|
+
original observation time. Poll 429 (`rate-limited`), host renewal failure
|
|
47
|
+
(`refresh-failed`) and credential refusals are distinct; usage never changes
|
|
48
|
+
account health or renews credentials.
|
|
18
49
|
|
|
19
|
-
|
|
50
|
+
Windows include kind, optional reported `usedPercent`, duration, reset, limit,
|
|
51
|
+
`limited` and `scope: { model?, surface? }`. Missing usage is unknown, never zero.
|
|
52
|
+
Claude `limits[]` session/weekly-all rows override corresponding legacy aggregates,
|
|
53
|
+
even if incomplete; dynamic weekly-scoped rows retain model and surface. Legacy
|
|
54
|
+
`five_hour`/`seven_day` are fallback for absent aggregate kinds. These are synthetic
|
|
55
|
+
contract fixtures; fresh live provider payload qualification has not been run.
|
|
20
56
|
|
|
21
|
-
|
|
57
|
+
Parsers are exported: `claudeWindows`, `codexWindows` (app-server),
|
|
58
|
+
`codexTokenWindows(raw, nowMs?)`, `codexHardLimit`, `goWindows`, `zaiWindows`,
|
|
59
|
+
`copilotWindows`, `grokWindows`, `minimaxWindows`, `geminiWindows`, `kimiWindows`.
|
|
60
|
+
Codex absolute reset takes precedence; relative seconds require a captured clock.
|
|
61
|
+
Hard flags do not replace the reported percentage or invent an absent window.
|
|
62
|
+
Hosts using exported Codex parsers must carry `codexHardLimit(raw)` into the reading
|
|
63
|
+
as `limited` to represent windowless blocks.
|
|
22
64
|
|
|
23
|
-
|
|
65
|
+
`roomOf(reading, nowMs)` returns `left`, observation `at?`, `ageMs?`, `freshness`
|
|
66
|
+
(`fresh`, `stale`, `future`, `unknown`), `poll?`, and the tightest row's `scope?`.
|
|
67
|
+
Numeric results include `span` and optional reset. Authoritative hard blocks give
|
|
68
|
+
zero eligibility room even without a percentage/window, with `limited: true`;
|
|
69
|
+
a predicted reset never clears a block. Otherwise undated/future/older-than-24h
|
|
70
|
+
readings and disconnected/expired/auth/no-plan readings give unknown room.
|
|
71
|
+
Incomplete windows cannot establish positive room; known exhaustion still stands.
|
|
72
|
+
Scope is conservative across all windows until hosts implement model demand for
|
|
73
|
+
every model/surface a run can use, including subagents and fallbacks.
|
|
74
|
+
Temporary poll failures may retain eligible last-good room and its age; a poll
|
|
75
|
+
failure itself never exhausts or moves an account. Auto's shared input contract is
|
|
76
|
+
`fixtures/conformance/usage-typescript.json` and runtime spec section 13.
|
|
24
77
|
|
|
25
|
-
|
|
78
|
+
`connected(source)` and `account(source)` are synchronous. `account` returns a salted
|
|
79
|
+
fingerprint of a non-secret host id, credential account UUID, token subject, or explicit
|
|
80
|
+
local folder identity. Token subjects provide cache identity only, never authentication.
|
|
81
|
+
For opaque tokens and plan keys, pass `accountId` to keep quota history across renewal.
|
|
82
|
+
Without it, reads still work in memory, but `account()` is undefined and no public or
|
|
83
|
+
disk store is used. Tokens never become persisted fingerprint inputs.
|
|
84
|
+
|
|
85
|
+
Good reads have no code. Bad sources throw `UsageError` (`code: 'bad-source'`); other
|
|
86
|
+
failures resolve codes without bodies or secrets. `lastKnown(source, { nowMs? })`
|
|
87
|
+
returns a connected account's last-good figures and most recent poll outcome.
|
|
88
|
+
Ordinary dated readings expire after 24 hours; authoritative blocks stay until a
|
|
89
|
+
new successful read replaces them. Undated figures remain available for display
|
|
90
|
+
with unknown room. Reads have a
|
|
91
|
+
60-second floor and concurrent deduplication per provider/account. Default 429
|
|
92
|
+
backoff honors Retry-After with a five-minute minimum. Unknown/transient and
|
|
93
|
+
refresh failures back off exponentially from one minute to one hour; these are
|
|
94
|
+
kit defaults, not universal provider policy. Failure counters are separate by
|
|
95
|
+
outcome and account, cleared by a successful quota poll. `UsageOptions.now` supplies
|
|
96
|
+
the default clock; a per-call clock overrides it. An injected `fetch` wins; otherwise
|
|
97
|
+
global fetch is resolved on each read.
|
|
98
|
+
|
|
99
|
+
The host may supply public synchronous persistence and backoff hooks:
|
|
100
|
+
|
|
101
|
+
```ts
|
|
102
|
+
import { usage, type UsageStore, type BackoffState } from '@byokit/usage';
|
|
103
|
+
// Replace these maps with your app's durable store to retain state across restart.
|
|
104
|
+
const appStore: UsageStore = {
|
|
105
|
+
get(provider, fingerprint) { return saved.get(`${provider}/${fingerprint}`); },
|
|
106
|
+
put(provider, fingerprint, reading) { saved.set(`${provider}/${fingerprint}`, reading); },
|
|
107
|
+
};
|
|
108
|
+
const saved = new Map<string, import('@byokit/usage').StoredReading>();
|
|
109
|
+
const appBackoff = new Map<string, BackoffState | number>();
|
|
110
|
+
const reader = usage({
|
|
111
|
+
store: {
|
|
112
|
+
get(provider, fingerprint) { return appStore.get(provider, fingerprint); },
|
|
113
|
+
put(provider, fingerprint, reading) { appStore.put(provider, fingerprint, reading); },
|
|
114
|
+
},
|
|
115
|
+
backoff: {
|
|
116
|
+
get(provider, fingerprint) { return appBackoff.get(`${provider}/${fingerprint}`); },
|
|
117
|
+
set(provider, fingerprint, untilMs, state) { appBackoff.set(`${provider}/${fingerprint}`, state ?? untilMs); },
|
|
118
|
+
delayMs(retryAfterMs, { outcome, failures }) { return Math.max(300_000, retryAfterMs ?? 0); },
|
|
119
|
+
},
|
|
120
|
+
});
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
`UsageStore` holds `{ at?, windows, limited?, poll? }`; only whitelisted normalized fields cross
|
|
124
|
+
this boundary. Exceptions from host hooks do not expose data or fail a provider read.
|
|
125
|
+
Internal backoff remains effective if a host backoff hook fails. The 60-second
|
|
126
|
+
minimum retry interval applies even if a policy selects a shorter delay. By default,
|
|
127
|
+
`stateDir` selects an atomic disk store (0700 directory, 0600 file, 256 KB cap), or
|
|
128
|
+
without `stateDir` an in-memory store is used. `memoryUsageStore()` is exported.
|
|
129
|
+
`store` overrides `stateDir`. Disk storage uses `plans-v2.json`, deliberately ignoring
|
|
130
|
+
old raw-payload stores so second-based and millisecond-based readings never mix.
|
|
131
|
+
The default salt is `byokit/usage/account`.
|
|
132
|
+
|
|
133
|
+
To replace a host's Claude quota adapter, pass the exact files the host already
|
|
134
|
+
selected. Reuse the backoff policy across collector instances:
|
|
135
|
+
|
|
136
|
+
```ts
|
|
137
|
+
import { usage, fileUsageStore, memoryBackoffPolicy, fingerprint } from '@byokit/usage';
|
|
138
|
+
const salt = 'my-app/usage/account';
|
|
139
|
+
const store = fileUsageStore('/app/state/usage');
|
|
140
|
+
const backoff = memoryBackoffPolicy();
|
|
141
|
+
const source = {
|
|
142
|
+
provider: 'claude' as const,
|
|
143
|
+
credentialsFile: '/app/sign-ins/claude/.credentials.json',
|
|
144
|
+
configFile: '/app/sign-ins/claude/.claude.json',
|
|
145
|
+
statuslineFile: '/app/sign-ins/claude/statusline.json',
|
|
146
|
+
};
|
|
147
|
+
const reader = usage({ store, backoff, salt });
|
|
148
|
+
const reading = await reader.read(source);
|
|
149
|
+
const lastGood = reader.lastKnown(source);
|
|
150
|
+
// For a host-owned non-secret account UUID, this matches reader.account(source).
|
|
151
|
+
declare const accountUuid: string;
|
|
152
|
+
const accountKey = fingerprint(salt)('claude', accountUuid);
|
|
153
|
+
const saved = store.get('claude', accountKey);
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
The kit replaces the credential/snapshot read, quota request, payload parsing,
|
|
157
|
+
account fingerprint, last-good disk writes and 429 rest tracking. The host selects
|
|
158
|
+
paths and displays the returned windows. A valid recent snapshot precedes the
|
|
159
|
+
request; the request uses the kit's own user agent. No credential refresh occurs.
|
|
160
|
+
Legacy raw stores require host migration into normalized millisecond readings;
|
|
161
|
+
they are never loaded automatically.
|
|
162
|
+
|
|
163
|
+
`fileUsageStore(absoluteStateDir)` exposes the same bounded atomic disk store as
|
|
164
|
+
`stateDir`; invalid directory paths throw `UsageError`. Its keys must be 64-character
|
|
165
|
+
hex fingerprints, and it persists normalized observations, hard blocks and poll metadata.
|
|
166
|
+
`fingerprint(salt)` returns `(provider, nonSecretIdentity) => string`; use the same
|
|
167
|
+
salt as the reader and never supply a token as the identity.
|
|
168
|
+
`memoryBackoffPolicy()` supplies shared per-provider/account rests, keeps the later
|
|
169
|
+
rest when updated, and writes no files. The host can use `retryAfterMs(header, nowMs)`
|
|
170
|
+
to parse Retry-After seconds or an HTTP date (invalid/absent values give `undefined`,
|
|
171
|
+
past dates clamp to zero), and `backoffDelayMs(retryAfterMs)` to apply the default
|
|
172
|
+
five-minute minimum. These helpers also support an app-owned Claude `read` hook
|
|
173
|
+
without duplicating fingerprint, persistence or retry logic.
|
|
174
|
+
|
|
175
|
+
`BackoffPolicy.set(provider, fingerprint, untilMs, state?)` receives a normalized
|
|
176
|
+
`state` with `{ untilMs, at, outcome, failures }` on a retryable failure, and zero
|
|
177
|
+
eligibility time without state after success. Persist and return that state from
|
|
178
|
+
`get` to preserve outcome and retry eligibility across restart. Legacy numeric
|
|
179
|
+
`get` values still work; their reason is unavailable when no stored poll supplies it.
|
|
180
|
+
`delayMs(retryAfterMs, { outcome, failures })` chooses host policy; a valid server
|
|
181
|
+
Retry-After is always a lower bound, alongside the existing one-minute floor.
|
|
182
|
+
|
|
183
|
+
`UsageOptions.pace({ provider, account, origin, signal })` is an optional async host
|
|
184
|
+
hook before each HTTP usage request (including provider discovery/fallback reads).
|
|
185
|
+
The host can share an origin-keyed queue across reader instances. Distinct origins
|
|
186
|
+
are independent; the kit adds no global queue or fixed origin spacing. Account is
|
|
187
|
+
a fingerprint, never a token. Host Claude readers opt in with `source.origin`.
|
|
188
|
+
The hook shares the request deadline and can be cancelled by `ReadOptions.signal`;
|
|
189
|
+
no request is sent when pacing fails or is cancelled. In-flight duplicate callers
|
|
190
|
+
share the first caller's operation/signal. There is no polling timer or inference ping.
|
|
191
|
+
|
|
192
|
+
Isolation: there is no home/path discovery or environment read. Only absolute files
|
|
193
|
+
and the Codex binary explicitly supplied by the app are opened/run. Credential files
|
|
194
|
+
are bounded regular files, with final symlinks rejected. The spawn uses argv and an
|
|
195
|
+
environment built from the host's explicit `env` plus `CODEX_HOME=home`; pass PATH
|
|
196
|
+
and HOME explicitly when needed. No tokens in logs, errors, readings or public hooks;
|
|
197
|
+
no credential write-back, telemetry, automatic refresh or reset-credit spend.
|
|
198
|
+
|
|
199
|
+
Built-in requests send `User-Agent: byokit/usage/0.3.0`, never another app's identity.
|
|
200
|
+
A refusal returns a code; with no last-good quota, room is unknown. Fixed endpoints
|
|
201
|
+
are Anthropic `api/oauth/usage`, ChatGPT `backend-api/wham/usage`, GitHub
|
|
202
|
+
`copilot_internal/user`, Grok `v1/billing` (weekly credits then monthly when needed),
|
|
203
|
+
MiniMax `v1/token_plan/remains`, Gemini `v1internal:retrieveUserQuota`, Kimi
|
|
204
|
+
`coding/v1/usages`, OpenCode `zen/go/v1/usage`, and z.ai `monitor/usage/quota/limit`.
|
|
205
|
+
HTTP requests reject redirects, have a ten-second deadline and a 64 KB body cap.
|
|
206
|
+
Claude includes `anthropic-beta: oauth-2025-04-20`; Codex includes the host's
|
|
207
|
+
`ChatGPT-Account-Id`. Codex app-server reads have a twenty-second deadline, a 64 KB
|
|
208
|
+
stdout cap and SIGTERM followed by SIGKILL after one second.
|
|
209
|
+
|
|
210
|
+
The Node-only main entry also exports `UsageError`, types, `words` and `usageWords`.
|
|
211
|
+
`./testing` exports `fakeFetch`, `fakeCodex` and `usageContract(make, { test? })`.
|
|
212
|
+
Tests use synthetic recorded protocol shapes and fakes behind the repository's
|
|
213
|
+
network guard. They never open real sign-ins or call provider endpoints.
|
|
214
|
+
|
|
215
|
+
`tokenLedger({ store?, cap? })` records measured token counts for host member ids.
|
|
216
|
+
`record(member, tokens, time)` accepts a nonnegative safe integer and epoch milliseconds;
|
|
217
|
+
`query(member, from, to)` uses `[from, to)` bounds and returns total `tokens`, sorted
|
|
218
|
+
local-calendar `days: [{ date, tokens }]`, and a `week` ending at `to`. The week spans
|
|
219
|
+
seven local calendar days (including DST), independent of `from`, and includes
|
|
220
|
+
`{ from, to, tokens, cap?, remaining? }`. Remaining allowance is clamped to zero.
|
|
221
|
+
`cap` is a seven-day token count or a synchronous member-to-cap function; omitted
|
|
222
|
+
means uncapped. `store` implements `record(member, { tokens, time })` and
|
|
223
|
+
`query(member, from, to)`, with `memoryTokenLedgerStore()` as the default. Reuse a
|
|
224
|
+
store to keep history across reader instances. The host owns durable storage and
|
|
225
|
+
retention; the default ledger never writes files. Invalid inputs and store failures
|
|
226
|
+
throw `TokenLedgerError` with `code: 'invalid' | 'store'`, without exposing member ids
|
|
227
|
+
or store exception text. Entries are counts only, never sign-in tokens.
|
|
228
|
+
|
|
229
|
+
`callLedger({ store?, prices? })` records runtime model calls through the same
|
|
230
|
+
`TokenLedgerStore` seam. `record(member, { provider, account, model, runId, time,
|
|
231
|
+
billing, usage?, payer?, durationMs?, state?, limits? })` returns and stores one
|
|
232
|
+
`CallRecord`. `billing` is `subscription` or `api`; `payer` defaults to the member.
|
|
233
|
+
`state` is `completed` (default), `cancelled` or `failed`. The host records each
|
|
234
|
+
actual model call, including retries, and supplies the provider's final usage when
|
|
235
|
+
available. No missing counts are inferred from words or decision sub-answers.
|
|
236
|
+
|
|
237
|
+
`normalizeTokens(provider, usage)` accepts native usage or its response envelope,
|
|
238
|
+
and normalized `{ input?, output?, cachedInput?, cacheWrite?, total? }` counts.
|
|
239
|
+
It returns only safe nonnegative integer counts with
|
|
240
|
+
`provenance: 'reported' | 'partial' | 'unknown'`. Input includes cache reads/writes,
|
|
241
|
+
which are subsets, so total is input plus output once. Claude's distinct cache
|
|
242
|
+
buckets are added to ordinary input ([Claude usage fields](https://platform.claude.com/docs/en/build-with-claude/prompt-caching));
|
|
243
|
+
OpenAI-compatible input already includes cache ([OpenAI caching](https://developers.openai.com/api/docs/guides/prompt-caching)).
|
|
244
|
+
Gemini output includes candidates and thinking tokens ([Gemini UsageMetadata](https://ai.google.dev/api/generate-content#UsageMetadata)).
|
|
245
|
+
Absent, invalid, explicitly estimated or inconsistent counts remain unknown; raw
|
|
246
|
+
response objects, prompts and credential fields are discarded.
|
|
247
|
+
|
|
248
|
+
Prices are host data keyed by provider then model: `{ billing, currency,
|
|
249
|
+
inputPerMillion, outputPerMillion, cachedInputPerMillion?, cacheWritePerMillion? }`.
|
|
250
|
+
`priceCall(tokens, price, billing)` and the ledger return a cost only when reported
|
|
251
|
+
counts and the matching app price row suffice. No vendor prices are bundled or
|
|
252
|
+
fetched. Costs carry `basis: 'app-prices'`, `estimated: true`, and billing labels
|
|
253
|
+
`Person's own plan` or `Person's API bill`; an estimate is not an invoice or an
|
|
254
|
+
extra charge against a subscription. Missing separate cache rates use the host's
|
|
255
|
+
input rate. If a separate rate requires an unknown cache count, cost is unknown.
|
|
256
|
+
|
|
257
|
+
`query(member, from, to)` returns time-sorted `calls`, aggregate `tokens`, `costs`
|
|
258
|
+
separated by currency and billing, and `unpricedCalls`. An aggregate field is
|
|
259
|
+
unknown if any call lacks that field. Store implementations must preserve the
|
|
260
|
+
entry's optional `call` metadata to replay calls; the default in-memory store does.
|
|
261
|
+
Sharing the store lets `tokenLedger.query` count these calls automatically. For
|
|
262
|
+
calls with unknown total counts, member/day/week results expose `unknownCalls`,
|
|
263
|
+
`tokens` is the known subtotal, and week `remaining` is omitted. Cap and price
|
|
264
|
+
policy remain the host's. All times, durations and quota reset timestamps are
|
|
265
|
+
milliseconds. There is no transport, credential discovery or automatic rotation.
|
|
266
|
+
|
|
267
|
+
When passing normalized windows to `@byokit/accounts`' structural helper, use
|
|
268
|
+
`roomOf(reading.windows, reading.at, 'milliseconds')`. Its two-argument form is for
|
|
269
|
+
legacy reset seconds; normalized usage windows in 0.2.0+ already use milliseconds.
|
|
270
|
+
Alternatively, this package's `roomOf(reading, nowMs)` returns a structural `Room`
|
|
271
|
+
that the accounts chooser accepts directly. Preserve the original measurement time.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
import type { BackoffPolicy } from './types.ts';
|
|
2
|
+
/** Parse Retry-After seconds or HTTP-date into a duration in milliseconds. */
|
|
3
|
+
export declare function retryAfterMs(header: string | null | undefined, nowMs: number): number | undefined;
|
|
4
|
+
/** Default quota rest: honor a longer server delay, otherwise wait five minutes. */
|
|
5
|
+
export declare function backoffDelayMs(retryAfter: number | undefined): number;
|
|
6
|
+
/** Share this policy between readers to preserve per-account 429 rests in memory. */
|
|
7
|
+
export declare function memoryBackoffPolicy(): BackoffPolicy;
|
package/dist/backoff.js
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/** Parse Retry-After seconds or HTTP-date into a duration in milliseconds. */
|
|
2
|
+
export function retryAfterMs(header, nowMs) {
|
|
3
|
+
if (!header?.trim())
|
|
4
|
+
return undefined;
|
|
5
|
+
const value = header.trim();
|
|
6
|
+
const delay = /^\d+(?:\.\d+)?$/.test(value) ? Number(value) * 1000 : Date.parse(value) - nowMs;
|
|
7
|
+
return Number.isFinite(delay) ? Math.max(0, delay) : undefined;
|
|
8
|
+
}
|
|
9
|
+
/** Default quota rest: honor a longer server delay, otherwise wait five minutes. */
|
|
10
|
+
export function backoffDelayMs(retryAfter) {
|
|
11
|
+
return Math.max(300_000, typeof retryAfter === 'number' && Number.isFinite(retryAfter) ? retryAfter : 0);
|
|
12
|
+
}
|
|
13
|
+
/** Share this policy between readers to preserve per-account 429 rests in memory. */
|
|
14
|
+
export function memoryBackoffPolicy() {
|
|
15
|
+
const rests = new Map();
|
|
16
|
+
return {
|
|
17
|
+
get: (provider, account) => rests.get(`${provider}\0${account}`),
|
|
18
|
+
set: (provider, account, untilMs) => {
|
|
19
|
+
if (!Number.isFinite(untilMs))
|
|
20
|
+
return;
|
|
21
|
+
const key = `${provider}\0${account}`;
|
|
22
|
+
rests.set(key, Math.max(rests.get(key) ?? 0, untilMs));
|
|
23
|
+
},
|
|
24
|
+
delayMs: backoffDelayMs,
|
|
25
|
+
};
|
|
26
|
+
}
|
package/dist/calls.d.ts
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import { type TokenLedgerStore } from './ledger.ts';
|
|
2
|
+
import type { Window } from './types.ts';
|
|
3
|
+
export interface NormalizedTokens {
|
|
4
|
+
/** Input includes cache reads and cache writes; cache counts are subsets. */
|
|
5
|
+
input?: number;
|
|
6
|
+
output?: number;
|
|
7
|
+
cachedInput?: number;
|
|
8
|
+
cacheWrite?: number;
|
|
9
|
+
total?: number;
|
|
10
|
+
provenance: 'reported' | 'partial' | 'unknown';
|
|
11
|
+
}
|
|
12
|
+
export interface ModelPrice {
|
|
13
|
+
billing: 'subscription' | 'api';
|
|
14
|
+
currency: string;
|
|
15
|
+
inputPerMillion: number;
|
|
16
|
+
outputPerMillion: number;
|
|
17
|
+
cachedInputPerMillion?: number;
|
|
18
|
+
cacheWritePerMillion?: number;
|
|
19
|
+
}
|
|
20
|
+
export type PriceTable = Readonly<Record<string, Readonly<Record<string, ModelPrice>>>>;
|
|
21
|
+
export interface CallCost {
|
|
22
|
+
amount: number;
|
|
23
|
+
currency: string;
|
|
24
|
+
billing: ModelPrice['billing'];
|
|
25
|
+
label: "Person's own plan" | "Person's API bill";
|
|
26
|
+
basis: 'app-prices';
|
|
27
|
+
estimated: true;
|
|
28
|
+
}
|
|
29
|
+
export interface CallInput {
|
|
30
|
+
provider: string;
|
|
31
|
+
account: string;
|
|
32
|
+
model: string;
|
|
33
|
+
runId: string;
|
|
34
|
+
time: number;
|
|
35
|
+
billing: 'subscription' | 'api';
|
|
36
|
+
/** Native provider usage/envelope, or normalized input/output/cachedInput/cacheWrite/total counts. */
|
|
37
|
+
usage?: unknown;
|
|
38
|
+
payer?: string;
|
|
39
|
+
durationMs?: number;
|
|
40
|
+
state?: 'completed' | 'cancelled' | 'failed';
|
|
41
|
+
limits?: readonly Window[];
|
|
42
|
+
}
|
|
43
|
+
export interface CallRecord extends Omit<CallInput, 'usage' | 'limits'> {
|
|
44
|
+
tokens: NormalizedTokens;
|
|
45
|
+
state: 'completed' | 'cancelled' | 'failed';
|
|
46
|
+
cost?: CallCost;
|
|
47
|
+
limits?: Window[];
|
|
48
|
+
}
|
|
49
|
+
export interface CallQuery {
|
|
50
|
+
calls: CallRecord[];
|
|
51
|
+
tokens: NormalizedTokens;
|
|
52
|
+
/** Separate subtotals by currency and billing; unpriced calls are counted explicitly. */
|
|
53
|
+
costs: CallCost[];
|
|
54
|
+
unpricedCalls: number;
|
|
55
|
+
}
|
|
56
|
+
export interface CallLedger {
|
|
57
|
+
record(member: string, call: CallInput): CallRecord;
|
|
58
|
+
query(member: string, from: number, to: number): CallQuery;
|
|
59
|
+
}
|
|
60
|
+
/** Pure normalization of reported counts, never estimates from text or shared decision invocations. */
|
|
61
|
+
export declare function normalizeTokens(provider: string, raw: unknown): NormalizedTokens;
|
|
62
|
+
/** App-supplied price estimates only, including explicit billing attribution. */
|
|
63
|
+
export declare function priceCall(tokens: NormalizedTokens, price: ModelPrice | undefined, billing: ModelPrice['billing']): CallCost | undefined;
|
|
64
|
+
/** Every recorded attempt is a call; the host records retries separately under its run id. */
|
|
65
|
+
export declare function callLedger(options?: {
|
|
66
|
+
store?: TokenLedgerStore;
|
|
67
|
+
prices?: PriceTable;
|
|
68
|
+
}): CallLedger;
|
package/dist/calls.js
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
import { record as isRecord } from "./windows.js";
|
|
2
|
+
import { memoryTokenLedgerStore, TokenLedgerError } from "./ledger.js";
|
|
3
|
+
import { safeWindows } from "./store.js";
|
|
4
|
+
const object = (value) => isRecord(value) ? value : {};
|
|
5
|
+
const count = (value) => typeof value === 'number' && Number.isSafeInteger(value) && value >= 0 ? value : undefined;
|
|
6
|
+
const text = (value) => typeof value === 'string' && !!value && value.length <= 1024 && !/[\0\r\n]/.test(value);
|
|
7
|
+
const time = (value) => typeof value === 'number' && Number.isFinite(value) && !Number.isNaN(new Date(value).getTime());
|
|
8
|
+
const sum = (...values) => count(values.reduce((a, b) => a + b, 0));
|
|
9
|
+
function normalized(input, output, cachedInput, cacheWrite, total) {
|
|
10
|
+
if (input !== undefined && ((cachedInput ?? 0) + (cacheWrite ?? 0) > input))
|
|
11
|
+
return { provenance: 'unknown' };
|
|
12
|
+
const derived = input !== undefined && output !== undefined ? sum(input, output) : undefined;
|
|
13
|
+
// Inconsistent reported totals are not a reliable measurement.
|
|
14
|
+
if (derived !== undefined && total !== undefined && derived !== total)
|
|
15
|
+
return { provenance: 'unknown' };
|
|
16
|
+
const measured = derived ?? total;
|
|
17
|
+
return { ...(input === undefined ? {} : { input }), ...(output === undefined ? {} : { output }),
|
|
18
|
+
...(cachedInput === undefined ? {} : { cachedInput }), ...(cacheWrite === undefined ? {} : { cacheWrite }),
|
|
19
|
+
...(measured === undefined ? {} : { total: measured }),
|
|
20
|
+
provenance: input !== undefined && output !== undefined && derived !== undefined ? 'reported' : [input, output, cachedInput, cacheWrite, measured].some((v) => v !== undefined) ? 'partial' : 'unknown' };
|
|
21
|
+
}
|
|
22
|
+
/** Pure normalization of reported counts, never estimates from text or shared decision invocations. */
|
|
23
|
+
export function normalizeTokens(provider, raw) {
|
|
24
|
+
const envelope = object(raw);
|
|
25
|
+
const usage = object(envelope.usage ?? envelope.usageMetadata ?? raw);
|
|
26
|
+
if (usage.provenance === 'unknown' || usage.provenance === 'estimated')
|
|
27
|
+
return { provenance: 'unknown' };
|
|
28
|
+
if (['input', 'output', 'total'].some((key) => key in usage))
|
|
29
|
+
return normalized(count(usage.input), count(usage.output), count(usage.cachedInput), count(usage.cacheWrite), count(usage.total));
|
|
30
|
+
if (provider === 'claude' || provider === 'anthropic') {
|
|
31
|
+
if (!['input_tokens', 'output_tokens', 'cache_read_input_tokens', 'cache_creation_input_tokens'].some((key) => count(usage[key]) !== undefined))
|
|
32
|
+
return { provenance: 'unknown' };
|
|
33
|
+
const uncached = count(usage.input_tokens);
|
|
34
|
+
const cached = count(usage.cache_read_input_tokens) ?? (usage.cache_read_input_tokens === undefined ? 0 : undefined);
|
|
35
|
+
const written = count(usage.cache_creation_input_tokens) ?? (usage.cache_creation_input_tokens === undefined ? 0 : undefined);
|
|
36
|
+
const input = uncached !== undefined && cached !== undefined && written !== undefined ? sum(uncached, cached, written) : undefined;
|
|
37
|
+
return normalized(input, count(usage.output_tokens), cached, written);
|
|
38
|
+
}
|
|
39
|
+
if (provider === 'gemini' || provider === 'google') {
|
|
40
|
+
const output = count(usage.candidatesTokenCount);
|
|
41
|
+
const thoughts = count(usage.thoughtsTokenCount) ?? (usage.thoughtsTokenCount === undefined ? 0 : undefined);
|
|
42
|
+
const input = count(usage.promptTokenCount);
|
|
43
|
+
return normalized(input, output !== undefined && thoughts !== undefined ? sum(output, thoughts) : undefined, count(usage.cachedContentTokenCount), undefined, count(usage.totalTokenCount));
|
|
44
|
+
}
|
|
45
|
+
const input = count(usage.input_tokens ?? usage.prompt_tokens);
|
|
46
|
+
const output = count(usage.output_tokens ?? usage.completion_tokens);
|
|
47
|
+
const details = object(usage.input_tokens_details ?? usage.prompt_tokens_details);
|
|
48
|
+
return normalized(input, output, count(details.cached_tokens), count(details.cache_write_tokens), count(usage.total_tokens));
|
|
49
|
+
}
|
|
50
|
+
/** App-supplied price estimates only, including explicit billing attribution. */
|
|
51
|
+
export function priceCall(tokens, price, billing) {
|
|
52
|
+
if (!price || price.billing !== billing || tokens.input === undefined || tokens.output === undefined || tokens.provenance !== 'reported' || !text(price.currency))
|
|
53
|
+
return undefined;
|
|
54
|
+
if ([tokens.input, tokens.output, tokens.cachedInput, tokens.cacheWrite, tokens.total].some((value) => value !== undefined && count(value) === undefined) || tokens.total !== undefined && tokens.input + tokens.output !== tokens.total)
|
|
55
|
+
return undefined;
|
|
56
|
+
const rates = [price.inputPerMillion, price.outputPerMillion, price.cachedInputPerMillion, price.cacheWritePerMillion];
|
|
57
|
+
if (rates.some((rate) => rate !== undefined && (typeof rate !== 'number' || !Number.isFinite(rate) || rate < 0)))
|
|
58
|
+
return undefined;
|
|
59
|
+
if (typeof price.inputPerMillion !== 'number' || typeof price.outputPerMillion !== 'number')
|
|
60
|
+
return undefined;
|
|
61
|
+
if (tokens.cachedInput === undefined && price.cachedInputPerMillion !== undefined && price.cachedInputPerMillion !== price.inputPerMillion
|
|
62
|
+
|| tokens.cacheWrite === undefined && price.cacheWritePerMillion !== undefined && price.cacheWritePerMillion !== price.inputPerMillion)
|
|
63
|
+
return undefined;
|
|
64
|
+
const cached = tokens.cachedInput ?? 0;
|
|
65
|
+
const written = tokens.cacheWrite ?? 0;
|
|
66
|
+
const ordinary = tokens.input - cached - written;
|
|
67
|
+
if (ordinary < 0)
|
|
68
|
+
return undefined;
|
|
69
|
+
const amount = (ordinary * price.inputPerMillion + cached * (price.cachedInputPerMillion ?? price.inputPerMillion)
|
|
70
|
+
+ written * (price.cacheWritePerMillion ?? price.inputPerMillion) + tokens.output * price.outputPerMillion) / 1_000_000;
|
|
71
|
+
if (!Number.isFinite(amount))
|
|
72
|
+
return undefined;
|
|
73
|
+
return { amount, currency: price.currency, billing, label: billing === 'subscription' ? "Person's own plan" : "Person's API bill", basis: 'app-prices', estimated: true };
|
|
74
|
+
}
|
|
75
|
+
function aggregate(calls) {
|
|
76
|
+
const field = (key) => calls.every((call) => call.tokens[key] !== undefined) ? count(calls.reduce((total, call) => total + call.tokens[key], 0)) : undefined;
|
|
77
|
+
return normalized(field('input'), field('output'), field('cachedInput'), field('cacheWrite'), field('total'));
|
|
78
|
+
}
|
|
79
|
+
/** Every recorded attempt is a call; the host records retries separately under its run id. */
|
|
80
|
+
export function callLedger(options = {}) {
|
|
81
|
+
const store = options.store ?? memoryTokenLedgerStore();
|
|
82
|
+
return {
|
|
83
|
+
record(member, call) {
|
|
84
|
+
if (!text(member) || !call || ![call.provider, call.account, call.model, call.runId].every(text) || !time(call.time)
|
|
85
|
+
|| !['subscription', 'api'].includes(call.billing) || call.payer !== undefined && !text(call.payer)
|
|
86
|
+
|| call.durationMs !== undefined && (typeof call.durationMs !== 'number' || !Number.isFinite(call.durationMs) || call.durationMs < 0)
|
|
87
|
+
|| call.state !== undefined && !['completed', 'cancelled', 'failed'].includes(call.state) || call.limits !== undefined && !Array.isArray(call.limits))
|
|
88
|
+
throw new TokenLedgerError('invalid');
|
|
89
|
+
const tokens = normalizeTokens(call.provider, call.usage);
|
|
90
|
+
const cost = priceCall(tokens, options.prices?.[call.provider]?.[call.model], call.billing);
|
|
91
|
+
const limits = call.limits?.flatMap((window) => safeWindows(window.provider, [window]));
|
|
92
|
+
const result = { provider: call.provider, account: call.account, model: call.model, runId: call.runId, time: call.time,
|
|
93
|
+
billing: call.billing, payer: call.payer ?? member, state: call.state ?? 'completed', tokens,
|
|
94
|
+
...(call.durationMs === undefined ? {} : { durationMs: call.durationMs }), ...(cost ? { cost } : {}), ...(limits ? { limits } : {}) };
|
|
95
|
+
try {
|
|
96
|
+
store.record(member, { time: call.time, tokens: tokens.total ?? 0, call: copyCall(result) });
|
|
97
|
+
}
|
|
98
|
+
catch {
|
|
99
|
+
throw new TokenLedgerError('store');
|
|
100
|
+
}
|
|
101
|
+
return result;
|
|
102
|
+
},
|
|
103
|
+
query(member, from, to) {
|
|
104
|
+
if (!text(member) || !time(from) || !time(to) || to < from)
|
|
105
|
+
throw new TokenLedgerError('invalid');
|
|
106
|
+
let entries;
|
|
107
|
+
try {
|
|
108
|
+
entries = store.query(member, from, to);
|
|
109
|
+
}
|
|
110
|
+
catch {
|
|
111
|
+
throw new TokenLedgerError('store');
|
|
112
|
+
}
|
|
113
|
+
if (!Array.isArray(entries))
|
|
114
|
+
throw new TokenLedgerError('invalid');
|
|
115
|
+
const calls = entries.filter((entry) => entry.call && entry.time >= from && entry.time < to).map((entry) => copyCall(entry.call)).sort((a, b) => a.time - b.time);
|
|
116
|
+
const costs = new Map();
|
|
117
|
+
for (const call of calls) {
|
|
118
|
+
if (!call.cost)
|
|
119
|
+
continue;
|
|
120
|
+
const key = `${call.cost.currency}\0${call.cost.billing}`;
|
|
121
|
+
const previous = costs.get(key);
|
|
122
|
+
costs.set(key, { ...call.cost, amount: (previous?.amount ?? 0) + call.cost.amount });
|
|
123
|
+
}
|
|
124
|
+
return { calls, tokens: aggregate(calls), costs: [...costs.values()], unpricedCalls: calls.filter((call) => !call.cost).length };
|
|
125
|
+
},
|
|
126
|
+
};
|
|
127
|
+
}
|
|
128
|
+
function copyCall(call) {
|
|
129
|
+
return { ...call, tokens: { ...call.tokens }, ...(call.cost ? { cost: { ...call.cost } } : {}),
|
|
130
|
+
...(call.limits ? { limits: call.limits.map((window) => ({ ...window })) } : {}) };
|
|
131
|
+
}
|
package/dist/index.d.ts
CHANGED
|
@@ -1,6 +1,12 @@
|
|
|
1
1
|
import { type Usage, type UsageOptions } from './types.ts';
|
|
2
2
|
export * from './types.ts';
|
|
3
|
+
export { callLedger, normalizeTokens, priceCall, type CallLedger, type CallInput, type CallRecord, type CallQuery, type NormalizedTokens, type ModelPrice, type PriceTable, type CallCost } from './calls.ts';
|
|
4
|
+
export { tokenLedger, memoryTokenLedgerStore, TokenLedgerError, type TokenLedger, type TokenLedgerStore, type TokenLedgerOptions, type TokenEntry, type TokenQuery } from './ledger.ts';
|
|
5
|
+
export { roomOf } from './room.ts';
|
|
6
|
+
export { fingerprint, store as fileUsageStore, memoryUsageStore } from './store.ts';
|
|
7
|
+
export { retryAfterMs, backoffDelayMs, memoryBackoffPolicy } from './backoff.ts';
|
|
3
8
|
export { claudeWindows, codexWindows, goWindows, zaiWindows, type CodexRateLimitResult } from './windows.ts';
|
|
9
|
+
export { codexHardLimit, codexTokenWindows, copilotWindows, grokWindows, minimaxWindows, geminiWindows, kimiWindows } from './quota.ts';
|
|
4
10
|
export { WORDS, words, usageWords, type WordKey } from './words.ts';
|
|
5
11
|
/** One reader owns per-account backoff and concurrent-read deduplication. */
|
|
6
12
|
export declare function usage(options: UsageOptions): Usage;
|