@rayadesu/dsh-llm-billing 0.3.8 → 0.3.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -21,25 +21,77 @@ export const DEFAULT_PEAK_HOURS = [
21
21
  { start: 9, end: 12 },
22
22
  { start: 14, end: 18 },
23
23
  ];
24
- /** Official peak/off-peak rates (CNY per 1M tokens), effective 2026-08-17. */
24
+ /**
25
+ * Inclusive epoch ms of the published V4 Flash series re-pricing:
26
+ * 2026-09-10 12:00 Beijing time (UTC+8, no DST) = 04:00 UTC. Samples before
27
+ * this instant keep the base rates; samples at or after it bill at the second
28
+ * revision.
29
+ */
30
+ export const FLASH_SERIES_RATE_CHANGE_AT = Date.UTC(2026, 8, 10, 4, 0, 0);
31
+ /**
32
+ * Inclusive epoch ms of the announced V4 Pro route switch: 2026-09-14 12:00
33
+ * Beijing time (UTC+8, no DST) = 04:00 UTC. From that instant the V4 Pro route
34
+ * is served by V4.1 Flash and billed at the V4.1 Flash rates.
35
+ */
36
+ export const V4_PRO_ROUTE_SWITCH_AT = Date.UTC(2026, 8, 14, 4, 0, 0);
37
+ /** The V4 Flash series' base rates (effective 2026-08-17), CNY per 1M tokens. */
38
+ const FLASH_BASE_RATES = {
39
+ peak: { cacheHitInput: 0.10, cacheMissInput: 3.0, output: 9.0 },
40
+ offPeak: { cacheHitInput: 0.05, cacheMissInput: 1.5, output: 4.5 },
41
+ };
42
+ /**
43
+ * The V4 Flash series' second revision (effective
44
+ * {@link FLASH_SERIES_RATE_CHANGE_AT}): off-peak 0.02 / 1.0 / 4.0, peak at
45
+ * twice those prices.
46
+ */
47
+ const FLASH_REPRICED_RATES = {
48
+ effectiveFrom: FLASH_SERIES_RATE_CHANGE_AT,
49
+ peak: { cacheHitInput: 0.04, cacheMissInput: 2.0, output: 8.0 },
50
+ offPeak: { cacheHitInput: 0.02, cacheMissInput: 1.0, output: 4.0 },
51
+ };
52
+ /**
53
+ * The V4.1 Flash rates as they reach the retired V4 Pro route from
54
+ * {@link V4_PRO_ROUTE_SWITCH_AT}: the same price pair as the flash series'
55
+ * second revision, carried at its own effective instant.
56
+ */
57
+ const V4_PRO_SWITCHED_RATES = {
58
+ effectiveFrom: V4_PRO_ROUTE_SWITCH_AT,
59
+ peak: FLASH_REPRICED_RATES.peak,
60
+ offPeak: FLASH_REPRICED_RATES.offPeak,
61
+ };
62
+ /**
63
+ * Official peak/off-peak rates (CNY per 1M tokens) per model, as dated
64
+ * revisions. Base rows are the schedule effective 2026-08-17; the V4 Flash
65
+ * series (V4.1 Flash, V4 Flash, V4 Flash Vision Exp) carries the second
66
+ * revision effective 2026-09-10 12:00 Beijing, and the V4 Pro row the V4.1
67
+ * Flash rates from its announced route switch (2026-09-14 12:00 Beijing) —
68
+ * the MiMo-V2.5 series is untouched by either adjustment. Rows sharing a model
69
+ * are that model's rate history.
70
+ */
25
71
  export const DEFAULT_MODEL_PRICING = [
26
- {
27
- model: 'deepseek-v4-flash',
28
- peak: { cacheHitInput: 0.10, cacheMissInput: 3.0, output: 9.0 },
29
- offPeak: { cacheHitInput: 0.05, cacheMissInput: 1.5, output: 4.5 },
30
- },
72
+ // deepseek-flash is the V4.1 Flash route, DSH's default catalog entry; image
73
+ // inputs are converted to tokens at the same per-token price.
74
+ { model: 'deepseek-flash', ...FLASH_BASE_RATES },
75
+ { model: 'deepseek-flash', ...FLASH_REPRICED_RATES },
76
+ { model: 'deepseek-v4-flash', ...FLASH_BASE_RATES },
77
+ { model: 'deepseek-v4-flash', ...FLASH_REPRICED_RATES },
78
+ // deepseek-v4.1-flash-expires-on-0910 was the V4.1 Flash preview route,
79
+ // retired when the model was released on 2026-09-10; its rows stay so the
80
+ // logs that used it keep pricing.
81
+ { model: 'deepseek-v4.1-flash-expires-on-0910', ...FLASH_BASE_RATES },
82
+ { model: 'deepseek-v4.1-flash-expires-on-0910', ...FLASH_REPRICED_RATES },
31
83
  {
32
84
  model: 'deepseek-v4-pro',
33
85
  peak: { cacheHitInput: 0.30, cacheMissInput: 9.0, output: 27.0 },
34
86
  offPeak: { cacheHitInput: 0.15, cacheMissInput: 4.5, output: 13.5 },
35
87
  },
88
+ // From the announced route switch V4 Pro is served by V4.1 Flash and billed
89
+ // at the V4.1 Flash rates.
90
+ { model: 'deepseek-v4-pro', ...V4_PRO_SWITCHED_RATES },
36
91
  // deepseek-v4-flash-vision-exp bills at the same rates as deepseek-v4-flash;
37
92
  // images are converted to tokens at the same per-token price.
38
- {
39
- model: 'deepseek-v4-flash-vision-exp',
40
- peak: { cacheHitInput: 0.10, cacheMissInput: 3.0, output: 9.0 },
41
- offPeak: { cacheHitInput: 0.05, cacheMissInput: 1.5, output: 4.5 },
42
- },
93
+ { model: 'deepseek-v4-flash-vision-exp', ...FLASH_BASE_RATES },
94
+ { model: 'deepseek-v4-flash-vision-exp', ...FLASH_REPRICED_RATES },
43
95
  // MiMo-V2.5 series (Xiaomi): flat rate, no peak/off-peak distinction.
44
96
  {
45
97
  model: 'mimo-v2.5-pro',
@@ -58,8 +110,14 @@ export const DEFAULT_MODEL_PRICING = [
58
110
  * `z.array` as `[]` rather than `undefined`, so emptiness — not just absence —
59
111
  * selects the defaults. Explicit non-empty rows override the same model; a
60
112
  * supplied non-empty `models` list is authoritative.
113
+ *
114
+ * Rows sharing a model are that model's rate revisions, kept in ascending
115
+ * `effectiveFrom` order (an undated base revision first). Two rows declaring
116
+ * the same effective instant are one revision and the later row wins — the
117
+ * historical override rule — so re-declaring a model can neither duplicate a
118
+ * revision nor install a second undated base.
61
119
  * @param config - optional raw billing configuration.
62
- * @returns the resolved table and peak-hour windows.
120
+ * @returns the resolved table (per model: its revisions plus the newest rates) and peak-hour windows.
63
121
  */
64
122
  export function resolveBilling(config) {
65
123
  const peakHours = config?.peakHours !== undefined && config.peakHours.length > 0
@@ -68,28 +126,107 @@ export function resolveBilling(config) {
68
126
  const rows = config?.models !== undefined && config.models.length > 0
69
127
  ? config.models
70
128
  : DEFAULT_MODEL_PRICING;
129
+ const schedules = new Map();
130
+ for (const row of rows) {
131
+ // Spelled out per branch: `exactOptionalPropertyTypes` forbids handing an
132
+ // explicit `undefined` to an optional field.
133
+ const revision = row.effectiveFrom === undefined
134
+ ? { peak: row.peak, offPeak: row.offPeak }
135
+ : { effectiveFrom: row.effectiveFrom, peak: row.peak, offPeak: row.offPeak };
136
+ const revisions = schedules.get(row.model);
137
+ if (revisions === undefined) {
138
+ schedules.set(row.model, [revision]);
139
+ continue;
140
+ }
141
+ const duplicate = revisions.findIndex(candidate => candidate.effectiveFrom === revision.effectiveFrom);
142
+ if (duplicate >= 0)
143
+ revisions[duplicate] = revision;
144
+ else
145
+ revisions.push(revision);
146
+ }
71
147
  const models = new Map();
72
- for (const row of rows)
73
- models.set(row.model, { peak: row.peak, offPeak: row.offPeak });
148
+ for (const [model, revisions] of schedules) {
149
+ revisions.sort((left, right) => (left.effectiveFrom ?? Number.NEGATIVE_INFINITY) - (right.effectiveFrom ?? Number.NEGATIVE_INFINITY));
150
+ const newest = revisions[revisions.length - 1];
151
+ models.set(model, { peak: newest.peak, offPeak: newest.offPeak, revisions });
152
+ }
74
153
  return { peakHours, models };
75
154
  }
76
155
  /**
77
- * Derive the Beijing hour, weekday, and calendar-day key of one timestamp from
78
- * a single shifted `Date` — every timezone-sensitive read shares this one
79
- * implementation, so the pieces cannot drift apart.
156
+ * The rate revision in effect at one instant: the newest revision that took
157
+ * effect at or before it. Revisions are ascending, so the scan stops at the
158
+ * first future one. An instant before the earliest dated revision bills at that
159
+ * earliest revision — a model with only dated rows is never left unpriced.
160
+ */
161
+ function ratesAt(revisions, time) {
162
+ let chosen = revisions[0];
163
+ for (let index = 1; index < revisions.length; index += 1) {
164
+ const revision = revisions[index];
165
+ if (revision.effectiveFrom === undefined || revision.effectiveFrom > time)
166
+ break;
167
+ chosen = revision;
168
+ }
169
+ return chosen;
170
+ }
171
+ /** Beijing is a fixed UTC+8 offset with no DST. */
172
+ const BEIJING_OFFSET_MS = 8 * 3_600_000;
173
+ /** Milliseconds in one day. */
174
+ const DAY_MS = 86_400_000;
175
+ /** Epoch day of 1970-01-01 in the civil-date algorithm below. */
176
+ const CIVIL_EPOCH_DAY = 719_468;
177
+ /** Two-digit zero pad for a calendar field. */
178
+ function pad2(value) {
179
+ return value < 10 ? `0${value}` : String(value);
180
+ }
181
+ /**
182
+ * Civil date of an epoch day (Howard Hinnant's days-from-civil inverse):
183
+ * pure integer arithmetic, no `Date` allocation and no ISO-string slicing.
184
+ */
185
+ function civilDateOf(epochDay) {
186
+ const shifted = epochDay + CIVIL_EPOCH_DAY;
187
+ const era = Math.floor(shifted / 146_097);
188
+ const dayOfEra = shifted - era * 146_097;
189
+ const yearOfEra = Math.floor((dayOfEra - Math.floor(dayOfEra / 1_460) + Math.floor(dayOfEra / 36_524) - Math.floor(dayOfEra / 146_096)) / 365);
190
+ const year = yearOfEra + era * 400;
191
+ const dayOfYear = dayOfEra - (365 * yearOfEra + Math.floor(yearOfEra / 4) - Math.floor(yearOfEra / 100));
192
+ const monthPrime = Math.floor((5 * dayOfYear + 2) / 153);
193
+ const month = monthPrime + (monthPrime < 10 ? 3 : -9);
194
+ return {
195
+ // January/February belong to the civil year AFTER the era year.
196
+ year: month <= 2 ? year + 1 : year,
197
+ month,
198
+ day: dayOfYear - Math.floor((153 * monthPrime + 2) / 5) + 1,
199
+ };
200
+ }
201
+ /**
202
+ * Derive the Beijing hour, weekday, and calendar-day key of one timestamp with
203
+ * pure integer arithmetic — every timezone-sensitive read shares this one
204
+ * implementation, so the pieces cannot drift apart. Callers that filter by
205
+ * day and then price the same event reuse the returned view, so each event is
206
+ * parsed exactly once. (The hot fold path runs this per committed event; the
207
+ * previous `Date` + `toISOString().slice()` version allocated a `Date` and a
208
+ * 24-character string per call.)
80
209
  * @param time - epoch milliseconds.
210
+ * @throws {RangeError} when `time` is not a finite number.
81
211
  */
82
- function beijingParts(time) {
83
- const shifted = new Date(time + 8 * 3_600_000);
212
+ export function beijingPartsOf(time) {
213
+ if (!Number.isFinite(time))
214
+ throw new RangeError(`billing: event time is not finite (${String(time)})`);
215
+ const shifted = time + BEIJING_OFFSET_MS;
216
+ const epochDay = Math.floor(shifted / DAY_MS);
217
+ const msOfDay = shifted - epochDay * DAY_MS;
218
+ const civil = civilDateOf(epochDay);
84
219
  return {
85
- hour: shifted.getUTCHours(),
86
- weekday: shifted.getUTCDay(),
87
- dayKey: shifted.toISOString().slice(0, 10),
220
+ time,
221
+ hour: Math.floor(msOfDay / 3_600_000),
222
+ // 1970-01-01 was a Thursday (4).
223
+ weekday: ((epochDay + 4) % 7 + 7) % 7,
224
+ dayKey: `${civil.year}-${pad2(civil.month)}-${pad2(civil.day)}`,
88
225
  };
89
226
  }
90
227
  /** The Beijing (Asia/Shanghai, UTC+8, no DST) calendar-day key of a timestamp. */
91
228
  export function beijingDayKey(now) {
92
- return beijingParts(now.getTime()).dayKey;
229
+ return beijingPartsOf(now.getTime()).dayKey;
93
230
  }
94
231
  /**
95
232
  * The durable inherited-prefix boundary of one session: the number of leading
@@ -134,44 +271,73 @@ function isPeakParts(billing, hour, weekday) {
134
271
  * @returns true during a weekday peak hour.
135
272
  */
136
273
  export function isPeak(billing, now) {
137
- const { hour, weekday } = beijingParts(now.getTime());
274
+ const { hour, weekday } = beijingPartsOf(now.getTime());
138
275
  return isPeakParts(billing, hour, weekday);
139
276
  }
140
277
  /**
141
278
  * Price one event at the official per-model rates, applying the peak/off-peak
142
279
  * table by its Beijing-time hour and weekday (peak windows apply Monday–Friday
143
- * only; weekends are off-peak). Each `assistant/message` event with usage
280
+ * only; weekends are off-peak) and the rate revision in effect at its own
281
+ * timestamp. Each `assistant/message` event with usage
144
282
  * contributes cache-hit input, cache-miss input (uncached input plus cache
145
283
  * writes), and output (reasoning included) tokens at the rate of its own
146
- * timestamp; a model with usage but no pricing row contributes nothing (the
147
- * published table prices only the two V4 rows).
284
+ * timestamp; a model with usage but no pricing row contributes nothing.
148
285
  * @param event - the event to price.
149
286
  * @param billing - resolved pricing with peak-hour windows.
150
287
  * @param names - model id → display label.
151
288
  * @returns the priced contribution, or `undefined` when the event has no priced usage.
152
289
  */
153
290
  export function priceEvent(event, billing, names) {
291
+ return priceEventAt(beijingPartsOf(event.time), event, billing, names);
292
+ }
293
+ /**
294
+ * Price one event at the official per-model rates using a precomputed
295
+ * Beijing-time view — the day-filtering and pricing of one event share a
296
+ * single timezone parse (see {@link beijingPartsOf}). Semantics are identical
297
+ * to {@link priceEvent}.
298
+ * @param parts - the event's Beijing-time view.
299
+ * @param event - the event to price.
300
+ * @param billing - resolved pricing with peak-hour windows.
301
+ * @param names - model id → display label.
302
+ * @returns the priced contribution, or `undefined` when the event has no priced usage.
303
+ */
304
+ export function priceEventAt(parts, event, billing, names) {
154
305
  if (event.type !== 'assistant/message')
155
306
  return undefined;
156
307
  const reported = event.data.usage;
157
308
  if (reported === undefined)
158
309
  return undefined;
159
- const model = event.data.message.source.model;
310
+ return priceUsage(parts, reported, event.data.message.source.model, billing, names);
311
+ }
312
+ /**
313
+ * Price one provider-reported usage sample for one model at the rates of the
314
+ * sample's own Beijing-time hour and weekday — the peak or off-peak price of
315
+ * the rate revision in effect at the sample's own timestamp (a re-priced series
316
+ * bills its history at the rates that applied then). `undefined` when the model
317
+ * has no pricing row.
318
+ * @param parts - the sample's Beijing-time view.
319
+ * @param usage - the reported token buckets.
320
+ * @param model - the wire model id the sample belongs to.
321
+ * @param billing - resolved pricing with peak-hour windows.
322
+ * @param names - model id → display label.
323
+ * @returns the priced contribution, or `undefined` when the model has no rate row.
324
+ */
325
+ export function priceUsage(parts, usage, model, billing, names) {
160
326
  const pricing = billing.models.get(model);
161
327
  if (pricing === undefined)
162
328
  return undefined;
163
- const { hour, weekday, dayKey } = beijingParts(event.time);
164
- const peak = isPeakParts(billing, hour, weekday);
165
- const price = peak ? pricing.peak : pricing.offPeak;
166
- const hit = reported.cacheReadTokens ?? 0;
167
- const miss = reported.inputTokens + (reported.cacheWriteTokens ?? 0);
168
- const output = reported.outputTokens;
329
+ const peak = isPeakParts(billing, parts.hour, parts.weekday);
330
+ const revision = ratesAt(pricing.revisions, parts.time);
331
+ const price = peak ? revision.peak : revision.offPeak;
332
+ const hit = usage.cacheReadTokens ?? 0;
333
+ const miss = usage.inputTokens + (usage.cacheWriteTokens ?? 0);
334
+ const output = usage.outputTokens;
169
335
  const hitCost = (hit * price.cacheHitInput) / 1_000_000;
170
336
  const missCost = (miss * price.cacheMissInput) / 1_000_000;
171
337
  const outputCost = (output * price.output) / 1_000_000;
172
338
  const cost = hitCost + missCost + outputCost;
173
339
  return {
174
- dayKey,
340
+ dayKey: parts.dayKey,
175
341
  model,
176
342
  displayName: names.get(model) ?? model,
177
343
  cost,
@@ -244,6 +410,202 @@ export class SpendAccumulator {
244
410
  return { total: this.total, models: [...this.rows.values()] };
245
411
  }
246
412
  }
413
+ /** The additive inverse of one spend (pure): used to replace a priced sample. */
414
+ export function negateSpend(spend) {
415
+ const negate = (value) => -value;
416
+ return {
417
+ total: negate(spend.total),
418
+ models: spend.models.map(row => ({
419
+ ...row,
420
+ cost: negate(row.cost),
421
+ peakCost: negate(row.peakCost),
422
+ offPeakCost: negate(row.offPeakCost),
423
+ cacheHitInputTokens: negate(row.cacheHitInputTokens),
424
+ cacheMissInputTokens: negate(row.cacheMissInputTokens),
425
+ outputTokens: negate(row.outputTokens),
426
+ cacheHitInputCost: negate(row.cacheHitInputCost),
427
+ cacheMissInputCost: negate(row.cacheMissInputCost),
428
+ outputCost: negate(row.outputCost),
429
+ })),
430
+ };
431
+ }
432
+ /**
433
+ * Subtract one spend from another (pure). Rows that cancel out completely are
434
+ * dropped so a replaced sample leaves no zero row behind.
435
+ * @param target - the spend to subtract from.
436
+ * @param source - the spend to remove.
437
+ * @returns the difference.
438
+ */
439
+ export function subtractSpend(target, source) {
440
+ const rows = new Map();
441
+ for (const row of target.models)
442
+ rows.set(row.model, row);
443
+ for (const row of source.models) {
444
+ const existing = rows.get(row.model);
445
+ if (existing === undefined)
446
+ continue;
447
+ const next = mergeModelRows(existing, negateSpend({ total: 0, models: [row] }).models[0]);
448
+ if (next.cost === 0 && next.cacheHitInputTokens === 0 && next.cacheMissInputTokens === 0 && next.outputTokens === 0) {
449
+ rows.delete(row.model);
450
+ }
451
+ else {
452
+ rows.set(row.model, next);
453
+ }
454
+ }
455
+ return { total: target.total - source.total, models: [...rows.values()] };
456
+ }
457
+ /** The empty fold state for one fork boundary. */
458
+ export function emptyBillingFoldState(inheritedEventCount = 0) {
459
+ return {
460
+ dayKey: '',
461
+ spend: emptyTodaySpend(),
462
+ session: emptyTodaySpend(),
463
+ inheritedEventCount,
464
+ model: '',
465
+ last: null,
466
+ };
467
+ }
468
+ /** Whether an unknown value looks like a provider usage report. */
469
+ function isTokenUsage(value) {
470
+ if (typeof value !== 'object' || value === null)
471
+ return false;
472
+ const candidate = value;
473
+ return typeof candidate.inputTokens === 'number' && typeof candidate.outputTokens === 'number';
474
+ }
475
+ /**
476
+ * The last `usage` sample embedded in an event's stream, if any. `assistant/
477
+ * attempt` and the embedded streams are newer than the plugin's npm baseline,
478
+ * so the stream is read structurally (a failed/retried attempt reports its
479
+ * usage only there).
480
+ */
481
+ function streamUsageOf(event) {
482
+ const stream = event.data === undefined
483
+ ? undefined
484
+ : event.data.stream;
485
+ if (!Array.isArray(stream))
486
+ return undefined;
487
+ for (let index = stream.length - 1; index >= 0; index -= 1) {
488
+ const chunk = stream[index]?.chunk;
489
+ if (chunk === undefined || chunk.type !== 'usage')
490
+ continue;
491
+ return isTokenUsage(chunk.usage) ? chunk.usage : undefined;
492
+ }
493
+ return undefined;
494
+ }
495
+ /** The contribution as a one-row spend (the shape a sample keeps for replacement). */
496
+ function contributionSpend(priced) {
497
+ return { total: priced.cost, models: [contributionModel(priced)] };
498
+ }
499
+ /**
500
+ * Fold one committed event into a session's billed-spend state.
501
+ *
502
+ * Priced samples come from `assistant/message` (its own reported usage, or the
503
+ * stream's last usage chunk) and `assistant/attempt` (the stream's last usage
504
+ * chunk, priced with the model of the latest `request/header`, since an
505
+ * attempt carries no route). A sample for the same `(turn, step)` replaces the
506
+ * previous one; `llm/retry-started` closes the replacement slot so a retried
507
+ * attempt adds. Every other event is inert and returns the same state
508
+ * reference.
509
+ * @param state - the previous fold state.
510
+ * @param event - the committed event.
511
+ * @param billing - resolved pricing with peak-hour windows.
512
+ * @param names - model id → display label.
513
+ * @returns the next state (the same reference when nothing was priced).
514
+ */
515
+ export function applyBillingEvent(state, event, billing, names) {
516
+ if (event.seq < state.inheritedEventCount)
517
+ return state;
518
+ // `assistant/attempt`, `llm/retry-started`, and the embedded stream are all
519
+ // newer than the npm baseline this package builds against, so their fields
520
+ // are read structurally.
521
+ const type = event.type;
522
+ if (type === 'request/header') {
523
+ const model = event.data
524
+ ?.header?.config?.model;
525
+ return typeof model === 'string' && model.length > 0 && model !== state.model ? { ...state, model } : state;
526
+ }
527
+ const data = event.data;
528
+ if (type === 'llm/retry-started') {
529
+ if (typeof data?.turn !== 'number' || typeof data.step !== 'number')
530
+ return state;
531
+ const last = state.last;
532
+ if (last === null || last.turn !== data.turn || last.step !== data.step)
533
+ return state;
534
+ return { ...state, last: null };
535
+ }
536
+ if (type !== 'assistant/message' && type !== 'assistant/attempt')
537
+ return state;
538
+ const usage = (type === 'assistant/message' ? data?.usage : undefined) ?? streamUsageOf(event);
539
+ if (!isTokenUsage(usage))
540
+ return state;
541
+ const model = type === 'assistant/message' ? data?.message?.source?.model : state.model;
542
+ if (typeof model !== 'string' || model.length === 0)
543
+ return state;
544
+ const priced = priceUsage(beijingPartsOf(event.time), usage, model, billing, names);
545
+ if (priced === undefined)
546
+ return state;
547
+ let session = state.session;
548
+ let spend = state.spend;
549
+ let dayKey = state.dayKey;
550
+ const last = state.last;
551
+ const turn = typeof data?.turn === 'number' ? data.turn : 0;
552
+ const step = typeof data?.step === 'number' ? data.step : 0;
553
+ if (last !== null && last.turn === turn && last.step === step) {
554
+ session = subtractSpend(session, last.spend);
555
+ if (last.dayKey === dayKey)
556
+ spend = subtractSpend(spend, last.spend);
557
+ }
558
+ session = addEventContribution(session, priced);
559
+ if (dayKey === priced.dayKey) {
560
+ spend = addEventContribution(spend, priced);
561
+ }
562
+ else if (dayKey === '' || priced.dayKey > dayKey) {
563
+ // The session log is append-only and chronological, so a strictly older
564
+ // day cannot legally follow; ignore it for the latest-day state (the
565
+ // whole-session total still accrues).
566
+ dayKey = priced.dayKey;
567
+ spend = addEventContribution(emptyTodaySpend(), priced);
568
+ }
569
+ return {
570
+ ...state,
571
+ dayKey,
572
+ spend,
573
+ session,
574
+ last: { turn, step, dayKey: priced.dayKey, spend: contributionSpend(priced) },
575
+ };
576
+ }
577
+ /**
578
+ * Mutable wrapper over {@link applyBillingEvent} for the pure pricing paths:
579
+ * feed events in order, read the folded spend.
580
+ */
581
+ export class BillingFolder {
582
+ billing;
583
+ state;
584
+ /**
585
+ * @param billing - resolved pricing with peak-hour windows.
586
+ * @param catalog - model display rows, in presentation order.
587
+ * @param inheritedEventCount - fork boundary to skip (default 0).
588
+ */
589
+ constructor(billing, catalog, inheritedEventCount = 0) {
590
+ this.billing = billing;
591
+ this.names = new Map(catalog.map(model => [model.id, model.name]));
592
+ this.state = emptyBillingFoldState(inheritedEventCount);
593
+ }
594
+ names;
595
+ /** Fold one event. */
596
+ add(event) {
597
+ this.state = applyBillingEvent(this.state, event, this.billing, this.names);
598
+ }
599
+ /** Fold every event, in order. */
600
+ addAll(events) {
601
+ for (const event of events)
602
+ this.add(event);
603
+ }
604
+ /** The folded state (live reference; do not mutate). */
605
+ get fold() {
606
+ return this.state;
607
+ }
608
+ }
247
609
  /**
248
610
  * Merge one priced event's contribution into an accumulator spend (pure:
249
611
  * returns a new spend, never mutates its input).
@@ -276,36 +638,11 @@ export function mergeTodaySpend(target, source) {
276
638
  return { total: target.total + source.total, models: [...rows.values()] };
277
639
  }
278
640
  /**
279
- * Price a set of billed events at the official per-model rates, applying the
280
- * peak/off-peak table per event by its Beijing-time hour and weekday (peak
281
- * windows apply Monday–Friday only; weekends are off-peak). Each
282
- * `assistant/message` event with usage contributes cache-hit input, cache-miss
283
- * input (uncached input plus cache writes), and output (reasoning included)
284
- * tokens at the rate of its own timestamp, with the three component costs
285
- * carried separately; a model with usage but no pricing row is omitted (the
286
- * published table prices only the two V4 rows).
287
- * @param events - the events to price.
288
- * @param billing - resolved pricing with peak-hour windows.
289
- * @param names - model id → display label.
290
- * @param dayKey - when provided, only events on this Beijing calendar day contribute.
291
- * @param startSeq - when provided, only events with `seq >= startSeq` contribute
292
- * (a forked session's inherited prefix, `seq < startSeq`, is skipped).
293
- * @returns the total cost plus one row per priced model.
294
- */
295
- function priceEvents(events, billing, names, dayKey, startSeq = 0) {
296
- const accumulator = new SpendAccumulator();
297
- for (const event of events) {
298
- if (event.seq < startSeq)
299
- continue;
300
- const priced = priceEvent(event, billing, names);
301
- if (priced === undefined || (dayKey !== undefined && priced.dayKey !== dayKey))
302
- continue;
303
- accumulator.add(priced);
304
- }
305
- return accumulator.finish();
306
- }
307
- /**
308
- * Price one session's complete event log at the official per-model rates.
641
+ * Price one session's complete event log at the official per-model rates,
642
+ * with DSH's attempt semantics: every provider-reported sample (an
643
+ * `assistant/message`'s usage, or an `assistant/attempt`'s stream usage)
644
+ * contributes, a later sample for the same `(turn, step)` replaces the earlier
645
+ * one, and `llm/retry-started` makes the retried attempt add.
309
646
  * @param events - one session's complete event log.
310
647
  * @param billing - resolved pricing with peak-hour windows.
311
648
  * @param catalog - model display rows, in presentation order.
@@ -316,17 +653,18 @@ function priceEvents(events, billing, names, dayKey, startSeq = 0) {
316
653
  * @returns the session's total cost plus one row per priced model.
317
654
  */
318
655
  export function computeSessionSpend(events, billing, catalog, startSeq = 0) {
319
- const names = new Map(catalog.map(model => [model.id, model.name]));
320
- return priceEvents(events, billing, names, undefined, startSeq);
656
+ const folder = new BillingFolder(billing, catalog, startSeq);
657
+ folder.addAll(events);
658
+ return folder.fold.session;
321
659
  }
322
660
  /**
323
- * Price one completed Turn's billed usage at the official per-model rates,
324
- * identified by its closing assistant message id. The turn's events are those
325
- * between its `turn/start` and `turn/end` (both matched by the message's own
326
- * turn coordinate); each priced event applies the peak/off-peak table by its
327
- * Beijing-time hour and weekday. A message that cannot be located, a turn
328
- * without bracketing `turn/start` / `turn/end` events (for example after
329
- * compaction), or a session with no priced usage prices to zero.
661
+ * Price one completed Turn's billed usage, identified by its closing
662
+ * assistant message id. The turn's events are those between its `turn/start`
663
+ * and `turn/end` (both matched by the message's own turn coordinate), priced
664
+ * with the same attempt semantics as {@link computeSessionSpend}. A message
665
+ * that cannot be located, a turn without bracketing `turn/start` / `turn/end`
666
+ * events (for example after compaction), or a session with no priced usage
667
+ * prices to zero.
330
668
  * @param events - one session's complete event log.
331
669
  * @param billing - resolved pricing with peak-hour windows.
332
670
  * @param catalog - model display rows, in presentation order.
@@ -334,7 +672,18 @@ export function computeSessionSpend(events, billing, catalog, startSeq = 0) {
334
672
  * @returns the turn's total cost in CNY.
335
673
  */
336
674
  export function computeTurnSpend(events, billing, catalog, messageId) {
337
- const names = new Map(catalog.map(model => [model.id, model.name]));
675
+ return { total: turnCostOf(events, billing, catalog, messageId) };
676
+ }
677
+ /**
678
+ * The total cost of the Turn containing `messageId`, folded with the shared
679
+ * attempt semantics (see {@link applyBillingEvent}).
680
+ * @param events - one session's complete event log.
681
+ * @param billing - resolved pricing with peak-hour windows.
682
+ * @param catalog - model display rows, in presentation order.
683
+ * @param messageId - one assistant message inside the Turn.
684
+ * @returns the Turn's total cost in CNY, or 0 when the Turn cannot be located.
685
+ */
686
+ function turnCostOf(events, billing, catalog, messageId) {
338
687
  let turn;
339
688
  for (const event of events) {
340
689
  if (event.type !== 'assistant/message')
@@ -345,8 +694,8 @@ export function computeTurnSpend(events, billing, catalog, messageId) {
345
694
  break;
346
695
  }
347
696
  if (turn === undefined)
348
- return { total: 0 };
349
- const accumulator = new SpendAccumulator();
697
+ return 0;
698
+ const folder = new BillingFolder(billing, catalog);
350
699
  let active = false;
351
700
  for (const event of events) {
352
701
  if (event.type === 'turn/start' && event.data.turn === turn) {
@@ -357,17 +706,117 @@ export function computeTurnSpend(events, billing, catalog, messageId) {
357
706
  break;
358
707
  if (!active)
359
708
  continue;
360
- const priced = priceEvent(event, billing, names);
361
- if (priced !== undefined)
362
- accumulator.add(priced);
709
+ folder.add(event);
710
+ }
711
+ return folder.fold.session.total;
712
+ }
713
+ /**
714
+ * Incremental single-pass fold of one session's completed-Turn costs, keyed by
715
+ * the id of every assistant message inside each Turn. Feeding the fold only
716
+ * the appended tail keeps a growing session's map current in O(new events)
717
+ * instead of re-scanning the whole log per message.
718
+ *
719
+ * Semantics are exactly {@link computeTurnSpend}'s: a Turn is the
720
+ * `turn/start`..`turn/end` range (matched by the event's own turn coordinate),
721
+ * every priced event inside it contributes at its own timestamp's rate, and a
722
+ * message outside any bracket contributes nothing.
723
+ */
724
+ export class SessionTurnSpendFolder {
725
+ billing;
726
+ catalog;
727
+ rows = [];
728
+ ids = [];
729
+ /** Events of the open Turn, folded with the shared attempt semantics on close. */
730
+ events = [];
731
+ open = false;
732
+ /** Events already fed; a shorter log resets the fold. */
733
+ cursor = 0;
734
+ /**
735
+ * @param billing - resolved pricing with peak-hour windows.
736
+ * @param catalog - model display rows, in presentation order.
737
+ */
738
+ constructor(billing, catalog) {
739
+ this.billing = billing;
740
+ this.catalog = catalog;
741
+ }
742
+ /** How many events have been folded so far (the host's incremental cursor). */
743
+ get processed() {
744
+ return this.cursor;
363
745
  }
364
- return { total: accumulator.finish().total };
746
+ /**
747
+ * Fold every event from the cursor to the end of the log. A log shorter than
748
+ * the cursor (rewritten session) restarts the fold from an empty state.
749
+ * @param events - the session's complete event log, in seq order.
750
+ */
751
+ feed(events) {
752
+ if (events.length < this.cursor)
753
+ this.reset();
754
+ for (let index = this.cursor; index < events.length; index += 1) {
755
+ const event = events[index];
756
+ if (event.type === 'turn/start') {
757
+ this.open = true;
758
+ this.ids = [];
759
+ this.events = [];
760
+ continue;
761
+ }
762
+ if (event.type === 'turn/end') {
763
+ if (this.open) {
764
+ const folder = new BillingFolder(this.billing, this.catalog);
765
+ folder.addAll(this.events);
766
+ const total = folder.fold.session.total;
767
+ for (const messageId of this.ids)
768
+ this.rows.push({ messageId, total });
769
+ }
770
+ this.open = false;
771
+ this.ids = [];
772
+ this.events = [];
773
+ continue;
774
+ }
775
+ if (!this.open)
776
+ continue;
777
+ if (event.type === 'assistant/message')
778
+ this.ids.push(event.data.message.id);
779
+ this.events.push(event);
780
+ }
781
+ this.cursor = events.length;
782
+ }
783
+ /** The folded map; the fold stays usable afterwards. */
784
+ finish() {
785
+ return { turns: [...this.rows] };
786
+ }
787
+ /** Drop the fold state so the next feed starts from the log's beginning. */
788
+ reset() {
789
+ this.rows.length = 0;
790
+ this.ids = [];
791
+ this.events = [];
792
+ this.open = false;
793
+ this.cursor = 0;
794
+ }
795
+ }
796
+ /**
797
+ * Price every completed Turn of one session in a single pass (the pure
798
+ * equivalent of {@link SessionTurnSpendFolder}).
799
+ * @param events - one session's complete event log.
800
+ * @param billing - resolved pricing with peak-hour windows.
801
+ * @param catalog - model display rows, in presentation order.
802
+ * @returns one row per assistant message inside a completed Turn, in log order.
803
+ */
804
+ export function computeSessionTurnSpends(events, billing, catalog) {
805
+ const folder = new SessionTurnSpendFolder(billing, catalog);
806
+ folder.feed(events);
807
+ return folder.finish();
365
808
  }
366
809
  /**
367
- * Price every event whose Beijing-time calendar day is the day of `now`,
368
- * aggregating across every session's event log. Events from other Beijing
369
- * days are ignored, so a caller passes the concatenated logs of all sessions.
370
- * @param events - every session's complete event log, concatenated.
810
+ * Price one session's log for the Beijing-time calendar day of `now`. Events
811
+ * after the reference day are ignored; the fold's latest-day state then
812
+ * answers the query exactly (empty when the session's latest priced day is not
813
+ * the reference day). Pricing follows {@link applyBillingEvent} (attempt
814
+ * samples with same-step replacement).
815
+ *
816
+ * The fold's `(turn, step)` replacement slot is per session, so callers must
817
+ * pass ONE session's log; aggregate across sessions with
818
+ * {@link mergeTodaySpend}.
819
+ * @param events - one session's complete event log.
371
820
  * @param billing - resolved pricing with peak-hour windows.
372
821
  * @param catalog - model display rows, in presentation order.
373
822
  * @param now - the reference moment whose Beijing-time calendar day is "today".
@@ -375,7 +824,13 @@ export function computeTurnSpend(events, billing, catalog, messageId) {
375
824
  */
376
825
  export function computeTodaySpend(events, billing, catalog, now = new Date()) {
377
826
  const day = beijingDayKey(now);
378
- const names = new Map(catalog.map(model => [model.id, model.name]));
379
- return priceEvents(events, billing, names, day);
827
+ const folder = new BillingFolder(billing, catalog);
828
+ for (const event of events) {
829
+ // Only events up to the reference day can contribute.
830
+ if (beijingPartsOf(event.time).dayKey > day)
831
+ continue;
832
+ folder.add(event);
833
+ }
834
+ return folder.fold.dayKey === day ? folder.fold.spend : emptyTodaySpend();
380
835
  }
381
836
  //# sourceMappingURL=billing.js.map