@genee/omp-opsx-addon 0.7.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +441 -71
- package/index.ts +1030 -208
- package/lib/agent-defs.ts +55 -82
- package/lib/complexity-router.ts +336 -0
- package/lib/concurrency-control.ts +1246 -0
- package/lib/concurrency-multiproc.worker.ts +376 -0
- package/lib/concurrency-state.ts +397 -0
- package/lib/concurrency-tuner.ts +773 -0
- package/lib/concurrency-writer.ts +894 -0
- package/lib/continuity-guard.ts +117 -0
- package/lib/direct-fetchers.ts +18 -3
- package/lib/family-filter.ts +115 -2
- package/lib/model-roles.ts +41 -39
- package/lib/model-selector.ts +300 -36
- package/lib/model-speed.ts +189 -0
- package/lib/pick-model-render.ts +414 -0
- package/lib/pipe-core.ts +325 -98
- package/lib/pipe-push.ts +136 -5
- package/lib/provider-variants.ts +154 -0
- package/lib/selection-filters.ts +42 -1
- package/lib/stall-detector.ts +185 -0
- package/lib/system-prompt.ts +17 -9
- package/lib/tiers-data.ts +40 -5
- package/lib/tiers-updater.ts +51 -8
- package/lib/unified-config.ts +732 -27
- package/lib/usage-poller.ts +44 -1
- package/lib/usage-render.ts +57 -4
- package/lib/usage-widget.ts +18 -4
- package/package.json +1 -1
- package/skills/opsx-orchestration-protocol/SKILL.md +87 -0
|
@@ -0,0 +1,773 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pure AIMD concurrency controller + host-signal parsing/aggregation/summary.
|
|
3
|
+
*
|
|
4
|
+
* Zero IO: every export here is deterministic and takes an explicit options
|
|
5
|
+
* object plus an injectable clock (`now`); nothing reads the clock on its own.
|
|
6
|
+
* Host log/DB/config access lives in `lib/concurrency-control.ts`. This module
|
|
7
|
+
* deliberately never imports `ResolvedOpsxConfig` — the assembly layer injects
|
|
8
|
+
* the parsed `concurrency_*` values as {@link TunerOptions}.
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
/** Clean evaluation windows required before the additive increase (`M`). */
|
|
12
|
+
export const UP_WINDOWS = 2;
|
|
13
|
+
/** Consecutive bad evaluation windows required before the multiplicative decrease (`N`). */
|
|
14
|
+
export const DOWN_BAD_STREAK = 2;
|
|
15
|
+
|
|
16
|
+
/** Host log messages that bracket a queue wait. */
|
|
17
|
+
export const BLOCKED_MESSAGE = 'Provider in-flight limit blocked request';
|
|
18
|
+
export const WAIT_COMPLETED_MESSAGE = 'Provider in-flight limit wait completed';
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Injected configuration for the whole control loop. Field names mirror the
|
|
22
|
+
* `concurrency_*` opsx.yml keys; defaults are the spec defaults and are the
|
|
23
|
+
* single injection point for tests and for the assembly layer.
|
|
24
|
+
*/
|
|
25
|
+
export interface TunerOptions {
|
|
26
|
+
/**
|
|
27
|
+
* `concurrency_floor` — global lower bound of the adaptive range; MUST be >= 1.
|
|
28
|
+
* Each provider's *effective* floor is `min(floor, initial)` (see
|
|
29
|
+
* {@link effectiveFloor}): this knob sets the range but MUST NOT raise a
|
|
30
|
+
* provider's own static value.
|
|
31
|
+
*/
|
|
32
|
+
floor: number;
|
|
33
|
+
/** `concurrency_ceiling` — global upper bound of the adaptive range. */
|
|
34
|
+
ceiling: number;
|
|
35
|
+
/** `concurrency_start` — start value used when no shared value exists yet. */
|
|
36
|
+
start: number;
|
|
37
|
+
/**
|
|
38
|
+
* `concurrency_up_ms` — clean-window length. The collector's evaluation
|
|
39
|
+
* cadence comes from `evalIntervalMs`; this is how much *clean time* must
|
|
40
|
+
* accumulate before {@link evaluateWindow} credits one clean window, so
|
|
41
|
+
* the two knobs move independently.
|
|
42
|
+
*/
|
|
43
|
+
upMs: number;
|
|
44
|
+
/** `concurrency_down_ms` — cooldown after a multiplicative decrease. */
|
|
45
|
+
downMs: number;
|
|
46
|
+
/** `concurrency_latency_k` — inflation ratio that marks a window latency-bad. */
|
|
47
|
+
k: number;
|
|
48
|
+
/** `concurrency_min_samples` — latency samples below which latency is not judged. */
|
|
49
|
+
minSamples: number;
|
|
50
|
+
/** `concurrency_queue_wait_high_ms` — guardrail p50 threshold. */
|
|
51
|
+
queueWaitHighMs: number;
|
|
52
|
+
/** `concurrency_queue_wait_window_ms` — guardrail percentile window. */
|
|
53
|
+
queueWaitWindowMs: number;
|
|
54
|
+
/**
|
|
55
|
+
* `concurrency_eval_interval_ms` — evaluation cadence. The collector's
|
|
56
|
+
* absolute-time gate enforces it; carried here for injection parity.
|
|
57
|
+
*/
|
|
58
|
+
evalIntervalMs: number;
|
|
59
|
+
/**
|
|
60
|
+
* `concurrency_decay_after_ms` — idle period after which `L` decays one
|
|
61
|
+
* notch toward `initial` per elapsed period.
|
|
62
|
+
*/
|
|
63
|
+
decayAfterMs: number;
|
|
64
|
+
/** EWMA smoothing factor for the latency baseline (internal constant, not a config key). */
|
|
65
|
+
ewmaAlpha: number;
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** Spec defaults for every knob the control loop consumes. */
|
|
69
|
+
export const TUNER_DEFAULTS: TunerOptions = {
|
|
70
|
+
floor: 4,
|
|
71
|
+
ceiling: 16,
|
|
72
|
+
start: 4,
|
|
73
|
+
upMs: 30_000,
|
|
74
|
+
downMs: 60_000,
|
|
75
|
+
k: 2,
|
|
76
|
+
minSamples: 5,
|
|
77
|
+
queueWaitHighMs: 2_000,
|
|
78
|
+
queueWaitWindowMs: 60_000,
|
|
79
|
+
evalIntervalMs: 30_000,
|
|
80
|
+
decayAfterMs: 86_400_000,
|
|
81
|
+
ewmaAlpha: 0.3,
|
|
82
|
+
};
|
|
83
|
+
|
|
84
|
+
/** Bad-signal classes, ordered by fixed priority (highest first). */
|
|
85
|
+
export type BadEventKind = 'rate-limit' | 'latency' | 'throughput';
|
|
86
|
+
|
|
87
|
+
/** Per-provider control state. Moved across eval windows by {@link evaluateWindow}. */
|
|
88
|
+
export interface ProviderConcurrencyState {
|
|
89
|
+
/** Current adaptive limit `L`, always within `[floor, ceiling]`. */
|
|
90
|
+
limit: number;
|
|
91
|
+
/** Value `L` started from (decay target, restart fallback). */
|
|
92
|
+
initial: number;
|
|
93
|
+
floor: number;
|
|
94
|
+
ceiling: number;
|
|
95
|
+
/** Consecutive clean windows accumulated toward the additive increase. */
|
|
96
|
+
cleanWindows: number;
|
|
97
|
+
/** Clean time (ms) accumulated toward the next {@link cleanWindows} credit. */
|
|
98
|
+
cleanWindowMs: number;
|
|
99
|
+
/** Consecutive bad windows accumulated toward the multiplicative decrease. */
|
|
100
|
+
badStreak: number;
|
|
101
|
+
/** Epoch ms until which the additive increase is disabled (0 = not cooling). */
|
|
102
|
+
cooldownUntil: number;
|
|
103
|
+
/** EWMA baseline of provider generation duration in ms (`null` until observed). */
|
|
104
|
+
ewmaGenMs: number | null;
|
|
105
|
+
/** Epoch ms of the last window that carried any observation. */
|
|
106
|
+
lastObservationAt: number;
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
/** One host-log `blocked` / `wait completed` pair. */
|
|
110
|
+
export interface QueueWaitSample {
|
|
111
|
+
provider: string;
|
|
112
|
+
/** Completion timestamp (epoch ms). */
|
|
113
|
+
at: number;
|
|
114
|
+
/** Wait duration in ms (`completion - blocked`). */
|
|
115
|
+
ms: number;
|
|
116
|
+
/** `limit` reported on the blocked line, when present. */
|
|
117
|
+
limit: number | null;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/** Signals observed for one provider during one evaluation window. */
|
|
121
|
+
export interface WindowSignals {
|
|
122
|
+
/**
|
|
123
|
+
* R1 observation evidence: `true` when this window carried at least one
|
|
124
|
+
* generation sample for the provider. A window without evidence MUST NOT be
|
|
125
|
+
* credited as a clean window (see {@link evaluateWindow}).
|
|
126
|
+
*
|
|
127
|
+
* Criterion — the single source of truth for the freeze — is: the provider's
|
|
128
|
+
* `model_perf` delta for this window yields a non-empty {@link genSamplesMs},
|
|
129
|
+
* i.e. it really generated at least one sample in this window (the control
|
|
130
|
+
* layer derives it with `genSamplesFromPerfWindow`). Two deliberate
|
|
131
|
+
* exclusions: ① rate-limit events never qualify — such a window is
|
|
132
|
+
* classified bad before the additive branch is reached, so it can never be a
|
|
133
|
+
* *clean* window anyway; ② queue waits are load evidence, not service
|
|
134
|
+
* evidence (`L` may rise only on proof the provider served load well). This
|
|
135
|
+
* is narrower on purpose than the R5 decay anchor (`lastObservationAt`),
|
|
136
|
+
* which counts any signal as an observation.
|
|
137
|
+
*/
|
|
138
|
+
observed: boolean;
|
|
139
|
+
/** ① explicit provider-level failures (429 / rate-limit / 5xx / timeout) in this window. */
|
|
140
|
+
rateLimitCount: number;
|
|
141
|
+
/** ③ generation-duration samples for this provider in this window (ms). */
|
|
142
|
+
genSamplesMs: number[];
|
|
143
|
+
/** ④ aggregated `output_tokens / gen_ms` for this window, or `null` when unavailable. */
|
|
144
|
+
tokensPerMs: number | null;
|
|
145
|
+
/** ④ EWMA baseline for {@link tokensPerMs}, or `null` until a baseline exists. */
|
|
146
|
+
tokensPerMsEwma: number | null;
|
|
147
|
+
/** ② queue-wait observations (already restricted to this provider). */
|
|
148
|
+
queueWait: QueueWaitSample[];
|
|
149
|
+
/**
|
|
150
|
+
* ② The guardrail statistic of this window, computed by the caller over
|
|
151
|
+
* `concurrency_queue_wait_window_ms` (the assembly layer passes exactly the
|
|
152
|
+
* p50 it also shows in the summary — one number, one verdict). `null` = the
|
|
153
|
+
* guardrail window carried no sample. When omitted, the evaluator derives it
|
|
154
|
+
* from {@link queueWait} with {@link queueWaitPercentiles}, so a pure caller
|
|
155
|
+
* that only holds the raw samples stays exact.
|
|
156
|
+
*/
|
|
157
|
+
queueWaitP50Ms?: number | null;
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
/** Result of moving a provider state through one evaluation window. */
|
|
161
|
+
export interface WindowEvaluation {
|
|
162
|
+
/** Next state. The input state is never mutated. */
|
|
163
|
+
state: ProviderConcurrencyState;
|
|
164
|
+
/** Merged bad-event class for this window (`null` = no bad signal). */
|
|
165
|
+
badKind: BadEventKind | null;
|
|
166
|
+
/** `true` when this window actually moved `limit` down. */
|
|
167
|
+
decreased: boolean;
|
|
168
|
+
/** `true` when this window raised `limit` by 1. */
|
|
169
|
+
increased: boolean;
|
|
170
|
+
/**
|
|
171
|
+
* `true` when the ② queue guardrail was detected *and* blocked this window's
|
|
172
|
+
* additive increase (it says nothing about ③④: those are masked instead, see
|
|
173
|
+
* {@link latencySuppressedByQueue}).
|
|
174
|
+
*/
|
|
175
|
+
queueSuppressed: boolean;
|
|
176
|
+
/**
|
|
177
|
+
* `true` when a ③ latency / ④ throughput verdict was discarded because the
|
|
178
|
+
* queue was high: the extra generation time is attributable to our own gate,
|
|
179
|
+
* not to provider capacity. Signal ① (rate limit / explicit timeout) is
|
|
180
|
+
* never masked.
|
|
181
|
+
*/
|
|
182
|
+
latencySuppressedByQueue: boolean;
|
|
183
|
+
/**
|
|
184
|
+
* `true` when this window may fold into the EWMA baselines (latency and
|
|
185
|
+
* throughput): healthy (no bad event) and not queue-masked. A masked or bad
|
|
186
|
+
* window MUST NOT move the baseline that the very next verdict is measured
|
|
187
|
+
* against.
|
|
188
|
+
*/
|
|
189
|
+
baselineEligible: boolean;
|
|
190
|
+
/** `true` when the additive increase is blocked by the post-decrease cooldown. */
|
|
191
|
+
cooldownActive: boolean;
|
|
192
|
+
/**
|
|
193
|
+
* `true` when this window was frozen for lack of an observation sample
|
|
194
|
+
* (R1): no clean time accrued, clean streak left untouched.
|
|
195
|
+
*/
|
|
196
|
+
frozen: boolean;
|
|
197
|
+
/** Guardrail p50 over {@link TunerOptions.queueWaitWindowMs} (`null` = no samples). */
|
|
198
|
+
queueWaitP50Ms: number | null;
|
|
199
|
+
/** Median of this window's generation samples (`null` = no samples). */
|
|
200
|
+
latencyMedianMs: number | null;
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
function truncInt(value: number, fallback: number): number {
|
|
204
|
+
return Number.isFinite(value) ? Math.trunc(value) : fallback;
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
/** Clamp an integer into `[max(1, floor), max(floor, ceiling)]`. */
|
|
208
|
+
export function clampLimit(value: number, floor: number, ceiling: number): number {
|
|
209
|
+
const lo = Math.max(1, truncInt(floor, 1));
|
|
210
|
+
const hi = Math.max(lo, truncInt(ceiling, lo));
|
|
211
|
+
return Math.min(hi, Math.max(lo, truncInt(value, lo)));
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/** Exponential moving average; the first observation seeds the baseline. */
|
|
215
|
+
export function ewmaUpdate(previous: number | null, value: number, alpha: number): number {
|
|
216
|
+
if (previous === null || !Number.isFinite(previous)) return value;
|
|
217
|
+
const a = Number.isFinite(alpha) ? Math.min(1, Math.max(0, alpha)) : 1;
|
|
218
|
+
return previous + a * (value - previous);
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/**
|
|
222
|
+
* Inclusive linear-interpolation percentile over an ascending-sorted array.
|
|
223
|
+
* `p` is in `[0, 1]`; returns `null` for an empty array.
|
|
224
|
+
*/
|
|
225
|
+
function percentile(sortedAsc: number[], p: number): number | null {
|
|
226
|
+
const n = sortedAsc.length;
|
|
227
|
+
if (n === 0) return null;
|
|
228
|
+
if (n === 1) return sortedAsc[0];
|
|
229
|
+
const idx = (n - 1) * Math.min(1, Math.max(0, p));
|
|
230
|
+
const lo = Math.floor(idx);
|
|
231
|
+
const hi = Math.ceil(idx);
|
|
232
|
+
if (lo === hi) return sortedAsc[lo];
|
|
233
|
+
return sortedAsc[lo] + (sortedAsc[hi] - sortedAsc[lo]) * (idx - lo);
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
/** Median of an unsorted sample list (`null` when empty). */
|
|
237
|
+
function median(values: number[]): number | null {
|
|
238
|
+
if (values.length === 0) return null;
|
|
239
|
+
return percentile([...values].sort((a, b) => a - b), 0.5);
|
|
240
|
+
}
|
|
241
|
+
|
|
242
|
+
/**
|
|
243
|
+
* Effective lower bound of one provider's adaptive range: `min(concurrency_floor,
|
|
244
|
+
* initial)`.
|
|
245
|
+
*
|
|
246
|
+
* The provider's own starting value wins whenever it is smaller: the global floor
|
|
247
|
+
* may shape the range but MUST NOT raise a user's static value, so a provider
|
|
248
|
+
* sitting at 3 keeps `L >= 3` — as both its floor and its start — instead of
|
|
249
|
+
* being pulled up to `concurrency_floor`.
|
|
250
|
+
*/
|
|
251
|
+
export function effectiveFloor(configuredFloor: number, initial: number): number {
|
|
252
|
+
const configured = Math.max(1, truncInt(configuredFloor, 1));
|
|
253
|
+
const seed = Number.isFinite(initial) ? Math.trunc(initial) : configured;
|
|
254
|
+
return Math.max(1, Math.min(configured, seed));
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* Start state for a provider.
|
|
259
|
+
*
|
|
260
|
+
* `initial = clamp(sharedLimit ?? start, effectiveFloor(floor, seed), ceiling)`
|
|
261
|
+
* and `limit = initial`, so a provider that already has a static
|
|
262
|
+
* `maxInFlightRequests` value starts there (yielding `|new - old| = 0`, i.e. no
|
|
263
|
+
* write), while a provider only listed in `concurrency_providers` starts at the
|
|
264
|
+
* conservative `start`. `floor` is the effective floor, capped by the provider's
|
|
265
|
+
* own start and by its (possibly overridden) ceiling.
|
|
266
|
+
*/
|
|
267
|
+
export function createProviderState(
|
|
268
|
+
init: {
|
|
269
|
+
/** Current `providers.maxInFlightRequests[provider]`, or `null` when absent. */
|
|
270
|
+
sharedLimit: number | null;
|
|
271
|
+
/** `concurrency_start` fallback. */
|
|
272
|
+
start: number;
|
|
273
|
+
/** `concurrency_ceiling_by_provider[provider]` override. */
|
|
274
|
+
ceilingOverride?: number | null;
|
|
275
|
+
options: TunerOptions;
|
|
276
|
+
},
|
|
277
|
+
): ProviderConcurrencyState {
|
|
278
|
+
const ceiling = Math.max(1, truncInt(init.ceilingOverride ?? init.options.ceiling, init.options.ceiling));
|
|
279
|
+
const seed = init.sharedLimit === null || init.sharedLimit === undefined ? init.start : init.sharedLimit;
|
|
280
|
+
const floor = Math.min(ceiling, effectiveFloor(init.options.floor, seed));
|
|
281
|
+
const limit = clampLimit(seed, floor, ceiling);
|
|
282
|
+
return {
|
|
283
|
+
limit,
|
|
284
|
+
initial: limit,
|
|
285
|
+
floor,
|
|
286
|
+
ceiling,
|
|
287
|
+
cleanWindows: 0,
|
|
288
|
+
cleanWindowMs: 0,
|
|
289
|
+
badStreak: 0,
|
|
290
|
+
cooldownUntil: 0,
|
|
291
|
+
ewmaGenMs: null,
|
|
292
|
+
lastObservationAt: 0,
|
|
293
|
+
};
|
|
294
|
+
}
|
|
295
|
+
|
|
296
|
+
/** Samples within `[now - windowMs, now]`. */
|
|
297
|
+
function queueWaitSamplesInWindow(
|
|
298
|
+
samples: QueueWaitSample[],
|
|
299
|
+
now: number,
|
|
300
|
+
windowMs: number,
|
|
301
|
+
): number[] {
|
|
302
|
+
const since = now - windowMs;
|
|
303
|
+
const values: number[] = [];
|
|
304
|
+
for (const sample of samples) {
|
|
305
|
+
if (sample.at >= since && sample.at <= now) values.push(sample.ms);
|
|
306
|
+
}
|
|
307
|
+
return values;
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
/** p50/p90 of the queue waits observed in the guardrail window. */
|
|
311
|
+
export function queueWaitPercentiles(
|
|
312
|
+
samples: QueueWaitSample[],
|
|
313
|
+
now: number,
|
|
314
|
+
windowMs: number,
|
|
315
|
+
): { p50: number | null; p90: number | null; count: number } {
|
|
316
|
+
const values = queueWaitSamplesInWindow(samples, now, windowMs).sort((a, b) => a - b);
|
|
317
|
+
return { p50: percentile(values, 0.5), p90: percentile(values, 0.9), count: values.length };
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
/** ③ latency inflation: median above `k × EWMA` with at least `minSamples` samples. */
|
|
321
|
+
export function isLatencyBad(
|
|
322
|
+
medianMs: number | null,
|
|
323
|
+
sampleCount: number,
|
|
324
|
+
baselineMs: number | null,
|
|
325
|
+
options: TunerOptions,
|
|
326
|
+
): boolean {
|
|
327
|
+
if (medianMs === null || baselineMs === null || baselineMs <= 0) return false;
|
|
328
|
+
if (sampleCount < options.minSamples) return false;
|
|
329
|
+
return medianMs > options.k * baselineMs;
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
/**
|
|
333
|
+
* ④ throughput fallback: below `0.5 × EWMA`, and only while latency samples are
|
|
334
|
+
* sparse (`< minSamples`) — a provider with enough latency samples is never
|
|
335
|
+
* judged on throughput.
|
|
336
|
+
*/
|
|
337
|
+
export function isThroughputBad(
|
|
338
|
+
tokensPerMs: number | null,
|
|
339
|
+
baselineTokensPerMs: number | null,
|
|
340
|
+
latencySampleCount: number,
|
|
341
|
+
options: TunerOptions,
|
|
342
|
+
): boolean {
|
|
343
|
+
if (tokensPerMs === null || baselineTokensPerMs === null || baselineTokensPerMs <= 0) return false;
|
|
344
|
+
if (latencySampleCount >= options.minSamples) return false;
|
|
345
|
+
return tokensPerMs < 0.5 * baselineTokensPerMs;
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
/**
|
|
349
|
+
* Merge one window's signals into the single bad-event class that drives debounce
|
|
350
|
+
* and the multiplicative decrease. Priority ① > ③ > ④; queue wait (②) is not a
|
|
351
|
+
* bad-event source — it only feeds the guardrail.
|
|
352
|
+
*
|
|
353
|
+
* `queueHigh` is the ② verdict over `concurrency_queue_wait_window_ms`, applied
|
|
354
|
+
* *before* ③④ are judged: while the queue is high the extra generation time
|
|
355
|
+
* belongs to our own gate, so treating it as provider-capacity evidence is
|
|
356
|
+
* exactly the self-reinforcing spiral that
|
|
357
|
+
* drove three live providers down to `L = 1`. Signal ① is provider-side evidence
|
|
358
|
+
* and is never masked.
|
|
359
|
+
*/
|
|
360
|
+
export function classifyBadEvent(
|
|
361
|
+
signals: WindowSignals,
|
|
362
|
+
state: ProviderConcurrencyState,
|
|
363
|
+
options: TunerOptions,
|
|
364
|
+
queueHigh: boolean,
|
|
365
|
+
): BadEventKind | null {
|
|
366
|
+
if (signals.rateLimitCount > 0) return 'rate-limit';
|
|
367
|
+
if (queueHigh) return null;
|
|
368
|
+
return classifyLatencyThroughputBad(signals, state, options);
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
/**
|
|
372
|
+
* The ③④ merge *without* the ② mask: what {@link classifyBadEvent} would return
|
|
373
|
+
* if the queue were not high. Shared by the classifier and by
|
|
374
|
+
* {@link WindowEvaluation.latencySuppressedByQueue}, so the masked verdict is
|
|
375
|
+
* never a second implementation of the same judgement.
|
|
376
|
+
*/
|
|
377
|
+
function classifyLatencyThroughputBad(
|
|
378
|
+
signals: WindowSignals,
|
|
379
|
+
state: ProviderConcurrencyState,
|
|
380
|
+
options: TunerOptions,
|
|
381
|
+
): BadEventKind | null {
|
|
382
|
+
if (isLatencyBad(median(signals.genSamplesMs), signals.genSamplesMs.length, state.ewmaGenMs, options)) {
|
|
383
|
+
return 'latency';
|
|
384
|
+
}
|
|
385
|
+
if (isThroughputBad(signals.tokensPerMs, signals.tokensPerMsEwma, signals.genSamplesMs.length, options)) {
|
|
386
|
+
return 'throughput';
|
|
387
|
+
}
|
|
388
|
+
return null;
|
|
389
|
+
}
|
|
390
|
+
|
|
391
|
+
/**
|
|
392
|
+
* Advance one provider's state by one evaluation window.
|
|
393
|
+
*
|
|
394
|
+
* - bad window → clear clean counter, grow the bad streak; at `N = 2` halve
|
|
395
|
+
* `L = max(floor, floor(L / 2))` and start the cooldown. At most one decrease
|
|
396
|
+
* per cooldown cycle: bad events inside the cooldown keep accruing in the
|
|
397
|
+
* streak but MUST NOT move `L` again until it expires (anti-windup).
|
|
398
|
+
* - queue-wait p50 above the threshold → mask ③④ for this window *before* they
|
|
399
|
+
* are judged (signal ① still classifies and may decrease), suppress the
|
|
400
|
+
* increase, restart the clean-window count, and leave the EWMA baselines
|
|
401
|
+
* untouched (never a decrease on its own).
|
|
402
|
+
* - cooldown active → suppress the increase.
|
|
403
|
+
* - no observation sample in the window (R1) → freeze: no clean time accrues and
|
|
404
|
+
* the clean streak keeps its previous value. Climbing `L` is a bet that the
|
|
405
|
+
* provider served load well, and a window with nothing observed is no
|
|
406
|
+
* evidence, so a purely idle provider never walks up to `ceiling` (and never
|
|
407
|
+
* fights the R5 decay back down). See {@link WindowSignals.observed}.
|
|
408
|
+
* - otherwise the window's elapsed clean time accrues toward one
|
|
409
|
+
* `concurrency_up_ms` clean window (at most one per call — see `elapsedMs`);
|
|
410
|
+
* at `M = 2` clean windows `L += 1`, capped at `ceiling`.
|
|
411
|
+
*
|
|
412
|
+
* `elapsedMs` is the observed gap since the previous evaluation: it is what the
|
|
413
|
+
* clean-window length is measured against, so `concurrency_up_ms` (this
|
|
414
|
+
* function) and `concurrency_eval_interval_ms` (the caller's cadence) stay
|
|
415
|
+
* independent knobs. It defaults to one evaluation interval, which reproduces
|
|
416
|
+
* "one call == one clean window" for a caller that ticks at the default cadence.
|
|
417
|
+
*/
|
|
418
|
+
export function evaluateWindow(
|
|
419
|
+
state: ProviderConcurrencyState,
|
|
420
|
+
signals: WindowSignals,
|
|
421
|
+
now: number,
|
|
422
|
+
options: TunerOptions,
|
|
423
|
+
elapsedMs: number = options.evalIntervalMs,
|
|
424
|
+
): WindowEvaluation {
|
|
425
|
+
const next: ProviderConcurrencyState = { ...state };
|
|
426
|
+
// ② The caller may hand in the guardrail p50 it already computed (the number it
|
|
427
|
+
// also shows in the summary); otherwise it is derived from the raw samples, so
|
|
428
|
+
// a pure caller holding only those stays exact.
|
|
429
|
+
const explicitP50 = signals.queueWaitP50Ms;
|
|
430
|
+
const guardrail = explicitP50 === undefined
|
|
431
|
+
? queueWaitPercentiles(signals.queueWait, now, options.queueWaitWindowMs)
|
|
432
|
+
: null;
|
|
433
|
+
const p50 = guardrail ? guardrail.p50 : explicitP50 ?? null;
|
|
434
|
+
// ② guardrail verdict: no sample is never "high".
|
|
435
|
+
const queueHigh = p50 !== null && p50 > options.queueWaitHighMs;
|
|
436
|
+
const latencyMedianMs = median(signals.genSamplesMs);
|
|
437
|
+
// The mask is applied *before* the ③④ verdict is taken, not in the branch
|
|
438
|
+
// chain below: under load the latency inflation is manufactured by our own
|
|
439
|
+
// gate, and letting it win the classification is what made the queue
|
|
440
|
+
// guardrail unreachable during the incident.
|
|
441
|
+
const maskedKind = queueHigh ? classifyLatencyThroughputBad(signals, state, options) : null;
|
|
442
|
+
const badKind = classifyBadEvent(signals, state, options, queueHigh);
|
|
443
|
+
|
|
444
|
+
if (badKind !== null || latencyMedianMs !== null || signals.rateLimitCount > 0 || p50 !== null) {
|
|
445
|
+
next.lastObservationAt = now;
|
|
446
|
+
}
|
|
447
|
+
// Baseline hygiene: only a healthy, non-masked window may move the latency
|
|
448
|
+
// EWMA. A window we just judged inflated would blind the next verdict, and a
|
|
449
|
+
// masked window is our own gate's inflation — folding it in would raise the
|
|
450
|
+
// baseline and blunt the mask itself.
|
|
451
|
+
const baselineEligible = badKind === null && !queueHigh;
|
|
452
|
+
if (latencyMedianMs !== null && baselineEligible) {
|
|
453
|
+
next.ewmaGenMs = ewmaUpdate(next.ewmaGenMs, latencyMedianMs, options.ewmaAlpha);
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
let decreased = false;
|
|
457
|
+
let increased = false;
|
|
458
|
+
let queueSuppressed = false;
|
|
459
|
+
let frozen = false;
|
|
460
|
+
if (badKind !== null) {
|
|
461
|
+
next.cleanWindows = 0;
|
|
462
|
+
next.cleanWindowMs = 0;
|
|
463
|
+
next.badStreak += 1;
|
|
464
|
+
// Anti-windup: at most ONE multiplicative decrease per `concurrency_down_ms`
|
|
465
|
+
// cycle. Bad events observed inside the cooldown are not discarded — they
|
|
466
|
+
// keep accruing in `badStreak`, which is what decides the *next* decrease
|
|
467
|
+
// once the cooldown expires — but they MUST NOT halve `L` again while it
|
|
468
|
+
// runs: each extra halving tightens the very gate that produced the queue.
|
|
469
|
+
if (next.badStreak >= DOWN_BAD_STREAK && now >= next.cooldownUntil) {
|
|
470
|
+
next.badStreak = 0;
|
|
471
|
+
const halved = Math.max(next.floor, Math.floor(next.limit / 2));
|
|
472
|
+
decreased = halved !== next.limit;
|
|
473
|
+
next.limit = halved;
|
|
474
|
+
next.cooldownUntil = now + options.downMs;
|
|
475
|
+
}
|
|
476
|
+
} else if (queueHigh) {
|
|
477
|
+
next.cleanWindows = 0;
|
|
478
|
+
next.cleanWindowMs = 0;
|
|
479
|
+
queueSuppressed = true;
|
|
480
|
+
} else if (now < next.cooldownUntil) {
|
|
481
|
+
next.cleanWindows = 0;
|
|
482
|
+
next.cleanWindowMs = 0;
|
|
483
|
+
} else if (!signals.observed) {
|
|
484
|
+
// R1 freeze (see the doc above and {@link WindowSignals.observed}): a
|
|
485
|
+
// window without an observation sample neither earns clean time nor
|
|
486
|
+
// resets the streak — it is not evidence of a bad provider either, so
|
|
487
|
+
// the counters simply stay where the last observed window left them.
|
|
488
|
+
frozen = true;
|
|
489
|
+
} else {
|
|
490
|
+
// Effective clean window = max(up_ms, eval_interval_ms): a window can only
|
|
491
|
+
// be completed by an evaluation, and each evaluation credits at most one
|
|
492
|
+
// window (the remainder is dropped below), so any `up_ms` under the
|
|
493
|
+
// evaluation cadence behaves exactly like `eval_interval_ms`. The config
|
|
494
|
+
// layer already folds `up_ms < eval_interval_ms` into that value; this
|
|
495
|
+
// clamp keeps a direct (test/injected) `TunerOptions` equally honest.
|
|
496
|
+
const windowMs = Number.isFinite(options.upMs) && options.upMs > 0 ? options.upMs : TUNER_DEFAULTS.upMs;
|
|
497
|
+
const elapsed = Number.isFinite(elapsedMs) && elapsedMs > 0 ? elapsedMs : windowMs;
|
|
498
|
+
// A state that predates the accumulator (or came from a partial literal)
|
|
499
|
+
// starts its window now instead of poisoning the sum with NaN.
|
|
500
|
+
const accrued = Number.isFinite(next.cleanWindowMs) ? next.cleanWindowMs : 0;
|
|
501
|
+
next.cleanWindowMs = accrued + elapsed;
|
|
502
|
+
if (next.cleanWindowMs >= windowMs) {
|
|
503
|
+
next.cleanWindowMs = 0;
|
|
504
|
+
next.cleanWindows += 1;
|
|
505
|
+
if (next.cleanWindows >= UP_WINDOWS) {
|
|
506
|
+
next.cleanWindows = 0;
|
|
507
|
+
if (next.limit < next.ceiling) {
|
|
508
|
+
next.limit += 1;
|
|
509
|
+
increased = true;
|
|
510
|
+
}
|
|
511
|
+
}
|
|
512
|
+
}
|
|
513
|
+
}
|
|
514
|
+
next.limit = clampLimit(next.limit, next.floor, next.ceiling);
|
|
515
|
+
|
|
516
|
+
return {
|
|
517
|
+
state: next,
|
|
518
|
+
badKind,
|
|
519
|
+
decreased,
|
|
520
|
+
increased,
|
|
521
|
+
queueSuppressed,
|
|
522
|
+
latencySuppressedByQueue: maskedKind !== null && badKind === null,
|
|
523
|
+
baselineEligible,
|
|
524
|
+
cooldownActive: now < next.cooldownUntil,
|
|
525
|
+
frozen,
|
|
526
|
+
queueWaitP50Ms: p50,
|
|
527
|
+
latencyMedianMs,
|
|
528
|
+
};
|
|
529
|
+
}
|
|
530
|
+
|
|
531
|
+
/** Parse outcome for a batch of host log lines. */
|
|
532
|
+
export interface InFlightLogParseResult {
|
|
533
|
+
/** Paired waits, ascending by completion time. */
|
|
534
|
+
samples: QueueWaitSample[];
|
|
535
|
+
/** Lines that are not JSON objects or carry unusable `message`/`provider`/`timestamp`. */
|
|
536
|
+
malformedLines: number;
|
|
537
|
+
/** Well-formed lines that are neither a `blocked` nor a `wait completed` record. */
|
|
538
|
+
ignoredLines: number;
|
|
539
|
+
/** `blocked` records left without a completion. */
|
|
540
|
+
unpairedBlocked: number;
|
|
541
|
+
/** Completions left without a preceding `blocked` record. */
|
|
542
|
+
unpairedCompletions: number;
|
|
543
|
+
}
|
|
544
|
+
|
|
545
|
+
function parseTimestamp(raw: unknown): number | null {
|
|
546
|
+
if (typeof raw === 'number' && Number.isFinite(raw)) return raw;
|
|
547
|
+
if (typeof raw === 'string') {
|
|
548
|
+
const ms = Date.parse(raw);
|
|
549
|
+
if (Number.isFinite(ms)) return ms;
|
|
550
|
+
}
|
|
551
|
+
return null;
|
|
552
|
+
}
|
|
553
|
+
|
|
554
|
+
/**
|
|
555
|
+
* Parse host JSONL records into per-provider queue-wait samples.
|
|
556
|
+
*
|
|
557
|
+
* `blocked` and `wait completed` records are paired FIFO per provider; the wait
|
|
558
|
+
* is their timestamp gap. Unpaired records are dropped and counted, malformed
|
|
559
|
+
* lines are skipped — nothing throws. Callers pass the lines of every
|
|
560
|
+
* `omp.*.log` file under the same config root so other processes' waits count.
|
|
561
|
+
*/
|
|
562
|
+
export function parseInFlightLogLines(lines: Iterable<string>): InFlightLogParseResult {
|
|
563
|
+
const pending = new Map<string, Array<{ at: number; limit: number | null }>>();
|
|
564
|
+
const samples: QueueWaitSample[] = [];
|
|
565
|
+
let malformedLines = 0;
|
|
566
|
+
let ignoredLines = 0;
|
|
567
|
+
let unpairedCompletions = 0;
|
|
568
|
+
|
|
569
|
+
for (const line of lines) {
|
|
570
|
+
const text = line.trim();
|
|
571
|
+
if (text === '') continue;
|
|
572
|
+
let rec: Record<string, unknown>;
|
|
573
|
+
try {
|
|
574
|
+
const parsed: unknown = JSON.parse(text);
|
|
575
|
+
if (typeof parsed !== 'object' || parsed === null || Array.isArray(parsed)) {
|
|
576
|
+
malformedLines += 1;
|
|
577
|
+
continue;
|
|
578
|
+
}
|
|
579
|
+
rec = parsed as Record<string, unknown>;
|
|
580
|
+
} catch {
|
|
581
|
+
malformedLines += 1;
|
|
582
|
+
continue;
|
|
583
|
+
}
|
|
584
|
+
const message = rec.message;
|
|
585
|
+
const isBlocked = message === BLOCKED_MESSAGE;
|
|
586
|
+
const isCompleted = message === WAIT_COMPLETED_MESSAGE;
|
|
587
|
+
if (!isBlocked && !isCompleted) {
|
|
588
|
+
ignoredLines += 1;
|
|
589
|
+
continue;
|
|
590
|
+
}
|
|
591
|
+
const provider = rec.provider;
|
|
592
|
+
const at = parseTimestamp(rec.timestamp);
|
|
593
|
+
if (typeof provider !== 'string' || provider === '' || at === null) {
|
|
594
|
+
malformedLines += 1;
|
|
595
|
+
continue;
|
|
596
|
+
}
|
|
597
|
+
const limit = typeof rec.limit === 'number' && Number.isFinite(rec.limit)
|
|
598
|
+
? Math.trunc(rec.limit)
|
|
599
|
+
: null;
|
|
600
|
+
if (isBlocked) {
|
|
601
|
+
const queue = pending.get(provider) ?? [];
|
|
602
|
+
queue.push({ at, limit });
|
|
603
|
+
pending.set(provider, queue);
|
|
604
|
+
continue;
|
|
605
|
+
}
|
|
606
|
+
const queue = pending.get(provider);
|
|
607
|
+
const blocked = queue?.shift();
|
|
608
|
+
if (!blocked) {
|
|
609
|
+
unpairedCompletions += 1;
|
|
610
|
+
continue;
|
|
611
|
+
}
|
|
612
|
+
samples.push({
|
|
613
|
+
provider,
|
|
614
|
+
at,
|
|
615
|
+
ms: Math.max(0, at - blocked.at),
|
|
616
|
+
limit: blocked.limit ?? limit,
|
|
617
|
+
});
|
|
618
|
+
}
|
|
619
|
+
|
|
620
|
+
let unpairedBlocked = 0;
|
|
621
|
+
for (const queue of pending.values()) unpairedBlocked += queue.length;
|
|
622
|
+
samples.sort((a, b) => a.at - b.at || a.provider.localeCompare(b.provider));
|
|
623
|
+
return { samples, malformedLines, ignoredLines, unpairedBlocked, unpairedCompletions };
|
|
624
|
+
}
|
|
625
|
+
|
|
626
|
+
/** Provider id of a `model_perf.model_key` (`<provider>/<model>`). */
|
|
627
|
+
export function providerOfModelKey(modelKey: string): string {
|
|
628
|
+
const slash = modelKey.indexOf('/');
|
|
629
|
+
return slash <= 0 ? modelKey : modelKey.slice(0, slash);
|
|
630
|
+
}
|
|
631
|
+
|
|
632
|
+
/** Cumulative `model_perf` counters for one model_key. */
|
|
633
|
+
export interface ModelPerfRow {
|
|
634
|
+
modelKey: string;
|
|
635
|
+
samples: number;
|
|
636
|
+
outputTokens: number;
|
|
637
|
+
genMs: number;
|
|
638
|
+
}
|
|
639
|
+
|
|
640
|
+
/** Per-provider window delta derived from cumulative `model_perf` counters. */
|
|
641
|
+
export interface ProviderPerfWindow {
|
|
642
|
+
provider: string;
|
|
643
|
+
/** Δsamples for the window. */
|
|
644
|
+
samples: number;
|
|
645
|
+
/** Δoutput_tokens for the window. */
|
|
646
|
+
outputTokens: number;
|
|
647
|
+
/** Δgen_ms for the window. */
|
|
648
|
+
genMs: number;
|
|
649
|
+
/** Aggregated generation duration per sample (ms). */
|
|
650
|
+
genMsPerSample: number;
|
|
651
|
+
/** Aggregated throughput (tokens per ms). */
|
|
652
|
+
tokensPerMs: number;
|
|
653
|
+
/** `false` when `samples < concurrency_min_samples` — source unusable this window. */
|
|
654
|
+
sufficientSamples: boolean;
|
|
655
|
+
}
|
|
656
|
+
|
|
657
|
+
function delta(current: number, previous: number): number {
|
|
658
|
+
const d = current - previous;
|
|
659
|
+
return d > 0 ? d : 0; // counter reset / rollover → no window delta
|
|
660
|
+
}
|
|
661
|
+
|
|
662
|
+
/**
|
|
663
|
+
* Difference cumulative `model_perf` snapshots into per-provider windows.
|
|
664
|
+
*
|
|
665
|
+
* Rows without a previous snapshot are treated as all-new. Deltas are clamped at
|
|
666
|
+
* 0 so a table reset never produces a negative window.
|
|
667
|
+
*/
|
|
668
|
+
export function diffModelPerf(
|
|
669
|
+
previous: ModelPerfRow[],
|
|
670
|
+
current: ModelPerfRow[],
|
|
671
|
+
options: TunerOptions,
|
|
672
|
+
): ProviderPerfWindow[] {
|
|
673
|
+
const before = new Map<string, ModelPerfRow>();
|
|
674
|
+
for (const row of previous) before.set(row.modelKey, row);
|
|
675
|
+
|
|
676
|
+
const agg = new Map<string, { samples: number; outputTokens: number; genMs: number }>();
|
|
677
|
+
for (const row of current) {
|
|
678
|
+
const prev = before.get(row.modelKey);
|
|
679
|
+
const samples = delta(row.samples, prev?.samples ?? 0);
|
|
680
|
+
const outputTokens = delta(row.outputTokens, prev?.outputTokens ?? 0);
|
|
681
|
+
const genMs = delta(row.genMs, prev?.genMs ?? 0);
|
|
682
|
+
const provider = providerOfModelKey(row.modelKey);
|
|
683
|
+
const bucket = agg.get(provider) ?? { samples: 0, outputTokens: 0, genMs: 0 };
|
|
684
|
+
bucket.samples += samples;
|
|
685
|
+
bucket.outputTokens += outputTokens;
|
|
686
|
+
bucket.genMs += genMs;
|
|
687
|
+
agg.set(provider, bucket);
|
|
688
|
+
}
|
|
689
|
+
|
|
690
|
+
const windows: ProviderPerfWindow[] = [];
|
|
691
|
+
for (const [provider, bucket] of agg) {
|
|
692
|
+
windows.push({
|
|
693
|
+
provider,
|
|
694
|
+
samples: bucket.samples,
|
|
695
|
+
outputTokens: bucket.outputTokens,
|
|
696
|
+
genMs: bucket.genMs,
|
|
697
|
+
genMsPerSample: bucket.samples > 0 ? bucket.genMs / bucket.samples : 0,
|
|
698
|
+
tokensPerMs: bucket.genMs > 0 ? bucket.outputTokens / bucket.genMs : 0,
|
|
699
|
+
sufficientSamples: bucket.samples >= options.minSamples,
|
|
700
|
+
});
|
|
701
|
+
}
|
|
702
|
+
windows.sort((a, b) => a.provider.localeCompare(b.provider));
|
|
703
|
+
return windows;
|
|
704
|
+
}
|
|
705
|
+
|
|
706
|
+
/** EWMA baselines kept alongside the per-provider control state. */
|
|
707
|
+
export interface PerfBaseline {
|
|
708
|
+
/** EWMA of `gen_ms / samples`. */
|
|
709
|
+
genMsPerSample: number | null;
|
|
710
|
+
/** EWMA of `output_tokens / gen_ms`. */
|
|
711
|
+
tokensPerMs: number | null;
|
|
712
|
+
}
|
|
713
|
+
|
|
714
|
+
/** Fold a provider's perf window into its EWMA baselines (no-op when unusable). */
|
|
715
|
+
export function updatePerfBaseline(
|
|
716
|
+
previous: PerfBaseline | null,
|
|
717
|
+
window: ProviderPerfWindow,
|
|
718
|
+
options: TunerOptions,
|
|
719
|
+
): PerfBaseline {
|
|
720
|
+
const base = previous ?? { genMsPerSample: null, tokensPerMs: null };
|
|
721
|
+
if (!window.sufficientSamples) return base;
|
|
722
|
+
return {
|
|
723
|
+
genMsPerSample: ewmaUpdate(base.genMsPerSample, window.genMsPerSample, options.ewmaAlpha),
|
|
724
|
+
tokensPerMs: ewmaUpdate(base.tokensPerMs, window.tokensPerMs, options.ewmaAlpha),
|
|
725
|
+
};
|
|
726
|
+
}
|
|
727
|
+
|
|
728
|
+
/** Per-provider row shown in the concurrency summary. */
|
|
729
|
+
export interface ProviderConcurrencyView {
|
|
730
|
+
provider: string;
|
|
731
|
+
limit: number;
|
|
732
|
+
floor: number;
|
|
733
|
+
ceiling: number;
|
|
734
|
+
/** Explicit rate-limit (429 / rate-limit error) observations in the last window. */
|
|
735
|
+
rateLimitCount: number;
|
|
736
|
+
/** Timeout observations in the last window. */
|
|
737
|
+
timeoutCount: number;
|
|
738
|
+
queueWaitP50Ms: number | null;
|
|
739
|
+
queueWaitP90Ms: number | null;
|
|
740
|
+
/** Epoch ms of the last successful `maxInFlightRequests` write (`null` = never). */
|
|
741
|
+
lastWriteAt: number | null;
|
|
742
|
+
}
|
|
743
|
+
|
|
744
|
+
export interface ConcurrencySummaryInput {
|
|
745
|
+
providers: ProviderConcurrencyView[];
|
|
746
|
+
/** Lease holder id, or `null` when no process holds it. */
|
|
747
|
+
leaseOwner: string | null;
|
|
748
|
+
/** This process's id; compared against {@link leaseOwner} for the writer identity. */
|
|
749
|
+
selfId: string;
|
|
750
|
+
}
|
|
751
|
+
|
|
752
|
+
/**
|
|
753
|
+
* Format the concurrency summary as one line per provider prefixed by a lease
|
|
754
|
+
* header line. Writer identity is derived from `leaseOwner === selfId`, so a
|
|
755
|
+
* writer and a non-writer describe themselves differently. Returns `[]` when
|
|
756
|
+
* there is no managed provider (empty summary).
|
|
757
|
+
*/
|
|
758
|
+
export function formatConcurrencySummary(input: ConcurrencySummaryInput): string[] {
|
|
759
|
+
if (input.providers.length === 0) return [];
|
|
760
|
+
const isWriter = input.leaseOwner !== null && input.leaseOwner === input.selfId;
|
|
761
|
+
const owner = isWriter ? '本进程' : input.leaseOwner ?? '无';
|
|
762
|
+
const lines = [`并发控制: 租约持有者=${owner};本进程${isWriter ? '为写者' : '非写者'}`];
|
|
763
|
+
for (const view of input.providers) {
|
|
764
|
+
lines.push(
|
|
765
|
+
`${view.provider}: L=${view.limit} 区间=[${view.floor},${view.ceiling}] `
|
|
766
|
+
+ `429=${view.rateLimitCount} 超时=${view.timeoutCount} `
|
|
767
|
+
+ `排队等待p50=${view.queueWaitP50Ms === null ? '-' : `${Math.round(view.queueWaitP50Ms)}ms`} `
|
|
768
|
+
+ `p90=${view.queueWaitP90Ms === null ? '-' : `${Math.round(view.queueWaitP90Ms)}ms`} `
|
|
769
|
+
+ `最近写入=${view.lastWriteAt === null ? '无' : new Date(view.lastWriteAt).toISOString()}`,
|
|
770
|
+
);
|
|
771
|
+
}
|
|
772
|
+
return lines;
|
|
773
|
+
}
|