pi-live-speed 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +46 -0
- package/docs/configuration.md +36 -0
- package/docs/demo.gif +0 -0
- package/docs/demo.mp4 +0 -0
- package/docs/measurements.md +63 -0
- package/docs/screenshot.png +0 -0
- package/package.json +47 -0
- package/scripts/token-speed-stats.sh +28 -0
- package/src/index.ts +111 -0
- package/src/meter.ts +109 -0
- package/src/storage.ts +37 -0
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 aisensiy
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# pi-live-speed
|
|
2
|
+
|
|
3
|
+
**See how fast your model is generating, while it's still generating.**
|
|
4
|
+
|
|
5
|
+
Live token speed in [Pi](https://pi.dev/)'s footer. No commands to run, no dashboard to open, no waiting for the response to finish.
|
|
6
|
+
|
|
7
|
+
```text
|
|
8
|
+
⚡ ~42.0 tok/s · ttft 2.3s · gen 4.1s
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+

|
|
12
|
+
|
|
13
|
+
- **Live speed:** refreshes every 250 ms, even when the stream pauses.
|
|
14
|
+
- **Live waiting time:** see how long you've been waiting for the first token.
|
|
15
|
+
- **Final speed:** replaces the `~` estimate with provider-reported usage when the response ends.
|
|
16
|
+
- **Stays out of the way:** uses Pi's status area without replacing the footer.
|
|
17
|
+
|
|
18
|
+
## Install
|
|
19
|
+
|
|
20
|
+
Requires Node.js 22.19+ and Pi. Tested with Pi 0.87.1.
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
pi install git:github.com/aisensiy/pi-live-speed
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
Restart Pi or run `/reload`, then send a message. If you previously installed `token-speed.ts`, disable or remove it first to avoid duplicate displays.
|
|
27
|
+
|
|
28
|
+
The repository currently contains the **0.1.0 release candidate**, available through Git; it has not been published to npm. Live display, response logs and cancellation have been checked on the author's machine. Testing on a second machine is next.
|
|
29
|
+
|
|
30
|
+
## Good to know
|
|
31
|
+
|
|
32
|
+
`~` means an estimate during streaming. Final speed uses reported output tokens over the client-observed generation time, including stalls. It is not server-side decoding speed.
|
|
33
|
+
|
|
34
|
+
Performance logging is enabled by default, without conversation text. See [logging and configuration](docs/configuration.md) to disable it, and [measurement details](docs/measurements.md) for the exact timing and log format.
|
|
35
|
+
|
|
36
|
+
## Development
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
npm ci
|
|
40
|
+
npm run check
|
|
41
|
+
npm pack --dry-run
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## License
|
|
45
|
+
|
|
46
|
+
[MIT](LICENSE)
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# Logging and configuration
|
|
2
|
+
|
|
3
|
+
The footer works without configuration. This page covers optional performance history settings; changing them does not change the live display.
|
|
4
|
+
|
|
5
|
+
## Settings
|
|
6
|
+
|
|
7
|
+
Optional `~/.pi/agent/pi-live-speed.json` (or under `PI_CODING_AGENT_DIR`):
|
|
8
|
+
|
|
9
|
+
```json
|
|
10
|
+
{
|
|
11
|
+
"logging": true,
|
|
12
|
+
"logPath": "/absolute/path/to/pi-live-speed.jsonl"
|
|
13
|
+
}
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
Default: logging enabled, writing `pi-live-speed.jsonl` inside Pi's agent directory. Set `logging` to `false` to retain live display without writing history. `logPath` must be absolute; omit it for the default. Reload after edits.
|
|
17
|
+
|
|
18
|
+
Invalid configuration disables logging and displays a warning rather than silently using defaults. Write failures also show a warning; they do not interrupt the agent.
|
|
19
|
+
|
|
20
|
+
## Privacy and retention
|
|
21
|
+
|
|
22
|
+
Logs contain provider/model names, session identifiers and timestamps, which can still be sensitive. Prompts, answers, reasoning text, tool arguments and raw error messages are not saved by this plugin.
|
|
23
|
+
|
|
24
|
+
Newly created log files use mode `0600`; existing permissions are unchanged. Logs grow until you archive or delete them: automatic retention is not implemented.
|
|
25
|
+
|
|
26
|
+
See [measurement contract and schema](measurements.md) before comparing providers. Speed and availability do not measure answer correctness.
|
|
27
|
+
|
|
28
|
+
## Legacy history
|
|
29
|
+
|
|
30
|
+
The original `scripts/token-speed-stats.sh [log-path]` is retained for **legacy `token-speed.jsonl` only**. It requires Bash, jq, column, awk and GNU date. Do not use its unweighted averages or null handling to analyze the new schema.
|
|
31
|
+
|
|
32
|
+
Existing legacy logs are not modified by this plugin.
|
|
33
|
+
|
|
34
|
+
## Display compatibility
|
|
35
|
+
|
|
36
|
+
Custom footers must render extension statuses; a narrow terminal can truncate them. If you previously loaded a personal `token-speed.ts` extension, disable or remove that copy before loading this package.
|
package/docs/demo.gif
ADDED
|
Binary file
|
package/docs/demo.mp4
ADDED
|
Binary file
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# Measurement contract (schema v1)
|
|
2
|
+
|
|
3
|
+
## Scope and event boundaries
|
|
4
|
+
|
|
5
|
+
One row measures one **assistant response observed through Pi's main extension event stream**, not a whole user prompt, tool batch, or HTTP request attempt.
|
|
6
|
+
|
|
7
|
+
Pi 0.87.1 emits `turn_start` before preparing and streaming each response. `message_start` may arrive only after the provider starts streaming, and is also emitted for immediate errors. `message_end` closes the response before tool execution begins. A tool-following response starts a new turn.
|
|
8
|
+
|
|
9
|
+
Source: [Pi v0.87.1 agent-loop.ts](https://github.com/earendil-works/pi/blob/v0.87.1/packages/agent/src/agent-loop.ts), particularly `runLoop` and `streamAssistantResponse`.
|
|
10
|
+
|
|
11
|
+
We deliberately do not use provider request hooks as a primary clock: their public events do not carry a response/request correlation ID, and nested model calls or internal retries cannot safely be attributed to one assistant response. This first version favors an explicit client-observed wait over a misleading network-only TTFT.
|
|
12
|
+
|
|
13
|
+
- Start: monotonic clock at `turn_start`; fallback to `message_start` if a host omits the turn event. `timingSource` records the distinction.
|
|
14
|
+
- First content: first non-empty `text_delta`, `thinking_delta` or `toolcall_delta`. Start events and empty deltas do not count.
|
|
15
|
+
- TTFT: first content minus start. Includes client preparation and transport, potentially provider-internal retries.
|
|
16
|
+
- Generation seconds: `message_end` minus first content. Includes stream stalls and final usage/stream-finalization delays. Never subtracts stalls.
|
|
17
|
+
- Final TPS: positive, finite provider `usage.output` divided by generation seconds. Output includes reasoning if the provider accounts for it there.
|
|
18
|
+
- Live TPS: accumulated CJK units plus other UTF-16 units divided by four, divided by generation seconds. Fractional counts are accumulated, not rounded per chunk. This is a heuristic, not tokenization; always marked `~`.
|
|
19
|
+
- Fewer than 50 ms of generation, missing positive output usage, or no content delta: TPS is `null`, never a fabricated zero. Zero usage may mean an unreported value, not necessarily zero work.
|
|
20
|
+
|
|
21
|
+
Durations use `performance.now()` and survive wall-clock changes. Epoch timestamps use `Date.now()` for historical grouping. Client buffering, hidden reasoning and bursty delivery can distort measured speed; it is not the server's internal decoding throughput.
|
|
22
|
+
|
|
23
|
+
## Lifecycle and coverage
|
|
24
|
+
|
|
25
|
+
`error` and `aborted` terminal messages are logged even without content or positive usage. A response with no terminal message is recorded as `incomplete` at `agent_end`, session shutdown, session replacement or supersession. Timers are stopped on each completion and shutdown; no timer starts during module discovery or in headless sessions.
|
|
26
|
+
|
|
27
|
+
Automatic retries that Pi exposes as separate turns get separate rows and response IDs. **Internal HTTP retries within a provider are not separate rows.** Calls by other extensions that do not emit main assistant events are outside this log. Abrupt process termination (for example SIGKILL) cannot flush an active response. This is not a complete HTTP request ledger, and its failure rate must be labeled as an observed assistant-response failure rate.
|
|
28
|
+
|
|
29
|
+
We do not record raw error messages: they can contain credentials, URLs or conversation fragments. `status` and the controlled `stopReason` category are the first version's failure information; HTTP error classification and attempt correlation are deferred.
|
|
30
|
+
|
|
31
|
+
## JSONL fields
|
|
32
|
+
|
|
33
|
+
Each line is a standalone object. No prior line or process-global state is needed.
|
|
34
|
+
|
|
35
|
+
| Field | Meaning |
|
|
36
|
+
| --- | --- |
|
|
37
|
+
| `schemaVersion` | `1` |
|
|
38
|
+
| `scope` | `assistant-response` |
|
|
39
|
+
| `sessionId`, `responseId` | Pi session identifier and a fresh response UUID |
|
|
40
|
+
| `provider`, `model` | Terminal assistant identity, or initial context identity when no terminal message exists; nullable |
|
|
41
|
+
| `timingSource` | `turn_start` or fallback `message_start` |
|
|
42
|
+
| `startedAt`, `ts` | Start/end epoch milliseconds |
|
|
43
|
+
| `elapsedSec`, `ttftSec`, `genSec` | Monotonic durations in seconds; unavailable TTFT/generation is null |
|
|
44
|
+
| `outputTokens`, `inputTokens`, `cacheReadTokens`, `cacheWriteTokens` | Non-negative provider usage, or null if missing/invalid |
|
|
45
|
+
| `tokenSource` | `provider-usage` when output is positive, otherwise `unavailable` |
|
|
46
|
+
| `tps` | Final average, or null |
|
|
47
|
+
| `unavailableReason` | Null when measurable; otherwise `no-content-delta`, `short-generation`, or `no-output-usage` |
|
|
48
|
+
| `status` | `completed`, `error`, `aborted`, `incomplete` |
|
|
49
|
+
| `stopReason` | Pi terminal reason or a lifecycle category such as `session-shutdown` |
|
|
50
|
+
|
|
51
|
+
Raw prompts, deltas, reasoning, tool arguments, endpoint URLs and error text are never added to these records. Existing legacy history is not rewritten or mixed with schema v1 by default.
|
|
52
|
+
|
|
53
|
+
## Comparing suppliers
|
|
54
|
+
|
|
55
|
+
Separate successes, errors, user cancellations and incomplete samples. Report counts and missing-value coverage. For valid comparable samples, aggregate throughput as `sum(outputTokens) / sum(genSec)`, not the average of response TPS values. TTFT and speed percentiles are useful alongside throughput.
|
|
56
|
+
|
|
57
|
+
Compare equivalent provider/model, timing source, input/cache and output length ranges. Thinking settings and task difficulty can still confound results; v1 does not record every such dimension and is not a controlled benchmark. Answer quality needs separate tests, task outcomes or human ratings.
|
|
58
|
+
|
|
59
|
+
## Acceptance status
|
|
60
|
+
|
|
61
|
+
The author's Pi 0.87.1 installation has been migrated and reloaded. Single-copy footer display and continuous updates were visually confirmed; recorded output usage matched real provider session messages. A manual cancellation produced an `aborted` record, and the footer timer stopped.
|
|
62
|
+
|
|
63
|
+
Automated tests additionally cover waits and stalls without incoming chunks, lifecycle cleanup, configuration and log failures. Real narrow-width behavior and installation on a second machine remain unverified; the author will perform the latter before the release announcement.
|
|
Binary file
|
package/package.json
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "pi-live-speed",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "See token speed live in Pi's footer, while the model is still generating.",
|
|
5
|
+
"license": "MIT",
|
|
6
|
+
"type": "module",
|
|
7
|
+
"engines": {
|
|
8
|
+
"node": ">=22.19.0"
|
|
9
|
+
},
|
|
10
|
+
"scripts": {
|
|
11
|
+
"test": "vitest run",
|
|
12
|
+
"typecheck": "tsc --noEmit",
|
|
13
|
+
"check:load": "node scripts/check-load.mjs",
|
|
14
|
+
"check": "npm run typecheck && npm test && npm run check:load"
|
|
15
|
+
},
|
|
16
|
+
"keywords": [
|
|
17
|
+
"pi-package",
|
|
18
|
+
"tokens",
|
|
19
|
+
"performance"
|
|
20
|
+
],
|
|
21
|
+
"repository": {
|
|
22
|
+
"type": "git",
|
|
23
|
+
"url": "https://github.com/aisensiy/pi-live-speed.git"
|
|
24
|
+
},
|
|
25
|
+
"files": [
|
|
26
|
+
"src",
|
|
27
|
+
"scripts/token-speed-stats.sh",
|
|
28
|
+
"README.md",
|
|
29
|
+
"docs"
|
|
30
|
+
],
|
|
31
|
+
"pi": {
|
|
32
|
+
"extensions": [
|
|
33
|
+
"./src/index.ts"
|
|
34
|
+
],
|
|
35
|
+
"image": "./docs/screenshot.png",
|
|
36
|
+
"video": "./docs/demo.mp4"
|
|
37
|
+
},
|
|
38
|
+
"peerDependencies": {
|
|
39
|
+
"@earendil-works/pi-coding-agent": "*"
|
|
40
|
+
},
|
|
41
|
+
"devDependencies": {
|
|
42
|
+
"@earendil-works/pi-coding-agent": "^0.87.1",
|
|
43
|
+
"@types/node": "^26.6.3",
|
|
44
|
+
"typescript": "^7.0.2",
|
|
45
|
+
"vitest": "^5.0.2"
|
|
46
|
+
}
|
|
47
|
+
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# 汇总 token-speed.jsonl:按模型看速度分布,并列出最近 10 次
|
|
3
|
+
LOG="${1:-$HOME/.pi/agent/token-speed.jsonl}"
|
|
4
|
+
[ -f "$LOG" ] || { echo "没有日志:$LOG"; exit 1; }
|
|
5
|
+
|
|
6
|
+
echo "=== 按模型汇总 ==="
|
|
7
|
+
jq -rs '
|
|
8
|
+
group_by(.provider + "/" + .model)
|
|
9
|
+
| map({
|
|
10
|
+
model: .[0].provider + "/" + .[0].model,
|
|
11
|
+
runs: length,
|
|
12
|
+
ttft_avg: ((map(.ttftSec // 0) | add / length) * 100 | round / 100),
|
|
13
|
+
tps_avg: ((map(.tps) | add / length) * 10 | round / 10),
|
|
14
|
+
tps_min: (map(.tps) | min),
|
|
15
|
+
tps_max: (map(.tps) | max),
|
|
16
|
+
out_avg: ((map(.outputTokens) | add / length) | round)
|
|
17
|
+
})
|
|
18
|
+
| sort_by(-.runs)
|
|
19
|
+
| (["模型","次数","首字均值s","吞吐均值","吞吐最低","吞吐最高","输出均值"]
|
|
20
|
+
| @tsv),
|
|
21
|
+
(.[] | [.model, .runs, .ttft_avg, .tps_avg, .tps_min, .tps_max, .out_avg] | @tsv)
|
|
22
|
+
' "$LOG" | column -t -s $'\t'
|
|
23
|
+
|
|
24
|
+
echo
|
|
25
|
+
echo "=== 最近 10 次 ==="
|
|
26
|
+
tail -10 "$LOG" | jq -r '[.ts, .model, .ttftSec, .genSec, .outputTokens, .tps, .stopReason] | @tsv' \
|
|
27
|
+
| awk -F'\t' 'BEGIN{print "时间\t模型\t首字s\t生成s\ttoken\t吞吐\t结束原因"} {cmd="date -d @" int($1/1000) " +%H:%M:%S"; cmd | getline t; close(cmd); $1=t; print}' \
|
|
28
|
+
| column -t -s $'\t'
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
import { randomUUID } from "node:crypto";
|
|
2
|
+
import { getAgentDir, type ExtensionAPI, type ExtensionContext } from "@earendil-works/pi-coding-agent";
|
|
3
|
+
import { finalStatus, Meter, type Outcome, type Usage } from "./meter.ts";
|
|
4
|
+
import { appendRecord, loadConfig, type Config } from "./storage.ts";
|
|
5
|
+
|
|
6
|
+
const KEY = "pi-live-speed";
|
|
7
|
+
const WARNING_KEY = "pi-live-speed-log";
|
|
8
|
+
|
|
9
|
+
export default function liveSpeed(pi: ExtensionAPI): void {
|
|
10
|
+
let active: Meter | undefined;
|
|
11
|
+
let timer: ReturnType<typeof setInterval> | undefined;
|
|
12
|
+
let config: Config = { logging: false, logPath: "" };
|
|
13
|
+
let warned = false;
|
|
14
|
+
|
|
15
|
+
function warn(ctx: ExtensionContext, message: string): void {
|
|
16
|
+
if (warned) return;
|
|
17
|
+
warned = true;
|
|
18
|
+
// Do not expose filesystem errors: they may contain private paths or values.
|
|
19
|
+
ctx.ui.setStatus(WARNING_KEY, "⚠ live-speed log unavailable");
|
|
20
|
+
ctx.ui.notify(message, "warning");
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
function stopTimer(): void {
|
|
24
|
+
if (timer) clearInterval(timer);
|
|
25
|
+
timer = undefined;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
function finish(ctx: ExtensionContext, status: Outcome, stopReason: string | null, usage?: Usage): void {
|
|
29
|
+
if (!active) return;
|
|
30
|
+
const record = active.finish(performance.now(), Date.now(), status, stopReason, usage);
|
|
31
|
+
active = undefined;
|
|
32
|
+
stopTimer();
|
|
33
|
+
ctx.ui.setStatus(KEY, finalStatus(record));
|
|
34
|
+
if (config.logging) {
|
|
35
|
+
try {
|
|
36
|
+
appendRecord(config.logPath, record);
|
|
37
|
+
} catch {
|
|
38
|
+
warn(ctx, "pi-live-speed could not append its performance log. Check the configured path and permissions.");
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
function start(ctx: ExtensionContext, source: "turn_start" | "message_start"): void {
|
|
44
|
+
if (active) finish(ctx, "incomplete", "superseded");
|
|
45
|
+
active = new Meter({
|
|
46
|
+
sessionId: ctx.sessionManager.getSessionId(),
|
|
47
|
+
responseId: randomUUID(),
|
|
48
|
+
provider: ctx.model?.provider ?? null,
|
|
49
|
+
model: ctx.model?.id ?? null,
|
|
50
|
+
}, performance.now(), Date.now(), source);
|
|
51
|
+
ctx.ui.setStatus(KEY, active.live(performance.now()));
|
|
52
|
+
if (ctx.hasUI) {
|
|
53
|
+
timer = setInterval(() => {
|
|
54
|
+
if (active) ctx.ui.setStatus(KEY, active.live(performance.now()));
|
|
55
|
+
}, 250);
|
|
56
|
+
timer.unref();
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
pi.on("session_start", (_event, ctx) => {
|
|
61
|
+
finish(ctx, "incomplete", "session-switch");
|
|
62
|
+
stopTimer();
|
|
63
|
+
warned = false;
|
|
64
|
+
ctx.ui.setStatus(KEY, undefined);
|
|
65
|
+
ctx.ui.setStatus(WARNING_KEY, undefined);
|
|
66
|
+
try {
|
|
67
|
+
config = loadConfig(getAgentDir());
|
|
68
|
+
} catch {
|
|
69
|
+
config = { logging: false, logPath: "" };
|
|
70
|
+
warn(ctx, "pi-live-speed config could not be read. Logging is disabled until a successful reload; check pi-live-speed.json.");
|
|
71
|
+
}
|
|
72
|
+
});
|
|
73
|
+
|
|
74
|
+
// A Pi turn is one assistant response plus any subsequent tools, not a whole user prompt.
|
|
75
|
+
// Starting here measures client-observed wait, including context preparation, not network-only TTFT.
|
|
76
|
+
pi.on("turn_start", (_event, ctx) => start(ctx, "turn_start"));
|
|
77
|
+
pi.on("message_start", (event, ctx) => {
|
|
78
|
+
if (event.message.role !== "assistant") return;
|
|
79
|
+
if (!active) start(ctx, "message_start");
|
|
80
|
+
if (active) {
|
|
81
|
+
active.identity.provider = event.message.provider;
|
|
82
|
+
active.identity.model = event.message.model;
|
|
83
|
+
}
|
|
84
|
+
});
|
|
85
|
+
pi.on("message_update", (event, ctx) => {
|
|
86
|
+
if (!active || event.message.role !== "assistant") return;
|
|
87
|
+
const delta = event.assistantMessageEvent;
|
|
88
|
+
if (delta.type === "text_delta" || delta.type === "thinking_delta" || delta.type === "toolcall_delta") {
|
|
89
|
+
active.delta(delta.delta, performance.now());
|
|
90
|
+
// Ticker handles steady rendering; headless clients still receive an updated status.
|
|
91
|
+
if (!timer) ctx.ui.setStatus(KEY, active.live(performance.now()));
|
|
92
|
+
}
|
|
93
|
+
});
|
|
94
|
+
pi.on("message_end", (event, ctx) => {
|
|
95
|
+
const message = event.message;
|
|
96
|
+
if (message.role !== "assistant") return;
|
|
97
|
+
if (active) {
|
|
98
|
+
active.identity.provider = message.provider;
|
|
99
|
+
active.identity.model = message.model;
|
|
100
|
+
}
|
|
101
|
+
const outcome = message.stopReason === "error" || message.stopReason === "aborted" ? message.stopReason : "completed";
|
|
102
|
+
finish(ctx, outcome, message.stopReason, message.usage);
|
|
103
|
+
});
|
|
104
|
+
pi.on("agent_end", (_event, ctx) => finish(ctx, "incomplete", "agent-end-without-message"));
|
|
105
|
+
pi.on("session_shutdown", (_event, ctx) => {
|
|
106
|
+
finish(ctx, "incomplete", "session-shutdown");
|
|
107
|
+
stopTimer();
|
|
108
|
+
ctx.ui.setStatus(KEY, undefined);
|
|
109
|
+
ctx.ui.setStatus(WARNING_KEY, undefined);
|
|
110
|
+
});
|
|
111
|
+
}
|
package/src/meter.ts
ADDED
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
export interface Identity {
|
|
2
|
+
sessionId: string;
|
|
3
|
+
responseId: string;
|
|
4
|
+
provider: string | null;
|
|
5
|
+
model: string | null;
|
|
6
|
+
}
|
|
7
|
+
|
|
8
|
+
export interface Usage {
|
|
9
|
+
input?: number;
|
|
10
|
+
output?: number;
|
|
11
|
+
cacheRead?: number;
|
|
12
|
+
cacheWrite?: number;
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
export type Outcome = "completed" | "error" | "aborted" | "incomplete";
|
|
16
|
+
|
|
17
|
+
export interface RecordV1 extends Identity {
|
|
18
|
+
schemaVersion: 1;
|
|
19
|
+
scope: "assistant-response";
|
|
20
|
+
timingSource: "turn_start" | "message_start";
|
|
21
|
+
startedAt: number;
|
|
22
|
+
ts: number;
|
|
23
|
+
elapsedSec: number;
|
|
24
|
+
ttftSec: number | null;
|
|
25
|
+
genSec: number | null;
|
|
26
|
+
outputTokens: number | null;
|
|
27
|
+
inputTokens: number | null;
|
|
28
|
+
cacheReadTokens: number | null;
|
|
29
|
+
cacheWriteTokens: number | null;
|
|
30
|
+
tokenSource: "provider-usage" | "unavailable";
|
|
31
|
+
tps: number | null;
|
|
32
|
+
unavailableReason: string | null;
|
|
33
|
+
status: Outcome;
|
|
34
|
+
stopReason: string | null;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
const count = (n: number | undefined): number | null =>
|
|
38
|
+
n !== undefined && Number.isFinite(n) && n >= 0 ? n : null;
|
|
39
|
+
const fmt = (n: number) => n >= 100 ? Math.round(n).toString() : n.toFixed(1);
|
|
40
|
+
|
|
41
|
+
/** A response clock; all durations use injected monotonic milliseconds. Never retains content. */
|
|
42
|
+
export class Meter {
|
|
43
|
+
private firstDelta: number | null = null;
|
|
44
|
+
private cjk = 0;
|
|
45
|
+
private other = 0;
|
|
46
|
+
|
|
47
|
+
constructor(
|
|
48
|
+
readonly identity: Identity,
|
|
49
|
+
readonly start: number,
|
|
50
|
+
readonly startedAt: number,
|
|
51
|
+
readonly timingSource: RecordV1["timingSource"],
|
|
52
|
+
) {}
|
|
53
|
+
|
|
54
|
+
delta(text: string, now: number): void {
|
|
55
|
+
if (!text) return;
|
|
56
|
+
this.firstDelta ??= now;
|
|
57
|
+
// UTF-16 units keep the estimate invariant even when a surrogate pair is split across chunks.
|
|
58
|
+
for (let i = 0; i < text.length; i++) {
|
|
59
|
+
const code = text.charCodeAt(i);
|
|
60
|
+
if ((code >= 0x3000 && code <= 0x9fff) || (code >= 0xff00 && code <= 0xffef)) this.cjk++;
|
|
61
|
+
else this.other++;
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
live(now: number): string {
|
|
66
|
+
if (this.firstDelta === null) return `⏳ ttft ${fmt((now - this.start) / 1000)}s`;
|
|
67
|
+
const gen = (now - this.firstDelta) / 1000;
|
|
68
|
+
const speed = gen >= 0.05 ? `~${fmt((this.cjk + this.other / 4) / gen)}` : "--";
|
|
69
|
+
return `⚡ ${speed} tok/s · ttft ${fmt((this.firstDelta - this.start) / 1000)}s · gen ${fmt(gen)}s`;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
finish(now: number, wall: number, status: Outcome, stopReason: string | null, usage?: Usage): RecordV1 {
|
|
73
|
+
const gen = this.firstDelta === null ? null : (now - this.firstDelta) / 1000;
|
|
74
|
+
const output = count(usage?.output);
|
|
75
|
+
let unavailableReason: string | null = null;
|
|
76
|
+
let tps: number | null = null;
|
|
77
|
+
if (gen === null) unavailableReason = "no-content-delta";
|
|
78
|
+
else if (gen < 0.05) unavailableReason = "short-generation";
|
|
79
|
+
else if (output === null || output === 0) unavailableReason = "no-output-usage";
|
|
80
|
+
else tps = output / gen;
|
|
81
|
+
return {
|
|
82
|
+
...this.identity,
|
|
83
|
+
schemaVersion: 1,
|
|
84
|
+
scope: "assistant-response",
|
|
85
|
+
timingSource: this.timingSource,
|
|
86
|
+
startedAt: this.startedAt,
|
|
87
|
+
ts: wall,
|
|
88
|
+
elapsedSec: (now - this.start) / 1000,
|
|
89
|
+
ttftSec: this.firstDelta === null ? null : (this.firstDelta - this.start) / 1000,
|
|
90
|
+
genSec: gen,
|
|
91
|
+
outputTokens: output,
|
|
92
|
+
inputTokens: count(usage?.input),
|
|
93
|
+
cacheReadTokens: count(usage?.cacheRead),
|
|
94
|
+
cacheWriteTokens: count(usage?.cacheWrite),
|
|
95
|
+
tokenSource: output !== null && output > 0 ? "provider-usage" : "unavailable",
|
|
96
|
+
tps,
|
|
97
|
+
unavailableReason,
|
|
98
|
+
status,
|
|
99
|
+
stopReason,
|
|
100
|
+
};
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export function finalStatus(record: RecordV1): string {
|
|
105
|
+
const outcome = record.status === "completed" ? "" : ` · ${record.status}`;
|
|
106
|
+
return `⚡ ${record.tps === null ? "--" : fmt(record.tps)} tok/s`
|
|
107
|
+
+ ` · ttft ${record.ttftSec === null ? "--" : fmt(record.ttftSec)}s`
|
|
108
|
+
+ ` · gen ${record.genSec === null ? "--" : fmt(record.genSec)}s${outcome}`;
|
|
109
|
+
}
|
package/src/storage.ts
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
import { appendFileSync, mkdirSync, readFileSync } from "node:fs";
|
|
2
|
+
import { dirname, isAbsolute, join } from "node:path";
|
|
3
|
+
import type { RecordV1 } from "./meter.ts";
|
|
4
|
+
|
|
5
|
+
export interface Config {
|
|
6
|
+
logging: boolean;
|
|
7
|
+
logPath: string;
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
/** Invalid configuration disables logging rather than silently writing to an unintended location. */
|
|
11
|
+
export function loadConfig(agentDir: string): Config {
|
|
12
|
+
const defaults: Config = { logging: true, logPath: join(agentDir, "pi-live-speed.jsonl") };
|
|
13
|
+
let text: string;
|
|
14
|
+
try {
|
|
15
|
+
text = readFileSync(join(agentDir, "pi-live-speed.json"), "utf8");
|
|
16
|
+
} catch (error) {
|
|
17
|
+
if ((error as NodeJS.ErrnoException).code === "ENOENT") return defaults;
|
|
18
|
+
throw error;
|
|
19
|
+
}
|
|
20
|
+
const value: unknown = JSON.parse(text);
|
|
21
|
+
if (!value || typeof value !== "object" || Array.isArray(value)) throw new Error("Invalid config");
|
|
22
|
+
const config = value as Record<string, unknown>;
|
|
23
|
+
if (Object.keys(config).some((key) => key !== "logging" && key !== "logPath")) throw new Error("Unknown setting");
|
|
24
|
+
if (config.logging !== undefined && typeof config.logging !== "boolean") throw new Error("Invalid logging setting");
|
|
25
|
+
if (config.logPath !== undefined && (typeof config.logPath !== "string" || !isAbsolute(config.logPath))) {
|
|
26
|
+
throw new Error("logPath must be absolute");
|
|
27
|
+
}
|
|
28
|
+
return {
|
|
29
|
+
logging: config.logging === undefined ? defaults.logging : config.logging as boolean,
|
|
30
|
+
logPath: config.logPath === undefined ? defaults.logPath : config.logPath as string,
|
|
31
|
+
};
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
export function appendRecord(path: string, record: RecordV1): void {
|
|
35
|
+
mkdirSync(dirname(path), { recursive: true, mode: 0o700 });
|
|
36
|
+
appendFileSync(path, JSON.stringify(record) + "\n", { encoding: "utf8", mode: 0o600 });
|
|
37
|
+
}
|