pi-advisor-flow 0.3.6 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/README.md +5 -2
- package/extensions/index.ts +0 -10
- package/package.json +1 -3
- package/src/commands.ts +32 -9
- package/src/config.ts +7 -3
- package/src/scout-context.ts +34 -8
- package/src/scout.ts +7 -34
- package/src/session-state.ts +22 -0
- package/src/tools.ts +108 -96
- package/src/usage.ts +199 -0
- package/src/telemetry.ts +0 -270
package/src/telemetry.ts
DELETED
|
@@ -1,270 +0,0 @@
|
|
|
1
|
-
import type { EventBus } from "@earendil-works/pi-coding-agent";
|
|
2
|
-
import { redactAndCapText } from "./conversation.js";
|
|
3
|
-
|
|
4
|
-
export const BENCHMARK_TELEMETRY_CHANNEL = "pi-advisor:benchmark";
|
|
5
|
-
const BENCHMARK_CONTEXT = "PI_ADVISOR_BENCHMARK_CONTEXT";
|
|
6
|
-
const BENCHMARK_RUN_ID = "PI_ADVISOR_BENCHMARK_RUN_ID";
|
|
7
|
-
const BENCHMARK_TOKEN = "PI_ADVISOR_BENCHMARK_TOKEN";
|
|
8
|
-
const MAX_TEXT_BYTES = 2000;
|
|
9
|
-
const MAX_LABELS = 32;
|
|
10
|
-
const MAX_LABEL_BYTES = 160;
|
|
11
|
-
|
|
12
|
-
export interface BenchmarkAdvisorStart {
|
|
13
|
-
model: string;
|
|
14
|
-
question?: string;
|
|
15
|
-
trigger?: string;
|
|
16
|
-
}
|
|
17
|
-
|
|
18
|
-
export interface BenchmarkAdvisorEnd extends BenchmarkAdvisorStart {
|
|
19
|
-
outcome?: string;
|
|
20
|
-
response?: string;
|
|
21
|
-
usage?: unknown;
|
|
22
|
-
}
|
|
23
|
-
|
|
24
|
-
export interface BenchmarkAdvisorError extends BenchmarkAdvisorStart {
|
|
25
|
-
category: "provider-error" | "empty-response" | "cancelled" | "unknown";
|
|
26
|
-
}
|
|
27
|
-
|
|
28
|
-
export type BenchmarkScoutEvent =
|
|
29
|
-
| { model: string; type: "call" }
|
|
30
|
-
| {
|
|
31
|
-
availableCount?: number;
|
|
32
|
-
latencyMs?: number;
|
|
33
|
-
model: string;
|
|
34
|
-
omittedBeforeScout?: number;
|
|
35
|
-
selectedCount?: number;
|
|
36
|
-
selectedLabels?: string[];
|
|
37
|
-
synthesis?: string;
|
|
38
|
-
type: "success";
|
|
39
|
-
usage?: unknown;
|
|
40
|
-
}
|
|
41
|
-
| {
|
|
42
|
-
availableCount?: number;
|
|
43
|
-
fallback?: string;
|
|
44
|
-
latencyMs?: number;
|
|
45
|
-
model: string;
|
|
46
|
-
omittedBeforeScout?: number;
|
|
47
|
-
selectedCount?: number;
|
|
48
|
-
type: "fallback";
|
|
49
|
-
usage?: unknown;
|
|
50
|
-
}
|
|
51
|
-
| { type: "cancelled" };
|
|
52
|
-
|
|
53
|
-
export type BenchmarkTelemetryEvent =
|
|
54
|
-
| {
|
|
55
|
-
type: "advisor:start";
|
|
56
|
-
runId: string;
|
|
57
|
-
model: string;
|
|
58
|
-
question?: string;
|
|
59
|
-
trigger?: string;
|
|
60
|
-
timestamp: string;
|
|
61
|
-
}
|
|
62
|
-
| {
|
|
63
|
-
type: "advisor:end";
|
|
64
|
-
runId: string;
|
|
65
|
-
model: string;
|
|
66
|
-
outcome?: string;
|
|
67
|
-
response?: string;
|
|
68
|
-
timestamp: string;
|
|
69
|
-
usage?: Record<string, unknown>;
|
|
70
|
-
}
|
|
71
|
-
| {
|
|
72
|
-
type: "advisor:error";
|
|
73
|
-
category: BenchmarkAdvisorError["category"];
|
|
74
|
-
runId: string;
|
|
75
|
-
model: string;
|
|
76
|
-
timestamp: string;
|
|
77
|
-
}
|
|
78
|
-
| {
|
|
79
|
-
event: BenchmarkScoutEvent;
|
|
80
|
-
runId: string;
|
|
81
|
-
timestamp: string;
|
|
82
|
-
type: "scout";
|
|
83
|
-
}
|
|
84
|
-
| {
|
|
85
|
-
fields: Record<string, unknown>;
|
|
86
|
-
runId: string;
|
|
87
|
-
timestamp: string;
|
|
88
|
-
type: "provider-request";
|
|
89
|
-
};
|
|
90
|
-
|
|
91
|
-
type BenchmarkTelemetryPayload = {
|
|
92
|
-
[K in BenchmarkTelemetryEvent["type"]]: Omit<
|
|
93
|
-
Extract<BenchmarkTelemetryEvent, { type: K }>,
|
|
94
|
-
"runId" | "timestamp"
|
|
95
|
-
>;
|
|
96
|
-
}[BenchmarkTelemetryEvent["type"]];
|
|
97
|
-
|
|
98
|
-
export interface BenchmarkTelemetry {
|
|
99
|
-
advisorEnd: (event: BenchmarkAdvisorEnd) => void;
|
|
100
|
-
advisorError: (event: BenchmarkAdvisorError) => void;
|
|
101
|
-
advisorStart: (event: BenchmarkAdvisorStart) => void;
|
|
102
|
-
providerRequest: (payload: unknown) => void;
|
|
103
|
-
scout: (event: BenchmarkScoutEvent) => void;
|
|
104
|
-
}
|
|
105
|
-
|
|
106
|
-
const finite = (value: unknown) =>
|
|
107
|
-
typeof value === "number" && Number.isFinite(value) ? value : undefined;
|
|
108
|
-
|
|
109
|
-
const usageSnapshot = (usage: unknown): Record<string, unknown> | undefined => {
|
|
110
|
-
if (!usage || typeof usage !== "object") {
|
|
111
|
-
return undefined;
|
|
112
|
-
}
|
|
113
|
-
const source = usage as Record<string, unknown>;
|
|
114
|
-
const { cost } = source;
|
|
115
|
-
const costSource =
|
|
116
|
-
cost && typeof cost === "object"
|
|
117
|
-
? (cost as Record<string, unknown>)
|
|
118
|
-
: undefined;
|
|
119
|
-
const result = Object.fromEntries(
|
|
120
|
-
["input", "output", "cacheRead", "cacheWrite", "totalTokens"]
|
|
121
|
-
.map((key) => [key, finite(source[key])] as const)
|
|
122
|
-
.filter(([, value]) => value !== undefined)
|
|
123
|
-
) as Record<string, unknown>;
|
|
124
|
-
const providerCost = finite(costSource?.total);
|
|
125
|
-
if (providerCost !== undefined) {
|
|
126
|
-
result.cost = { total: providerCost };
|
|
127
|
-
}
|
|
128
|
-
return Object.keys(result).length > 0 ? result : undefined;
|
|
129
|
-
};
|
|
130
|
-
|
|
131
|
-
const text = (value: string | undefined, maxBytes = MAX_TEXT_BYTES) =>
|
|
132
|
-
value ? redactAndCapText(value, maxBytes, true) : undefined;
|
|
133
|
-
|
|
134
|
-
const labels = (values: string[] | undefined) =>
|
|
135
|
-
values
|
|
136
|
-
?.slice(0, MAX_LABELS)
|
|
137
|
-
.map((value) => text(value, MAX_LABEL_BYTES))
|
|
138
|
-
.filter((value): value is string => Boolean(value));
|
|
139
|
-
|
|
140
|
-
const PROVIDER_SENSITIVE_KEY = /message|prompt|content|input|system/i;
|
|
141
|
-
const SAFE_PROVIDER_FIELDS = new Set([
|
|
142
|
-
"max_completion_tokens",
|
|
143
|
-
"max_tokens",
|
|
144
|
-
"parallel_tool_calls",
|
|
145
|
-
"reasoning_effort",
|
|
146
|
-
"temperature",
|
|
147
|
-
"top_p",
|
|
148
|
-
"tool_choice",
|
|
149
|
-
]);
|
|
150
|
-
const sanitizeProviderValue = (value: unknown, depth = 0): unknown => {
|
|
151
|
-
if (depth > 3 || value === null) {
|
|
152
|
-
return value;
|
|
153
|
-
}
|
|
154
|
-
if (
|
|
155
|
-
typeof value === "string" ||
|
|
156
|
-
typeof value === "number" ||
|
|
157
|
-
typeof value === "boolean"
|
|
158
|
-
) {
|
|
159
|
-
return typeof value === "string" ? text(value, 320) : value;
|
|
160
|
-
}
|
|
161
|
-
if (Array.isArray(value)) {
|
|
162
|
-
return value
|
|
163
|
-
.slice(0, 16)
|
|
164
|
-
.map((item) => sanitizeProviderValue(item, depth + 1));
|
|
165
|
-
}
|
|
166
|
-
if (typeof value === "object") {
|
|
167
|
-
return Object.fromEntries(
|
|
168
|
-
Object.entries(value as Record<string, unknown>)
|
|
169
|
-
.filter(([key]) => !PROVIDER_SENSITIVE_KEY.test(key))
|
|
170
|
-
.slice(0, 32)
|
|
171
|
-
.map(([key, item]) => [key, sanitizeProviderValue(item, depth + 1)])
|
|
172
|
-
);
|
|
173
|
-
}
|
|
174
|
-
return undefined;
|
|
175
|
-
};
|
|
176
|
-
|
|
177
|
-
const providerFields = (payload: unknown): Record<string, unknown> => {
|
|
178
|
-
const value =
|
|
179
|
-
payload && typeof payload === "object"
|
|
180
|
-
? (payload as Record<string, unknown>)
|
|
181
|
-
: {};
|
|
182
|
-
return Object.fromEntries(
|
|
183
|
-
[...SAFE_PROVIDER_FIELDS]
|
|
184
|
-
.filter((key) => key in value)
|
|
185
|
-
.map((key) => [key, sanitizeProviderValue(value[key])])
|
|
186
|
-
);
|
|
187
|
-
};
|
|
188
|
-
|
|
189
|
-
const enabledCapability = () => {
|
|
190
|
-
const context = process.env[BENCHMARK_CONTEXT] === "1";
|
|
191
|
-
const runId = process.env[BENCHMARK_RUN_ID];
|
|
192
|
-
const token = process.env[BENCHMARK_TOKEN];
|
|
193
|
-
if (!(context && runId && token && token.length >= 32)) {
|
|
194
|
-
return;
|
|
195
|
-
}
|
|
196
|
-
return { runId, token };
|
|
197
|
-
};
|
|
198
|
-
|
|
199
|
-
export const createBenchmarkTelemetry = (
|
|
200
|
-
events: EventBus
|
|
201
|
-
): BenchmarkTelemetry | undefined => {
|
|
202
|
-
const capability = enabledCapability();
|
|
203
|
-
if (!capability) {
|
|
204
|
-
return undefined;
|
|
205
|
-
}
|
|
206
|
-
const emit = (event: BenchmarkTelemetryPayload) => {
|
|
207
|
-
try {
|
|
208
|
-
events.emit(BENCHMARK_TELEMETRY_CHANNEL, {
|
|
209
|
-
...event,
|
|
210
|
-
runId: capability.runId,
|
|
211
|
-
timestamp: new Date().toISOString(),
|
|
212
|
-
} satisfies BenchmarkTelemetryEvent);
|
|
213
|
-
} catch {
|
|
214
|
-
// Benchmark diagnostics must never change Advisor/Scout behavior.
|
|
215
|
-
}
|
|
216
|
-
};
|
|
217
|
-
return {
|
|
218
|
-
advisorEnd: (event) =>
|
|
219
|
-
emit({
|
|
220
|
-
model: event.model,
|
|
221
|
-
outcome: text(event.outcome, 160),
|
|
222
|
-
response: text(event.response),
|
|
223
|
-
type: "advisor:end",
|
|
224
|
-
usage: usageSnapshot(event.usage),
|
|
225
|
-
}),
|
|
226
|
-
advisorError: (event) =>
|
|
227
|
-
emit({
|
|
228
|
-
category: event.category,
|
|
229
|
-
model: event.model,
|
|
230
|
-
type: "advisor:error",
|
|
231
|
-
}),
|
|
232
|
-
advisorStart: (event) =>
|
|
233
|
-
emit({
|
|
234
|
-
model: event.model,
|
|
235
|
-
question: text(event.question),
|
|
236
|
-
trigger: text(event.trigger, 160),
|
|
237
|
-
type: "advisor:start",
|
|
238
|
-
}),
|
|
239
|
-
providerRequest: (payload) => {
|
|
240
|
-
emit({
|
|
241
|
-
fields: providerFields(payload),
|
|
242
|
-
type: "provider-request",
|
|
243
|
-
});
|
|
244
|
-
},
|
|
245
|
-
scout: (event) => {
|
|
246
|
-
if (event.type === "call" || event.type === "cancelled") {
|
|
247
|
-
emit({ event, type: "scout" });
|
|
248
|
-
return;
|
|
249
|
-
}
|
|
250
|
-
emit({
|
|
251
|
-
event: {
|
|
252
|
-
...(event.type === "success"
|
|
253
|
-
? {
|
|
254
|
-
selectedLabels: labels(event.selectedLabels),
|
|
255
|
-
synthesis: text(event.synthesis),
|
|
256
|
-
}
|
|
257
|
-
: { fallback: text(event.fallback, 320) }),
|
|
258
|
-
availableCount: finite(event.availableCount),
|
|
259
|
-
latencyMs: finite(event.latencyMs),
|
|
260
|
-
model: event.model,
|
|
261
|
-
omittedBeforeScout: finite(event.omittedBeforeScout),
|
|
262
|
-
selectedCount: finite(event.selectedCount),
|
|
263
|
-
type: event.type,
|
|
264
|
-
usage: usageSnapshot(event.usage),
|
|
265
|
-
},
|
|
266
|
-
type: "scout",
|
|
267
|
-
});
|
|
268
|
-
},
|
|
269
|
-
};
|
|
270
|
-
};
|