grix-connector 3.22.0 → 3.24.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/adapter/acp/acp-adapter.js +14 -12
- package/dist/adapter/claude/claude-adapter.js +18 -18
- package/dist/adapter/claude/protocol-contract.js +1 -1
- package/dist/adapter/opencode/format-error.js +1 -0
- package/dist/adapter/opencode/opencode-adapter.js +6 -6
- package/dist/adapter/opencode/opencode-transport.js +3 -3
- package/dist/adapter/opencode/opencode-types.js +1 -0
- package/dist/adapter/shared/session-context.js +2 -0
- package/dist/bridge/acp-toolbar-persist.js +1 -1
- package/dist/bridge/adapter-pool.js +1 -1
- package/dist/bridge/bridge.js +10 -11
- package/dist/core/aibot/binding-card-meta-cache.js +1 -1
- package/dist/core/config/claude-direct-env.js +1 -1
- package/dist/core/config/provider-env.js +1 -1
- package/dist/core/persistence/agent-global-config-store.js +1 -1
- package/dist/core/persistence/session-binding-store.js +1 -1
- package/dist/core/protocol/interaction-parser.js +1 -1
- package/dist/core/provider-quota/index.js +1 -1
- package/dist/core/provider-quota/providers.js +2 -2
- package/dist/default-skills/grix-agent-dispatch/SKILL.md +130 -58
- package/dist/default-skills/grix-owner-relay/SKILL.md +7 -0
- package/dist/manager.js +1 -1
- package/dist/mcp/stream-http/security.js +1 -1
- package/openclaw-plugin/index.js +26 -2
- package/package.json +3 -1
- package/scripts/analyze-audit-replays.mjs +1416 -0
- package/scripts/upgrade-guardian.sh +64 -29
|
@@ -0,0 +1,1416 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
|
|
3
|
+
import { readFile, readdir, writeFile } from 'node:fs/promises';
|
|
4
|
+
import { homedir } from 'node:os';
|
|
5
|
+
import { join, resolve } from 'node:path';
|
|
6
|
+
import { fileURLToPath } from 'node:url';
|
|
7
|
+
|
|
8
|
+
const DEFAULT_ROOT = join(homedir(), '.grix', 'data', 'audit-replay');
|
|
9
|
+
const DEFAULT_TOP = 10;
|
|
10
|
+
const OUTCOMES = ['completed', 'failed', 'cancelled'];
|
|
11
|
+
const SPAN_STATUSES = ['completed', 'failed', 'cancelled', 'running'];
|
|
12
|
+
|
|
13
|
+
function asNumber(value) {
|
|
14
|
+
return typeof value === 'number' && Number.isFinite(value) ? value : 0;
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
function parseTimestamp(value) {
|
|
18
|
+
if (typeof value !== 'string') return null;
|
|
19
|
+
const timestamp = Date.parse(value);
|
|
20
|
+
return Number.isFinite(timestamp) ? timestamp : null;
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
function countBy(items, keyOf) {
|
|
24
|
+
const counts = new Map();
|
|
25
|
+
for (const item of items) {
|
|
26
|
+
const key = keyOf(item);
|
|
27
|
+
counts.set(key, (counts.get(key) ?? 0) + 1);
|
|
28
|
+
}
|
|
29
|
+
return counts;
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
function sortedCounts(counts) {
|
|
33
|
+
return [...counts.entries()]
|
|
34
|
+
.map(([name, count]) => ({ name, count }))
|
|
35
|
+
.sort((left, right) => right.count - left.count || left.name.localeCompare(right.name));
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
function sum(values) {
|
|
39
|
+
return values.reduce((total, value) => total + asNumber(value), 0);
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
function mean(values) {
|
|
43
|
+
return values.length === 0 ? null : sum(values) / values.length;
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
export function percentile(values, ratio) {
|
|
47
|
+
if (values.length === 0) return null;
|
|
48
|
+
const ordered = values
|
|
49
|
+
.filter((value) => typeof value === 'number' && Number.isFinite(value))
|
|
50
|
+
.sort((left, right) => left - right);
|
|
51
|
+
if (ordered.length === 0) return null;
|
|
52
|
+
const index = (ordered.length - 1) * ratio;
|
|
53
|
+
const lower = Math.floor(index);
|
|
54
|
+
const upper = Math.ceil(index);
|
|
55
|
+
if (lower === upper) return ordered[lower];
|
|
56
|
+
const weight = index - lower;
|
|
57
|
+
return ordered[lower] * (1 - weight) + ordered[upper] * weight;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
function distribution(values) {
|
|
61
|
+
return {
|
|
62
|
+
mean: mean(values),
|
|
63
|
+
p50: percentile(values, 0.5),
|
|
64
|
+
p90: percentile(values, 0.9),
|
|
65
|
+
max: values.length === 0 ? null : Math.max(...values),
|
|
66
|
+
};
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
function safeRatio(numerator, denominator) {
|
|
70
|
+
return denominator > 0 ? numerator / denominator : null;
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
function pearson(left, right) {
|
|
74
|
+
if (left.length !== right.length || left.length < 2) return null;
|
|
75
|
+
const leftMean = mean(left);
|
|
76
|
+
const rightMean = mean(right);
|
|
77
|
+
if (leftMean === null || rightMean === null) return null;
|
|
78
|
+
let numerator = 0;
|
|
79
|
+
let leftVariance = 0;
|
|
80
|
+
let rightVariance = 0;
|
|
81
|
+
for (let index = 0; index < left.length; index += 1) {
|
|
82
|
+
const leftDelta = left[index] - leftMean;
|
|
83
|
+
const rightDelta = right[index] - rightMean;
|
|
84
|
+
numerator += leftDelta * rightDelta;
|
|
85
|
+
leftVariance += leftDelta * leftDelta;
|
|
86
|
+
rightVariance += rightDelta * rightDelta;
|
|
87
|
+
}
|
|
88
|
+
const denominator = Math.sqrt(leftVariance * rightVariance);
|
|
89
|
+
return denominator > 0 ? numerator / denominator : null;
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
export function canonicalToolName(name) {
|
|
93
|
+
const normalized = String(name ?? 'unknown')
|
|
94
|
+
.trim()
|
|
95
|
+
.replace(/^mcp__/, '')
|
|
96
|
+
.replaceAll('__', '/')
|
|
97
|
+
.toLowerCase();
|
|
98
|
+
const aliases = new Map([
|
|
99
|
+
['commandexecution', 'shell'],
|
|
100
|
+
['shell', 'shell'],
|
|
101
|
+
['bash', 'shell'],
|
|
102
|
+
['functions/exec_command', 'shell'],
|
|
103
|
+
['filechange', 'file_change'],
|
|
104
|
+
['edit', 'file_change'],
|
|
105
|
+
['write', 'file_change'],
|
|
106
|
+
['functions/apply_patch', 'file_change'],
|
|
107
|
+
['read', 'file_read'],
|
|
108
|
+
['grep', 'file_search'],
|
|
109
|
+
['glob', 'file_search'],
|
|
110
|
+
['websearch', 'web_search'],
|
|
111
|
+
['fetchurl', 'fetch_url'],
|
|
112
|
+
['imageview', 'image_view'],
|
|
113
|
+
['todolist', 'planning'],
|
|
114
|
+
['updatetodos', 'planning'],
|
|
115
|
+
]);
|
|
116
|
+
return aliases.get(normalized) ?? (normalized || 'unknown');
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
export function classifyTool(name) {
|
|
120
|
+
const tool = canonicalToolName(name);
|
|
121
|
+
if (
|
|
122
|
+
tool === 'agent'
|
|
123
|
+
|| tool === 'await'
|
|
124
|
+
|| tool === 'taskoutput'
|
|
125
|
+
|| tool.includes('dispatch_agent')
|
|
126
|
+
|| tool.includes('subagent')
|
|
127
|
+
) {
|
|
128
|
+
return 'agent_orchestration';
|
|
129
|
+
}
|
|
130
|
+
if (
|
|
131
|
+
tool.includes('message_send')
|
|
132
|
+
|| tool.includes('session_send')
|
|
133
|
+
|| tool.includes('owner_relay')
|
|
134
|
+
|| tool.includes('group')
|
|
135
|
+
) {
|
|
136
|
+
return 'communication';
|
|
137
|
+
}
|
|
138
|
+
if (
|
|
139
|
+
tool.includes('grix_query')
|
|
140
|
+
|| tool.includes('chat_state')
|
|
141
|
+
|| tool.includes('audit_data')
|
|
142
|
+
|| tool.includes('file_link')
|
|
143
|
+
) {
|
|
144
|
+
return 'platform_query';
|
|
145
|
+
}
|
|
146
|
+
if (
|
|
147
|
+
tool === 'file_search'
|
|
148
|
+
|| tool.includes('codebase-memory')
|
|
149
|
+
|| tool.includes('search_graph')
|
|
150
|
+
|| tool.includes('search_code')
|
|
151
|
+
|| tool.includes('get_code_snippet')
|
|
152
|
+
|| tool.includes('get_architecture')
|
|
153
|
+
) {
|
|
154
|
+
return 'code_search';
|
|
155
|
+
}
|
|
156
|
+
if (tool === 'file_read' || tool.endsWith('/read_file')) return 'file_read';
|
|
157
|
+
if (tool === 'file_change' || tool.includes('apply_patch')) return 'file_change';
|
|
158
|
+
if (tool === 'shell' || tool.includes('exec_command')) return 'shell';
|
|
159
|
+
if (
|
|
160
|
+
tool === 'web_search'
|
|
161
|
+
|| tool === 'fetch_url'
|
|
162
|
+
|| tool.includes('browser')
|
|
163
|
+
|| tool.includes('web_search')
|
|
164
|
+
) {
|
|
165
|
+
return 'web';
|
|
166
|
+
}
|
|
167
|
+
if (tool === 'image_view' || tool.includes('imagegen') || tool.includes('image_gen')) {
|
|
168
|
+
return 'visual';
|
|
169
|
+
}
|
|
170
|
+
if (tool === 'planning' || tool.includes('update_plan')) return 'planning';
|
|
171
|
+
if (tool === 'skill' || tool.includes('skill')) return 'skill';
|
|
172
|
+
if (tool === 'mcp' || tool === 'getmcptools') return 'generic_mcp';
|
|
173
|
+
return 'other';
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
function dominantModel(replay) {
|
|
177
|
+
const counts = countBy(
|
|
178
|
+
(replay.spans ?? []).filter(
|
|
179
|
+
(span) => span?.kind === 'llm_request' && typeof span.modelId === 'string' && span.modelId.trim(),
|
|
180
|
+
),
|
|
181
|
+
(span) => span.modelId.trim(),
|
|
182
|
+
);
|
|
183
|
+
const ordered = sortedCounts(counts);
|
|
184
|
+
return ordered[0]?.name ?? 'unknown';
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
function durationMs(replay) {
|
|
188
|
+
const startedAt = parseTimestamp(replay.startedAt);
|
|
189
|
+
const endedAt = parseTimestamp(replay.endedAt);
|
|
190
|
+
if (startedAt === null || endedAt === null || endedAt < startedAt) return null;
|
|
191
|
+
return endedAt - startedAt;
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
function replayMetrics(replay) {
|
|
195
|
+
const spans = Array.isArray(replay.spans) ? replay.spans : [];
|
|
196
|
+
const toolSpans = spans.filter((span) => span?.kind === 'tool_call');
|
|
197
|
+
const canonicalTools = toolSpans.map((span) => canonicalToolName(span.name));
|
|
198
|
+
const toolFamilies = toolSpans.map((span) => classifyTool(span.name));
|
|
199
|
+
const statusCounts = Object.fromEntries(SPAN_STATUSES.map((status) => [status, 0]));
|
|
200
|
+
for (const span of toolSpans) {
|
|
201
|
+
const status = SPAN_STATUSES.includes(span.status) ? span.status : 'running';
|
|
202
|
+
statusCounts[status] += 1;
|
|
203
|
+
}
|
|
204
|
+
const input = replay.usage?.input ?? {};
|
|
205
|
+
const output = replay.usage?.output ?? {};
|
|
206
|
+
const inputTokens = asNumber(input.total);
|
|
207
|
+
const outputTokens = asNumber(output.total);
|
|
208
|
+
const totalTokens = asNumber(replay.usage?.totalProcessed) || inputTokens + outputTokens;
|
|
209
|
+
const llmRequestCount = spans.filter((span) => span?.kind === 'llm_request').length;
|
|
210
|
+
const toolCallCount = toolSpans.length;
|
|
211
|
+
|
|
212
|
+
return {
|
|
213
|
+
replay,
|
|
214
|
+
auditId: String(replay.auditId ?? 'unknown'),
|
|
215
|
+
inputContentId: replay.input?.reference?.contentId ?? null,
|
|
216
|
+
provider: String(replay.provider ?? 'unknown'),
|
|
217
|
+
model: dominantModel(replay),
|
|
218
|
+
outcome: String(replay.outcome ?? 'unknown'),
|
|
219
|
+
qualityStatus: String(replay.quality?.status ?? 'unknown'),
|
|
220
|
+
rawRequestsStatus: replay.quality?.rawRequestsComplete === true
|
|
221
|
+
? 'complete'
|
|
222
|
+
: (replay.quality?.rawRequestsComplete === false ? 'incomplete' : 'unknown'),
|
|
223
|
+
usageCompleteness: String(replay.usage?.completeness ?? 'unknown'),
|
|
224
|
+
durationMs: durationMs(replay),
|
|
225
|
+
spanCount: spans.length,
|
|
226
|
+
llmRequestCount,
|
|
227
|
+
toolCallCount,
|
|
228
|
+
toolStatusCounts: statusCounts,
|
|
229
|
+
uniqueTools: new Set(canonicalTools),
|
|
230
|
+
canonicalTools,
|
|
231
|
+
toolFamilies,
|
|
232
|
+
subagentCount: spans.filter((span) => span?.kind === 'subagent').length,
|
|
233
|
+
compactionCount: spans.filter((span) => span?.kind === 'compaction').length,
|
|
234
|
+
retryCount: spans.filter((span) => span?.kind === 'retry').length,
|
|
235
|
+
permissionCount: spans.filter((span) => span?.kind === 'permission').length,
|
|
236
|
+
inputTokens,
|
|
237
|
+
uncachedInputTokens: asNumber(input.uncached),
|
|
238
|
+
cacheReadTokens: asNumber(input.cacheRead),
|
|
239
|
+
cacheWriteTokens: asNumber(input.cacheWrite),
|
|
240
|
+
outputTokens,
|
|
241
|
+
reasoningTokens: asNumber(output.reasoning),
|
|
242
|
+
visibleOutputTokens: asNumber(output.visible),
|
|
243
|
+
totalTokens,
|
|
244
|
+
requestCount: asNumber(replay.usage?.requestCount),
|
|
245
|
+
startedAt: replay.startedAt,
|
|
246
|
+
finalizedAt: replay.finalizedAt,
|
|
247
|
+
toolsPerLlmRequest: safeRatio(toolCallCount, llmRequestCount),
|
|
248
|
+
tokensPerToolCall: safeRatio(totalTokens, toolCallCount),
|
|
249
|
+
};
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
function aggregateMetrics(metrics, label) {
|
|
253
|
+
const outcomes = Object.fromEntries(OUTCOMES.map((outcome) => [outcome, 0]));
|
|
254
|
+
const tools = [];
|
|
255
|
+
const toolFamilies = [];
|
|
256
|
+
const uniqueTools = new Set();
|
|
257
|
+
for (const item of metrics) {
|
|
258
|
+
outcomes[item.outcome] = (outcomes[item.outcome] ?? 0) + 1;
|
|
259
|
+
tools.push(...item.canonicalTools);
|
|
260
|
+
toolFamilies.push(...item.toolFamilies);
|
|
261
|
+
for (const tool of item.uniqueTools) uniqueTools.add(tool);
|
|
262
|
+
}
|
|
263
|
+
const inputTokens = sum(metrics.map((item) => item.inputTokens));
|
|
264
|
+
const outputTokens = sum(metrics.map((item) => item.outputTokens));
|
|
265
|
+
const totalTokens = sum(metrics.map((item) => item.totalTokens));
|
|
266
|
+
const cacheReadTokens = sum(metrics.map((item) => item.cacheReadTokens));
|
|
267
|
+
const reasoningTokens = sum(metrics.map((item) => item.reasoningTokens));
|
|
268
|
+
const toolCalls = sum(metrics.map((item) => item.toolCallCount));
|
|
269
|
+
const toolFailures = sum(metrics.map((item) => item.toolStatusCounts.failed));
|
|
270
|
+
const toolCancelled = sum(metrics.map((item) => item.toolStatusCounts.cancelled));
|
|
271
|
+
const completedTurns = outcomes.completed ?? 0;
|
|
272
|
+
const tokensByOutcome = Object.fromEntries(
|
|
273
|
+
[...new Set([...OUTCOMES, ...metrics.map((item) => item.outcome)])]
|
|
274
|
+
.map((outcome) => [
|
|
275
|
+
outcome,
|
|
276
|
+
sum(metrics.filter((item) => item.outcome === outcome).map((item) => item.totalTokens)),
|
|
277
|
+
]),
|
|
278
|
+
);
|
|
279
|
+
const nonCompletedTokens = totalTokens - asNumber(tokensByOutcome.completed);
|
|
280
|
+
const durations = metrics
|
|
281
|
+
.map((item) => item.durationMs)
|
|
282
|
+
.filter((value) => value !== null);
|
|
283
|
+
|
|
284
|
+
return {
|
|
285
|
+
label,
|
|
286
|
+
turns: metrics.length,
|
|
287
|
+
outcomes,
|
|
288
|
+
successRate: safeRatio(completedTurns, metrics.length),
|
|
289
|
+
qualityCompleteTurns: metrics.filter((item) => item.qualityStatus === 'complete').length,
|
|
290
|
+
qualityCompleteRate: safeRatio(
|
|
291
|
+
metrics.filter((item) => item.qualityStatus === 'complete').length,
|
|
292
|
+
metrics.length,
|
|
293
|
+
),
|
|
294
|
+
usageCompleteTurns: metrics.filter((item) => item.usageCompleteness === 'complete').length,
|
|
295
|
+
usageCompleteRate: safeRatio(
|
|
296
|
+
metrics.filter((item) => item.usageCompleteness === 'complete').length,
|
|
297
|
+
metrics.length,
|
|
298
|
+
),
|
|
299
|
+
durationMs: distribution(durations),
|
|
300
|
+
llmRequests: {
|
|
301
|
+
total: sum(metrics.map((item) => item.llmRequestCount)),
|
|
302
|
+
...distribution(metrics.map((item) => item.llmRequestCount)),
|
|
303
|
+
},
|
|
304
|
+
toolCalls: {
|
|
305
|
+
total: toolCalls,
|
|
306
|
+
...distribution(metrics.map((item) => item.toolCallCount)),
|
|
307
|
+
zeroToolTurnRate: safeRatio(
|
|
308
|
+
metrics.filter((item) => item.toolCallCount === 0).length,
|
|
309
|
+
metrics.length,
|
|
310
|
+
),
|
|
311
|
+
failed: toolFailures,
|
|
312
|
+
cancelled: toolCancelled,
|
|
313
|
+
failureRate: safeRatio(toolFailures, toolCalls),
|
|
314
|
+
},
|
|
315
|
+
uniqueToolCount: uniqueTools.size,
|
|
316
|
+
topTools: sortedCounts(countBy(tools, (tool) => tool)),
|
|
317
|
+
topToolFamilies: sortedCounts(countBy(toolFamilies, (family) => family)),
|
|
318
|
+
subagents: {
|
|
319
|
+
total: sum(metrics.map((item) => item.subagentCount)),
|
|
320
|
+
turns: metrics.filter((item) => item.subagentCount > 0).length,
|
|
321
|
+
turnRate: safeRatio(
|
|
322
|
+
metrics.filter((item) => item.subagentCount > 0).length,
|
|
323
|
+
metrics.length,
|
|
324
|
+
),
|
|
325
|
+
},
|
|
326
|
+
compactions: {
|
|
327
|
+
total: sum(metrics.map((item) => item.compactionCount)),
|
|
328
|
+
turns: metrics.filter((item) => item.compactionCount > 0).length,
|
|
329
|
+
turnRate: safeRatio(
|
|
330
|
+
metrics.filter((item) => item.compactionCount > 0).length,
|
|
331
|
+
metrics.length,
|
|
332
|
+
),
|
|
333
|
+
},
|
|
334
|
+
retries: sum(metrics.map((item) => item.retryCount)),
|
|
335
|
+
permissions: sum(metrics.map((item) => item.permissionCount)),
|
|
336
|
+
tokens: {
|
|
337
|
+
input: inputTokens,
|
|
338
|
+
output: outputTokens,
|
|
339
|
+
total: totalTokens,
|
|
340
|
+
reasoning: reasoningTokens,
|
|
341
|
+
cacheRead: cacheReadTokens,
|
|
342
|
+
perTurn: distribution(metrics.map((item) => item.totalTokens)),
|
|
343
|
+
inputPerTurn: distribution(metrics.map((item) => item.inputTokens)),
|
|
344
|
+
outputPerTurn: distribution(metrics.map((item) => item.outputTokens)),
|
|
345
|
+
perCompletedTurn: safeRatio(totalTokens, completedTurns),
|
|
346
|
+
perToolCall: safeRatio(totalTokens, toolCalls),
|
|
347
|
+
cacheReadRate: safeRatio(cacheReadTokens, inputTokens),
|
|
348
|
+
reasoningShare: safeRatio(reasoningTokens, outputTokens),
|
|
349
|
+
byOutcome: tokensByOutcome,
|
|
350
|
+
nonCompleted: nonCompletedTokens,
|
|
351
|
+
nonCompletedShare: safeRatio(nonCompletedTokens, totalTokens),
|
|
352
|
+
},
|
|
353
|
+
requests: {
|
|
354
|
+
total: sum(metrics.map((item) => item.requestCount)),
|
|
355
|
+
...distribution(metrics.map((item) => item.requestCount)),
|
|
356
|
+
},
|
|
357
|
+
};
|
|
358
|
+
}
|
|
359
|
+
|
|
360
|
+
function groupMetrics(metrics, keyOf) {
|
|
361
|
+
const groups = new Map();
|
|
362
|
+
for (const item of metrics) {
|
|
363
|
+
const key = keyOf(item);
|
|
364
|
+
const group = groups.get(key) ?? [];
|
|
365
|
+
group.push(item);
|
|
366
|
+
groups.set(key, group);
|
|
367
|
+
}
|
|
368
|
+
return [...groups.entries()]
|
|
369
|
+
.map(([label, items]) => aggregateMetrics(items, label))
|
|
370
|
+
.sort((left, right) => right.turns - left.turns || left.label.localeCompare(right.label));
|
|
371
|
+
}
|
|
372
|
+
|
|
373
|
+
function rankOutliers(metrics, key, top) {
|
|
374
|
+
return metrics
|
|
375
|
+
.filter((item) => typeof item[key] === 'number' && Number.isFinite(item[key]))
|
|
376
|
+
.sort((left, right) => right[key] - left[key])
|
|
377
|
+
.slice(0, top)
|
|
378
|
+
.map((item) => ({
|
|
379
|
+
auditId: item.auditId,
|
|
380
|
+
provider: item.provider,
|
|
381
|
+
model: item.model,
|
|
382
|
+
outcome: item.outcome,
|
|
383
|
+
value: item[key],
|
|
384
|
+
}));
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
export function selectLatestRevisions(replays) {
|
|
388
|
+
const latest = new Map();
|
|
389
|
+
for (const replay of replays) {
|
|
390
|
+
const auditId = String(replay?.auditId ?? '').trim();
|
|
391
|
+
if (!auditId) continue;
|
|
392
|
+
// Session-scoped audits reuse one auditId across turns, so the dedupe key
|
|
393
|
+
// must include turnId or every turn but one would be collapsed away.
|
|
394
|
+
const key = `${auditId}/${String(replay.turnId ?? '').trim()}`;
|
|
395
|
+
const current = latest.get(key);
|
|
396
|
+
if (
|
|
397
|
+
!current
|
|
398
|
+
|| asNumber(replay.revision) > asNumber(current.revision)
|
|
399
|
+
|| (
|
|
400
|
+
asNumber(replay.revision) === asNumber(current.revision)
|
|
401
|
+
&& (parseTimestamp(replay.finalizedAt) ?? 0) > (parseTimestamp(current.finalizedAt) ?? 0)
|
|
402
|
+
)
|
|
403
|
+
) {
|
|
404
|
+
latest.set(key, replay);
|
|
405
|
+
}
|
|
406
|
+
}
|
|
407
|
+
return [...latest.values()];
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
export function analyzeReplays(replays, options = {}) {
|
|
411
|
+
const top = options.top ?? DEFAULT_TOP;
|
|
412
|
+
const latest = selectLatestRevisions(replays);
|
|
413
|
+
const metrics = latest.map(replayMetrics);
|
|
414
|
+
const overall = aggregateMetrics(metrics, 'all');
|
|
415
|
+
const totalTokens = metrics.map((item) => item.totalTokens);
|
|
416
|
+
const toolCalls = metrics.map((item) => item.toolCallCount);
|
|
417
|
+
const llmRequests = metrics.map((item) => item.llmRequestCount);
|
|
418
|
+
const pairedDurations = metrics.filter((item) => item.durationMs !== null);
|
|
419
|
+
const rawStatuses = sortedCounts(countBy(metrics, (item) => item.rawRequestsStatus));
|
|
420
|
+
const orderedTokenCosts = metrics
|
|
421
|
+
.map((item) => item.totalTokens)
|
|
422
|
+
.sort((left, right) => right - left);
|
|
423
|
+
const tokenCostTotal = sum(orderedTokenCosts);
|
|
424
|
+
const concentration = (count) => safeRatio(
|
|
425
|
+
sum(orderedTokenCosts.slice(0, count)),
|
|
426
|
+
tokenCostTotal,
|
|
427
|
+
);
|
|
428
|
+
const inputCohorts = new Map();
|
|
429
|
+
for (const item of metrics) {
|
|
430
|
+
if (!item.inputContentId) continue;
|
|
431
|
+
const cohort = inputCohorts.get(item.inputContentId) ?? [];
|
|
432
|
+
cohort.push(item);
|
|
433
|
+
inputCohorts.set(item.inputContentId, cohort);
|
|
434
|
+
}
|
|
435
|
+
const matchedCohorts = [...inputCohorts.values()].filter(
|
|
436
|
+
(items) => new Set(items.map((item) => item.provider)).size >= 2,
|
|
437
|
+
);
|
|
438
|
+
const matchedMetrics = matchedCohorts.flat();
|
|
439
|
+
|
|
440
|
+
return {
|
|
441
|
+
summary: {
|
|
442
|
+
sourceReplayCount: replays.length,
|
|
443
|
+
latestAuditCount: latest.length,
|
|
444
|
+
sessionCount: new Set(latest.map((replay) => replay.sessionId).filter(Boolean)).size,
|
|
445
|
+
startedAtMin: [...metrics.map((item) => item.startedAt).filter(Boolean)].sort()[0] ?? null,
|
|
446
|
+
finalizedAtMax: [...metrics.map((item) => item.finalizedAt).filter(Boolean)].sort().at(-1) ?? null,
|
|
447
|
+
},
|
|
448
|
+
dataQuality: {
|
|
449
|
+
parseErrors: options.parseErrors ?? [],
|
|
450
|
+
skippedWithoutAuditId: replays.filter(
|
|
451
|
+
(replay) => !String(replay?.auditId ?? '').trim(),
|
|
452
|
+
).length,
|
|
453
|
+
rawRequestStatuses: rawStatuses,
|
|
454
|
+
qualityStatuses: sortedCounts(countBy(metrics, (item) => item.qualityStatus)),
|
|
455
|
+
usageCompleteness: sortedCounts(countBy(metrics, (item) => item.usageCompleteness)),
|
|
456
|
+
},
|
|
457
|
+
overall,
|
|
458
|
+
byProvider: groupMetrics(metrics, (item) => item.provider),
|
|
459
|
+
byProviderModel: groupMetrics(metrics, (item) => `${item.provider} / ${item.model}`),
|
|
460
|
+
matchedInputs: {
|
|
461
|
+
cohortCount: matchedCohorts.length,
|
|
462
|
+
turnCount: matchedMetrics.length,
|
|
463
|
+
providerSets: sortedCounts(countBy(
|
|
464
|
+
matchedCohorts,
|
|
465
|
+
(items) => [...new Set(items.map((item) => item.provider))].sort().join(' + '),
|
|
466
|
+
)),
|
|
467
|
+
byProvider: groupMetrics(matchedMetrics, (item) => item.provider),
|
|
468
|
+
},
|
|
469
|
+
correlations: {
|
|
470
|
+
toolCallsVsTotalTokens: pearson(toolCalls, totalTokens),
|
|
471
|
+
llmRequestsVsTotalTokens: pearson(llmRequests, totalTokens),
|
|
472
|
+
toolCallsVsDuration: pearson(
|
|
473
|
+
pairedDurations.map((item) => item.toolCallCount),
|
|
474
|
+
pairedDurations.map((item) => item.durationMs),
|
|
475
|
+
),
|
|
476
|
+
},
|
|
477
|
+
costConcentration: {
|
|
478
|
+
top1TurnShare: concentration(1),
|
|
479
|
+
top5TurnsShare: concentration(5),
|
|
480
|
+
top10TurnsShare: concentration(10),
|
|
481
|
+
},
|
|
482
|
+
outliers: {
|
|
483
|
+
totalTokens: rankOutliers(metrics, 'totalTokens', top),
|
|
484
|
+
toolCalls: rankOutliers(metrics, 'toolCallCount', top),
|
|
485
|
+
durationMs: rankOutliers(metrics, 'durationMs', top),
|
|
486
|
+
},
|
|
487
|
+
};
|
|
488
|
+
}
|
|
489
|
+
|
|
490
|
+
async function collectJsonFiles(root) {
|
|
491
|
+
const files = [];
|
|
492
|
+
async function visit(directory) {
|
|
493
|
+
const entries = await readdir(directory, { withFileTypes: true });
|
|
494
|
+
for (const entry of entries) {
|
|
495
|
+
const path = join(directory, entry.name);
|
|
496
|
+
if (entry.isDirectory()) await visit(path);
|
|
497
|
+
else if (entry.isFile() && entry.name.endsWith('.json')) files.push(path);
|
|
498
|
+
}
|
|
499
|
+
}
|
|
500
|
+
await visit(root);
|
|
501
|
+
return files.sort();
|
|
502
|
+
}
|
|
503
|
+
|
|
504
|
+
async function loadReplays(root) {
|
|
505
|
+
const replayRoot = join(root, 'replays');
|
|
506
|
+
const files = await collectJsonFiles(replayRoot);
|
|
507
|
+
const replays = [];
|
|
508
|
+
const parseErrors = [];
|
|
509
|
+
for (const path of files) {
|
|
510
|
+
try {
|
|
511
|
+
const value = JSON.parse(await readFile(path, 'utf8'));
|
|
512
|
+
if (!value || typeof value !== 'object' || Array.isArray(value)) {
|
|
513
|
+
parseErrors.push({ path, error: 'root JSON value is not an object' });
|
|
514
|
+
continue;
|
|
515
|
+
}
|
|
516
|
+
replays.push(value);
|
|
517
|
+
} catch (error) {
|
|
518
|
+
parseErrors.push({
|
|
519
|
+
path,
|
|
520
|
+
error: error instanceof Error ? error.message : String(error),
|
|
521
|
+
});
|
|
522
|
+
}
|
|
523
|
+
}
|
|
524
|
+
return { replays, parseErrors, files };
|
|
525
|
+
}
|
|
526
|
+
|
|
527
|
+
function parseDateBound(value, endOfDay) {
|
|
528
|
+
if (!value) return null;
|
|
529
|
+
const dateOnly = /^\d{4}-\d{2}-\d{2}$/.test(value);
|
|
530
|
+
const timestamp = Date.parse(dateOnly
|
|
531
|
+
? `${value}T${endOfDay ? '23:59:59.999' : '00:00:00.000'}Z`
|
|
532
|
+
: value);
|
|
533
|
+
if (!Number.isFinite(timestamp)) throw new Error(`Invalid date: ${value}`);
|
|
534
|
+
return timestamp;
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
function filterReplays(replays, options) {
|
|
538
|
+
const since = parseDateBound(options.since, false);
|
|
539
|
+
const until = parseDateBound(options.until, true);
|
|
540
|
+
return replays.filter((replay) => {
|
|
541
|
+
if (options.provider && replay.provider !== options.provider) return false;
|
|
542
|
+
const startedAt = parseTimestamp(replay.startedAt);
|
|
543
|
+
if (since !== null && (startedAt === null || startedAt < since)) return false;
|
|
544
|
+
if (until !== null && (startedAt === null || startedAt > until)) return false;
|
|
545
|
+
return true;
|
|
546
|
+
});
|
|
547
|
+
}
|
|
548
|
+
|
|
549
|
+
function fmtInteger(value) {
|
|
550
|
+
return Math.round(asNumber(value)).toLocaleString('en-US');
|
|
551
|
+
}
|
|
552
|
+
|
|
553
|
+
function fmtDecimal(value, digits = 1) {
|
|
554
|
+
return value === null || value === undefined ? '-' : value.toFixed(digits);
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
function fmtPercent(value) {
|
|
558
|
+
return value === null || value === undefined ? '-' : `${(value * 100).toFixed(1)}%`;
|
|
559
|
+
}
|
|
560
|
+
|
|
561
|
+
function fmtDuration(value) {
|
|
562
|
+
if (value === null || value === undefined) return '-';
|
|
563
|
+
if (value < 1_000) return `${Math.round(value)}ms`;
|
|
564
|
+
if (value < 60_000) return `${(value / 1_000).toFixed(1)}s`;
|
|
565
|
+
return `${(value / 60_000).toFixed(1)}m`;
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
function escapeCell(value) {
|
|
569
|
+
return String(value).replaceAll('|', '\\|').replaceAll('\n', ' ');
|
|
570
|
+
}
|
|
571
|
+
|
|
572
|
+
function table(headers, rows) {
|
|
573
|
+
return [
|
|
574
|
+
`| ${headers.map(escapeCell).join(' | ')} |`,
|
|
575
|
+
`| ${headers.map(() => '---').join(' | ')} |`,
|
|
576
|
+
...rows.map((row) => `| ${row.map(escapeCell).join(' | ')} |`),
|
|
577
|
+
].join('\n');
|
|
578
|
+
}
|
|
579
|
+
|
|
580
|
+
function groupTable(groups) {
|
|
581
|
+
return table(
|
|
582
|
+
[
|
|
583
|
+
'Group',
|
|
584
|
+
'Turns',
|
|
585
|
+
'Success',
|
|
586
|
+
'Quality',
|
|
587
|
+
'Tools avg/p50/p90',
|
|
588
|
+
'Tool fail',
|
|
589
|
+
'Tokens p50/p90',
|
|
590
|
+
'Cache read',
|
|
591
|
+
'Duration p50',
|
|
592
|
+
'Subagent turns',
|
|
593
|
+
],
|
|
594
|
+
groups.map((group) => [
|
|
595
|
+
group.label,
|
|
596
|
+
fmtInteger(group.turns),
|
|
597
|
+
fmtPercent(group.successRate),
|
|
598
|
+
fmtPercent(group.qualityCompleteRate),
|
|
599
|
+
[
|
|
600
|
+
fmtDecimal(group.toolCalls.mean),
|
|
601
|
+
fmtDecimal(group.toolCalls.p50),
|
|
602
|
+
fmtDecimal(group.toolCalls.p90),
|
|
603
|
+
].join('/'),
|
|
604
|
+
fmtPercent(group.toolCalls.failureRate),
|
|
605
|
+
`${fmtInteger(group.tokens.perTurn.p50)}/${fmtInteger(group.tokens.perTurn.p90)}`,
|
|
606
|
+
fmtPercent(group.tokens.cacheReadRate),
|
|
607
|
+
fmtDuration(group.durationMs.p50),
|
|
608
|
+
`${fmtInteger(group.subagents.turns)} (${fmtPercent(group.subagents.turnRate)})`,
|
|
609
|
+
]),
|
|
610
|
+
);
|
|
611
|
+
}
|
|
612
|
+
|
|
613
|
+
function outlierTable(items, formatter) {
|
|
614
|
+
return table(
|
|
615
|
+
['Audit ID', 'Provider / model', 'Outcome', 'Value'],
|
|
616
|
+
items.map((item) => [
|
|
617
|
+
item.auditId,
|
|
618
|
+
`${item.provider} / ${item.model}`,
|
|
619
|
+
item.outcome,
|
|
620
|
+
formatter(item.value),
|
|
621
|
+
]),
|
|
622
|
+
);
|
|
623
|
+
}
|
|
624
|
+
|
|
625
|
+
function escapeHtml(value) {
|
|
626
|
+
return String(value)
|
|
627
|
+
.replaceAll('&', '&')
|
|
628
|
+
.replaceAll('<', '<')
|
|
629
|
+
.replaceAll('>', '>')
|
|
630
|
+
.replaceAll('"', '"')
|
|
631
|
+
.replaceAll("'", ''');
|
|
632
|
+
}
|
|
633
|
+
|
|
634
|
+
function htmlTable(headers, rows, className = '') {
|
|
635
|
+
return `<div class="table-wrap"><table class="${escapeHtml(className)}">
|
|
636
|
+
<thead><tr>${headers.map((header) => `<th>${escapeHtml(header)}</th>`).join('')}</tr></thead>
|
|
637
|
+
<tbody>${rows.map((row) => `<tr>${row.map((cell) => `<td>${escapeHtml(cell)}</td>`).join('')}</tr>`).join('')}</tbody>
|
|
638
|
+
</table></div>`;
|
|
639
|
+
}
|
|
640
|
+
|
|
641
|
+
function htmlGroupTable(groups) {
|
|
642
|
+
return htmlTable(
|
|
643
|
+
[
|
|
644
|
+
'分组',
|
|
645
|
+
'轮次',
|
|
646
|
+
'完成率',
|
|
647
|
+
'审计完整率',
|
|
648
|
+
'工具调用 平均/P50/P90',
|
|
649
|
+
'工具失败率',
|
|
650
|
+
'Token P50/P90',
|
|
651
|
+
'缓存读取率',
|
|
652
|
+
'耗时 P50',
|
|
653
|
+
'子 Agent 轮次',
|
|
654
|
+
],
|
|
655
|
+
groups.map((group) => [
|
|
656
|
+
group.label,
|
|
657
|
+
fmtInteger(group.turns),
|
|
658
|
+
fmtPercent(group.successRate),
|
|
659
|
+
fmtPercent(group.qualityCompleteRate),
|
|
660
|
+
[
|
|
661
|
+
fmtDecimal(group.toolCalls.mean),
|
|
662
|
+
fmtDecimal(group.toolCalls.p50),
|
|
663
|
+
fmtDecimal(group.toolCalls.p90),
|
|
664
|
+
].join(' / '),
|
|
665
|
+
fmtPercent(group.toolCalls.failureRate),
|
|
666
|
+
`${fmtInteger(group.tokens.perTurn.p50)} / ${fmtInteger(group.tokens.perTurn.p90)}`,
|
|
667
|
+
fmtPercent(group.tokens.cacheReadRate),
|
|
668
|
+
fmtDuration(group.durationMs.p50),
|
|
669
|
+
`${fmtInteger(group.subagents.turns)} (${fmtPercent(group.subagents.turnRate)})`,
|
|
670
|
+
]),
|
|
671
|
+
'comparison-table',
|
|
672
|
+
);
|
|
673
|
+
}
|
|
674
|
+
|
|
675
|
+
function htmlOutlierTable(items, formatter) {
|
|
676
|
+
return htmlTable(
|
|
677
|
+
['审计 ID', 'Provider / 模型', '结果', '数值'],
|
|
678
|
+
items.map((item) => [
|
|
679
|
+
item.auditId,
|
|
680
|
+
`${item.provider} / ${item.model}`,
|
|
681
|
+
item.outcome,
|
|
682
|
+
formatter(item.value),
|
|
683
|
+
]),
|
|
684
|
+
);
|
|
685
|
+
}
|
|
686
|
+
|
|
687
|
+
function metricCard(label, value, note = '') {
|
|
688
|
+
return `<article class="metric-card">
|
|
689
|
+
<div class="metric-label">${escapeHtml(label)}</div>
|
|
690
|
+
<div class="metric-value">${escapeHtml(value)}</div>
|
|
691
|
+
${note ? `<div class="metric-note">${escapeHtml(note)}</div>` : ''}
|
|
692
|
+
</article>`;
|
|
693
|
+
}
|
|
694
|
+
|
|
695
|
+
function barChart(groups, valueOf, formatter) {
|
|
696
|
+
const values = groups.map((group) => asNumber(valueOf(group)));
|
|
697
|
+
const max = Math.max(...values, 0);
|
|
698
|
+
return `<div class="bars">${groups.map((group, index) => {
|
|
699
|
+
const value = values[index];
|
|
700
|
+
const width = max > 0 ? Math.max((value / max) * 100, 1.5) : 0;
|
|
701
|
+
return `<div class="bar-row">
|
|
702
|
+
<div class="bar-label">${escapeHtml(group.label)}</div>
|
|
703
|
+
<div class="bar-track"><div class="bar-fill" style="width:${width.toFixed(2)}%"></div></div>
|
|
704
|
+
<div class="bar-value">${escapeHtml(formatter(value))}</div>
|
|
705
|
+
</div>`;
|
|
706
|
+
}).join('')}</div>`;
|
|
707
|
+
}
|
|
708
|
+
|
|
709
|
+
function capabilityCards(groups) {
|
|
710
|
+
return `<div class="capability-grid">${groups.map((group) => `
|
|
711
|
+
<article class="capability-card">
|
|
712
|
+
<div class="capability-head">
|
|
713
|
+
<h3>${escapeHtml(group.label)}</h3>
|
|
714
|
+
<span>${escapeHtml(fmtInteger(group.uniqueToolCount))} 种工具</span>
|
|
715
|
+
</div>
|
|
716
|
+
<p class="eyebrow">主要工具族</p>
|
|
717
|
+
<div class="tags">${group.topToolFamilies.slice(0, 6).map(
|
|
718
|
+
(item) => `<span>${escapeHtml(item.name)} <b>${escapeHtml(fmtInteger(item.count))}</b></span>`,
|
|
719
|
+
).join('')}</div>
|
|
720
|
+
<p class="eyebrow">高频工具</p>
|
|
721
|
+
<ol class="tool-list">${group.topTools.slice(0, 8).map(
|
|
722
|
+
(item) => `<li><span>${escapeHtml(item.name)}</span><b>${escapeHtml(fmtInteger(item.count))}</b></li>`,
|
|
723
|
+
).join('')}</ol>
|
|
724
|
+
</article>`).join('')}</div>`;
|
|
725
|
+
}
|
|
726
|
+
|
|
727
|
+
export function renderHtml(report, context) {
|
|
728
|
+
const overall = report.overall;
|
|
729
|
+
const filterSummary = Object.entries(context.filters ?? {})
|
|
730
|
+
.filter(([, value]) => value)
|
|
731
|
+
.map(([key, value]) => `${key}=${value}`)
|
|
732
|
+
.join(', ') || '无';
|
|
733
|
+
const providerSets = report.matchedInputs.providerSets
|
|
734
|
+
.map((item) => `${item.name}:${item.count}`)
|
|
735
|
+
.join(', ') || '-';
|
|
736
|
+
const qualitySummary = report.dataQuality.qualityStatuses
|
|
737
|
+
.map((item) => `${item.name}:${item.count}`)
|
|
738
|
+
.join(', ') || '-';
|
|
739
|
+
const usageSummary = report.dataQuality.usageCompleteness
|
|
740
|
+
.map((item) => `${item.name}:${item.count}`)
|
|
741
|
+
.join(', ') || '-';
|
|
742
|
+
const rawCaptureSummary = report.dataQuality.rawRequestStatuses
|
|
743
|
+
.map((item) => `${item.name}:${item.count}`)
|
|
744
|
+
.join(', ') || '-';
|
|
745
|
+
|
|
746
|
+
return `<!doctype html>
|
|
747
|
+
<html lang="zh-CN">
|
|
748
|
+
<head>
|
|
749
|
+
<meta charset="utf-8">
|
|
750
|
+
<meta name="viewport" content="width=device-width, initial-scale=1">
|
|
751
|
+
<meta name="color-scheme" content="light">
|
|
752
|
+
<title>Grix Agent 审计分析报告</title>
|
|
753
|
+
<style>
|
|
754
|
+
:root {
|
|
755
|
+
--ink: #15263b;
|
|
756
|
+
--muted: #607086;
|
|
757
|
+
--line: #dbe4ee;
|
|
758
|
+
--paper: #ffffff;
|
|
759
|
+
--wash: #f3f7fb;
|
|
760
|
+
--navy: #153a5b;
|
|
761
|
+
--blue: #2176ae;
|
|
762
|
+
--teal: #1f9d8a;
|
|
763
|
+
--amber: #d98e04;
|
|
764
|
+
--red: #c94b50;
|
|
765
|
+
--shadow: 0 16px 45px rgba(21, 58, 91, .09);
|
|
766
|
+
}
|
|
767
|
+
* { box-sizing: border-box; }
|
|
768
|
+
html { scroll-behavior: smooth; }
|
|
769
|
+
body {
|
|
770
|
+
margin: 0;
|
|
771
|
+
color: var(--ink);
|
|
772
|
+
background:
|
|
773
|
+
radial-gradient(circle at 90% 0, rgba(33,118,174,.10), transparent 30rem),
|
|
774
|
+
var(--wash);
|
|
775
|
+
font-family: Inter, ui-sans-serif, -apple-system, BlinkMacSystemFont, "Segoe UI",
|
|
776
|
+
"PingFang SC", "Hiragino Sans GB", "Microsoft YaHei", sans-serif;
|
|
777
|
+
line-height: 1.55;
|
|
778
|
+
}
|
|
779
|
+
a { color: inherit; }
|
|
780
|
+
.hero {
|
|
781
|
+
color: white;
|
|
782
|
+
background:
|
|
783
|
+
linear-gradient(120deg, rgba(21,58,91,.98), rgba(21,78,104,.96)),
|
|
784
|
+
var(--navy);
|
|
785
|
+
padding: 64px max(24px, calc((100vw - 1240px) / 2));
|
|
786
|
+
position: relative;
|
|
787
|
+
overflow: hidden;
|
|
788
|
+
}
|
|
789
|
+
.hero::after {
|
|
790
|
+
content: "";
|
|
791
|
+
position: absolute;
|
|
792
|
+
width: 440px;
|
|
793
|
+
height: 440px;
|
|
794
|
+
border: 1px solid rgba(255,255,255,.16);
|
|
795
|
+
border-radius: 50%;
|
|
796
|
+
right: -110px;
|
|
797
|
+
top: -250px;
|
|
798
|
+
box-shadow: 0 0 0 70px rgba(255,255,255,.035), 0 0 0 140px rgba(255,255,255,.02);
|
|
799
|
+
}
|
|
800
|
+
.kicker {
|
|
801
|
+
margin: 0 0 12px;
|
|
802
|
+
color: #8de3d2;
|
|
803
|
+
font-size: 13px;
|
|
804
|
+
font-weight: 800;
|
|
805
|
+
letter-spacing: .15em;
|
|
806
|
+
text-transform: uppercase;
|
|
807
|
+
}
|
|
808
|
+
h1 {
|
|
809
|
+
max-width: 780px;
|
|
810
|
+
margin: 0;
|
|
811
|
+
font-size: clamp(36px, 5vw, 66px);
|
|
812
|
+
line-height: 1.05;
|
|
813
|
+
letter-spacing: -.04em;
|
|
814
|
+
}
|
|
815
|
+
.hero-copy {
|
|
816
|
+
max-width: 760px;
|
|
817
|
+
margin: 22px 0 0;
|
|
818
|
+
color: rgba(255,255,255,.78);
|
|
819
|
+
font-size: 17px;
|
|
820
|
+
}
|
|
821
|
+
.hero-meta {
|
|
822
|
+
display: flex;
|
|
823
|
+
flex-wrap: wrap;
|
|
824
|
+
gap: 10px 24px;
|
|
825
|
+
margin-top: 28px;
|
|
826
|
+
font-size: 13px;
|
|
827
|
+
color: rgba(255,255,255,.68);
|
|
828
|
+
}
|
|
829
|
+
.nav {
|
|
830
|
+
position: sticky;
|
|
831
|
+
top: 0;
|
|
832
|
+
z-index: 20;
|
|
833
|
+
display: flex;
|
|
834
|
+
gap: 8px;
|
|
835
|
+
overflow-x: auto;
|
|
836
|
+
padding: 12px max(20px, calc((100vw - 1240px) / 2));
|
|
837
|
+
background: rgba(255,255,255,.92);
|
|
838
|
+
border-bottom: 1px solid var(--line);
|
|
839
|
+
backdrop-filter: blur(16px);
|
|
840
|
+
}
|
|
841
|
+
.nav a {
|
|
842
|
+
flex: none;
|
|
843
|
+
padding: 7px 11px;
|
|
844
|
+
color: var(--muted);
|
|
845
|
+
font-size: 13px;
|
|
846
|
+
font-weight: 700;
|
|
847
|
+
text-decoration: none;
|
|
848
|
+
border-radius: 7px;
|
|
849
|
+
}
|
|
850
|
+
.nav a:hover { color: var(--navy); background: #e9f1f8; }
|
|
851
|
+
main {
|
|
852
|
+
width: min(1240px, calc(100% - 32px));
|
|
853
|
+
margin: 0 auto;
|
|
854
|
+
padding: 42px 0 72px;
|
|
855
|
+
}
|
|
856
|
+
section { margin-top: 52px; scroll-margin-top: 72px; }
|
|
857
|
+
.section-head {
|
|
858
|
+
display: flex;
|
|
859
|
+
align-items: end;
|
|
860
|
+
justify-content: space-between;
|
|
861
|
+
gap: 24px;
|
|
862
|
+
margin-bottom: 18px;
|
|
863
|
+
}
|
|
864
|
+
.section-head h2 {
|
|
865
|
+
margin: 0;
|
|
866
|
+
font-size: clamp(25px, 3vw, 36px);
|
|
867
|
+
line-height: 1.1;
|
|
868
|
+
letter-spacing: -.025em;
|
|
869
|
+
}
|
|
870
|
+
.section-head p { max-width: 620px; margin: 0; color: var(--muted); }
|
|
871
|
+
.metrics {
|
|
872
|
+
display: grid;
|
|
873
|
+
grid-template-columns: repeat(4, minmax(0, 1fr));
|
|
874
|
+
gap: 14px;
|
|
875
|
+
}
|
|
876
|
+
.metric-card, .panel, .capability-card {
|
|
877
|
+
background: var(--paper);
|
|
878
|
+
border: 1px solid var(--line);
|
|
879
|
+
border-radius: 14px;
|
|
880
|
+
box-shadow: var(--shadow);
|
|
881
|
+
}
|
|
882
|
+
.metric-card { padding: 22px; min-height: 142px; }
|
|
883
|
+
.metric-label {
|
|
884
|
+
color: var(--muted);
|
|
885
|
+
font-size: 12px;
|
|
886
|
+
font-weight: 800;
|
|
887
|
+
letter-spacing: .06em;
|
|
888
|
+
text-transform: uppercase;
|
|
889
|
+
}
|
|
890
|
+
.metric-value {
|
|
891
|
+
margin-top: 13px;
|
|
892
|
+
font-size: clamp(27px, 3vw, 39px);
|
|
893
|
+
font-weight: 850;
|
|
894
|
+
line-height: 1;
|
|
895
|
+
letter-spacing: -.035em;
|
|
896
|
+
}
|
|
897
|
+
.metric-note { margin-top: 12px; color: var(--muted); font-size: 12px; }
|
|
898
|
+
.insight {
|
|
899
|
+
display: grid;
|
|
900
|
+
grid-template-columns: 4px 1fr;
|
|
901
|
+
gap: 18px;
|
|
902
|
+
margin-top: 16px;
|
|
903
|
+
padding: 20px 22px;
|
|
904
|
+
background: #eaf5f2;
|
|
905
|
+
border: 1px solid #c9e8df;
|
|
906
|
+
border-radius: 12px;
|
|
907
|
+
}
|
|
908
|
+
.insight::before { content: ""; background: var(--teal); border-radius: 4px; }
|
|
909
|
+
.insight strong { display: block; margin-bottom: 4px; }
|
|
910
|
+
.insight p { margin: 0; color: #41655e; }
|
|
911
|
+
.split {
|
|
912
|
+
display: grid;
|
|
913
|
+
grid-template-columns: 1fr 1fr;
|
|
914
|
+
gap: 16px;
|
|
915
|
+
margin-bottom: 16px;
|
|
916
|
+
}
|
|
917
|
+
.panel { padding: 22px; overflow: hidden; }
|
|
918
|
+
.panel h3 { margin: 0 0 18px; font-size: 17px; }
|
|
919
|
+
.bars { display: grid; gap: 14px; }
|
|
920
|
+
.bar-row {
|
|
921
|
+
display: grid;
|
|
922
|
+
grid-template-columns: minmax(80px, 130px) 1fr 72px;
|
|
923
|
+
align-items: center;
|
|
924
|
+
gap: 12px;
|
|
925
|
+
font-size: 13px;
|
|
926
|
+
}
|
|
927
|
+
.bar-label { overflow: hidden; text-overflow: ellipsis; white-space: nowrap; font-weight: 700; }
|
|
928
|
+
.bar-track { height: 9px; overflow: hidden; background: #e6edf4; border-radius: 99px; }
|
|
929
|
+
.bar-fill {
|
|
930
|
+
height: 100%;
|
|
931
|
+
background: linear-gradient(90deg, var(--blue), var(--teal));
|
|
932
|
+
border-radius: inherit;
|
|
933
|
+
}
|
|
934
|
+
.bar-value { text-align: right; color: var(--muted); font-variant-numeric: tabular-nums; }
|
|
935
|
+
.table-wrap {
|
|
936
|
+
overflow-x: auto;
|
|
937
|
+
background: var(--paper);
|
|
938
|
+
border: 1px solid var(--line);
|
|
939
|
+
border-radius: 12px;
|
|
940
|
+
box-shadow: var(--shadow);
|
|
941
|
+
}
|
|
942
|
+
table { width: 100%; border-collapse: collapse; font-size: 13px; }
|
|
943
|
+
th, td {
|
|
944
|
+
padding: 12px 14px;
|
|
945
|
+
text-align: left;
|
|
946
|
+
white-space: nowrap;
|
|
947
|
+
border-bottom: 1px solid #e7edf3;
|
|
948
|
+
font-variant-numeric: tabular-nums;
|
|
949
|
+
}
|
|
950
|
+
th {
|
|
951
|
+
position: sticky;
|
|
952
|
+
top: 0;
|
|
953
|
+
color: #4f6075;
|
|
954
|
+
background: #f7fafc;
|
|
955
|
+
font-size: 11px;
|
|
956
|
+
letter-spacing: .035em;
|
|
957
|
+
text-transform: uppercase;
|
|
958
|
+
}
|
|
959
|
+
tbody tr:last-child td { border-bottom: 0; }
|
|
960
|
+
tbody tr:hover td { background: #f8fbfd; }
|
|
961
|
+
.capability-grid {
|
|
962
|
+
display: grid;
|
|
963
|
+
grid-template-columns: repeat(2, minmax(0, 1fr));
|
|
964
|
+
gap: 16px;
|
|
965
|
+
}
|
|
966
|
+
.capability-card { padding: 22px; }
|
|
967
|
+
.capability-head {
|
|
968
|
+
display: flex;
|
|
969
|
+
align-items: baseline;
|
|
970
|
+
justify-content: space-between;
|
|
971
|
+
gap: 12px;
|
|
972
|
+
}
|
|
973
|
+
.capability-head h3 { margin: 0; font-size: 24px; text-transform: capitalize; }
|
|
974
|
+
.capability-head span { color: var(--muted); font-size: 12px; }
|
|
975
|
+
.eyebrow {
|
|
976
|
+
margin: 22px 0 8px;
|
|
977
|
+
color: var(--muted);
|
|
978
|
+
font-size: 11px;
|
|
979
|
+
font-weight: 800;
|
|
980
|
+
letter-spacing: .08em;
|
|
981
|
+
text-transform: uppercase;
|
|
982
|
+
}
|
|
983
|
+
.tags { display: flex; flex-wrap: wrap; gap: 7px; }
|
|
984
|
+
.tags span {
|
|
985
|
+
padding: 5px 9px;
|
|
986
|
+
color: #225f57;
|
|
987
|
+
background: #e8f6f3;
|
|
988
|
+
border-radius: 7px;
|
|
989
|
+
font-size: 12px;
|
|
990
|
+
}
|
|
991
|
+
.tags b { margin-left: 5px; }
|
|
992
|
+
.tool-list { margin: 0; padding-left: 24px; }
|
|
993
|
+
.tool-list li { padding: 5px 0; color: var(--muted); }
|
|
994
|
+
.tool-list li span { color: var(--ink); overflow-wrap: anywhere; }
|
|
995
|
+
.tool-list li b { float: right; color: var(--muted); }
|
|
996
|
+
.subsection { margin: 28px 0 12px; font-size: 19px; }
|
|
997
|
+
.quality-list {
|
|
998
|
+
display: grid;
|
|
999
|
+
grid-template-columns: repeat(2, 1fr);
|
|
1000
|
+
gap: 10px 22px;
|
|
1001
|
+
margin: 0;
|
|
1002
|
+
padding: 0;
|
|
1003
|
+
list-style: none;
|
|
1004
|
+
}
|
|
1005
|
+
.quality-list li { padding: 11px 0; border-bottom: 1px solid var(--line); }
|
|
1006
|
+
.quality-list b { display: block; font-size: 12px; color: var(--muted); }
|
|
1007
|
+
.guardrails {
|
|
1008
|
+
margin: 0;
|
|
1009
|
+
padding-left: 20px;
|
|
1010
|
+
color: var(--muted);
|
|
1011
|
+
}
|
|
1012
|
+
.guardrails li { margin: 8px 0; }
|
|
1013
|
+
footer {
|
|
1014
|
+
padding: 28px;
|
|
1015
|
+
color: var(--muted);
|
|
1016
|
+
text-align: center;
|
|
1017
|
+
font-size: 12px;
|
|
1018
|
+
border-top: 1px solid var(--line);
|
|
1019
|
+
}
|
|
1020
|
+
@media (max-width: 900px) {
|
|
1021
|
+
.metrics { grid-template-columns: repeat(2, minmax(0, 1fr)); }
|
|
1022
|
+
.split, .capability-grid { grid-template-columns: 1fr; }
|
|
1023
|
+
.section-head { align-items: start; flex-direction: column; }
|
|
1024
|
+
}
|
|
1025
|
+
@media (max-width: 560px) {
|
|
1026
|
+
.hero { padding-top: 48px; padding-bottom: 48px; }
|
|
1027
|
+
.metrics { grid-template-columns: 1fr; }
|
|
1028
|
+
.quality-list { grid-template-columns: 1fr; }
|
|
1029
|
+
.bar-row { grid-template-columns: 86px 1fr 58px; }
|
|
1030
|
+
main { width: min(100% - 20px, 1240px); }
|
|
1031
|
+
}
|
|
1032
|
+
@media print {
|
|
1033
|
+
.nav { display: none; }
|
|
1034
|
+
body { background: white; }
|
|
1035
|
+
.hero { padding: 36px; print-color-adjust: exact; }
|
|
1036
|
+
main { width: 100%; padding: 20px; }
|
|
1037
|
+
section { break-inside: avoid; }
|
|
1038
|
+
.metric-card, .panel, .capability-card, .table-wrap { box-shadow: none; }
|
|
1039
|
+
}
|
|
1040
|
+
</style>
|
|
1041
|
+
</head>
|
|
1042
|
+
<body>
|
|
1043
|
+
<header class="hero">
|
|
1044
|
+
<p class="kicker">Grix · Audit Intelligence</p>
|
|
1045
|
+
<h1>Agent 审计分析报告</h1>
|
|
1046
|
+
<p class="hero-copy">从能力覆盖、工具行为、成本效率、稳定性和相同输入对照五个视角,对本地审计回放进行元数据分析。</p>
|
|
1047
|
+
<div class="hero-meta">
|
|
1048
|
+
<span>生成时间 ${escapeHtml(context.generatedAt)}</span>
|
|
1049
|
+
<span>数据范围 ${escapeHtml(report.summary.startedAtMin ?? '-')} → ${escapeHtml(report.summary.finalizedAtMax ?? '-')}</span>
|
|
1050
|
+
<span>筛选条件 ${escapeHtml(filterSummary)}</span>
|
|
1051
|
+
</div>
|
|
1052
|
+
</header>
|
|
1053
|
+
<nav class="nav" aria-label="报告导航">
|
|
1054
|
+
<a href="#overview">概览</a>
|
|
1055
|
+
<a href="#providers">Provider</a>
|
|
1056
|
+
<a href="#matched">相同输入</a>
|
|
1057
|
+
<a href="#capabilities">能力覆盖</a>
|
|
1058
|
+
<a href="#cost">成本行为</a>
|
|
1059
|
+
<a href="#outliers">异常轮次</a>
|
|
1060
|
+
<a href="#quality">数据质量</a>
|
|
1061
|
+
</nav>
|
|
1062
|
+
<main>
|
|
1063
|
+
<section id="overview">
|
|
1064
|
+
<div class="section-head">
|
|
1065
|
+
<h2>审计概览</h2>
|
|
1066
|
+
<p>仅使用 replay JSON 元数据和 span 名称;不读取 prompt、response 或 Blob 正文。</p>
|
|
1067
|
+
</div>
|
|
1068
|
+
<div class="metrics">
|
|
1069
|
+
${metricCard('最新审计轮次', fmtInteger(report.summary.latestAuditCount), `${fmtInteger(report.summary.sessionCount)} 个会话`)}
|
|
1070
|
+
${metricCard('完成率', fmtPercent(overall.successRate), `${fmtInteger(overall.outcomes.completed)} 完成 · ${fmtInteger(overall.outcomes.failed)} 失败 · ${fmtInteger(overall.outcomes.cancelled)} 取消`)}
|
|
1071
|
+
${metricCard('工具调用', fmtInteger(overall.toolCalls.total), `P50 ${fmtDecimal(overall.toolCalls.p50)} · P90 ${fmtDecimal(overall.toolCalls.p90)}`)}
|
|
1072
|
+
${metricCard('总处理 Token', fmtInteger(overall.tokens.total), `缓存读取率 ${fmtPercent(overall.tokens.cacheReadRate)}`)}
|
|
1073
|
+
${metricCard('耗时 P50', fmtDuration(overall.durationMs.p50), `P90 ${fmtDuration(overall.durationMs.p90)}`)}
|
|
1074
|
+
${metricCard('完整审计', fmtPercent(overall.qualityCompleteRate), `${fmtInteger(overall.qualityCompleteTurns)} 条完整`)}
|
|
1075
|
+
${metricCard('子 Agent 调用', fmtInteger(overall.subagents.total), `分布在 ${fmtInteger(overall.subagents.turns)} 轮`)}
|
|
1076
|
+
${metricCard('失败/取消 Token', fmtPercent(overall.tokens.nonCompletedShare), `${fmtInteger(overall.tokens.nonCompleted)} tokens`)}
|
|
1077
|
+
</div>
|
|
1078
|
+
<div class="insight">
|
|
1079
|
+
<div>
|
|
1080
|
+
<strong>成本首先集中在少数长链路任务。</strong>
|
|
1081
|
+
<p>Token 最高的前 5 轮占 ${escapeHtml(fmtPercent(report.costConcentration.top5TurnsShare))},前 10 轮占 ${escapeHtml(fmtPercent(report.costConcentration.top10TurnsShare))}。工具调用与 Token 的相关系数为 ${escapeHtml(fmtDecimal(report.correlations.toolCallsVsTotalTokens, 3))}。</p>
|
|
1082
|
+
</div>
|
|
1083
|
+
</div>
|
|
1084
|
+
</section>
|
|
1085
|
+
|
|
1086
|
+
<section id="providers">
|
|
1087
|
+
<div class="section-head">
|
|
1088
|
+
<h2>Provider 与模型对比</h2>
|
|
1089
|
+
<p>展示观测样本,不把不同任务组合下的完成率直接解释为绝对能力排名。</p>
|
|
1090
|
+
</div>
|
|
1091
|
+
<div class="split">
|
|
1092
|
+
<article class="panel">
|
|
1093
|
+
<h3>完成率</h3>
|
|
1094
|
+
${barChart(report.byProvider, (group) => group.successRate, fmtPercent)}
|
|
1095
|
+
</article>
|
|
1096
|
+
<article class="panel">
|
|
1097
|
+
<h3>单轮 Token P50</h3>
|
|
1098
|
+
${barChart(report.byProvider, (group) => group.tokens.perTurn.p50, fmtInteger)}
|
|
1099
|
+
</article>
|
|
1100
|
+
</div>
|
|
1101
|
+
${htmlGroupTable(report.byProvider)}
|
|
1102
|
+
<h3 class="subsection">Provider / 主模型</h3>
|
|
1103
|
+
${htmlGroupTable(report.byProviderModel)}
|
|
1104
|
+
</section>
|
|
1105
|
+
|
|
1106
|
+
<section id="matched">
|
|
1107
|
+
<div class="section-head">
|
|
1108
|
+
<h2>相同输入对照</h2>
|
|
1109
|
+
<p>使用 input contentId 匹配完全相同的输入,不读取输入内容。它能减少任务组合偏差,但无法消除会话上下文和环境状态差异。</p>
|
|
1110
|
+
</div>
|
|
1111
|
+
<div class="metrics">
|
|
1112
|
+
${metricCard('跨 Provider 输入组', fmtInteger(report.matchedInputs.cohortCount), `${fmtInteger(report.matchedInputs.turnCount)} 个匹配轮次`)}
|
|
1113
|
+
${metricCard('Provider 组合', fmtInteger(report.matchedInputs.providerSets.length), providerSets)}
|
|
1114
|
+
</div>
|
|
1115
|
+
<div style="height:16px"></div>
|
|
1116
|
+
${report.matchedInputs.byProvider.length > 0
|
|
1117
|
+
? htmlGroupTable(report.matchedInputs.byProvider)
|
|
1118
|
+
: '<div class="panel">没有发现跨 Provider 的相同输入组。</div>'}
|
|
1119
|
+
</section>
|
|
1120
|
+
|
|
1121
|
+
<section id="capabilities">
|
|
1122
|
+
<div class="section-head">
|
|
1123
|
+
<h2>能力覆盖与工具画像</h2>
|
|
1124
|
+
<p>工具别名经过跨 Provider 归一化;工具族是基于名称的保守分类,用于观察行为差异。</p>
|
|
1125
|
+
</div>
|
|
1126
|
+
${capabilityCards(report.byProvider)}
|
|
1127
|
+
</section>
|
|
1128
|
+
|
|
1129
|
+
<section id="cost">
|
|
1130
|
+
<div class="section-head">
|
|
1131
|
+
<h2>成本、行为与相关性</h2>
|
|
1132
|
+
<p>高相关性不等于因果关系,但可以帮助定位最值得优化的长链路、重复调用和上下文膨胀。</p>
|
|
1133
|
+
</div>
|
|
1134
|
+
<div class="metrics">
|
|
1135
|
+
${metricCard('输入 Token', fmtInteger(overall.tokens.input))}
|
|
1136
|
+
${metricCard('输出 Token', fmtInteger(overall.tokens.output), `推理占输出 ${fmtPercent(overall.tokens.reasoningShare)}`)}
|
|
1137
|
+
${metricCard('每完成轮 Token', fmtInteger(overall.tokens.perCompletedTurn), '包含失败与取消产生的成本')}
|
|
1138
|
+
${metricCard('每工具调用 Token', fmtInteger(overall.tokens.perToolCall))}
|
|
1139
|
+
${metricCard('LLM 请求', fmtInteger(overall.llmRequests.total), `每轮平均 ${fmtDecimal(overall.llmRequests.mean)}`)}
|
|
1140
|
+
${metricCard('压缩次数', fmtInteger(overall.compactions.total), `分布在 ${fmtInteger(overall.compactions.turns)} 轮`)}
|
|
1141
|
+
${metricCard('Top 1 轮成本占比', fmtPercent(report.costConcentration.top1TurnShare))}
|
|
1142
|
+
${metricCard('Top 10 轮成本占比', fmtPercent(report.costConcentration.top10TurnsShare))}
|
|
1143
|
+
</div>
|
|
1144
|
+
<h3 class="subsection">变量相关性</h3>
|
|
1145
|
+
${htmlTable(
|
|
1146
|
+
['变量组合', 'Pearson r', '解读'],
|
|
1147
|
+
[
|
|
1148
|
+
['工具调用 vs 总 Token', fmtDecimal(report.correlations.toolCallsVsTotalTokens, 3), '工具链越长,通常上下文与请求成本越高'],
|
|
1149
|
+
['LLM 请求 vs 总 Token', fmtDecimal(report.correlations.llmRequestsVsTotalTokens, 3), '多轮模型请求是 Token 成本的强信号'],
|
|
1150
|
+
['工具调用 vs 耗时', fmtDecimal(report.correlations.toolCallsVsDuration, 3), '工具密集型任务通常耗时更长'],
|
|
1151
|
+
],
|
|
1152
|
+
)}
|
|
1153
|
+
</section>
|
|
1154
|
+
|
|
1155
|
+
<section id="outliers">
|
|
1156
|
+
<div class="section-head">
|
|
1157
|
+
<h2>异常与高成本轮次</h2>
|
|
1158
|
+
<p>用于进一步按 auditId 回放定位;报告不包含对话内容。</p>
|
|
1159
|
+
</div>
|
|
1160
|
+
<h3 class="subsection">Token 最高</h3>
|
|
1161
|
+
${htmlOutlierTable(report.outliers.totalTokens, fmtInteger)}
|
|
1162
|
+
<h3 class="subsection">工具调用最多</h3>
|
|
1163
|
+
${htmlOutlierTable(report.outliers.toolCalls, fmtInteger)}
|
|
1164
|
+
<h3 class="subsection">耗时最长</h3>
|
|
1165
|
+
${htmlOutlierTable(report.outliers.durationMs, fmtDuration)}
|
|
1166
|
+
</section>
|
|
1167
|
+
|
|
1168
|
+
<section id="quality">
|
|
1169
|
+
<div class="section-head">
|
|
1170
|
+
<h2>数据质量与解释边界</h2>
|
|
1171
|
+
<p>任何能力判断都应同时查看样本量、捕获完整性和任务构成。</p>
|
|
1172
|
+
</div>
|
|
1173
|
+
<div class="panel">
|
|
1174
|
+
<ul class="quality-list">
|
|
1175
|
+
<li><b>JSON 解析错误</b>${escapeHtml(fmtInteger(report.dataQuality.parseErrors.length))}</li>
|
|
1176
|
+
<li><b>Usage 完整性</b>${escapeHtml(usageSummary)}</li>
|
|
1177
|
+
<li><b>审计完整性</b>${escapeHtml(qualitySummary)}</li>
|
|
1178
|
+
<li><b>原始请求捕获</b>${escapeHtml(rawCaptureSummary)}</li>
|
|
1179
|
+
</ul>
|
|
1180
|
+
</div>
|
|
1181
|
+
<div class="panel" style="margin-top:16px">
|
|
1182
|
+
<ul class="guardrails">
|
|
1183
|
+
<li>Provider/模型分组是观测队列,任务难度差异可能主导结果。</li>
|
|
1184
|
+
<li>相同输入组仍可能具有不同的会话上下文、代码状态和外部环境。</li>
|
|
1185
|
+
<li>“完成”表示审计 outcome 为 completed,不代表独立验证了答案正确性。</li>
|
|
1186
|
+
<li>工具别名经过归一化,工具族为启发式分类。</li>
|
|
1187
|
+
<li>分析器只读取 replay JSON 元数据,不读取 Blob 内容。</li>
|
|
1188
|
+
</ul>
|
|
1189
|
+
</div>
|
|
1190
|
+
</section>
|
|
1191
|
+
</main>
|
|
1192
|
+
<footer>Grix Agent Audit Analysis · ${escapeHtml(context.generatedAt)}</footer>
|
|
1193
|
+
</body>
|
|
1194
|
+
</html>
|
|
1195
|
+
`;
|
|
1196
|
+
}
|
|
1197
|
+
|
|
1198
|
+
export function renderMarkdown(report, context) {
|
|
1199
|
+
const overall = report.overall;
|
|
1200
|
+
const top = context.top;
|
|
1201
|
+
const lines = [
|
|
1202
|
+
'# Audit replay analysis',
|
|
1203
|
+
'',
|
|
1204
|
+
`Generated: ${context.generatedAt}`,
|
|
1205
|
+
'',
|
|
1206
|
+
`Root: \`${context.root}\``,
|
|
1207
|
+
'',
|
|
1208
|
+
'## Scope',
|
|
1209
|
+
'',
|
|
1210
|
+
`- Latest audits: ${fmtInteger(report.summary.latestAuditCount)} from ${fmtInteger(report.summary.sessionCount)} sessions`,
|
|
1211
|
+
`- Time range: ${report.summary.startedAtMin ?? '-'} → ${report.summary.finalizedAtMax ?? '-'}`,
|
|
1212
|
+
`- Outcomes: ${fmtInteger(overall.outcomes.completed)} completed, ${fmtInteger(overall.outcomes.failed)} failed, ${fmtInteger(overall.outcomes.cancelled)} cancelled`,
|
|
1213
|
+
`- Complete audit quality: ${fmtInteger(overall.qualityCompleteTurns)} (${fmtPercent(overall.qualityCompleteRate)})`,
|
|
1214
|
+
'',
|
|
1215
|
+
'## Provider comparison',
|
|
1216
|
+
'',
|
|
1217
|
+
groupTable(report.byProvider),
|
|
1218
|
+
'',
|
|
1219
|
+
'## Provider / primary model comparison',
|
|
1220
|
+
'',
|
|
1221
|
+
groupTable(report.byProviderModel),
|
|
1222
|
+
'',
|
|
1223
|
+
'## Matched exact-input cohorts',
|
|
1224
|
+
'',
|
|
1225
|
+
`Found ${fmtInteger(report.matchedInputs.cohortCount)} exact-input cohorts with ${fmtInteger(report.matchedInputs.turnCount)} turns spanning at least two providers.`,
|
|
1226
|
+
'',
|
|
1227
|
+
report.matchedInputs.byProvider.length > 0
|
|
1228
|
+
? groupTable(report.matchedInputs.byProvider)
|
|
1229
|
+
: 'No cross-provider exact-input cohorts were found.',
|
|
1230
|
+
'',
|
|
1231
|
+
`Provider sets: ${report.matchedInputs.providerSets.map((item) => `${item.name}:${item.count}`).join(', ') || '-'}`,
|
|
1232
|
+
'',
|
|
1233
|
+
'## Capability footprint',
|
|
1234
|
+
'',
|
|
1235
|
+
table(
|
|
1236
|
+
['Provider', 'Unique tools', 'Top tool families', 'Top canonical tools'],
|
|
1237
|
+
report.byProvider.map((group) => [
|
|
1238
|
+
group.label,
|
|
1239
|
+
fmtInteger(group.uniqueToolCount),
|
|
1240
|
+
group.topToolFamilies.slice(0, 5).map((item) => `${item.name}:${item.count}`).join(', '),
|
|
1241
|
+
group.topTools.slice(0, 8).map((item) => `${item.name}:${item.count}`).join(', '),
|
|
1242
|
+
]),
|
|
1243
|
+
),
|
|
1244
|
+
'',
|
|
1245
|
+
'## Overall cost and behavior',
|
|
1246
|
+
'',
|
|
1247
|
+
table(
|
|
1248
|
+
['Metric', 'Value'],
|
|
1249
|
+
[
|
|
1250
|
+
['Tool calls', fmtInteger(overall.toolCalls.total)],
|
|
1251
|
+
['Tool calls per turn (avg / p50 / p90)', `${fmtDecimal(overall.toolCalls.mean)} / ${fmtDecimal(overall.toolCalls.p50)} / ${fmtDecimal(overall.toolCalls.p90)}`],
|
|
1252
|
+
['LLM requests', fmtInteger(overall.llmRequests.total)],
|
|
1253
|
+
['Subagent calls / turns', `${fmtInteger(overall.subagents.total)} / ${fmtInteger(overall.subagents.turns)}`],
|
|
1254
|
+
['Compactions / turns', `${fmtInteger(overall.compactions.total)} / ${fmtInteger(overall.compactions.turns)}`],
|
|
1255
|
+
['Input / output / total tokens', `${fmtInteger(overall.tokens.input)} / ${fmtInteger(overall.tokens.output)} / ${fmtInteger(overall.tokens.total)}`],
|
|
1256
|
+
['Tokens per completed turn', fmtInteger(overall.tokens.perCompletedTurn)],
|
|
1257
|
+
['Tokens per tool call', fmtInteger(overall.tokens.perToolCall)],
|
|
1258
|
+
['Tokens on failed/cancelled turns', `${fmtInteger(overall.tokens.nonCompleted)} (${fmtPercent(overall.tokens.nonCompletedShare)})`],
|
|
1259
|
+
['Top 1 / 5 / 10 turns share of tokens', `${fmtPercent(report.costConcentration.top1TurnShare)} / ${fmtPercent(report.costConcentration.top5TurnsShare)} / ${fmtPercent(report.costConcentration.top10TurnsShare)}`],
|
|
1260
|
+
['Cache read rate', fmtPercent(overall.tokens.cacheReadRate)],
|
|
1261
|
+
['Reasoning share of output', fmtPercent(overall.tokens.reasoningShare)],
|
|
1262
|
+
['Duration p50 / p90', `${fmtDuration(overall.durationMs.p50)} / ${fmtDuration(overall.durationMs.p90)}`],
|
|
1263
|
+
],
|
|
1264
|
+
),
|
|
1265
|
+
'',
|
|
1266
|
+
'## Relationships',
|
|
1267
|
+
'',
|
|
1268
|
+
table(
|
|
1269
|
+
['Pair', 'Pearson r'],
|
|
1270
|
+
[
|
|
1271
|
+
['Tool calls vs total tokens', fmtDecimal(report.correlations.toolCallsVsTotalTokens, 3)],
|
|
1272
|
+
['LLM requests vs total tokens', fmtDecimal(report.correlations.llmRequestsVsTotalTokens, 3)],
|
|
1273
|
+
['Tool calls vs duration', fmtDecimal(report.correlations.toolCallsVsDuration, 3)],
|
|
1274
|
+
],
|
|
1275
|
+
),
|
|
1276
|
+
'',
|
|
1277
|
+
`## Top ${top} outliers`,
|
|
1278
|
+
'',
|
|
1279
|
+
'### Token usage',
|
|
1280
|
+
'',
|
|
1281
|
+
outlierTable(report.outliers.totalTokens, fmtInteger),
|
|
1282
|
+
'',
|
|
1283
|
+
'### Tool calls',
|
|
1284
|
+
'',
|
|
1285
|
+
outlierTable(report.outliers.toolCalls, fmtInteger),
|
|
1286
|
+
'',
|
|
1287
|
+
'### Duration',
|
|
1288
|
+
'',
|
|
1289
|
+
outlierTable(report.outliers.durationMs, fmtDuration),
|
|
1290
|
+
'',
|
|
1291
|
+
'## Data quality',
|
|
1292
|
+
'',
|
|
1293
|
+
`- Parse errors: ${fmtInteger(report.dataQuality.parseErrors.length)}`,
|
|
1294
|
+
`- Usage completeness: ${report.dataQuality.usageCompleteness.map((item) => `${item.name}:${item.count}`).join(', ') || '-'}`,
|
|
1295
|
+
`- Audit quality: ${report.dataQuality.qualityStatuses.map((item) => `${item.name}:${item.count}`).join(', ') || '-'}`,
|
|
1296
|
+
`- Raw request capture: ${report.dataQuality.rawRequestStatuses.map((item) => `${item.name}:${item.count}`).join(', ') || '-'}`,
|
|
1297
|
+
'',
|
|
1298
|
+
'## Interpretation guardrails',
|
|
1299
|
+
'',
|
|
1300
|
+
'- Provider/model groups are observational cohorts. Different task mixes can dominate the comparison.',
|
|
1301
|
+
'- Exact-input cohorts reduce prompt-mix bias, but session context and environment state can still differ.',
|
|
1302
|
+
'- Success means the recorded turn outcome is `completed`; it does not independently judge answer correctness.',
|
|
1303
|
+
'- Tool aliases are canonicalized for cross-provider comparison, while tool families are heuristic.',
|
|
1304
|
+
'- The script reads replay JSON metadata only. It does not read blob content or include prompt/response text.',
|
|
1305
|
+
'',
|
|
1306
|
+
];
|
|
1307
|
+
return lines.join('\n');
|
|
1308
|
+
}
|
|
1309
|
+
|
|
1310
|
+
function help() {
|
|
1311
|
+
return `Usage:
|
|
1312
|
+
node scripts/analyze-audit-replays.mjs [options]
|
|
1313
|
+
|
|
1314
|
+
Options:
|
|
1315
|
+
--root <path> Audit storage root (default: ~/.grix/data/audit-replay)
|
|
1316
|
+
--format <value> html, markdown, or json (default: markdown)
|
|
1317
|
+
--output <path> Write the report to a file instead of stdout
|
|
1318
|
+
--provider <name> Filter by provider
|
|
1319
|
+
--since <date> Include turns starting on/after an ISO date
|
|
1320
|
+
--until <date> Include turns starting on/before an ISO date
|
|
1321
|
+
--top <number> Number of outliers and top items (default: ${DEFAULT_TOP})
|
|
1322
|
+
--help Show this help
|
|
1323
|
+
|
|
1324
|
+
The analyzer reads replay JSON metadata only and never reads audit blob content.
|
|
1325
|
+
`;
|
|
1326
|
+
}
|
|
1327
|
+
|
|
1328
|
+
function parseArgs(argv) {
|
|
1329
|
+
const options = {
|
|
1330
|
+
root: process.env.GRIX_AUDIT_ROOT || DEFAULT_ROOT,
|
|
1331
|
+
format: 'markdown',
|
|
1332
|
+
output: null,
|
|
1333
|
+
provider: null,
|
|
1334
|
+
since: null,
|
|
1335
|
+
until: null,
|
|
1336
|
+
top: DEFAULT_TOP,
|
|
1337
|
+
help: false,
|
|
1338
|
+
};
|
|
1339
|
+
for (let index = 0; index < argv.length; index += 1) {
|
|
1340
|
+
const argument = argv[index];
|
|
1341
|
+
if (argument === '--help' || argument === '-h') {
|
|
1342
|
+
options.help = true;
|
|
1343
|
+
continue;
|
|
1344
|
+
}
|
|
1345
|
+
if (!argument.startsWith('--')) throw new Error(`Unknown argument: ${argument}`);
|
|
1346
|
+
const key = argument.slice(2);
|
|
1347
|
+
if (!['root', 'format', 'output', 'provider', 'since', 'until', 'top'].includes(key)) {
|
|
1348
|
+
throw new Error(`Unknown option: ${argument}`);
|
|
1349
|
+
}
|
|
1350
|
+
const value = argv[index + 1];
|
|
1351
|
+
if (!value || value.startsWith('--')) throw new Error(`Missing value for ${argument}`);
|
|
1352
|
+
index += 1;
|
|
1353
|
+
if (key === 'top') {
|
|
1354
|
+
const top = Number.parseInt(value, 10);
|
|
1355
|
+
if (!Number.isInteger(top) || top < 1 || top > 100) {
|
|
1356
|
+
throw new Error('--top must be an integer from 1 to 100');
|
|
1357
|
+
}
|
|
1358
|
+
options.top = top;
|
|
1359
|
+
} else {
|
|
1360
|
+
options[key] = value;
|
|
1361
|
+
}
|
|
1362
|
+
}
|
|
1363
|
+
if (!['html', 'markdown', 'json'].includes(options.format)) {
|
|
1364
|
+
throw new Error('--format must be html, markdown, or json');
|
|
1365
|
+
}
|
|
1366
|
+
options.root = resolve(options.root);
|
|
1367
|
+
if (options.output) options.output = resolve(options.output);
|
|
1368
|
+
return options;
|
|
1369
|
+
}
|
|
1370
|
+
|
|
1371
|
+
async function main() {
|
|
1372
|
+
const options = parseArgs(process.argv.slice(2));
|
|
1373
|
+
if (options.help) {
|
|
1374
|
+
process.stdout.write(help());
|
|
1375
|
+
return;
|
|
1376
|
+
}
|
|
1377
|
+
const { replays, parseErrors, files } = await loadReplays(options.root);
|
|
1378
|
+
const filtered = filterReplays(replays, options);
|
|
1379
|
+
const generatedAt = new Date().toISOString();
|
|
1380
|
+
const report = analyzeReplays(filtered, {
|
|
1381
|
+
top: options.top,
|
|
1382
|
+
parseErrors,
|
|
1383
|
+
});
|
|
1384
|
+
const context = {
|
|
1385
|
+
generatedAt,
|
|
1386
|
+
root: options.root,
|
|
1387
|
+
top: options.top,
|
|
1388
|
+
sourceFiles: files.length,
|
|
1389
|
+
filters: {
|
|
1390
|
+
provider: options.provider,
|
|
1391
|
+
since: options.since,
|
|
1392
|
+
until: options.until,
|
|
1393
|
+
},
|
|
1394
|
+
};
|
|
1395
|
+
const output = options.format === 'json'
|
|
1396
|
+
? `${JSON.stringify({ generatedAt, root: options.root, filters: context.filters, ...report }, null, 2)}\n`
|
|
1397
|
+
: options.format === 'html'
|
|
1398
|
+
? renderHtml(report, context)
|
|
1399
|
+
: renderMarkdown(report, context);
|
|
1400
|
+
if (options.output) {
|
|
1401
|
+
await writeFile(options.output, output, 'utf8');
|
|
1402
|
+
process.stdout.write(`Wrote ${options.format} report to ${options.output}\n`);
|
|
1403
|
+
} else {
|
|
1404
|
+
process.stdout.write(output);
|
|
1405
|
+
}
|
|
1406
|
+
}
|
|
1407
|
+
|
|
1408
|
+
const isMain = process.argv[1]
|
|
1409
|
+
&& resolve(process.argv[1]) === resolve(fileURLToPath(import.meta.url));
|
|
1410
|
+
|
|
1411
|
+
if (isMain) {
|
|
1412
|
+
main().catch((error) => {
|
|
1413
|
+
process.stderr.write(`audit analysis failed: ${error instanceof Error ? error.message : String(error)}\n`);
|
|
1414
|
+
process.exitCode = 1;
|
|
1415
|
+
});
|
|
1416
|
+
}
|