akm-cli 0.9.27 → 0.9.28-alpha.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +65 -0
- package/STABILITY.md +4 -0
- package/dist/assets/hints/cli-hints-full.md +3 -0
- package/dist/assets/hints/cli-hints-short.md +1 -0
- package/dist/assets/prompts/consolidate-pair.md +1 -1
- package/dist/assets/templates/html/metrics.html +977 -0
- package/dist/cli/shared.js +5 -4
- package/dist/cli.js +11 -1
- package/dist/commands/health/accept-rate.js +8 -4
- package/dist/commands/health/html-report.js +3 -8
- package/dist/commands/health/llm-usage.js +17 -6
- package/dist/commands/health/renderers.js +4 -4
- package/dist/commands/improve/improve-report.js +4 -2
- package/dist/commands/improve/improve.js +22 -10
- package/dist/commands/improve/loop-stages.js +8 -1
- package/dist/commands/improve/preparation.js +16 -0
- package/dist/commands/improve/stage.js +1 -1
- package/dist/commands/metrics/collect.js +439 -0
- package/dist/commands/metrics/html-report.js +82 -0
- package/dist/commands/metrics/md-report.js +44 -0
- package/dist/commands/metrics/metrics-cli.js +213 -0
- package/dist/commands/metrics/report-view.js +243 -0
- package/dist/commands/metrics/types.js +4 -0
- package/dist/commands/read/search.js +5 -0
- package/dist/indexer/indexer.js +64 -14
- package/dist/indexer/usage/usage-events.js +3 -1
- package/dist/integrations/session-logs/pre-filter.js +1 -0
- package/dist/llm/usage-persist.js +22 -11
- package/dist/llm/usage-telemetry.js +4 -0
- package/dist/output/html-render.js +15 -8
- package/dist/output/shapes/passthrough.js +1 -0
- package/dist/output/text/metrics.js +39 -0
- package/dist/output/text.js +2 -0
- package/dist/scripts/akm-migrate-node.js +281 -8
- package/dist/scripts/akm-migrate.js +281 -8
- package/dist/storage/repositories/index-utility-repository.js +24 -0
- package/dist/storage/repositories/metrics-repository.js +80 -0
- package/docs/reference/cli.md +63 -4
- package/docs/reference/data-and-telemetry.md +41 -8
- package/package.json +1 -1
|
@@ -27777,6 +27777,243 @@ registerMdRenderer("health", (result) => {
|
|
|
27777
27777
|
return null;
|
|
27778
27778
|
});
|
|
27779
27779
|
|
|
27780
|
+
// src/commands/metrics/report-view.ts
|
|
27781
|
+
var BRIEF_LIST_LIMIT = 5;
|
|
27782
|
+
function isMetricsResult(value) {
|
|
27783
|
+
if (value === null || typeof value !== "object")
|
|
27784
|
+
return false;
|
|
27785
|
+
const r = value;
|
|
27786
|
+
return r.schemaVersion === 1 && !!r.window && !!r.usage && !!r.feedback && !!r.llm;
|
|
27787
|
+
}
|
|
27788
|
+
var int = (n) => String(n);
|
|
27789
|
+
var dash = (v) => v === undefined || v === "" ? "-" : v;
|
|
27790
|
+
var pct = (n) => n === null ? "n/a" : `${(n * 100).toFixed(1)}%`;
|
|
27791
|
+
var ms = (n) => n === null ? "n/a" : `${Math.round(n)} ms`;
|
|
27792
|
+
var fixed = (n, digits) => n === null ? "n/a" : n.toFixed(digits);
|
|
27793
|
+
function sourceFacts(counts) {
|
|
27794
|
+
const entries = Object.entries(counts);
|
|
27795
|
+
return entries.length === 0 ? "-" : entries.map(([k, v]) => `${k}=${v}`).join(" ");
|
|
27796
|
+
}
|
|
27797
|
+
function llmRows(group) {
|
|
27798
|
+
return Object.entries(group).map(([name, a]) => [
|
|
27799
|
+
name,
|
|
27800
|
+
int(a.calls),
|
|
27801
|
+
int(a.failures),
|
|
27802
|
+
int(a.promptTokens),
|
|
27803
|
+
int(a.completionTokens),
|
|
27804
|
+
int(a.totalTokens),
|
|
27805
|
+
ms(a.totalDurationMs)
|
|
27806
|
+
]);
|
|
27807
|
+
}
|
|
27808
|
+
var LLM_HEADERS = ["name", "calls", "failed", "prompt", "completion", "total", "time"];
|
|
27809
|
+
function buildMetricsView(r, detail) {
|
|
27810
|
+
const brief = detail === "brief";
|
|
27811
|
+
const cut = (items) => brief ? items.slice(0, BRIEF_LIST_LIMIT) : items;
|
|
27812
|
+
const { usage, feedback, utility, llm } = r;
|
|
27813
|
+
const searches = usage.totals.searches;
|
|
27814
|
+
const sections = [
|
|
27815
|
+
{
|
|
27816
|
+
title: "Usage",
|
|
27817
|
+
facts: [
|
|
27818
|
+
["searches", int(searches)],
|
|
27819
|
+
["shows", int(usage.totals.shows)],
|
|
27820
|
+
["curates", int(usage.totals.curates)],
|
|
27821
|
+
["selects", int(usage.totals.selects)],
|
|
27822
|
+
["select rate", pct(usage.selectRate)],
|
|
27823
|
+
[
|
|
27824
|
+
"zero-result searches",
|
|
27825
|
+
`${usage.totals.zeroResultSearches} (${pct(searches === 0 ? null : usage.totals.zeroResultSearches / searches)})`
|
|
27826
|
+
],
|
|
27827
|
+
["search median", ms(usage.searchMedianMs)],
|
|
27828
|
+
["distinct assets", int(usage.totals.distinctAssets)],
|
|
27829
|
+
["distinct queries", int(usage.totals.distinctQueries)],
|
|
27830
|
+
["by source", sourceFacts(usage.bySource)]
|
|
27831
|
+
],
|
|
27832
|
+
tables: [
|
|
27833
|
+
{
|
|
27834
|
+
title: "Top assets",
|
|
27835
|
+
headers: ["ref", "shows", "search hits", "selects", "+", "-", "last used"],
|
|
27836
|
+
rows: cut(usage.topAssets).map((a) => [
|
|
27837
|
+
a.ref,
|
|
27838
|
+
int(a.shows),
|
|
27839
|
+
int(a.searchHits),
|
|
27840
|
+
int(a.selects),
|
|
27841
|
+
int(a.positive),
|
|
27842
|
+
int(a.negative),
|
|
27843
|
+
dash(a.lastUsedAt)
|
|
27844
|
+
])
|
|
27845
|
+
},
|
|
27846
|
+
{
|
|
27847
|
+
title: "Top queries",
|
|
27848
|
+
headers: ["query", "count", "avg results", "last"],
|
|
27849
|
+
rows: cut(usage.topQueries).map((q) => [q.query, int(q.count), fixed(q.avgResults, 1), q.lastAt])
|
|
27850
|
+
},
|
|
27851
|
+
{
|
|
27852
|
+
title: "Zero-result queries",
|
|
27853
|
+
headers: ["query", "count", "last"],
|
|
27854
|
+
rows: cut(usage.zeroResultQueries).map((q) => [q.query, int(q.count), q.lastAt])
|
|
27855
|
+
},
|
|
27856
|
+
{
|
|
27857
|
+
title: "Daily",
|
|
27858
|
+
headers: ["day", "search", "show", "curate", "feedback"],
|
|
27859
|
+
rows: brief ? [] : usage.daily.map((d) => [d.day, int(d.search), int(d.show), int(d.curate), int(d.feedback)]),
|
|
27860
|
+
hideWhenEmpty: true
|
|
27861
|
+
}
|
|
27862
|
+
]
|
|
27863
|
+
},
|
|
27864
|
+
{
|
|
27865
|
+
title: "Feedback",
|
|
27866
|
+
facts: [
|
|
27867
|
+
["positive", int(feedback.totals.positive)],
|
|
27868
|
+
["negative", int(feedback.totals.negative)]
|
|
27869
|
+
],
|
|
27870
|
+
tables: [
|
|
27871
|
+
{
|
|
27872
|
+
title: "By asset",
|
|
27873
|
+
headers: ["ref", "+", "-", "valence", "last"],
|
|
27874
|
+
rows: cut(feedback.byAsset).map((a) => [
|
|
27875
|
+
a.ref,
|
|
27876
|
+
int(a.positive),
|
|
27877
|
+
int(a.negative),
|
|
27878
|
+
fixed(a.valence, 2),
|
|
27879
|
+
a.lastAt
|
|
27880
|
+
])
|
|
27881
|
+
},
|
|
27882
|
+
{
|
|
27883
|
+
title: "By tag",
|
|
27884
|
+
headers: ["tag", "+", "-"],
|
|
27885
|
+
rows: cut(Object.entries(feedback.byTag)).map(([tag, c]) => [tag, int(c.positive), int(c.negative)]),
|
|
27886
|
+
hideWhenEmpty: true
|
|
27887
|
+
},
|
|
27888
|
+
{
|
|
27889
|
+
title: "Recent negative",
|
|
27890
|
+
headers: ["ref", "at", "reason", "tags"],
|
|
27891
|
+
rows: cut(feedback.recentNegative).map((n) => [n.ref, n.at, dash(n.reason), dash(n.tags?.join(", "))])
|
|
27892
|
+
}
|
|
27893
|
+
]
|
|
27894
|
+
},
|
|
27895
|
+
{
|
|
27896
|
+
title: "Utility",
|
|
27897
|
+
facts: [
|
|
27898
|
+
["scored assets", int(utility.count)],
|
|
27899
|
+
["never used", int(utility.neverUsed)]
|
|
27900
|
+
],
|
|
27901
|
+
tables: [
|
|
27902
|
+
{
|
|
27903
|
+
title: "Histogram",
|
|
27904
|
+
headers: ["bucket", "count"],
|
|
27905
|
+
rows: utility.histogram.map((h) => [h.bucket, int(h.count)]),
|
|
27906
|
+
hideWhenEmpty: true
|
|
27907
|
+
},
|
|
27908
|
+
{ title: "Lowest", headers: utilityHeaders(), rows: cut(utility.lowest).map(utilityRow) },
|
|
27909
|
+
{ title: "Highest", headers: utilityHeaders(), rows: cut(utility.highest).map(utilityRow) },
|
|
27910
|
+
{
|
|
27911
|
+
title: "Lowest outcome",
|
|
27912
|
+
headers: ["ref", "outcome", "retrievals", "negative", "accepted changes"],
|
|
27913
|
+
rows: cut(r.outcomes.lowestOutcome).map((o) => [
|
|
27914
|
+
o.ref,
|
|
27915
|
+
fixed(o.outcomeScore, 2),
|
|
27916
|
+
int(o.retrievalCount),
|
|
27917
|
+
int(o.negativeFeedbackCount),
|
|
27918
|
+
int(o.acceptedChangeCount)
|
|
27919
|
+
]),
|
|
27920
|
+
hideWhenEmpty: true
|
|
27921
|
+
}
|
|
27922
|
+
]
|
|
27923
|
+
},
|
|
27924
|
+
{
|
|
27925
|
+
title: "LLM",
|
|
27926
|
+
facts: [
|
|
27927
|
+
["calls", int(llm.calls)],
|
|
27928
|
+
["failed", int(llm.failures)],
|
|
27929
|
+
["prompt tokens", int(llm.promptTokens)],
|
|
27930
|
+
["completion tokens", int(llm.completionTokens)],
|
|
27931
|
+
["reasoning tokens", int(llm.reasoningTokens)],
|
|
27932
|
+
["total tokens", int(llm.totalTokens)],
|
|
27933
|
+
["time", ms(llm.totalDurationMs)]
|
|
27934
|
+
],
|
|
27935
|
+
tables: [
|
|
27936
|
+
{ title: "By engine", headers: LLM_HEADERS, rows: cut(llmRows(llm.byEngine)), hideWhenEmpty: true },
|
|
27937
|
+
{ title: "By process", headers: LLM_HEADERS, rows: cut(llmRows(llm.byProcess)), hideWhenEmpty: true },
|
|
27938
|
+
{ title: "By stage", headers: LLM_HEADERS, rows: cut(llmRows(llm.byStage)), hideWhenEmpty: true }
|
|
27939
|
+
]
|
|
27940
|
+
},
|
|
27941
|
+
{
|
|
27942
|
+
title: "Index",
|
|
27943
|
+
facts: [
|
|
27944
|
+
["runs", int(r.index.runs)],
|
|
27945
|
+
["median time", ms(r.index.medianMs)]
|
|
27946
|
+
],
|
|
27947
|
+
tables: [
|
|
27948
|
+
{
|
|
27949
|
+
title: "Recent runs",
|
|
27950
|
+
headers: ["at", "mode", "time"],
|
|
27951
|
+
rows: brief ? [] : r.index.recent.map((i) => [i.at, i.mode, ms(i.totalMs)]),
|
|
27952
|
+
hideWhenEmpty: true
|
|
27953
|
+
}
|
|
27954
|
+
]
|
|
27955
|
+
},
|
|
27956
|
+
{
|
|
27957
|
+
title: "Tasks",
|
|
27958
|
+
facts: [
|
|
27959
|
+
["runs", int(r.tasks.runs)],
|
|
27960
|
+
["failed", int(r.tasks.failed)],
|
|
27961
|
+
["fail rate", pct(r.tasks.failRate)]
|
|
27962
|
+
],
|
|
27963
|
+
tables: [
|
|
27964
|
+
{
|
|
27965
|
+
title: "By task",
|
|
27966
|
+
headers: ["task", "runs", "failed", "median"],
|
|
27967
|
+
rows: cut(r.tasks.byTask).map((t) => [t.taskId, int(t.runs), int(t.failed), ms(t.medianMs)]),
|
|
27968
|
+
hideWhenEmpty: true
|
|
27969
|
+
}
|
|
27970
|
+
]
|
|
27971
|
+
},
|
|
27972
|
+
{
|
|
27973
|
+
title: "Proposals",
|
|
27974
|
+
facts: [["by status", sourceFacts(r.proposals.byStatus)]],
|
|
27975
|
+
tables: [
|
|
27976
|
+
{
|
|
27977
|
+
title: "Accept rate by source",
|
|
27978
|
+
headers: ["source", "total", "accepted", "rejected", "pending", "accept rate"],
|
|
27979
|
+
rows: r.proposals.acceptRateBySource.map((p) => [
|
|
27980
|
+
p.source,
|
|
27981
|
+
int(p.total),
|
|
27982
|
+
int(p.accepted),
|
|
27983
|
+
int(p.rejected),
|
|
27984
|
+
int(p.pending),
|
|
27985
|
+
pct(p.acceptRate)
|
|
27986
|
+
]),
|
|
27987
|
+
hideWhenEmpty: true
|
|
27988
|
+
}
|
|
27989
|
+
]
|
|
27990
|
+
},
|
|
27991
|
+
{
|
|
27992
|
+
title: "Workflows",
|
|
27993
|
+
facts: [
|
|
27994
|
+
["runs", int(r.workflows.runs)],
|
|
27995
|
+
["by status", sourceFacts(r.workflows.byStatus)],
|
|
27996
|
+
["tokens", int(r.workflows.tokens)],
|
|
27997
|
+
["tokens by model", sourceFacts(r.workflows.byModel)]
|
|
27998
|
+
],
|
|
27999
|
+
tables: []
|
|
28000
|
+
}
|
|
28001
|
+
];
|
|
28002
|
+
const f = r.filters;
|
|
28003
|
+
return {
|
|
28004
|
+
window: `${r.window.since} to ${r.window.until}`,
|
|
28005
|
+
filters: `source=${f.source}`,
|
|
28006
|
+
sections,
|
|
28007
|
+
notes: r.notes
|
|
28008
|
+
};
|
|
28009
|
+
}
|
|
28010
|
+
function utilityHeaders() {
|
|
28011
|
+
return ["ref", "utility", "shows", "searches", "select rate", "last used"];
|
|
28012
|
+
}
|
|
28013
|
+
function utilityRow(u) {
|
|
28014
|
+
return [u.ref, fixed(u.utility, 2), int(u.showCount), int(u.searchCount), pct(u.selectRate), dash(u.lastUsedAt)];
|
|
28015
|
+
}
|
|
28016
|
+
|
|
27780
28017
|
// src/output/generic-render.ts
|
|
27781
28018
|
var ENVELOPE_META_KEYS = new Set(["shape", "schemaVersion"]);
|
|
27782
28019
|
function flattenForText(value, path11, lines) {
|
|
@@ -28377,6 +28614,7 @@ var PASSTHROUGH_COMMANDS = [
|
|
|
28377
28614
|
"info",
|
|
28378
28615
|
"lint",
|
|
28379
28616
|
"list",
|
|
28617
|
+
"metrics",
|
|
28380
28618
|
"models",
|
|
28381
28619
|
"proposal-accept-batch",
|
|
28382
28620
|
"proposal-drain",
|
|
@@ -29651,10 +29889,10 @@ function formatHealthPlain(r, detail) {
|
|
|
29651
29889
|
`).trim();
|
|
29652
29890
|
}
|
|
29653
29891
|
// src/output/text/lint-format.ts
|
|
29654
|
-
function glyphFor(
|
|
29655
|
-
if (
|
|
29892
|
+
function glyphFor(fixed2) {
|
|
29893
|
+
if (fixed2 === "failed")
|
|
29656
29894
|
return { glyph: "\u2717", severityRank: 0 };
|
|
29657
|
-
if (
|
|
29895
|
+
if (fixed2 === true)
|
|
29658
29896
|
return { glyph: "\u2713", severityRank: 2 };
|
|
29659
29897
|
return { glyph: "\u26A0", severityRank: 1 };
|
|
29660
29898
|
}
|
|
@@ -29673,17 +29911,17 @@ function renderIssueSection(title, issues) {
|
|
|
29673
29911
|
function formatLintPlain(r) {
|
|
29674
29912
|
if (r === null || typeof r !== "object")
|
|
29675
29913
|
return null;
|
|
29676
|
-
const
|
|
29914
|
+
const fixed2 = Array.isArray(r.fixed) ? r.fixed : [];
|
|
29677
29915
|
const flagged = Array.isArray(r.flagged) ? r.flagged : [];
|
|
29678
29916
|
const warnings = Array.isArray(r.warnings) ? r.warnings : [];
|
|
29679
29917
|
const summary = r.summary;
|
|
29680
29918
|
const lines = [];
|
|
29681
29919
|
if (typeof r.ok === "boolean")
|
|
29682
29920
|
lines.push(`ok: ${r.ok}`);
|
|
29683
|
-
lines.push(`summary: fixed=${summary?.fixed ??
|
|
29921
|
+
lines.push(`summary: fixed=${summary?.fixed ?? fixed2.length} flagged=${summary?.flagged ?? flagged.length}` + ` warnings=${summary?.warnings ?? warnings.length}`);
|
|
29684
29922
|
lines.push("", ...renderIssueSection("flagged", flagged));
|
|
29685
29923
|
lines.push("", ...renderIssueSection("warnings", warnings));
|
|
29686
|
-
lines.push("", ...renderIssueSection("fixed",
|
|
29924
|
+
lines.push("", ...renderIssueSection("fixed", fixed2));
|
|
29687
29925
|
return lines.join(`
|
|
29688
29926
|
`).trim();
|
|
29689
29927
|
}
|
|
@@ -36144,8 +36382,8 @@ function getConfigLockPath() {
|
|
|
36144
36382
|
}
|
|
36145
36383
|
var CONFIG_LOCK_MAX_RETRIES = 40;
|
|
36146
36384
|
var CONFIG_LOCK_RETRY_DELAY_MS = 50;
|
|
36147
|
-
function sleepSyncMs(
|
|
36148
|
-
sleepSync(
|
|
36385
|
+
function sleepSyncMs(ms2) {
|
|
36386
|
+
sleepSync(ms2);
|
|
36149
36387
|
}
|
|
36150
36388
|
function acquireConfigLock() {
|
|
36151
36389
|
const lockPath = getConfigLockPath();
|
|
@@ -53170,6 +53408,40 @@ var lintFormatters = [{ command: "lint", handler: (r) => formatLintPlain(r) }];
|
|
|
53170
53408
|
// src/output/text/list.ts
|
|
53171
53409
|
var listFormatters = [{ command: "list", handler: (r) => formatListPlain(r) }];
|
|
53172
53410
|
|
|
53411
|
+
// src/output/text/metrics.ts
|
|
53412
|
+
var flat = (s) => s.replace(/\s*\r?\n\s*/g, " ");
|
|
53413
|
+
function renderTable2(table) {
|
|
53414
|
+
const all = [table.headers, ...table.rows].map((r) => r.map(flat));
|
|
53415
|
+
const widths = table.headers.map((_, i) => Math.max(...all.map((r) => (r[i] ?? "").length)));
|
|
53416
|
+
const line = (r) => ` ${r.map((c, i) => c.padEnd(widths[i] ?? 0)).join(" ")}`.trimEnd();
|
|
53417
|
+
return [`${table.title}:`, ...all.map(line)];
|
|
53418
|
+
}
|
|
53419
|
+
function formatMetricsPlain(result, detail) {
|
|
53420
|
+
if (!isMetricsResult(result))
|
|
53421
|
+
return null;
|
|
53422
|
+
const view = buildMetricsView(result, detail);
|
|
53423
|
+
const lines = [`akm metrics ${view.window}`, `filters: ${view.filters}`];
|
|
53424
|
+
for (const section of view.sections) {
|
|
53425
|
+
lines.push("", section.title.toUpperCase());
|
|
53426
|
+
const labelWidth = Math.max(0, ...section.facts.map(([label]) => label.length));
|
|
53427
|
+
for (const [label, value] of section.facts)
|
|
53428
|
+
lines.push(` ${label.padEnd(labelWidth)} ${value}`);
|
|
53429
|
+
for (const table of section.tables) {
|
|
53430
|
+
if (table.rows.length > 0)
|
|
53431
|
+
lines.push("", ...renderTable2(table));
|
|
53432
|
+
else if (!table.hideWhenEmpty)
|
|
53433
|
+
lines.push("", `${table.title}: (none)`);
|
|
53434
|
+
}
|
|
53435
|
+
}
|
|
53436
|
+
if (view.notes.length > 0)
|
|
53437
|
+
lines.push("", "NOTES", ...view.notes.map((n) => ` - ${flat(n)}`));
|
|
53438
|
+
return lines.join(`
|
|
53439
|
+
`);
|
|
53440
|
+
}
|
|
53441
|
+
var metricsFormatters = [
|
|
53442
|
+
{ command: "metrics", handler: (r, detail) => formatMetricsPlain(r, detail) }
|
|
53443
|
+
];
|
|
53444
|
+
|
|
53173
53445
|
// src/output/text/migrate.ts
|
|
53174
53446
|
function planGlyph(status) {
|
|
53175
53447
|
switch (status) {
|
|
@@ -53304,6 +53576,7 @@ var BUILT_IN_TEXT_FORMATTERS = [
|
|
|
53304
53576
|
...proposalProducerFormatters,
|
|
53305
53577
|
...infoFormatters,
|
|
53306
53578
|
...healthFormatters,
|
|
53579
|
+
...metricsFormatters,
|
|
53307
53580
|
...improveReportFormatters,
|
|
53308
53581
|
...lintFormatters,
|
|
53309
53582
|
...configFormatters,
|
|
@@ -271,3 +271,27 @@ export function applyFeedbackToUtilityScore(db, entryId, positiveCount, negative
|
|
|
271
271
|
`).run(entryId, result.nextUtility, now, now, result.nextUtility, now);
|
|
272
272
|
return result;
|
|
273
273
|
}
|
|
274
|
+
/**
|
|
275
|
+
* Every indexed entry with its `utility_scores` row (when it has one), for
|
|
276
|
+
* `akm metrics`. `utility_scores` is keyed by the unstable `entry_id`, so the
|
|
277
|
+
* join to `entries` is what yields a durable ref.
|
|
278
|
+
*/
|
|
279
|
+
export function listUtilityWithRefs(db) {
|
|
280
|
+
const rows = db
|
|
281
|
+
.prepare(`SELECT e.item_ref AS ref, u.utility, u.show_count, u.search_count, u.select_rate, u.last_used_at
|
|
282
|
+
FROM entries e LEFT JOIN utility_scores u ON u.entry_id = e.id
|
|
283
|
+
ORDER BY e.item_ref`)
|
|
284
|
+
.all();
|
|
285
|
+
return rows.map((row) => row.utility === null
|
|
286
|
+
? { ref: row.ref }
|
|
287
|
+
: {
|
|
288
|
+
ref: row.ref,
|
|
289
|
+
score: {
|
|
290
|
+
utility: row.utility,
|
|
291
|
+
showCount: row.show_count ?? 0,
|
|
292
|
+
searchCount: row.search_count ?? 0,
|
|
293
|
+
selectRate: row.select_rate ?? 0,
|
|
294
|
+
lastUsedAt: row.last_used_at ?? undefined,
|
|
295
|
+
},
|
|
296
|
+
});
|
|
297
|
+
}
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/** Usage rows (search / show / curate / feedback) in the window, oldest first. */
|
|
5
|
+
export function listUsageEventRows(db, filter) {
|
|
6
|
+
const sourceSql = filter.source === undefined ? "" : " AND source = ?";
|
|
7
|
+
const sourceParams = filter.source === undefined ? [] : [filter.source];
|
|
8
|
+
// `created_at` has whole-second resolution, so a row written during the
|
|
9
|
+
// bound's own second (e.g. `--until` defaulting to now) is still inside the
|
|
10
|
+
// half-open window: round a fractional bound up to the next second.
|
|
11
|
+
const untilMs = Date.parse(filter.untilIso);
|
|
12
|
+
const untilBound = new Date(Math.ceil(untilMs / 1000) * 1000).toISOString();
|
|
13
|
+
return db
|
|
14
|
+
.prepare(`SELECT id, event_type, query, entry_id, entry_ref, signal, metadata, source, created_at
|
|
15
|
+
FROM usage_events
|
|
16
|
+
WHERE datetime(created_at) >= datetime(?) AND datetime(created_at) < datetime(?)${sourceSql}
|
|
17
|
+
ORDER BY datetime(created_at), id`)
|
|
18
|
+
.all(filter.sinceIso, untilBound, ...sourceParams);
|
|
19
|
+
}
|
|
20
|
+
/**
|
|
21
|
+
* `select` events (a show within 60 s of a search that returned the ref) in the
|
|
22
|
+
* window. The events stream stores the ref as the user typed it (often
|
|
23
|
+
* bundle-less), so the caller resolves it to the durable ref.
|
|
24
|
+
*/
|
|
25
|
+
export function listSelectEvents(db, sinceIso, untilIso) {
|
|
26
|
+
return db
|
|
27
|
+
.prepare(`SELECT ts, ref FROM events
|
|
28
|
+
WHERE event_type = 'select' AND ref IS NOT NULL AND ts >= ? AND ts < ?
|
|
29
|
+
ORDER BY id`)
|
|
30
|
+
.all(sinceIso, untilIso);
|
|
31
|
+
}
|
|
32
|
+
/** `index_completed` events (one per `akm index` run) in the window, oldest first. */
|
|
33
|
+
export function listIndexCompletedEvents(db, sinceIso, untilIso) {
|
|
34
|
+
return db
|
|
35
|
+
.prepare(`SELECT ts, metadata_json FROM events
|
|
36
|
+
WHERE event_type = 'index_completed' AND ts >= ? AND ts < ?
|
|
37
|
+
ORDER BY id`)
|
|
38
|
+
.all(sinceIso, untilIso);
|
|
39
|
+
}
|
|
40
|
+
/** The `limit` assets with the lowest `outcome_score`, ties broken by ref. */
|
|
41
|
+
export function listLowestOutcomeAssets(db, limit) {
|
|
42
|
+
return db
|
|
43
|
+
.prepare(`SELECT asset_ref, last_retrieved_at, retrieval_count, expected_retrieval_rate,
|
|
44
|
+
negative_feedback_count, accepted_change_count, outcome_score, updated_at
|
|
45
|
+
FROM asset_outcome
|
|
46
|
+
ORDER BY outcome_score ASC, asset_ref ASC LIMIT ?`)
|
|
47
|
+
.all(limit);
|
|
48
|
+
}
|
|
49
|
+
/** Proposal counts by status, for proposals last updated in the window. */
|
|
50
|
+
export function countProposalsByStatus(db, sinceIso, untilIso) {
|
|
51
|
+
const rows = db
|
|
52
|
+
.prepare(`SELECT status, COUNT(*) AS n FROM proposals
|
|
53
|
+
WHERE updated_at >= ? AND updated_at < ?
|
|
54
|
+
GROUP BY status ORDER BY status`)
|
|
55
|
+
.all(sinceIso, untilIso);
|
|
56
|
+
return Object.fromEntries(rows.map((row) => [row.status, row.n]));
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* Workflow runs created in the window by status, and the tokens their unit
|
|
60
|
+
* attempts spent by model. Attempts, not the unit projection, so a retried
|
|
61
|
+
* unit keeps the tokens of every attempt.
|
|
62
|
+
*/
|
|
63
|
+
export function summarizeWorkflowRuns(db, sinceIso, untilIso) {
|
|
64
|
+
const statusRows = db
|
|
65
|
+
.prepare(`SELECT status, COUNT(*) AS n FROM workflow_runs
|
|
66
|
+
WHERE created_at >= ? AND created_at < ? GROUP BY status ORDER BY status`)
|
|
67
|
+
.all(sinceIso, untilIso);
|
|
68
|
+
const modelRows = db
|
|
69
|
+
.prepare(`SELECT COALESCE(model, 'unattributed') AS model, SUM(tokens) AS tokens
|
|
70
|
+
FROM workflow_run_unit_attempts
|
|
71
|
+
WHERE tokens IS NOT NULL AND started_at >= ? AND started_at < ?
|
|
72
|
+
GROUP BY COALESCE(model, 'unattributed') ORDER BY model`)
|
|
73
|
+
.all(sinceIso, untilIso);
|
|
74
|
+
return {
|
|
75
|
+
runs: statusRows.reduce((sum, row) => sum + row.n, 0),
|
|
76
|
+
byStatus: Object.fromEntries(statusRows.map((row) => [row.status, row.n])),
|
|
77
|
+
tokens: modelRows.reduce((sum, row) => sum + row.tokens, 0),
|
|
78
|
+
byModel: Object.fromEntries(modelRows.map((row) => [row.model, row.tokens])),
|
|
79
|
+
};
|
|
80
|
+
}
|
package/docs/reference/cli.md
CHANGED
|
@@ -42,7 +42,8 @@ Useful for streaming consumption by scripts or agents.
|
|
|
42
42
|
A command may register a renderer for a document format when it has something
|
|
43
43
|
better to say than the generic one: `akm health --group-by run --format md`
|
|
44
44
|
emits its per-run table, and `akm health --report --format html` renders the
|
|
45
|
-
full report with KPI cards, charts, and advisories.
|
|
45
|
+
full report with KPI cards, charts, and advisories. `akm metrics` always carries
|
|
46
|
+
its window rows under `--format html`. The renderers are
|
|
46
47
|
data-driven — they fire when the result carries the report dataset, never on
|
|
47
48
|
the format alone, so the same dataset is available as JSON too. Every other command falls back to a
|
|
48
49
|
generic rendering derived from its own envelope — headings for the top-level
|
|
@@ -439,6 +440,64 @@ for an infrastructure reason (`llm_unavailable`, `read_failed`, `exception`,
|
|
|
439
440
|
`locked_concurrent`) — naming the reason and, when recorded, the engine — and
|
|
440
441
|
`pass` otherwise, with per-outcome counts.
|
|
441
442
|
|
|
443
|
+
### metrics
|
|
444
|
+
|
|
445
|
+
**Experimental** (see [STABILITY.md](../../STABILITY.md)): the report, its JSON shape and the
|
|
446
|
+
dashboard may change in any release.
|
|
447
|
+
|
|
448
|
+
Report what akm has recorded locally: asset usage (search, show, curate),
|
|
449
|
+
feedback, derived utility, LLM tokens, task runs, proposal
|
|
450
|
+
flow, workflow token spend, and index runs. Read-only, and every number comes
|
|
451
|
+
from `state.db` or `index.db`; nothing is collected that was not already
|
|
452
|
+
recorded.
|
|
453
|
+
|
|
454
|
+
```sh
|
|
455
|
+
akm metrics
|
|
456
|
+
akm metrics --since 7d
|
|
457
|
+
akm metrics --since 2026-05-01 --format yaml
|
|
458
|
+
akm metrics --format html --output metrics.html
|
|
459
|
+
```
|
|
460
|
+
|
|
461
|
+
| Flag | Description |
|
|
462
|
+
| --- | --- |
|
|
463
|
+
| `--since` | Window start. Accepts ISO 8601, `YYYY-MM-DD`, epoch milliseconds, or shorthand like `24h` / `7d`. Default: `30d`. |
|
|
464
|
+
|
|
465
|
+
The window always ends now. Only `user`-source usage rows are counted (matching
|
|
466
|
+
utility and retrieval counts), and every ranked list holds its top 20. For a
|
|
467
|
+
narrower view, open the HTML dashboard and filter by date, bundle, source and
|
|
468
|
+
event type in the browser.
|
|
469
|
+
|
|
470
|
+
Result sections (`schemaVersion: 1`):
|
|
471
|
+
|
|
472
|
+
| Field | Description |
|
|
473
|
+
| --- | --- |
|
|
474
|
+
| `window`, `filters` | The resolved window and the usage source counted |
|
|
475
|
+
| `usage` | Search, show, curate and select totals, `selectRate` (selects over searches that returned a hit), `searchMedianMs`, a daily series, top assets, top and zero-result queries, and rows by source |
|
|
476
|
+
| `feedback` | Positive and negative totals, per-asset valence, per-tag counts, and the most recent negatives with their reasons |
|
|
477
|
+
| `utility` | A ten-bucket histogram of `utility_scores`, the lowest and highest assets, and how many indexed entries were never used |
|
|
478
|
+
| `outcomes` | The assets with the lowest `outcome_score` |
|
|
479
|
+
| `llm` | `akm health`'s LLM usage aggregate (by stage, process and engine) |
|
|
480
|
+
| `index`, `tasks`, `proposals`, `workflows` | Index runs and median time, task runs and fail rate, proposals by status and accept rate by source, workflow runs and tokens by model |
|
|
481
|
+
| `rows` | The raw usage and LLM rows of the window; see below |
|
|
482
|
+
| `notes` | Anything that limits what the numbers mean |
|
|
483
|
+
|
|
484
|
+
A rate whose denominator is 0 is `null`, never `NaN`. A missing `state.db`
|
|
485
|
+
gives an empty report and a missing `index.db` an empty `utility` section, each
|
|
486
|
+
with a `notes` entry, and the exit code stays 0.
|
|
487
|
+
|
|
488
|
+
`rows` is present with `--format html` (the dashboard re-aggregates from it in
|
|
489
|
+
the browser) and with `--detail full` in any other format; it is absent
|
|
490
|
+
otherwise.
|
|
491
|
+
|
|
492
|
+
Notes you may see:
|
|
493
|
+
|
|
494
|
+
- A `--since` older than what a store keeps names the store, its retention
|
|
495
|
+
(`usage_events` keeps 90 days; `events`, which holds selects, LLM calls and
|
|
496
|
+
index runs, keeps `improve.eventRetentionDays`, default 90) and where its data
|
|
497
|
+
effectively starts, so a long window is never silently shorter than asked.
|
|
498
|
+
- Selects are read from the events stream, which records no source, so they are
|
|
499
|
+
counted whatever the usage source is.
|
|
500
|
+
|
|
442
501
|
### search
|
|
443
502
|
|
|
444
503
|
Search bundle assets, registries, or both.
|
|
@@ -2655,8 +2714,8 @@ precedent as the retired `canary` scope).
|
|
|
2655
2714
|
Every real (non-dry-run) `akm improve` invocation persists a `usageReport`
|
|
2656
2715
|
field on the result (`result_json` in `improve_runs`, and in the
|
|
2657
2716
|
`--json-to-stdout` / dry-run JSON): `{ byProcessEngineModel, noCalls }`.
|
|
2658
|
-
`byProcessEngineModel` is a cross-tab of this run's own
|
|
2659
|
-
(#576) — one row per distinct `(process, engine, model)` triple, each with
|
|
2717
|
+
`byProcessEngineModel` is a cross-tab of the LLM call records this run's own
|
|
2718
|
+
usage sink collects (#576) — one row per distinct `(process, engine, model)` triple, each with
|
|
2660
2719
|
`calls`, `failures`, `promptTokens`, `completionTokens`, `totalTokens`,
|
|
2661
2720
|
`reasoningTokens`, and `totalDurationMs`. `noCalls` lists every model-calling
|
|
2662
2721
|
process (`reflect`, `distill`, `consolidate`, `memoryInference`,
|
|
@@ -2678,7 +2737,7 @@ every real run in the window (`byProcessEngineModel` rows merged by
|
|
|
2678
2737
|
`(process, engine, model)`; `noCalls` lists a process only if it made zero
|
|
2679
2738
|
calls across every included run). A run recorded before 0.9.15 has no
|
|
2680
2739
|
persisted `usageReport` — the command recomputes `byProcessEngineModel` from
|
|
2681
|
-
that run's
|
|
2740
|
+
that run's stored `llm_usage` events (`summarizeLlmUsageCrossTab`) instead of erroring, sets `noCalls` to `[]`
|
|
2682
2741
|
(eligibility reasons are not reconstructable after the fact), and adds a
|
|
2683
2742
|
`notes` entry saying so rather than fabricating precision the old row can't
|
|
2684
2743
|
support.
|
|
@@ -162,6 +162,7 @@ the set of types the code actually emits at HEAD (verified against every
|
|
|
162
162
|
| `feedback` | `akm feedback <ref>` | `signal` (positive/negative), `reason`, `tags`, `fix` (`source`, the number of replacements and, for `--superseded-by` or `--outdated`, the `beliefState` the proposal leaves and the `supersededBy` ref, when a fix was attached), `contentHash` (sha256 of the asset's body, without its frontmatter, as it stood when the feedback was given: it lets reflect mark feedback given on an earlier version of the text, and the loop's distill pass tell that a memory flagged wrong still has it; left out for an env or secret file and when the file cannot be read) |
|
|
163
163
|
| `sync` | `akm sync` | `name`, `message`, `ok` |
|
|
164
164
|
| `index_db_vacuumed` | `akm index` VACUUMed index.db, after an index layout migration or because more than half its pages were free | `pagesBefore`, `pagesAfter`, `freelistRatioBefore` |
|
|
165
|
+
| `index_completed` | `akm index`, once when a run finishes | `mode`, `totalMs`, `walkMs`, `llmMs`, `embedMs`, `ftsMs`, `finalizeMs` (phase timings in milliseconds) |
|
|
165
166
|
| `stash_synced` | `akm improve`'s internal auto-sync pass (the `sync.push` feature), **distinct from** the `akm sync` command above | `committed`, `pushed`, `skipped`, `reason`, `attributed` (paths the run wrote and staged), `unattributed` (in-scope paths that went dirty during the run without the run writing them — left for their author) |
|
|
166
167
|
| `env_access` | `akm env run <name> -- <command>` (audit trail: key **names** only, values never recorded) | `ref`, `keys` |
|
|
167
168
|
| `secret_access` | `akm secret run <ref> <VAR> -- <command>` (audit trail: var **name** only, value never recorded) | `ref`, `var` |
|
|
@@ -188,7 +189,7 @@ the set of types the code actually emits at HEAD (verified against every
|
|
|
188
189
|
| `improve_invoked` | Start of an `akm improve` run | `ref` (scope); `strategy`, `scope`, `dryRun`, `eligibleCount` |
|
|
189
190
|
| `improve_completed` | `akm improve` run finished | run stats |
|
|
190
191
|
| `improve_failed` | `akm improve` run errored | error |
|
|
191
|
-
| `improve_skipped` | `akm improve` left a ref, a lane, or a group of refs out | `reason` (`no_new_signal`, `not_retrieved`, `distill_no_new_signal`, `budget_exhausted`, `budget_exhausted_batch`, `asset_missing_on_disk`, `strategy_filtered_all_passes`, `autonomy_gated`, `engine_unavailable`, `pool_below_min_size`, `consolidation_no_memory_updates`, `below_min_new_sessions`, `derived_memory_reflect_skipped`, `memory_distill_requires_feedback`, `distill_flagged_wrong`, `distill_positive_without_reason`); `count`, `remaining`, `strategy`, `lane` or `configKey` where they apply |
|
|
192
|
+
| `improve_skipped` | `akm improve` left a ref, a lane, or a group of refs out | `reason` (`no_new_signal`, `not_retrieved`, `distill_no_new_signal`, `budget_exhausted`, `budget_exhausted_batch`, `asset_missing_on_disk`, `strategy_filtered_all_passes`, `autonomy_gated`, `engine_unavailable`, `pool_below_min_size`, `consolidation_no_memory_updates`, `below_min_new_sessions`, `derived_memory_reflect_skipped`, `memory_distill_requires_feedback`, `distill_flagged_wrong`, `distill_positive_without_reason`, `distill_deprecated_or_superseded`); `count`, `remaining`, `strategy`, `lane` or `configKey` where they apply |
|
|
192
193
|
| `improve_lock_recovered` | Stale improve lock cleared at startup | |
|
|
193
194
|
| `improve_review_needed` | `akm feedback` pushed a high-utility asset's utility below the review threshold — a review-needed escalation is recorded (not a proposal, so it can't accidentally overwrite the asset) | `ref`, `previousUtility`, `nextUtility` |
|
|
194
195
|
| `reflect_invoked` | Start of reflect phase in `akm improve` | `ref`, engine |
|
|
@@ -222,17 +223,20 @@ the set of types the code actually emits at HEAD (verified against every
|
|
|
222
223
|
|
|
223
224
|
| Event type | When emitted | Key metadata fields |
|
|
224
225
|
|---|---|---|
|
|
225
|
-
| `llm_usage` | Per-attempt LLM call usage telemetry (#576) | model provenance, terminal outcome, duration, optional token usage |
|
|
226
|
-
| `llm_usage_summary` | The owning LLM telemetry sink's terminal-record count marker | `expectedTerminalRecords` |
|
|
226
|
+
| `llm_usage` | Per-attempt LLM call usage telemetry (#576), written by every akm command that makes an LLM call (`index`, `curate`, workflows, agent dispatch, `command run`, `improve`, `proposal drain`) | model provenance, terminal outcome, duration, optional token usage |
|
|
227
|
+
| `llm_usage_summary` | The owning LLM telemetry sink's terminal-record count marker. The process-wide sink that covers commands other than `improve` writes none when it saw no call | `expectedTerminalRecords` |
|
|
227
228
|
| `health_probe` | `akm health`'s state.db round-trip write/read probe. **Not durably retained**: the row is inserted then deleted within the same connection once the round trip is confirmed, so the net effect on the `events` table is always zero rows | n/a (ephemeral) |
|
|
228
229
|
|
|
229
230
|
`llm_usage` rows also carry `process`/`engine`/`stage` (each optional; a call
|
|
230
231
|
made outside any attributed scope carries none of them). `akm improve`
|
|
231
|
-
(#944)
|
|
232
|
-
|
|
233
|
-
— persisted on the run result as
|
|
234
|
-
queryable per-run or aggregated with
|
|
235
|
-
`docs/reference/cli.md`'s `#### improve report`
|
|
232
|
+
(#944) builds a process x engine x model cross-tab from the LLM call records
|
|
233
|
+
the run's own usage sink collects — `summarizeLlmUsageRecordsCrossTab` in
|
|
234
|
+
`src/commands/health/llm-usage.ts` — persisted on the run result as
|
|
235
|
+
`usageReport.byProcessEngineModel` and queryable per-run or aggregated with
|
|
236
|
+
`akm improve report`; see `docs/reference/cli.md`'s `#### improve report`
|
|
237
|
+
section. `summarizeLlmUsageCrossTab` is the events form of the same
|
|
238
|
+
aggregation, used by `improve report` to recompute the cross-tab from stored
|
|
239
|
+
`llm_usage` events when a run has no persisted `usageReport`.
|
|
236
240
|
|
|
237
241
|
### 2. Usage Events Table
|
|
238
242
|
|
|
@@ -248,6 +252,12 @@ configured endpoint.
|
|
|
248
252
|
Successful `search`, `curate`, and `show` commands always record usage. Machine
|
|
249
253
|
reads are stamped by source (below), so they never skew ranking or eval.
|
|
250
254
|
|
|
255
|
+
The `search` summary row (the one with no `entry_ref`) carries `resultCount`,
|
|
256
|
+
`stashHitCount`, `registryHitCount`, `resolvedCount` and `mode` in its metadata,
|
|
257
|
+
plus latency for `akm metrics`: `totalMs` for the whole search, and `rankMs` and
|
|
258
|
+
`embedMs` for the ranking and query-embedding phases when the local search
|
|
259
|
+
reported them (a registry-only search has `totalMs` alone).
|
|
260
|
+
|
|
251
261
|
Every runtime writer stamps provenance as `user`, `improve`, `task`, `audit`, or
|
|
252
262
|
`unknown`. Direct interactive CLI traffic defaults to `user`; internal improve,
|
|
253
263
|
scheduled-task, and eval subprocesses preserve their stamp across nested
|
|
@@ -279,6 +289,19 @@ higher-priority-wins `(type, entry.name)` dedup across sources: attribution
|
|
|
279
289
|
source-qualifies every indexed row but does not invent a lower-priority row for
|
|
280
290
|
an identity that production indexing omitted.
|
|
281
291
|
|
|
292
|
+
**Retention:** usage events older than 90 days are purged on every `akm index`.
|
|
293
|
+
`created_at` is `YYYY-MM-DD HH:MM:SS` in UTC (SQLite `datetime('now')`), not
|
|
294
|
+
ISO 8601, so time bounds compare `datetime(created_at)` with the bound rather
|
|
295
|
+
than the raw text.
|
|
296
|
+
|
|
297
|
+
`akm metrics` is the read surface for this table: it aggregates the rows of a
|
|
298
|
+
window into per-asset usage, queries (including searches that returned nothing),
|
|
299
|
+
feedback, and daily series, and `--format html` carries the window's rows in the
|
|
300
|
+
page so they can be filtered and exported in the browser. It reads
|
|
301
|
+
`usage_events`, the `select` events, `utility_scores`, `asset_outcome`,
|
|
302
|
+
`proposals`, the workflow tables, and `llm_usage` events, and writes nothing.
|
|
303
|
+
A window longer than a store's retention is reported in the result's `notes`.
|
|
304
|
+
|
|
282
305
|
### 3. Proposals Table
|
|
283
306
|
|
|
284
307
|
The proposal queue: pending, accepted, rejected, and reverted improvement proposals for your bundle assets. Generated by `akm improve`, `akm proposal new`, and related proposal-producing flows.
|
|
@@ -311,6 +334,16 @@ A record of scheduled task runs (from `akm task`):
|
|
|
311
334
|
|
|
312
335
|
## How to Inspect and Clear Local Data
|
|
313
336
|
|
|
337
|
+
### Report on usage
|
|
338
|
+
|
|
339
|
+
```sh
|
|
340
|
+
# What was searched, shown and rated in the last 30 days
|
|
341
|
+
akm metrics
|
|
342
|
+
|
|
343
|
+
# A dashboard you can open from disk
|
|
344
|
+
akm metrics --format html --output metrics.html
|
|
345
|
+
```
|
|
346
|
+
|
|
314
347
|
### Inspect events
|
|
315
348
|
|
|
316
349
|
```sh
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "akm-cli",
|
|
3
|
-
"version": "0.9.
|
|
3
|
+
"version": "0.9.28-alpha.2",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "akm (Agent Knowledge Manager) — a portable, local-first capability library for AI agents. Discover, load, share, and improve reusable skills, scripts, workflows, and knowledge across any shell-capable coding agent, including Claude Code, OpenCode, and Cursor.",
|
|
6
6
|
"keywords": [
|