akm-cli 0.9.27 → 0.9.28-alpha.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. package/CHANGELOG.md +65 -0
  2. package/STABILITY.md +4 -0
  3. package/dist/assets/hints/cli-hints-full.md +3 -0
  4. package/dist/assets/hints/cli-hints-short.md +1 -0
  5. package/dist/assets/prompts/consolidate-pair.md +1 -1
  6. package/dist/assets/templates/html/metrics.html +977 -0
  7. package/dist/cli/shared.js +5 -4
  8. package/dist/cli.js +11 -1
  9. package/dist/commands/health/accept-rate.js +8 -4
  10. package/dist/commands/health/html-report.js +3 -8
  11. package/dist/commands/health/llm-usage.js +17 -6
  12. package/dist/commands/health/renderers.js +4 -4
  13. package/dist/commands/improve/improve-report.js +4 -2
  14. package/dist/commands/improve/improve.js +22 -10
  15. package/dist/commands/improve/loop-stages.js +8 -1
  16. package/dist/commands/improve/preparation.js +16 -0
  17. package/dist/commands/improve/stage.js +1 -1
  18. package/dist/commands/metrics/collect.js +439 -0
  19. package/dist/commands/metrics/html-report.js +82 -0
  20. package/dist/commands/metrics/md-report.js +44 -0
  21. package/dist/commands/metrics/metrics-cli.js +213 -0
  22. package/dist/commands/metrics/report-view.js +243 -0
  23. package/dist/commands/metrics/types.js +4 -0
  24. package/dist/commands/read/search.js +5 -0
  25. package/dist/indexer/indexer.js +64 -14
  26. package/dist/indexer/usage/usage-events.js +3 -1
  27. package/dist/integrations/session-logs/pre-filter.js +1 -0
  28. package/dist/llm/usage-persist.js +22 -11
  29. package/dist/llm/usage-telemetry.js +4 -0
  30. package/dist/output/html-render.js +15 -8
  31. package/dist/output/shapes/passthrough.js +1 -0
  32. package/dist/output/text/metrics.js +39 -0
  33. package/dist/output/text.js +2 -0
  34. package/dist/scripts/akm-migrate-node.js +281 -8
  35. package/dist/scripts/akm-migrate.js +281 -8
  36. package/dist/storage/repositories/index-utility-repository.js +24 -0
  37. package/dist/storage/repositories/metrics-repository.js +80 -0
  38. package/docs/reference/cli.md +63 -4
  39. package/docs/reference/data-and-telemetry.md +41 -8
  40. package/package.json +1 -1
@@ -27777,6 +27777,243 @@ registerMdRenderer("health", (result) => {
27777
27777
  return null;
27778
27778
  });
27779
27779
 
27780
+ // src/commands/metrics/report-view.ts
27781
+ var BRIEF_LIST_LIMIT = 5;
27782
+ function isMetricsResult(value) {
27783
+ if (value === null || typeof value !== "object")
27784
+ return false;
27785
+ const r = value;
27786
+ return r.schemaVersion === 1 && !!r.window && !!r.usage && !!r.feedback && !!r.llm;
27787
+ }
27788
+ var int = (n) => String(n);
27789
+ var dash = (v) => v === undefined || v === "" ? "-" : v;
27790
+ var pct = (n) => n === null ? "n/a" : `${(n * 100).toFixed(1)}%`;
27791
+ var ms = (n) => n === null ? "n/a" : `${Math.round(n)} ms`;
27792
+ var fixed = (n, digits) => n === null ? "n/a" : n.toFixed(digits);
27793
+ function sourceFacts(counts) {
27794
+ const entries = Object.entries(counts);
27795
+ return entries.length === 0 ? "-" : entries.map(([k, v]) => `${k}=${v}`).join(" ");
27796
+ }
27797
+ function llmRows(group) {
27798
+ return Object.entries(group).map(([name, a]) => [
27799
+ name,
27800
+ int(a.calls),
27801
+ int(a.failures),
27802
+ int(a.promptTokens),
27803
+ int(a.completionTokens),
27804
+ int(a.totalTokens),
27805
+ ms(a.totalDurationMs)
27806
+ ]);
27807
+ }
27808
+ var LLM_HEADERS = ["name", "calls", "failed", "prompt", "completion", "total", "time"];
27809
+ function buildMetricsView(r, detail) {
27810
+ const brief = detail === "brief";
27811
+ const cut = (items) => brief ? items.slice(0, BRIEF_LIST_LIMIT) : items;
27812
+ const { usage, feedback, utility, llm } = r;
27813
+ const searches = usage.totals.searches;
27814
+ const sections = [
27815
+ {
27816
+ title: "Usage",
27817
+ facts: [
27818
+ ["searches", int(searches)],
27819
+ ["shows", int(usage.totals.shows)],
27820
+ ["curates", int(usage.totals.curates)],
27821
+ ["selects", int(usage.totals.selects)],
27822
+ ["select rate", pct(usage.selectRate)],
27823
+ [
27824
+ "zero-result searches",
27825
+ `${usage.totals.zeroResultSearches} (${pct(searches === 0 ? null : usage.totals.zeroResultSearches / searches)})`
27826
+ ],
27827
+ ["search median", ms(usage.searchMedianMs)],
27828
+ ["distinct assets", int(usage.totals.distinctAssets)],
27829
+ ["distinct queries", int(usage.totals.distinctQueries)],
27830
+ ["by source", sourceFacts(usage.bySource)]
27831
+ ],
27832
+ tables: [
27833
+ {
27834
+ title: "Top assets",
27835
+ headers: ["ref", "shows", "search hits", "selects", "+", "-", "last used"],
27836
+ rows: cut(usage.topAssets).map((a) => [
27837
+ a.ref,
27838
+ int(a.shows),
27839
+ int(a.searchHits),
27840
+ int(a.selects),
27841
+ int(a.positive),
27842
+ int(a.negative),
27843
+ dash(a.lastUsedAt)
27844
+ ])
27845
+ },
27846
+ {
27847
+ title: "Top queries",
27848
+ headers: ["query", "count", "avg results", "last"],
27849
+ rows: cut(usage.topQueries).map((q) => [q.query, int(q.count), fixed(q.avgResults, 1), q.lastAt])
27850
+ },
27851
+ {
27852
+ title: "Zero-result queries",
27853
+ headers: ["query", "count", "last"],
27854
+ rows: cut(usage.zeroResultQueries).map((q) => [q.query, int(q.count), q.lastAt])
27855
+ },
27856
+ {
27857
+ title: "Daily",
27858
+ headers: ["day", "search", "show", "curate", "feedback"],
27859
+ rows: brief ? [] : usage.daily.map((d) => [d.day, int(d.search), int(d.show), int(d.curate), int(d.feedback)]),
27860
+ hideWhenEmpty: true
27861
+ }
27862
+ ]
27863
+ },
27864
+ {
27865
+ title: "Feedback",
27866
+ facts: [
27867
+ ["positive", int(feedback.totals.positive)],
27868
+ ["negative", int(feedback.totals.negative)]
27869
+ ],
27870
+ tables: [
27871
+ {
27872
+ title: "By asset",
27873
+ headers: ["ref", "+", "-", "valence", "last"],
27874
+ rows: cut(feedback.byAsset).map((a) => [
27875
+ a.ref,
27876
+ int(a.positive),
27877
+ int(a.negative),
27878
+ fixed(a.valence, 2),
27879
+ a.lastAt
27880
+ ])
27881
+ },
27882
+ {
27883
+ title: "By tag",
27884
+ headers: ["tag", "+", "-"],
27885
+ rows: cut(Object.entries(feedback.byTag)).map(([tag, c]) => [tag, int(c.positive), int(c.negative)]),
27886
+ hideWhenEmpty: true
27887
+ },
27888
+ {
27889
+ title: "Recent negative",
27890
+ headers: ["ref", "at", "reason", "tags"],
27891
+ rows: cut(feedback.recentNegative).map((n) => [n.ref, n.at, dash(n.reason), dash(n.tags?.join(", "))])
27892
+ }
27893
+ ]
27894
+ },
27895
+ {
27896
+ title: "Utility",
27897
+ facts: [
27898
+ ["scored assets", int(utility.count)],
27899
+ ["never used", int(utility.neverUsed)]
27900
+ ],
27901
+ tables: [
27902
+ {
27903
+ title: "Histogram",
27904
+ headers: ["bucket", "count"],
27905
+ rows: utility.histogram.map((h) => [h.bucket, int(h.count)]),
27906
+ hideWhenEmpty: true
27907
+ },
27908
+ { title: "Lowest", headers: utilityHeaders(), rows: cut(utility.lowest).map(utilityRow) },
27909
+ { title: "Highest", headers: utilityHeaders(), rows: cut(utility.highest).map(utilityRow) },
27910
+ {
27911
+ title: "Lowest outcome",
27912
+ headers: ["ref", "outcome", "retrievals", "negative", "accepted changes"],
27913
+ rows: cut(r.outcomes.lowestOutcome).map((o) => [
27914
+ o.ref,
27915
+ fixed(o.outcomeScore, 2),
27916
+ int(o.retrievalCount),
27917
+ int(o.negativeFeedbackCount),
27918
+ int(o.acceptedChangeCount)
27919
+ ]),
27920
+ hideWhenEmpty: true
27921
+ }
27922
+ ]
27923
+ },
27924
+ {
27925
+ title: "LLM",
27926
+ facts: [
27927
+ ["calls", int(llm.calls)],
27928
+ ["failed", int(llm.failures)],
27929
+ ["prompt tokens", int(llm.promptTokens)],
27930
+ ["completion tokens", int(llm.completionTokens)],
27931
+ ["reasoning tokens", int(llm.reasoningTokens)],
27932
+ ["total tokens", int(llm.totalTokens)],
27933
+ ["time", ms(llm.totalDurationMs)]
27934
+ ],
27935
+ tables: [
27936
+ { title: "By engine", headers: LLM_HEADERS, rows: cut(llmRows(llm.byEngine)), hideWhenEmpty: true },
27937
+ { title: "By process", headers: LLM_HEADERS, rows: cut(llmRows(llm.byProcess)), hideWhenEmpty: true },
27938
+ { title: "By stage", headers: LLM_HEADERS, rows: cut(llmRows(llm.byStage)), hideWhenEmpty: true }
27939
+ ]
27940
+ },
27941
+ {
27942
+ title: "Index",
27943
+ facts: [
27944
+ ["runs", int(r.index.runs)],
27945
+ ["median time", ms(r.index.medianMs)]
27946
+ ],
27947
+ tables: [
27948
+ {
27949
+ title: "Recent runs",
27950
+ headers: ["at", "mode", "time"],
27951
+ rows: brief ? [] : r.index.recent.map((i) => [i.at, i.mode, ms(i.totalMs)]),
27952
+ hideWhenEmpty: true
27953
+ }
27954
+ ]
27955
+ },
27956
+ {
27957
+ title: "Tasks",
27958
+ facts: [
27959
+ ["runs", int(r.tasks.runs)],
27960
+ ["failed", int(r.tasks.failed)],
27961
+ ["fail rate", pct(r.tasks.failRate)]
27962
+ ],
27963
+ tables: [
27964
+ {
27965
+ title: "By task",
27966
+ headers: ["task", "runs", "failed", "median"],
27967
+ rows: cut(r.tasks.byTask).map((t) => [t.taskId, int(t.runs), int(t.failed), ms(t.medianMs)]),
27968
+ hideWhenEmpty: true
27969
+ }
27970
+ ]
27971
+ },
27972
+ {
27973
+ title: "Proposals",
27974
+ facts: [["by status", sourceFacts(r.proposals.byStatus)]],
27975
+ tables: [
27976
+ {
27977
+ title: "Accept rate by source",
27978
+ headers: ["source", "total", "accepted", "rejected", "pending", "accept rate"],
27979
+ rows: r.proposals.acceptRateBySource.map((p) => [
27980
+ p.source,
27981
+ int(p.total),
27982
+ int(p.accepted),
27983
+ int(p.rejected),
27984
+ int(p.pending),
27985
+ pct(p.acceptRate)
27986
+ ]),
27987
+ hideWhenEmpty: true
27988
+ }
27989
+ ]
27990
+ },
27991
+ {
27992
+ title: "Workflows",
27993
+ facts: [
27994
+ ["runs", int(r.workflows.runs)],
27995
+ ["by status", sourceFacts(r.workflows.byStatus)],
27996
+ ["tokens", int(r.workflows.tokens)],
27997
+ ["tokens by model", sourceFacts(r.workflows.byModel)]
27998
+ ],
27999
+ tables: []
28000
+ }
28001
+ ];
28002
+ const f = r.filters;
28003
+ return {
28004
+ window: `${r.window.since} to ${r.window.until}`,
28005
+ filters: `source=${f.source}`,
28006
+ sections,
28007
+ notes: r.notes
28008
+ };
28009
+ }
28010
+ function utilityHeaders() {
28011
+ return ["ref", "utility", "shows", "searches", "select rate", "last used"];
28012
+ }
28013
+ function utilityRow(u) {
28014
+ return [u.ref, fixed(u.utility, 2), int(u.showCount), int(u.searchCount), pct(u.selectRate), dash(u.lastUsedAt)];
28015
+ }
28016
+
27780
28017
  // src/output/generic-render.ts
27781
28018
  var ENVELOPE_META_KEYS = new Set(["shape", "schemaVersion"]);
27782
28019
  function flattenForText(value, path11, lines) {
@@ -28377,6 +28614,7 @@ var PASSTHROUGH_COMMANDS = [
28377
28614
  "info",
28378
28615
  "lint",
28379
28616
  "list",
28617
+ "metrics",
28380
28618
  "models",
28381
28619
  "proposal-accept-batch",
28382
28620
  "proposal-drain",
@@ -29651,10 +29889,10 @@ function formatHealthPlain(r, detail) {
29651
29889
  `).trim();
29652
29890
  }
29653
29891
  // src/output/text/lint-format.ts
29654
- function glyphFor(fixed) {
29655
- if (fixed === "failed")
29892
+ function glyphFor(fixed2) {
29893
+ if (fixed2 === "failed")
29656
29894
  return { glyph: "\u2717", severityRank: 0 };
29657
- if (fixed === true)
29895
+ if (fixed2 === true)
29658
29896
  return { glyph: "\u2713", severityRank: 2 };
29659
29897
  return { glyph: "\u26A0", severityRank: 1 };
29660
29898
  }
@@ -29673,17 +29911,17 @@ function renderIssueSection(title, issues) {
29673
29911
  function formatLintPlain(r) {
29674
29912
  if (r === null || typeof r !== "object")
29675
29913
  return null;
29676
- const fixed = Array.isArray(r.fixed) ? r.fixed : [];
29914
+ const fixed2 = Array.isArray(r.fixed) ? r.fixed : [];
29677
29915
  const flagged = Array.isArray(r.flagged) ? r.flagged : [];
29678
29916
  const warnings = Array.isArray(r.warnings) ? r.warnings : [];
29679
29917
  const summary = r.summary;
29680
29918
  const lines = [];
29681
29919
  if (typeof r.ok === "boolean")
29682
29920
  lines.push(`ok: ${r.ok}`);
29683
- lines.push(`summary: fixed=${summary?.fixed ?? fixed.length} flagged=${summary?.flagged ?? flagged.length}` + ` warnings=${summary?.warnings ?? warnings.length}`);
29921
+ lines.push(`summary: fixed=${summary?.fixed ?? fixed2.length} flagged=${summary?.flagged ?? flagged.length}` + ` warnings=${summary?.warnings ?? warnings.length}`);
29684
29922
  lines.push("", ...renderIssueSection("flagged", flagged));
29685
29923
  lines.push("", ...renderIssueSection("warnings", warnings));
29686
- lines.push("", ...renderIssueSection("fixed", fixed));
29924
+ lines.push("", ...renderIssueSection("fixed", fixed2));
29687
29925
  return lines.join(`
29688
29926
  `).trim();
29689
29927
  }
@@ -36144,8 +36382,8 @@ function getConfigLockPath() {
36144
36382
  }
36145
36383
  var CONFIG_LOCK_MAX_RETRIES = 40;
36146
36384
  var CONFIG_LOCK_RETRY_DELAY_MS = 50;
36147
- function sleepSyncMs(ms) {
36148
- sleepSync(ms);
36385
+ function sleepSyncMs(ms2) {
36386
+ sleepSync(ms2);
36149
36387
  }
36150
36388
  function acquireConfigLock() {
36151
36389
  const lockPath = getConfigLockPath();
@@ -53170,6 +53408,40 @@ var lintFormatters = [{ command: "lint", handler: (r) => formatLintPlain(r) }];
53170
53408
  // src/output/text/list.ts
53171
53409
  var listFormatters = [{ command: "list", handler: (r) => formatListPlain(r) }];
53172
53410
 
53411
+ // src/output/text/metrics.ts
53412
+ var flat = (s) => s.replace(/\s*\r?\n\s*/g, " ");
53413
+ function renderTable2(table) {
53414
+ const all = [table.headers, ...table.rows].map((r) => r.map(flat));
53415
+ const widths = table.headers.map((_, i) => Math.max(...all.map((r) => (r[i] ?? "").length)));
53416
+ const line = (r) => ` ${r.map((c, i) => c.padEnd(widths[i] ?? 0)).join(" ")}`.trimEnd();
53417
+ return [`${table.title}:`, ...all.map(line)];
53418
+ }
53419
+ function formatMetricsPlain(result, detail) {
53420
+ if (!isMetricsResult(result))
53421
+ return null;
53422
+ const view = buildMetricsView(result, detail);
53423
+ const lines = [`akm metrics ${view.window}`, `filters: ${view.filters}`];
53424
+ for (const section of view.sections) {
53425
+ lines.push("", section.title.toUpperCase());
53426
+ const labelWidth = Math.max(0, ...section.facts.map(([label]) => label.length));
53427
+ for (const [label, value] of section.facts)
53428
+ lines.push(` ${label.padEnd(labelWidth)} ${value}`);
53429
+ for (const table of section.tables) {
53430
+ if (table.rows.length > 0)
53431
+ lines.push("", ...renderTable2(table));
53432
+ else if (!table.hideWhenEmpty)
53433
+ lines.push("", `${table.title}: (none)`);
53434
+ }
53435
+ }
53436
+ if (view.notes.length > 0)
53437
+ lines.push("", "NOTES", ...view.notes.map((n) => ` - ${flat(n)}`));
53438
+ return lines.join(`
53439
+ `);
53440
+ }
53441
+ var metricsFormatters = [
53442
+ { command: "metrics", handler: (r, detail) => formatMetricsPlain(r, detail) }
53443
+ ];
53444
+
53173
53445
  // src/output/text/migrate.ts
53174
53446
  function planGlyph(status) {
53175
53447
  switch (status) {
@@ -53304,6 +53576,7 @@ var BUILT_IN_TEXT_FORMATTERS = [
53304
53576
  ...proposalProducerFormatters,
53305
53577
  ...infoFormatters,
53306
53578
  ...healthFormatters,
53579
+ ...metricsFormatters,
53307
53580
  ...improveReportFormatters,
53308
53581
  ...lintFormatters,
53309
53582
  ...configFormatters,
@@ -271,3 +271,27 @@ export function applyFeedbackToUtilityScore(db, entryId, positiveCount, negative
271
271
  `).run(entryId, result.nextUtility, now, now, result.nextUtility, now);
272
272
  return result;
273
273
  }
274
+ /**
275
+ * Every indexed entry with its `utility_scores` row (when it has one), for
276
+ * `akm metrics`. `utility_scores` is keyed by the unstable `entry_id`, so the
277
+ * join to `entries` is what yields a durable ref.
278
+ */
279
+ export function listUtilityWithRefs(db) {
280
+ const rows = db
281
+ .prepare(`SELECT e.item_ref AS ref, u.utility, u.show_count, u.search_count, u.select_rate, u.last_used_at
282
+ FROM entries e LEFT JOIN utility_scores u ON u.entry_id = e.id
283
+ ORDER BY e.item_ref`)
284
+ .all();
285
+ return rows.map((row) => row.utility === null
286
+ ? { ref: row.ref }
287
+ : {
288
+ ref: row.ref,
289
+ score: {
290
+ utility: row.utility,
291
+ showCount: row.show_count ?? 0,
292
+ searchCount: row.search_count ?? 0,
293
+ selectRate: row.select_rate ?? 0,
294
+ lastUsedAt: row.last_used_at ?? undefined,
295
+ },
296
+ });
297
+ }
@@ -0,0 +1,80 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ /** Usage rows (search / show / curate / feedback) in the window, oldest first. */
5
+ export function listUsageEventRows(db, filter) {
6
+ const sourceSql = filter.source === undefined ? "" : " AND source = ?";
7
+ const sourceParams = filter.source === undefined ? [] : [filter.source];
8
+ // `created_at` has whole-second resolution, so a row written during the
9
+ // bound's own second (e.g. `--until` defaulting to now) is still inside the
10
+ // half-open window: round a fractional bound up to the next second.
11
+ const untilMs = Date.parse(filter.untilIso);
12
+ const untilBound = new Date(Math.ceil(untilMs / 1000) * 1000).toISOString();
13
+ return db
14
+ .prepare(`SELECT id, event_type, query, entry_id, entry_ref, signal, metadata, source, created_at
15
+ FROM usage_events
16
+ WHERE datetime(created_at) >= datetime(?) AND datetime(created_at) < datetime(?)${sourceSql}
17
+ ORDER BY datetime(created_at), id`)
18
+ .all(filter.sinceIso, untilBound, ...sourceParams);
19
+ }
20
+ /**
21
+ * `select` events (a show within 60 s of a search that returned the ref) in the
22
+ * window. The events stream stores the ref as the user typed it (often
23
+ * bundle-less), so the caller resolves it to the durable ref.
24
+ */
25
+ export function listSelectEvents(db, sinceIso, untilIso) {
26
+ return db
27
+ .prepare(`SELECT ts, ref FROM events
28
+ WHERE event_type = 'select' AND ref IS NOT NULL AND ts >= ? AND ts < ?
29
+ ORDER BY id`)
30
+ .all(sinceIso, untilIso);
31
+ }
32
+ /** `index_completed` events (one per `akm index` run) in the window, oldest first. */
33
+ export function listIndexCompletedEvents(db, sinceIso, untilIso) {
34
+ return db
35
+ .prepare(`SELECT ts, metadata_json FROM events
36
+ WHERE event_type = 'index_completed' AND ts >= ? AND ts < ?
37
+ ORDER BY id`)
38
+ .all(sinceIso, untilIso);
39
+ }
40
+ /** The `limit` assets with the lowest `outcome_score`, ties broken by ref. */
41
+ export function listLowestOutcomeAssets(db, limit) {
42
+ return db
43
+ .prepare(`SELECT asset_ref, last_retrieved_at, retrieval_count, expected_retrieval_rate,
44
+ negative_feedback_count, accepted_change_count, outcome_score, updated_at
45
+ FROM asset_outcome
46
+ ORDER BY outcome_score ASC, asset_ref ASC LIMIT ?`)
47
+ .all(limit);
48
+ }
49
+ /** Proposal counts by status, for proposals last updated in the window. */
50
+ export function countProposalsByStatus(db, sinceIso, untilIso) {
51
+ const rows = db
52
+ .prepare(`SELECT status, COUNT(*) AS n FROM proposals
53
+ WHERE updated_at >= ? AND updated_at < ?
54
+ GROUP BY status ORDER BY status`)
55
+ .all(sinceIso, untilIso);
56
+ return Object.fromEntries(rows.map((row) => [row.status, row.n]));
57
+ }
58
+ /**
59
+ * Workflow runs created in the window by status, and the tokens their unit
60
+ * attempts spent by model. Attempts, not the unit projection, so a retried
61
+ * unit keeps the tokens of every attempt.
62
+ */
63
+ export function summarizeWorkflowRuns(db, sinceIso, untilIso) {
64
+ const statusRows = db
65
+ .prepare(`SELECT status, COUNT(*) AS n FROM workflow_runs
66
+ WHERE created_at >= ? AND created_at < ? GROUP BY status ORDER BY status`)
67
+ .all(sinceIso, untilIso);
68
+ const modelRows = db
69
+ .prepare(`SELECT COALESCE(model, 'unattributed') AS model, SUM(tokens) AS tokens
70
+ FROM workflow_run_unit_attempts
71
+ WHERE tokens IS NOT NULL AND started_at >= ? AND started_at < ?
72
+ GROUP BY COALESCE(model, 'unattributed') ORDER BY model`)
73
+ .all(sinceIso, untilIso);
74
+ return {
75
+ runs: statusRows.reduce((sum, row) => sum + row.n, 0),
76
+ byStatus: Object.fromEntries(statusRows.map((row) => [row.status, row.n])),
77
+ tokens: modelRows.reduce((sum, row) => sum + row.tokens, 0),
78
+ byModel: Object.fromEntries(modelRows.map((row) => [row.model, row.tokens])),
79
+ };
80
+ }
@@ -42,7 +42,8 @@ Useful for streaming consumption by scripts or agents.
42
42
  A command may register a renderer for a document format when it has something
43
43
  better to say than the generic one: `akm health --group-by run --format md`
44
44
  emits its per-run table, and `akm health --report --format html` renders the
45
- full report with KPI cards, charts, and advisories. The renderers are
45
+ full report with KPI cards, charts, and advisories. `akm metrics` always carries
46
+ its window rows under `--format html`. The renderers are
46
47
  data-driven — they fire when the result carries the report dataset, never on
47
48
  the format alone, so the same dataset is available as JSON too. Every other command falls back to a
48
49
  generic rendering derived from its own envelope — headings for the top-level
@@ -439,6 +440,64 @@ for an infrastructure reason (`llm_unavailable`, `read_failed`, `exception`,
439
440
  `locked_concurrent`) — naming the reason and, when recorded, the engine — and
440
441
  `pass` otherwise, with per-outcome counts.
441
442
 
443
+ ### metrics
444
+
445
+ **Experimental** (see [STABILITY.md](../../STABILITY.md)): the report, its JSON shape and the
446
+ dashboard may change in any release.
447
+
448
+ Report what akm has recorded locally: asset usage (search, show, curate),
449
+ feedback, derived utility, LLM tokens, task runs, proposal
450
+ flow, workflow token spend, and index runs. Read-only, and every number comes
451
+ from `state.db` or `index.db`; nothing is collected that was not already
452
+ recorded.
453
+
454
+ ```sh
455
+ akm metrics
456
+ akm metrics --since 7d
457
+ akm metrics --since 2026-05-01 --format yaml
458
+ akm metrics --format html --output metrics.html
459
+ ```
460
+
461
+ | Flag | Description |
462
+ | --- | --- |
463
+ | `--since` | Window start. Accepts ISO 8601, `YYYY-MM-DD`, epoch milliseconds, or shorthand like `24h` / `7d`. Default: `30d`. |
464
+
465
+ The window always ends now. Only `user`-source usage rows are counted (matching
466
+ utility and retrieval counts), and every ranked list holds its top 20. For a
467
+ narrower view, open the HTML dashboard and filter by date, bundle, source and
468
+ event type in the browser.
469
+
470
+ Result sections (`schemaVersion: 1`):
471
+
472
+ | Field | Description |
473
+ | --- | --- |
474
+ | `window`, `filters` | The resolved window and the usage source counted |
475
+ | `usage` | Search, show, curate and select totals, `selectRate` (selects over searches that returned a hit), `searchMedianMs`, a daily series, top assets, top and zero-result queries, and rows by source |
476
+ | `feedback` | Positive and negative totals, per-asset valence, per-tag counts, and the most recent negatives with their reasons |
477
+ | `utility` | A ten-bucket histogram of `utility_scores`, the lowest and highest assets, and how many indexed entries were never used |
478
+ | `outcomes` | The assets with the lowest `outcome_score` |
479
+ | `llm` | `akm health`'s LLM usage aggregate (by stage, process and engine) |
480
+ | `index`, `tasks`, `proposals`, `workflows` | Index runs and median time, task runs and fail rate, proposals by status and accept rate by source, workflow runs and tokens by model |
481
+ | `rows` | The raw usage and LLM rows of the window; see below |
482
+ | `notes` | Anything that limits what the numbers mean |
483
+
484
+ A rate whose denominator is 0 is `null`, never `NaN`. A missing `state.db`
485
+ gives an empty report and a missing `index.db` an empty `utility` section, each
486
+ with a `notes` entry, and the exit code stays 0.
487
+
488
+ `rows` is present with `--format html` (the dashboard re-aggregates from it in
489
+ the browser) and with `--detail full` in any other format; it is absent
490
+ otherwise.
491
+
492
+ Notes you may see:
493
+
494
+ - A `--since` older than what a store keeps names the store, its retention
495
+ (`usage_events` keeps 90 days; `events`, which holds selects, LLM calls and
496
+ index runs, keeps `improve.eventRetentionDays`, default 90) and where its data
497
+ effectively starts, so a long window is never silently shorter than asked.
498
+ - Selects are read from the events stream, which records no source, so they are
499
+ counted whatever the usage source is.
500
+
442
501
  ### search
443
502
 
444
503
  Search bundle assets, registries, or both.
@@ -2655,8 +2714,8 @@ precedent as the retired `canary` scope).
2655
2714
  Every real (non-dry-run) `akm improve` invocation persists a `usageReport`
2656
2715
  field on the result (`result_json` in `improve_runs`, and in the
2657
2716
  `--json-to-stdout` / dry-run JSON): `{ byProcessEngineModel, noCalls }`.
2658
- `byProcessEngineModel` is a cross-tab of this run's own `llm_usage` events
2659
- (#576) — one row per distinct `(process, engine, model)` triple, each with
2717
+ `byProcessEngineModel` is a cross-tab of the LLM call records this run's own
2718
+ usage sink collects (#576) — one row per distinct `(process, engine, model)` triple, each with
2660
2719
  `calls`, `failures`, `promptTokens`, `completionTokens`, `totalTokens`,
2661
2720
  `reasoningTokens`, and `totalDurationMs`. `noCalls` lists every model-calling
2662
2721
  process (`reflect`, `distill`, `consolidate`, `memoryInference`,
@@ -2678,7 +2737,7 @@ every real run in the window (`byProcessEngineModel` rows merged by
2678
2737
  `(process, engine, model)`; `noCalls` lists a process only if it made zero
2679
2738
  calls across every included run). A run recorded before 0.9.15 has no
2680
2739
  persisted `usageReport` — the command recomputes `byProcessEngineModel` from
2681
- that run's own `llm_usage` events instead of erroring, sets `noCalls` to `[]`
2740
+ that run's stored `llm_usage` events (`summarizeLlmUsageCrossTab`) instead of erroring, sets `noCalls` to `[]`
2682
2741
  (eligibility reasons are not reconstructable after the fact), and adds a
2683
2742
  `notes` entry saying so rather than fabricating precision the old row can't
2684
2743
  support.
@@ -162,6 +162,7 @@ the set of types the code actually emits at HEAD (verified against every
162
162
  | `feedback` | `akm feedback <ref>` | `signal` (positive/negative), `reason`, `tags`, `fix` (`source`, the number of replacements and, for `--superseded-by` or `--outdated`, the `beliefState` the proposal leaves and the `supersededBy` ref, when a fix was attached), `contentHash` (sha256 of the asset's body, without its frontmatter, as it stood when the feedback was given: it lets reflect mark feedback given on an earlier version of the text, and the loop's distill pass tell that a memory flagged wrong still has it; left out for an env or secret file and when the file cannot be read) |
163
163
  | `sync` | `akm sync` | `name`, `message`, `ok` |
164
164
  | `index_db_vacuumed` | `akm index` VACUUMed index.db, after an index layout migration or because more than half its pages were free | `pagesBefore`, `pagesAfter`, `freelistRatioBefore` |
165
+ | `index_completed` | `akm index`, once when a run finishes | `mode`, `totalMs`, `walkMs`, `llmMs`, `embedMs`, `ftsMs`, `finalizeMs` (phase timings in milliseconds) |
165
166
  | `stash_synced` | `akm improve`'s internal auto-sync pass (the `sync.push` feature), **distinct from** the `akm sync` command above | `committed`, `pushed`, `skipped`, `reason`, `attributed` (paths the run wrote and staged), `unattributed` (in-scope paths that went dirty during the run without the run writing them — left for their author) |
166
167
  | `env_access` | `akm env run <name> -- <command>` (audit trail: key **names** only, values never recorded) | `ref`, `keys` |
167
168
  | `secret_access` | `akm secret run <ref> <VAR> -- <command>` (audit trail: var **name** only, value never recorded) | `ref`, `var` |
@@ -188,7 +189,7 @@ the set of types the code actually emits at HEAD (verified against every
188
189
  | `improve_invoked` | Start of an `akm improve` run | `ref` (scope); `strategy`, `scope`, `dryRun`, `eligibleCount` |
189
190
  | `improve_completed` | `akm improve` run finished | run stats |
190
191
  | `improve_failed` | `akm improve` run errored | error |
191
- | `improve_skipped` | `akm improve` left a ref, a lane, or a group of refs out | `reason` (`no_new_signal`, `not_retrieved`, `distill_no_new_signal`, `budget_exhausted`, `budget_exhausted_batch`, `asset_missing_on_disk`, `strategy_filtered_all_passes`, `autonomy_gated`, `engine_unavailable`, `pool_below_min_size`, `consolidation_no_memory_updates`, `below_min_new_sessions`, `derived_memory_reflect_skipped`, `memory_distill_requires_feedback`, `distill_flagged_wrong`, `distill_positive_without_reason`); `count`, `remaining`, `strategy`, `lane` or `configKey` where they apply |
192
+ | `improve_skipped` | `akm improve` left a ref, a lane, or a group of refs out | `reason` (`no_new_signal`, `not_retrieved`, `distill_no_new_signal`, `budget_exhausted`, `budget_exhausted_batch`, `asset_missing_on_disk`, `strategy_filtered_all_passes`, `autonomy_gated`, `engine_unavailable`, `pool_below_min_size`, `consolidation_no_memory_updates`, `below_min_new_sessions`, `derived_memory_reflect_skipped`, `memory_distill_requires_feedback`, `distill_flagged_wrong`, `distill_positive_without_reason`, `distill_deprecated_or_superseded`); `count`, `remaining`, `strategy`, `lane` or `configKey` where they apply |
192
193
  | `improve_lock_recovered` | Stale improve lock cleared at startup | |
193
194
  | `improve_review_needed` | `akm feedback` pushed a high-utility asset's utility below the review threshold — a review-needed escalation is recorded (not a proposal, so it can't accidentally overwrite the asset) | `ref`, `previousUtility`, `nextUtility` |
194
195
  | `reflect_invoked` | Start of reflect phase in `akm improve` | `ref`, engine |
@@ -222,17 +223,20 @@ the set of types the code actually emits at HEAD (verified against every
222
223
 
223
224
  | Event type | When emitted | Key metadata fields |
224
225
  |---|---|---|
225
- | `llm_usage` | Per-attempt LLM call usage telemetry (#576) | model provenance, terminal outcome, duration, optional token usage |
226
- | `llm_usage_summary` | The owning LLM telemetry sink's terminal-record count marker | `expectedTerminalRecords` |
226
+ | `llm_usage` | Per-attempt LLM call usage telemetry (#576), written by every akm command that makes an LLM call (`index`, `curate`, workflows, agent dispatch, `command run`, `improve`, `proposal drain`) | model provenance, terminal outcome, duration, optional token usage |
227
+ | `llm_usage_summary` | The owning LLM telemetry sink's terminal-record count marker. The process-wide sink that covers commands other than `improve` writes none when it saw no call | `expectedTerminalRecords` |
227
228
  | `health_probe` | `akm health`'s state.db round-trip write/read probe. **Not durably retained**: the row is inserted then deleted within the same connection once the round trip is confirmed, so the net effect on the `events` table is always zero rows | n/a (ephemeral) |
228
229
 
229
230
  `llm_usage` rows also carry `process`/`engine`/`stage` (each optional; a call
230
231
  made outside any attributed scope carries none of them). `akm improve`
231
- (#944) aggregates a run's own `llm_usage` events into a process x engine x
232
- model cross-tab — `summarizeLlmUsageCrossTab` in `src/commands/health/llm-usage.ts`
233
- — persisted on the run result as `usageReport.byProcessEngineModel` and
234
- queryable per-run or aggregated with `akm improve report`; see
235
- `docs/reference/cli.md`'s `#### improve report` section.
232
+ (#944) builds a process x engine x model cross-tab from the LLM call records
233
+ the run's own usage sink collects — `summarizeLlmUsageRecordsCrossTab` in
234
+ `src/commands/health/llm-usage.ts` — persisted on the run result as
235
+ `usageReport.byProcessEngineModel` and queryable per-run or aggregated with
236
+ `akm improve report`; see `docs/reference/cli.md`'s `#### improve report`
237
+ section. `summarizeLlmUsageCrossTab` is the events form of the same
238
+ aggregation, used by `improve report` to recompute the cross-tab from stored
239
+ `llm_usage` events when a run has no persisted `usageReport`.
236
240
 
237
241
  ### 2. Usage Events Table
238
242
 
@@ -248,6 +252,12 @@ configured endpoint.
248
252
  Successful `search`, `curate`, and `show` commands always record usage. Machine
249
253
  reads are stamped by source (below), so they never skew ranking or eval.
250
254
 
255
+ The `search` summary row (the one with no `entry_ref`) carries `resultCount`,
256
+ `stashHitCount`, `registryHitCount`, `resolvedCount` and `mode` in its metadata,
257
+ plus latency for `akm metrics`: `totalMs` for the whole search, and `rankMs` and
258
+ `embedMs` for the ranking and query-embedding phases when the local search
259
+ reported them (a registry-only search has `totalMs` alone).
260
+
251
261
  Every runtime writer stamps provenance as `user`, `improve`, `task`, `audit`, or
252
262
  `unknown`. Direct interactive CLI traffic defaults to `user`; internal improve,
253
263
  scheduled-task, and eval subprocesses preserve their stamp across nested
@@ -279,6 +289,19 @@ higher-priority-wins `(type, entry.name)` dedup across sources: attribution
279
289
  source-qualifies every indexed row but does not invent a lower-priority row for
280
290
  an identity that production indexing omitted.
281
291
 
292
+ **Retention:** usage events older than 90 days are purged on every `akm index`.
293
+ `created_at` is `YYYY-MM-DD HH:MM:SS` in UTC (SQLite `datetime('now')`), not
294
+ ISO 8601, so time bounds compare `datetime(created_at)` with the bound rather
295
+ than the raw text.
296
+
297
+ `akm metrics` is the read surface for this table: it aggregates the rows of a
298
+ window into per-asset usage, queries (including searches that returned nothing),
299
+ feedback, and daily series, and `--format html` carries the window's rows in the
300
+ page so they can be filtered and exported in the browser. It reads
301
+ `usage_events`, the `select` events, `utility_scores`, `asset_outcome`,
302
+ `proposals`, the workflow tables, and `llm_usage` events, and writes nothing.
303
+ A window longer than a store's retention is reported in the result's `notes`.
304
+
282
305
  ### 3. Proposals Table
283
306
 
284
307
  The proposal queue: pending, accepted, rejected, and reverted improvement proposals for your bundle assets. Generated by `akm improve`, `akm proposal new`, and related proposal-producing flows.
@@ -311,6 +334,16 @@ A record of scheduled task runs (from `akm task`):
311
334
 
312
335
  ## How to Inspect and Clear Local Data
313
336
 
337
+ ### Report on usage
338
+
339
+ ```sh
340
+ # What was searched, shown and rated in the last 30 days
341
+ akm metrics
342
+
343
+ # A dashboard you can open from disk
344
+ akm metrics --format html --output metrics.html
345
+ ```
346
+
314
347
  ### Inspect events
315
348
 
316
349
  ```sh
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "akm-cli",
3
- "version": "0.9.27",
3
+ "version": "0.9.28-alpha.2",
4
4
  "type": "module",
5
5
  "description": "akm (Agent Knowledge Manager) — a portable, local-first capability library for AI agents. Discover, load, share, and improve reusable skills, scripts, workflows, and knowledge across any shell-capable coding agent, including Claude Code, OpenCode, and Cursor.",
6
6
  "keywords": [