akm-cli 0.9.17-alpha.5 → 0.9.17-alpha.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. package/CHANGELOG.md +281 -0
  2. package/STABILITY.md +2 -2
  3. package/dist/akm +7 -7
  4. package/dist/assets/prompts/retrieval-relevance-judge.md +6 -0
  5. package/dist/commands/improve/consolidate.js +11 -0
  6. package/dist/commands/improve/execution.js +5 -5
  7. package/dist/commands/improve/improve-cli.js +27 -7
  8. package/dist/commands/improve/improve-strategies.js +3 -0
  9. package/dist/commands/improve/ledger.js +7 -3
  10. package/dist/commands/improve/loop-stages.js +5 -4
  11. package/dist/commands/improve/preparation.js +40 -10
  12. package/dist/commands/improve/reflect.js +46 -22
  13. package/dist/commands/improve/retrieval-gate.js +127 -0
  14. package/dist/commands/improve/retrieval-scope.js +77 -0
  15. package/dist/commands/read/curate.js +15 -49
  16. package/dist/commands/read/show.js +2 -81
  17. package/dist/commands/tasks/tasks-cli.js +10 -12
  18. package/dist/commands/tasks/tasks.js +57 -56
  19. package/dist/commands/tasks/validate.js +27 -46
  20. package/dist/core/adapter/adapters/akm-task-adapter.js +29 -8
  21. package/dist/core/config/config.js +1 -1
  22. package/dist/core/config/schema/improve-processes.js +4 -3
  23. package/dist/core/config/schema/index-config.js +4 -23
  24. package/dist/core/improve-result.js +4 -1
  25. package/dist/core/non-task-input.js +20 -0
  26. package/dist/core/paths.js +0 -4
  27. package/dist/indexer/db/graph-db.js +8 -81
  28. package/dist/indexer/graph/graph-extraction.js +74 -227
  29. package/dist/indexer/graph/graph-related.js +5 -4
  30. package/dist/indexer/indexer.js +1 -3
  31. package/dist/indexer/search/db-search.js +10 -1
  32. package/dist/indexer/usage/usage-events.js +34 -0
  33. package/dist/llm/feature-gate.js +0 -3
  34. package/dist/llm/graph-extract.js +81 -41
  35. package/dist/scripts/akm-migrate-node.js +7148 -7174
  36. package/dist/scripts/akm-migrate.js +8718 -8744
  37. package/dist/storage/repositories/index-entries-repository.js +5 -7
  38. package/dist/storage/repositories/index-schema.js +16 -34
  39. package/dist/storage/repositories/proposals-repository.js +4 -0
  40. package/dist/tasks/backends/cron.js +80 -43
  41. package/dist/tasks/backends/launchd.js +28 -15
  42. package/dist/tasks/backends/schtasks.js +25 -10
  43. package/dist/tasks/run/load-task.js +1 -1
  44. package/dist/tasks/scheduler-binding.js +4 -2
  45. package/dist/tasks/scheduler-invocation.js +127 -235
  46. package/dist/tasks/scheduler-sync.js +13 -8
  47. package/dist/tasks/source/parse-task-source.js +22 -126
  48. package/dist/tasks/source/task-to-v3.js +1 -55
  49. package/dist/tasks/source/task-to-v4.js +1 -13
  50. package/docs/migration/v0.9.1-to-v0.9.2.md +7 -3
  51. package/docs/reference/cli.md +13 -4
  52. package/docs/reference/configuration.md +12 -0
  53. package/docs/reference/tasks.md +58 -40
  54. package/package.json +1 -1
  55. package/schemas/akm-config.json +0 -6
@@ -23,7 +23,7 @@ import { ConfigError, rethrowIfTestIsolationError } from "../../core/errors.js";
23
23
  import { appendEvent, readEvents } from "../../core/events.js";
24
24
  import { withStateDb } from "../../core/state-db.js";
25
25
  import { info, warn } from "../../core/warn.js";
26
- import { countUsageEventsByType } from "../../indexer/usage/usage-events.js";
26
+ import { countUsageEventsByType, USAGE_EVENT_RETENTION_DAYS } from "../../indexer/usage/usage-events.js";
27
27
  import { getAvailableHarnesses } from "../../integrations/session-logs/index.js";
28
28
  import { getZeroResultSearches } from "../../storage/repositories/index-entries-repository.js";
29
29
  import { getRetrievalCounts } from "../../storage/repositories/index-utility-repository.js";
@@ -41,6 +41,7 @@ import { applyMemoryCleanup } from "./memory/memory-improve.js";
41
41
  import { getAllAssetOutcomes, getAssetOutcome, getOutcomeScoresByRef, OUTCOME_SCORE_MAX, outcomeScoreToSalience, projectAssetOutcome, updateAssetOutcome, } from "./outcome-loop.js";
42
42
  import { projectMemoryCleanup, selectEffectiveImproveRefs } from "./planner.js";
43
43
  import { DEFAULT_DUE_DAYS, DEFAULT_MAX_PER_RUN, selectProactiveMaintenanceRefs } from "./proactive-maintenance.js";
44
+ import { isInRetrievalScope, loadRetrievalScope } from "./retrieval-scope.js";
44
45
  import { buildRankChangeReport, computeSalience, getAllRankScores, getAssetSalience, getLastUseMsByRef, isContentEncodingRow, SALIENCE_NO_OP_DAMPEN_FACTOR, SALIENCE_NO_OP_DAMPEN_THRESHOLD, upsertAssetSalience, } from "./salience.js";
45
46
  import { attributeStage, errMessage } from "./stage.js";
46
47
  /** The candidate's durable state key (salience, outcome, ledger). */
@@ -663,12 +664,23 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
663
664
  const processableRefs = [...partition.eligibleRefs, ...partition.distillOnlyRefs];
664
665
  const signalFiltered = processableRefs.filter((c) => snapshot.feedback.get(c.ref)?.hasSignal === true);
665
666
  const signalBearingSet = new Set(signalFiltered.map((r) => r.ref));
666
- const noFeedbackCandidates = dedupeRefs([
667
+ // The fallback lanes (proactive, high salience, forgetting safety) have no
668
+ // usage evidence of their own: they pick only what retrieval returned or new
669
+ // material improve never processed (#986). Evaluated once per candidate.
670
+ const fallbackEligible = postCleanupRefs.filter((c) => !validationFailureRefs.has(c.ref));
671
+ const allowFallbacks = options.requireFeedbackSignal !== true;
672
+ const retrievalScope = scope.mode === "ref" || !allowFallbacks
673
+ ? undefined
674
+ : loadRetrievalScope({ eventsCtx, ...(persist ? {} : { readOnly: true }) }, primaryStashDir ?? options.stashDir);
675
+ const unscoped = new Set(fallbackEligible.filter((c) => !isInRetrievalScope(retrievalScope, c.ref, c.filePath)).map((c) => c.ref));
676
+ const noFeedbackPool = dedupeRefs([
667
677
  ...processableRefs.filter((r) => !signalBearingSet.has(r.ref)),
668
678
  ...partition.noFeedbackPool,
669
679
  ]);
680
+ const noFeedbackCandidates = noFeedbackPool.filter((r) => !unscoped.has(r.ref));
681
+ // Only a ref no fallback lane may pick anymore is charged to the retrieval gate.
682
+ const outOfScope = new Set(noFeedbackPool.filter((r) => unscoped.has(r.ref)).map((r) => r.ref));
670
683
  const retrieval = fetchRetrievalSignals(options, signalFiltered, noFeedbackCandidates, eventsCtx, persist);
671
- const allowFallbacks = options.requireFeedbackSignal !== true;
672
684
  const proactive = allowFallbacks
673
685
  ? selectProactiveMaintenanceLane(args, noFeedbackCandidates, snapshot, retrieval, persist)
674
686
  : { proactiveRefs: [] };
@@ -692,10 +704,10 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
692
704
  sourceByRef.set(r.ref, "scope");
693
705
  for (const r of mergedRefs)
694
706
  r.eligibilitySource = sourceByRef.get(r.ref) ?? "unknown";
695
- // Forgetting safety may only reuse this plan's own surviving objects, and
696
- // never a ref whose reflect window is still open.
697
- const fallbackEligible = postCleanupRefs.filter((c) => !validationFailureRefs.has(c.ref));
698
- const forgettingEligible = fallbackEligible.filter((c) => !isLedgerBlocked(ledgerRowFor(snapshot.ledger, "reflect", c.ref, c.itemRef), snapshot.nowIso));
707
+ // Forgetting safety may only reuse this plan's own surviving objects inside
708
+ // the retrieval scope, and never a ref whose reflect window is still open.
709
+ const forgettingEligible = fallbackEligible.filter((c) => !unscoped.has(c.ref) &&
710
+ !isLedgerBlocked(ledgerRowFor(snapshot.ledger, "reflect", c.ref, c.itemRef), snapshot.nowIso));
699
711
  const scored = scoreSalience(args, mergedRefs, snapshot.feedback, retrieval.retrievalCounts, persist);
700
712
  mergedRefs = applyForgettingSafety({
701
713
  pendingForgettingRefs: scored.pendingForgettingRefs,
@@ -739,28 +751,46 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
739
751
  // Skip observability waits until every fallback lane has finalized the
740
752
  // survivors, so a rescued ref is never also reported skipped.
741
753
  const survivors = new Set(sorted.map((c) => c.ref));
742
- const signalSkipped = fallbackEligible.filter((c) => !survivors.has(c.ref));
754
+ const skipped = fallbackEligible.filter((c) => !survivors.has(c.ref));
755
+ const retrievalSkipped = skipped.filter((c) => outOfScope.has(c.ref));
756
+ const signalSkipped = skipped.filter((c) => !outOfScope.has(c.ref));
743
757
  for (const ref of partition.distillCooledRefs) {
744
758
  actions.push({ ref, mode: "distill-skipped", result: { ok: true, reason: "distill signal-delta" } });
745
759
  if (persist)
746
760
  recordImproveSkip(eventsCtx, ref, { reason: "distill_no_new_signal" });
747
761
  }
748
- for (const candidate of signalSkipped) {
762
+ for (const candidate of skipped) {
749
763
  actions.push({
750
764
  ref: candidate.ref,
751
765
  mode: "distill-skipped",
752
- result: { ok: true, reason: "no new signal since last proposal" },
766
+ result: {
767
+ ok: true,
768
+ reason: outOfScope.has(candidate.ref)
769
+ ? "not retrieved inside the usage window"
770
+ : "no new signal since last proposal",
771
+ },
753
772
  });
754
773
  }
755
774
  if (persist && signalSkipped.length > 0) {
756
775
  recordImproveSkip(eventsCtx, undefined, { reason: "no_new_signal", count: signalSkipped.length });
757
776
  }
777
+ if (persist && retrievalSkipped.length > 0) {
778
+ recordImproveSkip(eventsCtx, undefined, { reason: "not_retrieved", count: retrievalSkipped.length });
779
+ }
758
780
  const blocked = signalSkipped.length + partition.distillOnlyRefs.length;
759
781
  if (blocked > 0) {
760
782
  info(`[improve] ${blocked} of ${partition.preCooldownCount} indexed refs blocked by reflect signal-delta ` +
761
783
  `(${signalSkipped.length} fully skipped, ${partition.distillOnlyRefs.length} routed to distill-only)`);
762
784
  }
785
+ if (retrievalSkipped.length > 0) {
786
+ info(`[improve] ${retrievalSkipped.length} refs left out: not retrieved in the last ${USAGE_EVENT_RETENTION_DAYS} days, and not new material`);
787
+ }
763
788
  const gates = [
789
+ {
790
+ name: "retrieval",
791
+ removed: retrievalSkipped.length,
792
+ reason: `no feedback, and neither returned by search, curate or show in the last ${USAGE_EVENT_RETENTION_DAYS} days nor new material improve never processed`,
793
+ },
764
794
  {
765
795
  name: "signal",
766
796
  removed: signalSkipped.length,
@@ -43,6 +43,7 @@ import { findAssetFilePath } from "./eligibility.js";
43
43
  import { resolveImproveLlmExecution } from "./execution.js";
44
44
  import { recordLedgerAttempt } from "./ledger.js";
45
45
  import { classifyReflectChange, splitFrontmatter } from "./reflect-noise.js";
46
+ import { loadRetrievalQueries, runRetrievalRegressionGate } from "./retrieval-gate.js";
46
47
  import { callStage, mintProposal, noticeSet, rejectedProposalContext, runReflectQualityJudge, } from "./stage.js";
47
48
  const MAX_FEEDBACK_LINES = 10;
48
49
  const MAX_GLOBAL_FEEDBACK_LINES = 20;
@@ -960,6 +961,24 @@ async function finalizeReflectProposal(args) {
960
961
  }
961
962
  const flagged = Boolean(sanitized.sizeGuardRatio || sanitized.truncationMarkerLeaked);
962
963
  const judged = judge.enabled && !flagged;
964
+ /** A judge refused the revision: record it for the ledger's rejection window and stop. */
965
+ const refuse = (detail, metadata, message) => {
966
+ if (options.ref) {
967
+ recordLedgerAttempt({ proposalsCtx: options.ctx, eventsCtx: options.eventsCtx }, {
968
+ stashDir: run.stash,
969
+ ref: options.itemRef ?? options.ref,
970
+ source: "reflect",
971
+ outcome: "quality_rejected",
972
+ detail,
973
+ });
974
+ }
975
+ appendEvent({
976
+ eventType: "reflect_completed",
977
+ ref: payload.ref,
978
+ metadata: { source: "reflect", qualityRejected: true, ...metadata, ...telemetry },
979
+ }, options.eventsCtx);
980
+ return reflectFailure(run, result, "quality_rejected", message, false);
981
+ };
963
982
  if (judged) {
964
983
  const verdict = await runReflectQualityJudge(run.config, payload.content, assetContent ?? "", feedback, options.chat, {
965
984
  runnerSelectionFrozen: true,
@@ -969,28 +988,33 @@ async function finalizeReflectProposal(args) {
969
988
  onNotices: run.notices.add,
970
989
  });
971
990
  if (!verdict.pass) {
972
- if (options.ref) {
973
- recordLedgerAttempt({ proposalsCtx: options.ctx, eventsCtx: options.eventsCtx }, {
974
- stashDir: run.stash,
975
- ref: options.itemRef ?? options.ref,
976
- source: "reflect",
977
- outcome: "quality_rejected",
978
- detail: verdict.reason,
979
- });
980
- }
981
- appendEvent({
982
- eventType: "reflect_completed",
983
- ref: payload.ref,
984
- metadata: {
985
- source: "reflect",
986
- qualityRejected: true,
987
- qualityScore: verdict.score,
988
- qualityReason: verdict.reason,
989
- ...(verdict.criteria ? { qualityCriteria: verdict.criteria } : {}),
990
- ...telemetry,
991
- },
992
- }, options.eventsCtx);
993
- return reflectFailure(run, result, "quality_rejected", `Reflect proposal quality gate rejected: score=${verdict.score}, reason="${verdict.reason}"`, false);
991
+ return refuse(verdict.reason, {
992
+ qualityScore: verdict.score,
993
+ qualityReason: verdict.reason,
994
+ ...(verdict.criteria ? { qualityCriteria: verdict.criteria } : {}),
995
+ }, `Reflect proposal quality gate rejected: score=${verdict.score}, reason="${verdict.reason}"`);
996
+ }
997
+ }
998
+ // #722: a rewrite of an existing asset must not grade lower on its own retrieval queries.
999
+ if (judged && judge.runner && assetContent !== undefined) {
1000
+ const retrieval = await runRetrievalRegressionGate({
1001
+ ref: payload.ref,
1002
+ before: assetContent,
1003
+ after: payload.content,
1004
+ queries: loadRetrievalQueries({ proposalsCtx: options.ctx, eventsCtx: options.eventsCtx }, payload.ref),
1005
+ runner: judge.runner,
1006
+ ...(options.chat ? { chat: options.chat } : {}),
1007
+ ...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
1008
+ ...(options.signal ? { signal: options.signal } : {}),
1009
+ onNotices: run.notices.add,
1010
+ });
1011
+ if (!retrieval.pass) {
1012
+ return refuse(retrieval.reason, {
1013
+ retrievalRegression: true,
1014
+ retrievalQueries: retrieval.queries,
1015
+ ...(retrieval.oldMean !== undefined ? { retrievalGradeBefore: retrieval.oldMean } : {}),
1016
+ ...(retrieval.newMean !== undefined ? { retrievalGradeAfter: retrieval.newMean } : {}),
1017
+ }, `Reflect proposal refused: ${retrieval.reason}`);
994
1018
  }
995
1019
  }
996
1020
  // A lesson reflect wrote is marked so a later reflect on the same skill does
@@ -0,0 +1,127 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ /**
5
+ * The retrieval regression gate (#722): a reflect rewrite of an existing asset
6
+ * must not grade lower on the queries that actually retrieved it.
7
+ *
8
+ * Measured before it was built: of 60 accepted reflect rewrites judged against
9
+ * their own queries, 14 graded lower (23%, 95% CI 14–35%) and 12 higher. The
10
+ * old and new content are graded one query at a time, blind, with the
11
+ * retrieval-eval judge's prompt (kappa 0.83 against human grades) and its
12
+ * document shape: type, ref, name, description and the first 1,500 characters
13
+ * of the body.
14
+ */
15
+ import relevanceJudgePrompt from "../../assets/prompts/retrieval-relevance-judge.md" with { type: "text" };
16
+ import { parseFrontmatter } from "../../core/asset/frontmatter.js";
17
+ import { parseRefInput } from "../../core/asset/resolve-ref.js";
18
+ import { nonTaskInput } from "../../core/non-task-input.js";
19
+ import { parseEmbeddedJsonResponse } from "../../core/parse.js";
20
+ import { listRetrievalQueries } from "../../indexer/usage/usage-events.js";
21
+ import { readLedgerDb, stripBundle } from "./ledger.js";
22
+ import { callStage } from "./stage.js";
23
+ /** Queries graded per rewrite, as measured. */
24
+ const MAX_QUERIES = 5;
25
+ /** The retrieval suite treats a longer input as a paste, not a query; the measurement did the same. */
26
+ const MAX_QUERY_CHARS = 2000;
27
+ /** The judge sees this much of the body, as in the retrieval eval. */
28
+ const MAX_DOC_CHARS = 1500;
29
+ const GRADE_SCHEMA = {
30
+ type: "object",
31
+ required: ["grade", "reason"],
32
+ additionalProperties: false,
33
+ properties: { grade: { type: "integer", minimum: 0, maximum: 3 }, reason: { type: "string" } },
34
+ };
35
+ /** Up to five distinct task queries, in the given order, whitespace collapsed. */
36
+ export function usableRetrievalQueries(raw) {
37
+ const out = [];
38
+ for (const text of raw) {
39
+ const query = text.replace(/\s+/g, " ").trim();
40
+ if (!query || query.length > MAX_QUERY_CHARS || nonTaskInput(query) || out.includes(query))
41
+ continue;
42
+ out.push(query);
43
+ if (out.length === MAX_QUERIES)
44
+ break;
45
+ }
46
+ return out;
47
+ }
48
+ /** The asset's own retrieval queries from the usage log (none when state.db cannot be read). */
49
+ export function loadRetrievalQueries(access, ref) {
50
+ try {
51
+ return usableRetrievalQueries(readLedgerDb(access, (db) => listRetrievalQueries(db, stripBundle(ref))) ?? []);
52
+ }
53
+ catch {
54
+ return [];
55
+ }
56
+ }
57
+ function judgeDocument(ref, raw) {
58
+ const conceptId = stripBundle(ref);
59
+ const parsed = parseRefInput(conceptId);
60
+ const { data, content } = parseFrontmatter(raw);
61
+ const text = content
62
+ .replace(/\n{3,}/g, "\n\n")
63
+ .trim()
64
+ .slice(0, MAX_DOC_CHARS);
65
+ const name = typeof data.name === "string" && data.name ? data.name : parsed.name;
66
+ const description = typeof data.description === "string" ? data.description : "";
67
+ return [
68
+ "Candidate asset:",
69
+ `Type: ${parsed.type}`,
70
+ `Ref: ${conceptId}`,
71
+ `Name: ${name}`,
72
+ `Description: ${description}`,
73
+ "",
74
+ "Content:",
75
+ text,
76
+ ].join("\n");
77
+ }
78
+ /**
79
+ * Grade `before` and `after` on each query; refuse the rewrite when the new
80
+ * content's mean grade is lower. Fails closed, like the quality judge: a grade
81
+ * that cannot be obtained refuses the rewrite.
82
+ */
83
+ export async function runRetrievalRegressionGate(args) {
84
+ if (args.queries.length === 0)
85
+ return { pass: true, queries: 0, reason: "no retrieval queries to compare on" };
86
+ const documents = { old: judgeDocument(args.ref, args.before), new: judgeDocument(args.ref, args.after) };
87
+ const totals = { old: 0, new: 0 };
88
+ for (const query of args.queries) {
89
+ for (const version of ["old", "new"]) {
90
+ const outcome = await callStage({
91
+ feature: "proposal_quality_gate",
92
+ runner: args.runner,
93
+ system: relevanceJudgePrompt.trim(),
94
+ prompt: `Query: ${query}\n\n${documents[version]}`,
95
+ request: {
96
+ enableThinking: false,
97
+ temperature: 0,
98
+ responseSchema: GRADE_SCHEMA,
99
+ ...(args.timeoutMs !== undefined ? { timeoutMs: args.timeoutMs } : {}),
100
+ ...(args.signal ? { signal: args.signal } : {}),
101
+ ...(args.chat ? { chat: args.chat } : {}),
102
+ },
103
+ ...(args.onNotices ? { onNotices: args.onNotices } : {}),
104
+ });
105
+ const grade = outcome.ok ? parseEmbeddedJsonResponse(outcome.raw)?.grade : undefined;
106
+ if (typeof grade !== "number" || !Number.isInteger(grade) || grade < 0 || grade > 3) {
107
+ return {
108
+ pass: false,
109
+ queries: args.queries.length,
110
+ reason: `retrieval check could not grade the ${version} content${outcome.ok ? "" : ` (${outcome.reason})`}`,
111
+ };
112
+ }
113
+ totals[version] += grade;
114
+ }
115
+ }
116
+ const oldMean = totals.old / args.queries.length;
117
+ const newMean = totals.new / args.queries.length;
118
+ const pass = newMean >= oldMean;
119
+ const grades = `${oldMean.toFixed(2)} -> ${newMean.toFixed(2)} over ${args.queries.length} retrieval ${args.queries.length === 1 ? "query" : "queries"}`;
120
+ return {
121
+ pass,
122
+ queries: args.queries.length,
123
+ oldMean,
124
+ newMean,
125
+ reason: pass ? `retrieval grade ${grades}` : `retrieval regression: grade ${grades}`,
126
+ };
127
+ }
@@ -0,0 +1,77 @@
1
+ // This Source Code Form is subject to the terms of the Mozilla Public
2
+ // License, v. 2.0. If a copy of the MPL was not distributed with this
3
+ // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
+ /**
5
+ * What improve may rework (#986): an asset that retrieval returned inside the
6
+ * usage window, or newly captured material no improve stage has processed.
7
+ *
8
+ * Fresh feedback and an explicit `--scope <ref>` are usage evidence of their
9
+ * own, so the signal-delta and scope lanes need no check. The fallback lanes
10
+ * (proactive maintenance, high salience, forgetting safety) and consolidation
11
+ * pick assets without such evidence, so they pick only inside this scope.
12
+ */
13
+ import fs from "node:fs";
14
+ import { daysToMs } from "../../core/common.js";
15
+ import { warn } from "../../core/warn.js";
16
+ import { listUsedEntryRefs, USAGE_EVENT_RETENTION_DAYS } from "../../indexer/usage/usage-events.js";
17
+ import { listImproveLedgerRows } from "../../storage/repositories/improve-ledger-repository.js";
18
+ import { listProposalRefSources } from "../../storage/repositories/proposals-repository.js";
19
+ import { readLedgerDb, stripBundle } from "./ledger.js";
20
+ /**
21
+ * Ledger and proposal sources that bring material in. Every other source is an
22
+ * improve stage reworking an asset (reflect, distill, consolidate, schema repair).
23
+ */
24
+ const CAPTURE_SOURCES = new Set(["extract", "propose", "remember", "import"]);
25
+ /**
26
+ * Load the scope from state.db. The window is the usage log's retention: the
27
+ * log keeps nothing older, and anything shorter would drop assets read less
28
+ * often than the window. A `.derived` hit counts for its parent memory, whose
29
+ * facts it carries. `undefined` (every asset eligible, as before #986) only
30
+ * when state.db cannot be read.
31
+ */
32
+ export function loadRetrievalScope(access, stashDir) {
33
+ const sinceMs = Date.now() - daysToMs(USAGE_EVENT_RETENTION_DAYS);
34
+ const used = new Set();
35
+ const processed = new Set();
36
+ try {
37
+ readLedgerDb(access, (db) => {
38
+ for (const entryRef of listUsedEntryRefs(db, new Date(sinceMs).toISOString())) {
39
+ const conceptId = stripBundle(entryRef);
40
+ used.add(conceptId);
41
+ if (conceptId.endsWith(".derived"))
42
+ used.add(conceptId.slice(0, -".derived".length));
43
+ }
44
+ if (!stashDir)
45
+ return;
46
+ for (const row of [...listImproveLedgerRows(db, stashDir), ...listProposalRefSources(db, stashDir)]) {
47
+ if (!CAPTURE_SOURCES.has(row.source))
48
+ processed.add(stripBundle(row.ref));
49
+ }
50
+ });
51
+ }
52
+ catch (error) {
53
+ warn(`[improve] usage history unreadable, so every asset stays eligible: ${error instanceof Error ? error.message : String(error)}`);
54
+ return undefined;
55
+ }
56
+ return { used, processed, sinceMs };
57
+ }
58
+ /**
59
+ * Whether improve may rework `ref` (whose file is `filePath`) under `scope`. An
60
+ * unprocessed asset whose file cannot be read is left to the disk check that
61
+ * follows, which reports it as missing.
62
+ */
63
+ export function isInRetrievalScope(scope, ref, filePath) {
64
+ if (!scope)
65
+ return true;
66
+ const conceptId = stripBundle(ref);
67
+ if (scope.used.has(conceptId))
68
+ return true;
69
+ if (scope.processed.has(conceptId))
70
+ return false;
71
+ try {
72
+ return filePath === undefined || fs.statSync(filePath).mtimeMs >= scope.sinceMs;
73
+ }
74
+ catch {
75
+ return true;
76
+ }
77
+ }
@@ -15,17 +15,14 @@
15
15
  * The exported `akmCurate()` API is the single entry point; tests can also
16
16
  * drive `curateSearchResults` with a fixture search response.
17
17
  */
18
- import fs from "node:fs";
19
- import { parseFrontmatter } from "../../core/asset/frontmatter.js";
20
- import { getIndexPassConfig, loadConfig } from "../../core/config/config.js";
18
+ import { loadConfig } from "../../core/config/config.js";
21
19
  import { rethrowIfTestIsolationError, UsageError } from "../../core/errors.js";
22
20
  import { appendEvent } from "../../core/events.js";
21
+ import { nonTaskInput } from "../../core/non-task-input.js";
23
22
  import { redactCredentialPatterns } from "../../core/redaction.js";
24
23
  import { withStateDbTelemetry } from "../../core/state-db.js";
25
- import { enqueueGraphExtraction, hasGraphData } from "../../indexer/db/graph-db.js";
26
24
  import { searchHitContent } from "../../indexer/search/db-search.js";
27
25
  import { copySearchHitAttribution, getSearchHitAttribution, usageEventAttributionMetadata, } from "../../indexer/search/search-attribution.js";
28
- import { findSourceForPath, resolveSourceEntries } from "../../indexer/search/search-source.js";
29
26
  import { insertUsageEvent } from "../../indexer/usage/usage-events.js";
30
27
  import { estimateTokenCount } from "../../llm/embedders/remote.js";
31
28
  import { isLlmFeatureEnabled, tryLlmFeature } from "../../llm/feature-gate.js";
@@ -33,7 +30,6 @@ import { rerankDocuments } from "../../llm/rerank-client.js";
33
30
  import { truncateDescription } from "../../output/shapes/helpers.js";
34
31
  import { TELEMETRY_BUSY_TIMEOUT_MS, withIndexDb } from "../../storage/repositories/index-db.js";
35
32
  import { findEntryIdByRef, getItemRefById } from "../../storage/repositories/index-entries-repository.js";
36
- import { computeBodyHash } from "../../storage/repositories/index-llm-cache-repository.js";
37
33
  import { akmSearch, parseSearchSource } from "./search.js";
38
34
  import { akmShowUnified } from "./show.js";
39
35
  const DEFAULT_CURATE_LIMIT = 4;
@@ -104,6 +100,19 @@ export async function akmCurate(options) {
104
100
  if (!trimmedQuery) {
105
101
  throw new UsageError('A curation query is required. Usage: akm curate "<task or prompt>" [--type <type>] [--limit <n>]', "MISSING_REQUIRED_ARGUMENT");
106
102
  }
103
+ const nonTask = nonTaskInput(trimmedQuery);
104
+ if (nonTask) {
105
+ const abstained = {
106
+ query: options.query,
107
+ summary: `Curate abstained: the input is ${nonTask}, not a task.`,
108
+ items: [],
109
+ tip: 'Nothing was selected on purpose. To curate for it, pass the task itself: akm curate "<what you are trying to do>".',
110
+ };
111
+ if (!options.skipLogging) {
112
+ logCurateEvent(options.query, abstained, options.eventSource, options.attributionProjection);
113
+ }
114
+ return abstained;
115
+ }
107
116
  const limit = options.limit && options.limit > 0 ? options.limit : DEFAULT_CURATE_LIMIT;
108
117
  const source = options.source ?? parseSearchSource("local");
109
118
  const searchResponse = options.searchResponse ??
@@ -201,11 +210,6 @@ async function enrichCuratedStashHit(query, hit, selectedRefs, eventSource) {
201
210
  catch {
202
211
  shown = undefined;
203
212
  }
204
- // #624-P3: when lazy graph extraction is opted in, enqueue an ungraphed
205
- // asset for a later pass to extract. Fire-and-forget, non-blocking, NO inline
206
- // extraction and NO LLM call here. Default-off (flag unset) = byte-identical.
207
- if (shown?.path)
208
- maybeEnqueueLazyGraph(shown.path);
209
213
  const description = shown?.description ?? hit.description;
210
214
  const preview = buildCuratedPreview(shown, hit);
211
215
  const supportRefs = buildCurateSupportRefs(shown?.related?.hits, selectedRefs, hit.ref);
@@ -232,44 +236,6 @@ async function enrichCuratedStashHit(query, hit, selectedRefs, eventSource) {
232
236
  copySearchHitAttribution(hit, item, item.description);
233
237
  return item;
234
238
  }
235
- /**
236
- * #624-P3 — enqueue an ungraphed asset for lazy graph extraction when the
237
- * `index.graph.lazyGraphExtraction` flag is on. Pure side-effect, fully
238
- * best-effort: any failure (config, fs, db) is swallowed so curate never fails
239
- * on it. NO LLM call and NO inline extraction — only a cheap queue insert.
240
- * Default-off (flag unset) returns immediately = byte-identical behavior.
241
- */
242
- function maybeEnqueueLazyGraph(assetPath) {
243
- try {
244
- const config = loadConfig();
245
- if (getIndexPassConfig(config.index, "graph")?.lazyGraphExtraction !== true)
246
- return;
247
- const sources = resolveSourceEntries();
248
- const source = findSourceForPath(assetPath, sources);
249
- const stashRoot = source?.path;
250
- if (!stashRoot)
251
- return;
252
- let raw;
253
- try {
254
- raw = fs.readFileSync(assetPath, "utf8");
255
- }
256
- catch {
257
- return;
258
- }
259
- const body = parseFrontmatter(raw).content.trim();
260
- if (!body)
261
- return;
262
- const bodyHash = computeBodyHash(body);
263
- withIndexDb((db) => {
264
- if (!hasGraphData(db, stashRoot, assetPath)) {
265
- enqueueGraphExtraction(db, stashRoot, assetPath, bodyHash, 0);
266
- }
267
- }, { busyTimeoutMs: TELEMETRY_BUSY_TIMEOUT_MS });
268
- }
269
- catch (err) {
270
- rethrowIfTestIsolationError(err);
271
- }
272
- }
273
239
  function buildCuratedRegistryItem(query, hit) {
274
240
  return {
275
241
  source: "registry",
@@ -27,14 +27,12 @@ import { buildMarkdownLeadContext, fragmentForSelector, MARKDOWN_FRAGMENT_CONTEX
27
27
  import { displayRef, typeNameFromConceptId } from "../../core/asset/resolve-ref.js";
28
28
  import { META_DIR, parseMetaRef, readMetaFile } from "../../core/asset/stash-meta.js";
29
29
  import { asNonEmptyString, isWithin } from "../../core/common.js";
30
- import { getIndexPassConfig, loadConfig } from "../../core/config/config.js";
30
+ import { loadConfig } from "../../core/config/config.js";
31
31
  import { NotFoundError, rethrowIfDataDirUnreadable, rethrowIfTestIsolationError, UsageError } from "../../core/errors.js";
32
32
  import { appendEvent } from "../../core/events.js";
33
33
  import { SCRIPT_EXTENSIONS } from "../../core/recognition-util.js";
34
34
  import { presentationFor } from "../../core/type-presentation.js";
35
35
  import { warn, warnOnce } from "../../core/warn.js";
36
- import { hasGraphData } from "../../indexer/db/graph-db.js";
37
- import { extractGraphForSingleFile } from "../../indexer/graph/graph-extraction.js";
38
36
  import { listRelatedPathsForFile } from "../../indexer/graph/graph-related.js";
39
37
  import { lookupBundleRef, lookupBundleRefWithResolution } from "../../indexer/indexer.js";
40
38
  import { projectMarkdownFragmentContent } from "../../indexer/passes/metadata.js";
@@ -42,11 +40,8 @@ import { ensurePrimaryIndexForRead, resolveReadSources } from "../../indexer/rea
42
40
  import { buildEditHint, findSourceForPath, isEditable, resolveSourceEntries, } from "../../indexer/search/search-source.js";
43
41
  import { recentShowCount, recordShowUsage } from "../../indexer/usage/show-usage.js";
44
42
  import { buildFileContext, buildRenderContext, getRenderer, } from "../../indexer/walk/file-context.js";
45
- import { resolveIndexPassExecution } from "../../llm/index-passes.js";
46
43
  import { resolveSourcesForOrigin } from "../../registry/origin-resolve.js";
47
- import { resolveStorageLocations } from "../../storage/locations.js";
48
- import { closeDatabase, openExistingDatabase } from "../../storage/repositories/index-connection.js";
49
- import { TELEMETRY_BUSY_TIMEOUT_MS, withIndexDb } from "../../storage/repositories/index-db.js";
44
+ import { withIndexDb } from "../../storage/repositories/index-db.js";
50
45
  import { getIndexedMarkdownFragment } from "../../storage/repositories/index-fts-repository.js";
51
46
  import { getCurrentWorkflowScopeKey } from "../../workflows/authoring/scope-key.js";
52
47
  import { buildWorkflowAction } from "../../workflows/renderer.js";
@@ -405,15 +400,6 @@ export async function showLocal(input) {
405
400
  if (activeRun) {
406
401
  fullResponse.activeRun = activeRun;
407
402
  }
408
- // #624-P3: opt-in inline graph extraction. Default OFF — when the flag is
409
- // unset this whole block is skipped (no hasGraphData check, no LLM call), so
410
- // behavior is byte-identical to today. When ON, it extracts graph data for an
411
- // ungraphed asset, but ONLY when a model is configured (model-available
412
- // guard) and ALWAYS bounded by a 30s timeout so `show` can never hang. Any
413
- // timeout/model-unavailable/error path returns the response unchanged.
414
- if (getIndexPassConfig(config.index, "graph")?.lazyGraphExtraction === true) {
415
- await maybeExtractGraphInline(config, sourceStashDir, assetPath);
416
- }
417
403
  if (input.detail === "brief") {
418
404
  return buildBriefResponse(fullResponse, assetPath);
419
405
  }
@@ -470,71 +456,6 @@ function findUnrecognizedScriptSource(assetParts, sources) {
470
456
  }
471
457
  return undefined;
472
458
  }
473
- /**
474
- * #624-P3 — opt-in inline graph extraction for `akm show`. Best-effort and
475
- * timeout-bounded: never throws, never hangs, never mutates the response.
476
- *
477
- * Preconditions (caller already checked the flag): a model must be configured
478
- * (model-available guard via {@link resolveIndexPassExecution}) and the asset
479
- * must be ungraphed ({@link hasGraphData}). Extraction races a 30s timeout so
480
- * `show` cannot block on a slow provider; any timeout/error/missing-model path
481
- * is swallowed and `show` returns its already-assembled response unchanged.
482
- */
483
- async function maybeExtractGraphInline(config, sourceStashDir, assetPath) {
484
- try {
485
- // Resolve readiness and the symbolic runner once. The inline dispatch must
486
- // consume this same snapshot even if models.json changes while show runs.
487
- const graphExecution = resolveIndexPassExecution("graph", config);
488
- if (!graphExecution.runner)
489
- return;
490
- const emittedNoticeKeys = new Set();
491
- const reportNotices = (notices) => {
492
- for (const notice of notices) {
493
- const key = JSON.stringify(notice);
494
- if (emittedNoticeKeys.has(key))
495
- continue;
496
- emittedNoticeKeys.add(key);
497
- const field = typeof notice.field === "string" ? ` field=${notice.field}` : "";
498
- warn(`[akm] lazy graph extraction notice ${notice.code} adapter=${notice.adapter}${field}: ${notice.message}`);
499
- }
500
- };
501
- reportNotices(graphExecution.notices);
502
- let alreadyGraphed = false;
503
- withIndexDb((db) => {
504
- alreadyGraphed = hasGraphData(db, sourceStashDir, assetPath);
505
- }, { busyTimeoutMs: TELEMETRY_BUSY_TIMEOUT_MS });
506
- if (alreadyGraphed)
507
- return;
508
- // Open the db for the async extraction ourselves: `withIndexDb` is
509
- // synchronous and would close the connection the instant the async fn
510
- // returns its Promise (before extraction completes). Close it explicitly
511
- // after the race settles instead.
512
- const db = openExistingDatabase(resolveStorageLocations().indexDb);
513
- let timer;
514
- const timeout = new Promise((resolve) => {
515
- timer = setTimeout(resolve, 30_000);
516
- });
517
- try {
518
- await Promise.race([
519
- extractGraphForSingleFile(db, sourceStashDir, assetPath, {
520
- config,
521
- llmRunner: graphExecution.runner,
522
- onNotices: reportNotices,
523
- }),
524
- timeout,
525
- ]);
526
- }
527
- finally {
528
- if (timer)
529
- clearTimeout(timer);
530
- closeDatabase(db);
531
- }
532
- }
533
- catch (err) {
534
- rethrowIfTestIsolationError(err);
535
- // Any other failure: silently return the unchanged show response.
536
- }
537
- }
538
459
  /**
539
460
  * Minimal `show`: ref → indexer lookup → file contents. Used by callers that
540
461
  * just need the raw file (e.g. clone, write-source) and don't want the full