akm-cli 0.9.17-alpha.5 → 0.9.17-alpha.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +281 -0
- package/STABILITY.md +2 -2
- package/dist/akm +7 -7
- package/dist/assets/prompts/retrieval-relevance-judge.md +6 -0
- package/dist/commands/improve/consolidate.js +11 -0
- package/dist/commands/improve/execution.js +5 -5
- package/dist/commands/improve/improve-cli.js +27 -7
- package/dist/commands/improve/improve-strategies.js +3 -0
- package/dist/commands/improve/ledger.js +7 -3
- package/dist/commands/improve/loop-stages.js +5 -4
- package/dist/commands/improve/preparation.js +40 -10
- package/dist/commands/improve/reflect.js +46 -22
- package/dist/commands/improve/retrieval-gate.js +127 -0
- package/dist/commands/improve/retrieval-scope.js +77 -0
- package/dist/commands/read/curate.js +15 -49
- package/dist/commands/read/show.js +2 -81
- package/dist/commands/tasks/tasks-cli.js +10 -12
- package/dist/commands/tasks/tasks.js +57 -56
- package/dist/commands/tasks/validate.js +27 -46
- package/dist/core/adapter/adapters/akm-task-adapter.js +29 -8
- package/dist/core/config/config.js +1 -1
- package/dist/core/config/schema/improve-processes.js +4 -3
- package/dist/core/config/schema/index-config.js +4 -23
- package/dist/core/improve-result.js +4 -1
- package/dist/core/non-task-input.js +20 -0
- package/dist/core/paths.js +0 -4
- package/dist/indexer/db/graph-db.js +8 -81
- package/dist/indexer/graph/graph-extraction.js +74 -227
- package/dist/indexer/graph/graph-related.js +5 -4
- package/dist/indexer/indexer.js +1 -3
- package/dist/indexer/search/db-search.js +10 -1
- package/dist/indexer/usage/usage-events.js +34 -0
- package/dist/llm/feature-gate.js +0 -3
- package/dist/llm/graph-extract.js +81 -41
- package/dist/scripts/akm-migrate-node.js +7148 -7174
- package/dist/scripts/akm-migrate.js +8718 -8744
- package/dist/storage/repositories/index-entries-repository.js +5 -7
- package/dist/storage/repositories/index-schema.js +16 -34
- package/dist/storage/repositories/proposals-repository.js +4 -0
- package/dist/tasks/backends/cron.js +80 -43
- package/dist/tasks/backends/launchd.js +28 -15
- package/dist/tasks/backends/schtasks.js +25 -10
- package/dist/tasks/run/load-task.js +1 -1
- package/dist/tasks/scheduler-binding.js +4 -2
- package/dist/tasks/scheduler-invocation.js +127 -235
- package/dist/tasks/scheduler-sync.js +13 -8
- package/dist/tasks/source/parse-task-source.js +22 -126
- package/dist/tasks/source/task-to-v3.js +1 -55
- package/dist/tasks/source/task-to-v4.js +1 -13
- package/docs/migration/v0.9.1-to-v0.9.2.md +7 -3
- package/docs/reference/cli.md +13 -4
- package/docs/reference/configuration.md +12 -0
- package/docs/reference/tasks.md +58 -40
- package/package.json +1 -1
- package/schemas/akm-config.json +0 -6
|
@@ -23,7 +23,7 @@ import { ConfigError, rethrowIfTestIsolationError } from "../../core/errors.js";
|
|
|
23
23
|
import { appendEvent, readEvents } from "../../core/events.js";
|
|
24
24
|
import { withStateDb } from "../../core/state-db.js";
|
|
25
25
|
import { info, warn } from "../../core/warn.js";
|
|
26
|
-
import { countUsageEventsByType } from "../../indexer/usage/usage-events.js";
|
|
26
|
+
import { countUsageEventsByType, USAGE_EVENT_RETENTION_DAYS } from "../../indexer/usage/usage-events.js";
|
|
27
27
|
import { getAvailableHarnesses } from "../../integrations/session-logs/index.js";
|
|
28
28
|
import { getZeroResultSearches } from "../../storage/repositories/index-entries-repository.js";
|
|
29
29
|
import { getRetrievalCounts } from "../../storage/repositories/index-utility-repository.js";
|
|
@@ -41,6 +41,7 @@ import { applyMemoryCleanup } from "./memory/memory-improve.js";
|
|
|
41
41
|
import { getAllAssetOutcomes, getAssetOutcome, getOutcomeScoresByRef, OUTCOME_SCORE_MAX, outcomeScoreToSalience, projectAssetOutcome, updateAssetOutcome, } from "./outcome-loop.js";
|
|
42
42
|
import { projectMemoryCleanup, selectEffectiveImproveRefs } from "./planner.js";
|
|
43
43
|
import { DEFAULT_DUE_DAYS, DEFAULT_MAX_PER_RUN, selectProactiveMaintenanceRefs } from "./proactive-maintenance.js";
|
|
44
|
+
import { isInRetrievalScope, loadRetrievalScope } from "./retrieval-scope.js";
|
|
44
45
|
import { buildRankChangeReport, computeSalience, getAllRankScores, getAssetSalience, getLastUseMsByRef, isContentEncodingRow, SALIENCE_NO_OP_DAMPEN_FACTOR, SALIENCE_NO_OP_DAMPEN_THRESHOLD, upsertAssetSalience, } from "./salience.js";
|
|
45
46
|
import { attributeStage, errMessage } from "./stage.js";
|
|
46
47
|
/** The candidate's durable state key (salience, outcome, ledger). */
|
|
@@ -663,12 +664,23 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
|
|
|
663
664
|
const processableRefs = [...partition.eligibleRefs, ...partition.distillOnlyRefs];
|
|
664
665
|
const signalFiltered = processableRefs.filter((c) => snapshot.feedback.get(c.ref)?.hasSignal === true);
|
|
665
666
|
const signalBearingSet = new Set(signalFiltered.map((r) => r.ref));
|
|
666
|
-
|
|
667
|
+
// The fallback lanes (proactive, high salience, forgetting safety) have no
|
|
668
|
+
// usage evidence of their own: they pick only what retrieval returned or new
|
|
669
|
+
// material improve never processed (#986). Evaluated once per candidate.
|
|
670
|
+
const fallbackEligible = postCleanupRefs.filter((c) => !validationFailureRefs.has(c.ref));
|
|
671
|
+
const allowFallbacks = options.requireFeedbackSignal !== true;
|
|
672
|
+
const retrievalScope = scope.mode === "ref" || !allowFallbacks
|
|
673
|
+
? undefined
|
|
674
|
+
: loadRetrievalScope({ eventsCtx, ...(persist ? {} : { readOnly: true }) }, primaryStashDir ?? options.stashDir);
|
|
675
|
+
const unscoped = new Set(fallbackEligible.filter((c) => !isInRetrievalScope(retrievalScope, c.ref, c.filePath)).map((c) => c.ref));
|
|
676
|
+
const noFeedbackPool = dedupeRefs([
|
|
667
677
|
...processableRefs.filter((r) => !signalBearingSet.has(r.ref)),
|
|
668
678
|
...partition.noFeedbackPool,
|
|
669
679
|
]);
|
|
680
|
+
const noFeedbackCandidates = noFeedbackPool.filter((r) => !unscoped.has(r.ref));
|
|
681
|
+
// Only a ref no fallback lane may pick anymore is charged to the retrieval gate.
|
|
682
|
+
const outOfScope = new Set(noFeedbackPool.filter((r) => unscoped.has(r.ref)).map((r) => r.ref));
|
|
670
683
|
const retrieval = fetchRetrievalSignals(options, signalFiltered, noFeedbackCandidates, eventsCtx, persist);
|
|
671
|
-
const allowFallbacks = options.requireFeedbackSignal !== true;
|
|
672
684
|
const proactive = allowFallbacks
|
|
673
685
|
? selectProactiveMaintenanceLane(args, noFeedbackCandidates, snapshot, retrieval, persist)
|
|
674
686
|
: { proactiveRefs: [] };
|
|
@@ -692,10 +704,10 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
|
|
|
692
704
|
sourceByRef.set(r.ref, "scope");
|
|
693
705
|
for (const r of mergedRefs)
|
|
694
706
|
r.eligibilitySource = sourceByRef.get(r.ref) ?? "unknown";
|
|
695
|
-
// Forgetting safety may only reuse this plan's own surviving objects
|
|
696
|
-
// never a ref whose reflect window is still open.
|
|
697
|
-
const
|
|
698
|
-
|
|
707
|
+
// Forgetting safety may only reuse this plan's own surviving objects inside
|
|
708
|
+
// the retrieval scope, and never a ref whose reflect window is still open.
|
|
709
|
+
const forgettingEligible = fallbackEligible.filter((c) => !unscoped.has(c.ref) &&
|
|
710
|
+
!isLedgerBlocked(ledgerRowFor(snapshot.ledger, "reflect", c.ref, c.itemRef), snapshot.nowIso));
|
|
699
711
|
const scored = scoreSalience(args, mergedRefs, snapshot.feedback, retrieval.retrievalCounts, persist);
|
|
700
712
|
mergedRefs = applyForgettingSafety({
|
|
701
713
|
pendingForgettingRefs: scored.pendingForgettingRefs,
|
|
@@ -739,28 +751,46 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
|
|
|
739
751
|
// Skip observability waits until every fallback lane has finalized the
|
|
740
752
|
// survivors, so a rescued ref is never also reported skipped.
|
|
741
753
|
const survivors = new Set(sorted.map((c) => c.ref));
|
|
742
|
-
const
|
|
754
|
+
const skipped = fallbackEligible.filter((c) => !survivors.has(c.ref));
|
|
755
|
+
const retrievalSkipped = skipped.filter((c) => outOfScope.has(c.ref));
|
|
756
|
+
const signalSkipped = skipped.filter((c) => !outOfScope.has(c.ref));
|
|
743
757
|
for (const ref of partition.distillCooledRefs) {
|
|
744
758
|
actions.push({ ref, mode: "distill-skipped", result: { ok: true, reason: "distill signal-delta" } });
|
|
745
759
|
if (persist)
|
|
746
760
|
recordImproveSkip(eventsCtx, ref, { reason: "distill_no_new_signal" });
|
|
747
761
|
}
|
|
748
|
-
for (const candidate of
|
|
762
|
+
for (const candidate of skipped) {
|
|
749
763
|
actions.push({
|
|
750
764
|
ref: candidate.ref,
|
|
751
765
|
mode: "distill-skipped",
|
|
752
|
-
result: {
|
|
766
|
+
result: {
|
|
767
|
+
ok: true,
|
|
768
|
+
reason: outOfScope.has(candidate.ref)
|
|
769
|
+
? "not retrieved inside the usage window"
|
|
770
|
+
: "no new signal since last proposal",
|
|
771
|
+
},
|
|
753
772
|
});
|
|
754
773
|
}
|
|
755
774
|
if (persist && signalSkipped.length > 0) {
|
|
756
775
|
recordImproveSkip(eventsCtx, undefined, { reason: "no_new_signal", count: signalSkipped.length });
|
|
757
776
|
}
|
|
777
|
+
if (persist && retrievalSkipped.length > 0) {
|
|
778
|
+
recordImproveSkip(eventsCtx, undefined, { reason: "not_retrieved", count: retrievalSkipped.length });
|
|
779
|
+
}
|
|
758
780
|
const blocked = signalSkipped.length + partition.distillOnlyRefs.length;
|
|
759
781
|
if (blocked > 0) {
|
|
760
782
|
info(`[improve] ${blocked} of ${partition.preCooldownCount} indexed refs blocked by reflect signal-delta ` +
|
|
761
783
|
`(${signalSkipped.length} fully skipped, ${partition.distillOnlyRefs.length} routed to distill-only)`);
|
|
762
784
|
}
|
|
785
|
+
if (retrievalSkipped.length > 0) {
|
|
786
|
+
info(`[improve] ${retrievalSkipped.length} refs left out: not retrieved in the last ${USAGE_EVENT_RETENTION_DAYS} days, and not new material`);
|
|
787
|
+
}
|
|
763
788
|
const gates = [
|
|
789
|
+
{
|
|
790
|
+
name: "retrieval",
|
|
791
|
+
removed: retrievalSkipped.length,
|
|
792
|
+
reason: `no feedback, and neither returned by search, curate or show in the last ${USAGE_EVENT_RETENTION_DAYS} days nor new material improve never processed`,
|
|
793
|
+
},
|
|
764
794
|
{
|
|
765
795
|
name: "signal",
|
|
766
796
|
removed: signalSkipped.length,
|
|
@@ -43,6 +43,7 @@ import { findAssetFilePath } from "./eligibility.js";
|
|
|
43
43
|
import { resolveImproveLlmExecution } from "./execution.js";
|
|
44
44
|
import { recordLedgerAttempt } from "./ledger.js";
|
|
45
45
|
import { classifyReflectChange, splitFrontmatter } from "./reflect-noise.js";
|
|
46
|
+
import { loadRetrievalQueries, runRetrievalRegressionGate } from "./retrieval-gate.js";
|
|
46
47
|
import { callStage, mintProposal, noticeSet, rejectedProposalContext, runReflectQualityJudge, } from "./stage.js";
|
|
47
48
|
const MAX_FEEDBACK_LINES = 10;
|
|
48
49
|
const MAX_GLOBAL_FEEDBACK_LINES = 20;
|
|
@@ -960,6 +961,24 @@ async function finalizeReflectProposal(args) {
|
|
|
960
961
|
}
|
|
961
962
|
const flagged = Boolean(sanitized.sizeGuardRatio || sanitized.truncationMarkerLeaked);
|
|
962
963
|
const judged = judge.enabled && !flagged;
|
|
964
|
+
/** A judge refused the revision: record it for the ledger's rejection window and stop. */
|
|
965
|
+
const refuse = (detail, metadata, message) => {
|
|
966
|
+
if (options.ref) {
|
|
967
|
+
recordLedgerAttempt({ proposalsCtx: options.ctx, eventsCtx: options.eventsCtx }, {
|
|
968
|
+
stashDir: run.stash,
|
|
969
|
+
ref: options.itemRef ?? options.ref,
|
|
970
|
+
source: "reflect",
|
|
971
|
+
outcome: "quality_rejected",
|
|
972
|
+
detail,
|
|
973
|
+
});
|
|
974
|
+
}
|
|
975
|
+
appendEvent({
|
|
976
|
+
eventType: "reflect_completed",
|
|
977
|
+
ref: payload.ref,
|
|
978
|
+
metadata: { source: "reflect", qualityRejected: true, ...metadata, ...telemetry },
|
|
979
|
+
}, options.eventsCtx);
|
|
980
|
+
return reflectFailure(run, result, "quality_rejected", message, false);
|
|
981
|
+
};
|
|
963
982
|
if (judged) {
|
|
964
983
|
const verdict = await runReflectQualityJudge(run.config, payload.content, assetContent ?? "", feedback, options.chat, {
|
|
965
984
|
runnerSelectionFrozen: true,
|
|
@@ -969,28 +988,33 @@ async function finalizeReflectProposal(args) {
|
|
|
969
988
|
onNotices: run.notices.add,
|
|
970
989
|
});
|
|
971
990
|
if (!verdict.pass) {
|
|
972
|
-
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
|
|
978
|
-
|
|
979
|
-
|
|
980
|
-
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
return
|
|
991
|
+
return refuse(verdict.reason, {
|
|
992
|
+
qualityScore: verdict.score,
|
|
993
|
+
qualityReason: verdict.reason,
|
|
994
|
+
...(verdict.criteria ? { qualityCriteria: verdict.criteria } : {}),
|
|
995
|
+
}, `Reflect proposal quality gate rejected: score=${verdict.score}, reason="${verdict.reason}"`);
|
|
996
|
+
}
|
|
997
|
+
}
|
|
998
|
+
// #722: a rewrite of an existing asset must not grade lower on its own retrieval queries.
|
|
999
|
+
if (judged && judge.runner && assetContent !== undefined) {
|
|
1000
|
+
const retrieval = await runRetrievalRegressionGate({
|
|
1001
|
+
ref: payload.ref,
|
|
1002
|
+
before: assetContent,
|
|
1003
|
+
after: payload.content,
|
|
1004
|
+
queries: loadRetrievalQueries({ proposalsCtx: options.ctx, eventsCtx: options.eventsCtx }, payload.ref),
|
|
1005
|
+
runner: judge.runner,
|
|
1006
|
+
...(options.chat ? { chat: options.chat } : {}),
|
|
1007
|
+
...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
|
|
1008
|
+
...(options.signal ? { signal: options.signal } : {}),
|
|
1009
|
+
onNotices: run.notices.add,
|
|
1010
|
+
});
|
|
1011
|
+
if (!retrieval.pass) {
|
|
1012
|
+
return refuse(retrieval.reason, {
|
|
1013
|
+
retrievalRegression: true,
|
|
1014
|
+
retrievalQueries: retrieval.queries,
|
|
1015
|
+
...(retrieval.oldMean !== undefined ? { retrievalGradeBefore: retrieval.oldMean } : {}),
|
|
1016
|
+
...(retrieval.newMean !== undefined ? { retrievalGradeAfter: retrieval.newMean } : {}),
|
|
1017
|
+
}, `Reflect proposal refused: ${retrieval.reason}`);
|
|
994
1018
|
}
|
|
995
1019
|
}
|
|
996
1020
|
// A lesson reflect wrote is marked so a later reflect on the same skill does
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/**
|
|
5
|
+
* The retrieval regression gate (#722): a reflect rewrite of an existing asset
|
|
6
|
+
* must not grade lower on the queries that actually retrieved it.
|
|
7
|
+
*
|
|
8
|
+
* Measured before it was built: of 60 accepted reflect rewrites judged against
|
|
9
|
+
* their own queries, 14 graded lower (23%, 95% CI 14–35%) and 12 higher. The
|
|
10
|
+
* old and new content are graded one query at a time, blind, with the
|
|
11
|
+
* retrieval-eval judge's prompt (kappa 0.83 against human grades) and its
|
|
12
|
+
* document shape: type, ref, name, description and the first 1,500 characters
|
|
13
|
+
* of the body.
|
|
14
|
+
*/
|
|
15
|
+
import relevanceJudgePrompt from "../../assets/prompts/retrieval-relevance-judge.md" with { type: "text" };
|
|
16
|
+
import { parseFrontmatter } from "../../core/asset/frontmatter.js";
|
|
17
|
+
import { parseRefInput } from "../../core/asset/resolve-ref.js";
|
|
18
|
+
import { nonTaskInput } from "../../core/non-task-input.js";
|
|
19
|
+
import { parseEmbeddedJsonResponse } from "../../core/parse.js";
|
|
20
|
+
import { listRetrievalQueries } from "../../indexer/usage/usage-events.js";
|
|
21
|
+
import { readLedgerDb, stripBundle } from "./ledger.js";
|
|
22
|
+
import { callStage } from "./stage.js";
|
|
23
|
+
/** Queries graded per rewrite, as measured. */
|
|
24
|
+
const MAX_QUERIES = 5;
|
|
25
|
+
/** The retrieval suite treats a longer input as a paste, not a query; the measurement did the same. */
|
|
26
|
+
const MAX_QUERY_CHARS = 2000;
|
|
27
|
+
/** The judge sees this much of the body, as in the retrieval eval. */
|
|
28
|
+
const MAX_DOC_CHARS = 1500;
|
|
29
|
+
const GRADE_SCHEMA = {
|
|
30
|
+
type: "object",
|
|
31
|
+
required: ["grade", "reason"],
|
|
32
|
+
additionalProperties: false,
|
|
33
|
+
properties: { grade: { type: "integer", minimum: 0, maximum: 3 }, reason: { type: "string" } },
|
|
34
|
+
};
|
|
35
|
+
/** Up to five distinct task queries, in the given order, whitespace collapsed. */
|
|
36
|
+
export function usableRetrievalQueries(raw) {
|
|
37
|
+
const out = [];
|
|
38
|
+
for (const text of raw) {
|
|
39
|
+
const query = text.replace(/\s+/g, " ").trim();
|
|
40
|
+
if (!query || query.length > MAX_QUERY_CHARS || nonTaskInput(query) || out.includes(query))
|
|
41
|
+
continue;
|
|
42
|
+
out.push(query);
|
|
43
|
+
if (out.length === MAX_QUERIES)
|
|
44
|
+
break;
|
|
45
|
+
}
|
|
46
|
+
return out;
|
|
47
|
+
}
|
|
48
|
+
/** The asset's own retrieval queries from the usage log (none when state.db cannot be read). */
|
|
49
|
+
export function loadRetrievalQueries(access, ref) {
|
|
50
|
+
try {
|
|
51
|
+
return usableRetrievalQueries(readLedgerDb(access, (db) => listRetrievalQueries(db, stripBundle(ref))) ?? []);
|
|
52
|
+
}
|
|
53
|
+
catch {
|
|
54
|
+
return [];
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
function judgeDocument(ref, raw) {
|
|
58
|
+
const conceptId = stripBundle(ref);
|
|
59
|
+
const parsed = parseRefInput(conceptId);
|
|
60
|
+
const { data, content } = parseFrontmatter(raw);
|
|
61
|
+
const text = content
|
|
62
|
+
.replace(/\n{3,}/g, "\n\n")
|
|
63
|
+
.trim()
|
|
64
|
+
.slice(0, MAX_DOC_CHARS);
|
|
65
|
+
const name = typeof data.name === "string" && data.name ? data.name : parsed.name;
|
|
66
|
+
const description = typeof data.description === "string" ? data.description : "";
|
|
67
|
+
return [
|
|
68
|
+
"Candidate asset:",
|
|
69
|
+
`Type: ${parsed.type}`,
|
|
70
|
+
`Ref: ${conceptId}`,
|
|
71
|
+
`Name: ${name}`,
|
|
72
|
+
`Description: ${description}`,
|
|
73
|
+
"",
|
|
74
|
+
"Content:",
|
|
75
|
+
text,
|
|
76
|
+
].join("\n");
|
|
77
|
+
}
|
|
78
|
+
/**
|
|
79
|
+
* Grade `before` and `after` on each query; refuse the rewrite when the new
|
|
80
|
+
* content's mean grade is lower. Fails closed, like the quality judge: a grade
|
|
81
|
+
* that cannot be obtained refuses the rewrite.
|
|
82
|
+
*/
|
|
83
|
+
export async function runRetrievalRegressionGate(args) {
|
|
84
|
+
if (args.queries.length === 0)
|
|
85
|
+
return { pass: true, queries: 0, reason: "no retrieval queries to compare on" };
|
|
86
|
+
const documents = { old: judgeDocument(args.ref, args.before), new: judgeDocument(args.ref, args.after) };
|
|
87
|
+
const totals = { old: 0, new: 0 };
|
|
88
|
+
for (const query of args.queries) {
|
|
89
|
+
for (const version of ["old", "new"]) {
|
|
90
|
+
const outcome = await callStage({
|
|
91
|
+
feature: "proposal_quality_gate",
|
|
92
|
+
runner: args.runner,
|
|
93
|
+
system: relevanceJudgePrompt.trim(),
|
|
94
|
+
prompt: `Query: ${query}\n\n${documents[version]}`,
|
|
95
|
+
request: {
|
|
96
|
+
enableThinking: false,
|
|
97
|
+
temperature: 0,
|
|
98
|
+
responseSchema: GRADE_SCHEMA,
|
|
99
|
+
...(args.timeoutMs !== undefined ? { timeoutMs: args.timeoutMs } : {}),
|
|
100
|
+
...(args.signal ? { signal: args.signal } : {}),
|
|
101
|
+
...(args.chat ? { chat: args.chat } : {}),
|
|
102
|
+
},
|
|
103
|
+
...(args.onNotices ? { onNotices: args.onNotices } : {}),
|
|
104
|
+
});
|
|
105
|
+
const grade = outcome.ok ? parseEmbeddedJsonResponse(outcome.raw)?.grade : undefined;
|
|
106
|
+
if (typeof grade !== "number" || !Number.isInteger(grade) || grade < 0 || grade > 3) {
|
|
107
|
+
return {
|
|
108
|
+
pass: false,
|
|
109
|
+
queries: args.queries.length,
|
|
110
|
+
reason: `retrieval check could not grade the ${version} content${outcome.ok ? "" : ` (${outcome.reason})`}`,
|
|
111
|
+
};
|
|
112
|
+
}
|
|
113
|
+
totals[version] += grade;
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
const oldMean = totals.old / args.queries.length;
|
|
117
|
+
const newMean = totals.new / args.queries.length;
|
|
118
|
+
const pass = newMean >= oldMean;
|
|
119
|
+
const grades = `${oldMean.toFixed(2)} -> ${newMean.toFixed(2)} over ${args.queries.length} retrieval ${args.queries.length === 1 ? "query" : "queries"}`;
|
|
120
|
+
return {
|
|
121
|
+
pass,
|
|
122
|
+
queries: args.queries.length,
|
|
123
|
+
oldMean,
|
|
124
|
+
newMean,
|
|
125
|
+
reason: pass ? `retrieval grade ${grades}` : `retrieval regression: grade ${grades}`,
|
|
126
|
+
};
|
|
127
|
+
}
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/**
|
|
5
|
+
* What improve may rework (#986): an asset that retrieval returned inside the
|
|
6
|
+
* usage window, or newly captured material no improve stage has processed.
|
|
7
|
+
*
|
|
8
|
+
* Fresh feedback and an explicit `--scope <ref>` are usage evidence of their
|
|
9
|
+
* own, so the signal-delta and scope lanes need no check. The fallback lanes
|
|
10
|
+
* (proactive maintenance, high salience, forgetting safety) and consolidation
|
|
11
|
+
* pick assets without such evidence, so they pick only inside this scope.
|
|
12
|
+
*/
|
|
13
|
+
import fs from "node:fs";
|
|
14
|
+
import { daysToMs } from "../../core/common.js";
|
|
15
|
+
import { warn } from "../../core/warn.js";
|
|
16
|
+
import { listUsedEntryRefs, USAGE_EVENT_RETENTION_DAYS } from "../../indexer/usage/usage-events.js";
|
|
17
|
+
import { listImproveLedgerRows } from "../../storage/repositories/improve-ledger-repository.js";
|
|
18
|
+
import { listProposalRefSources } from "../../storage/repositories/proposals-repository.js";
|
|
19
|
+
import { readLedgerDb, stripBundle } from "./ledger.js";
|
|
20
|
+
/**
|
|
21
|
+
* Ledger and proposal sources that bring material in. Every other source is an
|
|
22
|
+
* improve stage reworking an asset (reflect, distill, consolidate, schema repair).
|
|
23
|
+
*/
|
|
24
|
+
const CAPTURE_SOURCES = new Set(["extract", "propose", "remember", "import"]);
|
|
25
|
+
/**
|
|
26
|
+
* Load the scope from state.db. The window is the usage log's retention: the
|
|
27
|
+
* log keeps nothing older, and anything shorter would drop assets read less
|
|
28
|
+
* often than the window. A `.derived` hit counts for its parent memory, whose
|
|
29
|
+
* facts it carries. `undefined` (every asset eligible, as before #986) only
|
|
30
|
+
* when state.db cannot be read.
|
|
31
|
+
*/
|
|
32
|
+
export function loadRetrievalScope(access, stashDir) {
|
|
33
|
+
const sinceMs = Date.now() - daysToMs(USAGE_EVENT_RETENTION_DAYS);
|
|
34
|
+
const used = new Set();
|
|
35
|
+
const processed = new Set();
|
|
36
|
+
try {
|
|
37
|
+
readLedgerDb(access, (db) => {
|
|
38
|
+
for (const entryRef of listUsedEntryRefs(db, new Date(sinceMs).toISOString())) {
|
|
39
|
+
const conceptId = stripBundle(entryRef);
|
|
40
|
+
used.add(conceptId);
|
|
41
|
+
if (conceptId.endsWith(".derived"))
|
|
42
|
+
used.add(conceptId.slice(0, -".derived".length));
|
|
43
|
+
}
|
|
44
|
+
if (!stashDir)
|
|
45
|
+
return;
|
|
46
|
+
for (const row of [...listImproveLedgerRows(db, stashDir), ...listProposalRefSources(db, stashDir)]) {
|
|
47
|
+
if (!CAPTURE_SOURCES.has(row.source))
|
|
48
|
+
processed.add(stripBundle(row.ref));
|
|
49
|
+
}
|
|
50
|
+
});
|
|
51
|
+
}
|
|
52
|
+
catch (error) {
|
|
53
|
+
warn(`[improve] usage history unreadable, so every asset stays eligible: ${error instanceof Error ? error.message : String(error)}`);
|
|
54
|
+
return undefined;
|
|
55
|
+
}
|
|
56
|
+
return { used, processed, sinceMs };
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* Whether improve may rework `ref` (whose file is `filePath`) under `scope`. An
|
|
60
|
+
* unprocessed asset whose file cannot be read is left to the disk check that
|
|
61
|
+
* follows, which reports it as missing.
|
|
62
|
+
*/
|
|
63
|
+
export function isInRetrievalScope(scope, ref, filePath) {
|
|
64
|
+
if (!scope)
|
|
65
|
+
return true;
|
|
66
|
+
const conceptId = stripBundle(ref);
|
|
67
|
+
if (scope.used.has(conceptId))
|
|
68
|
+
return true;
|
|
69
|
+
if (scope.processed.has(conceptId))
|
|
70
|
+
return false;
|
|
71
|
+
try {
|
|
72
|
+
return filePath === undefined || fs.statSync(filePath).mtimeMs >= scope.sinceMs;
|
|
73
|
+
}
|
|
74
|
+
catch {
|
|
75
|
+
return true;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
@@ -15,17 +15,14 @@
|
|
|
15
15
|
* The exported `akmCurate()` API is the single entry point; tests can also
|
|
16
16
|
* drive `curateSearchResults` with a fixture search response.
|
|
17
17
|
*/
|
|
18
|
-
import
|
|
19
|
-
import { parseFrontmatter } from "../../core/asset/frontmatter.js";
|
|
20
|
-
import { getIndexPassConfig, loadConfig } from "../../core/config/config.js";
|
|
18
|
+
import { loadConfig } from "../../core/config/config.js";
|
|
21
19
|
import { rethrowIfTestIsolationError, UsageError } from "../../core/errors.js";
|
|
22
20
|
import { appendEvent } from "../../core/events.js";
|
|
21
|
+
import { nonTaskInput } from "../../core/non-task-input.js";
|
|
23
22
|
import { redactCredentialPatterns } from "../../core/redaction.js";
|
|
24
23
|
import { withStateDbTelemetry } from "../../core/state-db.js";
|
|
25
|
-
import { enqueueGraphExtraction, hasGraphData } from "../../indexer/db/graph-db.js";
|
|
26
24
|
import { searchHitContent } from "../../indexer/search/db-search.js";
|
|
27
25
|
import { copySearchHitAttribution, getSearchHitAttribution, usageEventAttributionMetadata, } from "../../indexer/search/search-attribution.js";
|
|
28
|
-
import { findSourceForPath, resolveSourceEntries } from "../../indexer/search/search-source.js";
|
|
29
26
|
import { insertUsageEvent } from "../../indexer/usage/usage-events.js";
|
|
30
27
|
import { estimateTokenCount } from "../../llm/embedders/remote.js";
|
|
31
28
|
import { isLlmFeatureEnabled, tryLlmFeature } from "../../llm/feature-gate.js";
|
|
@@ -33,7 +30,6 @@ import { rerankDocuments } from "../../llm/rerank-client.js";
|
|
|
33
30
|
import { truncateDescription } from "../../output/shapes/helpers.js";
|
|
34
31
|
import { TELEMETRY_BUSY_TIMEOUT_MS, withIndexDb } from "../../storage/repositories/index-db.js";
|
|
35
32
|
import { findEntryIdByRef, getItemRefById } from "../../storage/repositories/index-entries-repository.js";
|
|
36
|
-
import { computeBodyHash } from "../../storage/repositories/index-llm-cache-repository.js";
|
|
37
33
|
import { akmSearch, parseSearchSource } from "./search.js";
|
|
38
34
|
import { akmShowUnified } from "./show.js";
|
|
39
35
|
const DEFAULT_CURATE_LIMIT = 4;
|
|
@@ -104,6 +100,19 @@ export async function akmCurate(options) {
|
|
|
104
100
|
if (!trimmedQuery) {
|
|
105
101
|
throw new UsageError('A curation query is required. Usage: akm curate "<task or prompt>" [--type <type>] [--limit <n>]', "MISSING_REQUIRED_ARGUMENT");
|
|
106
102
|
}
|
|
103
|
+
const nonTask = nonTaskInput(trimmedQuery);
|
|
104
|
+
if (nonTask) {
|
|
105
|
+
const abstained = {
|
|
106
|
+
query: options.query,
|
|
107
|
+
summary: `Curate abstained: the input is ${nonTask}, not a task.`,
|
|
108
|
+
items: [],
|
|
109
|
+
tip: 'Nothing was selected on purpose. To curate for it, pass the task itself: akm curate "<what you are trying to do>".',
|
|
110
|
+
};
|
|
111
|
+
if (!options.skipLogging) {
|
|
112
|
+
logCurateEvent(options.query, abstained, options.eventSource, options.attributionProjection);
|
|
113
|
+
}
|
|
114
|
+
return abstained;
|
|
115
|
+
}
|
|
107
116
|
const limit = options.limit && options.limit > 0 ? options.limit : DEFAULT_CURATE_LIMIT;
|
|
108
117
|
const source = options.source ?? parseSearchSource("local");
|
|
109
118
|
const searchResponse = options.searchResponse ??
|
|
@@ -201,11 +210,6 @@ async function enrichCuratedStashHit(query, hit, selectedRefs, eventSource) {
|
|
|
201
210
|
catch {
|
|
202
211
|
shown = undefined;
|
|
203
212
|
}
|
|
204
|
-
// #624-P3: when lazy graph extraction is opted in, enqueue an ungraphed
|
|
205
|
-
// asset for a later pass to extract. Fire-and-forget, non-blocking, NO inline
|
|
206
|
-
// extraction and NO LLM call here. Default-off (flag unset) = byte-identical.
|
|
207
|
-
if (shown?.path)
|
|
208
|
-
maybeEnqueueLazyGraph(shown.path);
|
|
209
213
|
const description = shown?.description ?? hit.description;
|
|
210
214
|
const preview = buildCuratedPreview(shown, hit);
|
|
211
215
|
const supportRefs = buildCurateSupportRefs(shown?.related?.hits, selectedRefs, hit.ref);
|
|
@@ -232,44 +236,6 @@ async function enrichCuratedStashHit(query, hit, selectedRefs, eventSource) {
|
|
|
232
236
|
copySearchHitAttribution(hit, item, item.description);
|
|
233
237
|
return item;
|
|
234
238
|
}
|
|
235
|
-
/**
|
|
236
|
-
* #624-P3 — enqueue an ungraphed asset for lazy graph extraction when the
|
|
237
|
-
* `index.graph.lazyGraphExtraction` flag is on. Pure side-effect, fully
|
|
238
|
-
* best-effort: any failure (config, fs, db) is swallowed so curate never fails
|
|
239
|
-
* on it. NO LLM call and NO inline extraction — only a cheap queue insert.
|
|
240
|
-
* Default-off (flag unset) returns immediately = byte-identical behavior.
|
|
241
|
-
*/
|
|
242
|
-
function maybeEnqueueLazyGraph(assetPath) {
|
|
243
|
-
try {
|
|
244
|
-
const config = loadConfig();
|
|
245
|
-
if (getIndexPassConfig(config.index, "graph")?.lazyGraphExtraction !== true)
|
|
246
|
-
return;
|
|
247
|
-
const sources = resolveSourceEntries();
|
|
248
|
-
const source = findSourceForPath(assetPath, sources);
|
|
249
|
-
const stashRoot = source?.path;
|
|
250
|
-
if (!stashRoot)
|
|
251
|
-
return;
|
|
252
|
-
let raw;
|
|
253
|
-
try {
|
|
254
|
-
raw = fs.readFileSync(assetPath, "utf8");
|
|
255
|
-
}
|
|
256
|
-
catch {
|
|
257
|
-
return;
|
|
258
|
-
}
|
|
259
|
-
const body = parseFrontmatter(raw).content.trim();
|
|
260
|
-
if (!body)
|
|
261
|
-
return;
|
|
262
|
-
const bodyHash = computeBodyHash(body);
|
|
263
|
-
withIndexDb((db) => {
|
|
264
|
-
if (!hasGraphData(db, stashRoot, assetPath)) {
|
|
265
|
-
enqueueGraphExtraction(db, stashRoot, assetPath, bodyHash, 0);
|
|
266
|
-
}
|
|
267
|
-
}, { busyTimeoutMs: TELEMETRY_BUSY_TIMEOUT_MS });
|
|
268
|
-
}
|
|
269
|
-
catch (err) {
|
|
270
|
-
rethrowIfTestIsolationError(err);
|
|
271
|
-
}
|
|
272
|
-
}
|
|
273
239
|
function buildCuratedRegistryItem(query, hit) {
|
|
274
240
|
return {
|
|
275
241
|
source: "registry",
|
|
@@ -27,14 +27,12 @@ import { buildMarkdownLeadContext, fragmentForSelector, MARKDOWN_FRAGMENT_CONTEX
|
|
|
27
27
|
import { displayRef, typeNameFromConceptId } from "../../core/asset/resolve-ref.js";
|
|
28
28
|
import { META_DIR, parseMetaRef, readMetaFile } from "../../core/asset/stash-meta.js";
|
|
29
29
|
import { asNonEmptyString, isWithin } from "../../core/common.js";
|
|
30
|
-
import {
|
|
30
|
+
import { loadConfig } from "../../core/config/config.js";
|
|
31
31
|
import { NotFoundError, rethrowIfDataDirUnreadable, rethrowIfTestIsolationError, UsageError } from "../../core/errors.js";
|
|
32
32
|
import { appendEvent } from "../../core/events.js";
|
|
33
33
|
import { SCRIPT_EXTENSIONS } from "../../core/recognition-util.js";
|
|
34
34
|
import { presentationFor } from "../../core/type-presentation.js";
|
|
35
35
|
import { warn, warnOnce } from "../../core/warn.js";
|
|
36
|
-
import { hasGraphData } from "../../indexer/db/graph-db.js";
|
|
37
|
-
import { extractGraphForSingleFile } from "../../indexer/graph/graph-extraction.js";
|
|
38
36
|
import { listRelatedPathsForFile } from "../../indexer/graph/graph-related.js";
|
|
39
37
|
import { lookupBundleRef, lookupBundleRefWithResolution } from "../../indexer/indexer.js";
|
|
40
38
|
import { projectMarkdownFragmentContent } from "../../indexer/passes/metadata.js";
|
|
@@ -42,11 +40,8 @@ import { ensurePrimaryIndexForRead, resolveReadSources } from "../../indexer/rea
|
|
|
42
40
|
import { buildEditHint, findSourceForPath, isEditable, resolveSourceEntries, } from "../../indexer/search/search-source.js";
|
|
43
41
|
import { recentShowCount, recordShowUsage } from "../../indexer/usage/show-usage.js";
|
|
44
42
|
import { buildFileContext, buildRenderContext, getRenderer, } from "../../indexer/walk/file-context.js";
|
|
45
|
-
import { resolveIndexPassExecution } from "../../llm/index-passes.js";
|
|
46
43
|
import { resolveSourcesForOrigin } from "../../registry/origin-resolve.js";
|
|
47
|
-
import {
|
|
48
|
-
import { closeDatabase, openExistingDatabase } from "../../storage/repositories/index-connection.js";
|
|
49
|
-
import { TELEMETRY_BUSY_TIMEOUT_MS, withIndexDb } from "../../storage/repositories/index-db.js";
|
|
44
|
+
import { withIndexDb } from "../../storage/repositories/index-db.js";
|
|
50
45
|
import { getIndexedMarkdownFragment } from "../../storage/repositories/index-fts-repository.js";
|
|
51
46
|
import { getCurrentWorkflowScopeKey } from "../../workflows/authoring/scope-key.js";
|
|
52
47
|
import { buildWorkflowAction } from "../../workflows/renderer.js";
|
|
@@ -405,15 +400,6 @@ export async function showLocal(input) {
|
|
|
405
400
|
if (activeRun) {
|
|
406
401
|
fullResponse.activeRun = activeRun;
|
|
407
402
|
}
|
|
408
|
-
// #624-P3: opt-in inline graph extraction. Default OFF — when the flag is
|
|
409
|
-
// unset this whole block is skipped (no hasGraphData check, no LLM call), so
|
|
410
|
-
// behavior is byte-identical to today. When ON, it extracts graph data for an
|
|
411
|
-
// ungraphed asset, but ONLY when a model is configured (model-available
|
|
412
|
-
// guard) and ALWAYS bounded by a 30s timeout so `show` can never hang. Any
|
|
413
|
-
// timeout/model-unavailable/error path returns the response unchanged.
|
|
414
|
-
if (getIndexPassConfig(config.index, "graph")?.lazyGraphExtraction === true) {
|
|
415
|
-
await maybeExtractGraphInline(config, sourceStashDir, assetPath);
|
|
416
|
-
}
|
|
417
403
|
if (input.detail === "brief") {
|
|
418
404
|
return buildBriefResponse(fullResponse, assetPath);
|
|
419
405
|
}
|
|
@@ -470,71 +456,6 @@ function findUnrecognizedScriptSource(assetParts, sources) {
|
|
|
470
456
|
}
|
|
471
457
|
return undefined;
|
|
472
458
|
}
|
|
473
|
-
/**
|
|
474
|
-
* #624-P3 — opt-in inline graph extraction for `akm show`. Best-effort and
|
|
475
|
-
* timeout-bounded: never throws, never hangs, never mutates the response.
|
|
476
|
-
*
|
|
477
|
-
* Preconditions (caller already checked the flag): a model must be configured
|
|
478
|
-
* (model-available guard via {@link resolveIndexPassExecution}) and the asset
|
|
479
|
-
* must be ungraphed ({@link hasGraphData}). Extraction races a 30s timeout so
|
|
480
|
-
* `show` cannot block on a slow provider; any timeout/error/missing-model path
|
|
481
|
-
* is swallowed and `show` returns its already-assembled response unchanged.
|
|
482
|
-
*/
|
|
483
|
-
async function maybeExtractGraphInline(config, sourceStashDir, assetPath) {
|
|
484
|
-
try {
|
|
485
|
-
// Resolve readiness and the symbolic runner once. The inline dispatch must
|
|
486
|
-
// consume this same snapshot even if models.json changes while show runs.
|
|
487
|
-
const graphExecution = resolveIndexPassExecution("graph", config);
|
|
488
|
-
if (!graphExecution.runner)
|
|
489
|
-
return;
|
|
490
|
-
const emittedNoticeKeys = new Set();
|
|
491
|
-
const reportNotices = (notices) => {
|
|
492
|
-
for (const notice of notices) {
|
|
493
|
-
const key = JSON.stringify(notice);
|
|
494
|
-
if (emittedNoticeKeys.has(key))
|
|
495
|
-
continue;
|
|
496
|
-
emittedNoticeKeys.add(key);
|
|
497
|
-
const field = typeof notice.field === "string" ? ` field=${notice.field}` : "";
|
|
498
|
-
warn(`[akm] lazy graph extraction notice ${notice.code} adapter=${notice.adapter}${field}: ${notice.message}`);
|
|
499
|
-
}
|
|
500
|
-
};
|
|
501
|
-
reportNotices(graphExecution.notices);
|
|
502
|
-
let alreadyGraphed = false;
|
|
503
|
-
withIndexDb((db) => {
|
|
504
|
-
alreadyGraphed = hasGraphData(db, sourceStashDir, assetPath);
|
|
505
|
-
}, { busyTimeoutMs: TELEMETRY_BUSY_TIMEOUT_MS });
|
|
506
|
-
if (alreadyGraphed)
|
|
507
|
-
return;
|
|
508
|
-
// Open the db for the async extraction ourselves: `withIndexDb` is
|
|
509
|
-
// synchronous and would close the connection the instant the async fn
|
|
510
|
-
// returns its Promise (before extraction completes). Close it explicitly
|
|
511
|
-
// after the race settles instead.
|
|
512
|
-
const db = openExistingDatabase(resolveStorageLocations().indexDb);
|
|
513
|
-
let timer;
|
|
514
|
-
const timeout = new Promise((resolve) => {
|
|
515
|
-
timer = setTimeout(resolve, 30_000);
|
|
516
|
-
});
|
|
517
|
-
try {
|
|
518
|
-
await Promise.race([
|
|
519
|
-
extractGraphForSingleFile(db, sourceStashDir, assetPath, {
|
|
520
|
-
config,
|
|
521
|
-
llmRunner: graphExecution.runner,
|
|
522
|
-
onNotices: reportNotices,
|
|
523
|
-
}),
|
|
524
|
-
timeout,
|
|
525
|
-
]);
|
|
526
|
-
}
|
|
527
|
-
finally {
|
|
528
|
-
if (timer)
|
|
529
|
-
clearTimeout(timer);
|
|
530
|
-
closeDatabase(db);
|
|
531
|
-
}
|
|
532
|
-
}
|
|
533
|
-
catch (err) {
|
|
534
|
-
rethrowIfTestIsolationError(err);
|
|
535
|
-
// Any other failure: silently return the unchanged show response.
|
|
536
|
-
}
|
|
537
|
-
}
|
|
538
459
|
/**
|
|
539
460
|
* Minimal `show`: ref → indexer lookup → file contents. Used by callers that
|
|
540
461
|
* just need the raw file (e.g. clone, write-source) and don't want the full
|