akm-cli 0.9.17-alpha.6 → 0.9.17-alpha.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +285 -0
- package/STABILITY.md +2 -2
- package/dist/akm +62 -29
- package/dist/akm-migrate +38 -19
- package/dist/assets/prompts/retrieval-relevance-judge.md +6 -0
- package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +5 -3
- package/dist/commands/improve/consolidate.js +11 -0
- package/dist/commands/improve/improve-cli.js +27 -7
- package/dist/commands/improve/ledger.js +7 -3
- package/dist/commands/improve/loop-stages.js +4 -3
- package/dist/commands/improve/preparation.js +40 -10
- package/dist/commands/improve/reflect.js +46 -22
- package/dist/commands/improve/retrieval-gate.js +127 -0
- package/dist/commands/improve/retrieval-scope.js +77 -0
- package/dist/commands/read/curate.js +41 -30
- package/dist/commands/read/show.js +55 -2
- package/dist/commands/sources/info.js +3 -0
- package/dist/commands/tasks/tasks-cli.js +10 -12
- package/dist/commands/tasks/tasks.js +57 -56
- package/dist/commands/tasks/validate.js +27 -46
- package/dist/core/adapter/adapters/akm-adapter.js +2 -0
- package/dist/core/adapter/adapters/akm-metadata.js +31 -0
- package/dist/core/adapter/adapters/akm-task-adapter.js +29 -8
- package/dist/core/improve-result.js +4 -1
- package/dist/core/non-task-input.js +20 -0
- package/dist/core/paths.js +0 -4
- package/dist/indexer/db/graph-db.js +0 -32
- package/dist/indexer/graph/graph-extraction.js +3 -1
- package/dist/indexer/indexer.js +1 -3
- package/dist/indexer/links/declared-links.js +90 -0
- package/dist/indexer/scan/doc-to-entry.js +1 -0
- package/dist/indexer/usage/usage-events.js +34 -0
- package/dist/llm/graph-extract.js +26 -37
- package/dist/output/shapes/helpers.js +3 -0
- package/dist/output/text/show-format.js +16 -0
- package/dist/scripts/akm-migrate-node.js +6377 -6437
- package/dist/scripts/akm-migrate.js +6860 -6920
- package/dist/storage/repositories/index-entries-repository.js +12 -6
- package/dist/storage/repositories/index-entry-schema.js +18 -1
- package/dist/storage/repositories/index-links-repository.js +143 -0
- package/dist/storage/repositories/index-schema.js +27 -0
- package/dist/storage/repositories/proposals-repository.js +4 -0
- package/dist/tasks/backends/cron.js +80 -43
- package/dist/tasks/backends/launchd.js +28 -15
- package/dist/tasks/backends/schtasks.js +25 -10
- package/dist/tasks/run/load-task.js +1 -1
- package/dist/tasks/scheduler-binding.js +4 -2
- package/dist/tasks/scheduler-invocation.js +127 -235
- package/dist/tasks/scheduler-sync.js +13 -8
- package/dist/tasks/source/parse-task-source.js +22 -126
- package/dist/tasks/source/task-to-v4.js +463 -87
- package/docs/migration/release-notes/0.9.17.md +7 -5
- package/docs/migration/v0.9.1-to-v0.9.2.md +7 -3
- package/docs/reference/cli.md +26 -10
- package/docs/reference/tasks.md +58 -40
- package/package.json +1 -1
- package/dist/tasks/source/task-to-v3.js +0 -507
|
@@ -14,6 +14,7 @@ import { getCacheDir } from "../../core/paths.js";
|
|
|
14
14
|
import { redactSensitiveText } from "../../core/redaction.js";
|
|
15
15
|
import { clearLogFile, setLogFile, warn } from "../../core/warn.js";
|
|
16
16
|
import { resolveWriteTarget } from "../../core/write-source.js";
|
|
17
|
+
import { DEFAULT_LLM_TIMEOUT_MS } from "../../integrations/agent/config.js";
|
|
17
18
|
import { collectEngineCredentialValues } from "../../integrations/agent/engine-resolution.js";
|
|
18
19
|
import { probeLlmReachable } from "../../llm/client.js";
|
|
19
20
|
import { getOutputMode } from "../../output/context.js";
|
|
@@ -77,25 +78,44 @@ function collectRequiredEngineTargets(plan) {
|
|
|
77
78
|
const targets = [];
|
|
78
79
|
for (const [processName, process] of Object.entries(plan.processes)) {
|
|
79
80
|
if (process.runner) {
|
|
80
|
-
targets.push({
|
|
81
|
+
targets.push({
|
|
82
|
+
process: processName,
|
|
83
|
+
engine: process.runner.engine,
|
|
84
|
+
connection: probeConnection(process.runner),
|
|
85
|
+
});
|
|
81
86
|
}
|
|
82
87
|
}
|
|
83
88
|
if (plan.triageJudgment?.kind === "llm") {
|
|
84
89
|
targets.push({
|
|
85
90
|
process: "triage.judgment",
|
|
86
91
|
engine: plan.triageJudgment.engine,
|
|
87
|
-
connection: plan.triageJudgment
|
|
92
|
+
connection: probeConnection(plan.triageJudgment),
|
|
88
93
|
});
|
|
89
94
|
}
|
|
90
95
|
return targets;
|
|
91
96
|
}
|
|
97
|
+
/** The resolved engine keeps its request timeout beside the connection (the runtime merges it in); the probe needs it on the connection. */
|
|
98
|
+
function probeConnection(runner) {
|
|
99
|
+
return runner.timeoutMs !== undefined ? { ...runner.connection, timeoutMs: runner.timeoutMs } : runner.connection;
|
|
100
|
+
}
|
|
101
|
+
/**
|
|
102
|
+
* The bound on `--require-engines`' probe: the connection's own request
|
|
103
|
+
* timeout, at most two minutes. A local server busy with another job queues
|
|
104
|
+
* the probe behind that job, and a fixed 3s bound failed every scheduled
|
|
105
|
+
* improve run on 2026-09-27 against a reachable endpoint; the cap still ends a
|
|
106
|
+
* hung endpoint (#957) long before a run's own multi-minute calls would.
|
|
107
|
+
*/
|
|
108
|
+
export function requiredEngineProbeTimeoutMs(connection) {
|
|
109
|
+
return Math.min(connection.timeoutMs ?? DEFAULT_LLM_TIMEOUT_MS, REQUIRED_ENGINE_PROBE_MAX_MS);
|
|
110
|
+
}
|
|
111
|
+
const REQUIRED_ENGINE_PROBE_MAX_MS = 120_000;
|
|
92
112
|
/**
|
|
93
113
|
* `--require-engines`, live: probe each connection's real completion path
|
|
94
|
-
* (a gateway can list a model whose completion route is dead, #980)
|
|
95
|
-
*
|
|
96
|
-
* result (R17); an unreachable one fails the run.
|
|
114
|
+
* (a gateway can list a model whose completion route is dead, #980), once per
|
|
115
|
+
* endpoint + model, within {@link requiredEngineProbeTimeoutMs}. Returns each
|
|
116
|
+
* target's latency for the run result (R17); an unreachable one fails the run.
|
|
97
117
|
*/
|
|
98
|
-
export async function assertRequiredEnginesReachable(plan, probeReachable = (connection) => probeLlmReachable(connection,
|
|
118
|
+
export async function assertRequiredEnginesReachable(plan, probeReachable = (connection) => probeLlmReachable(connection, requiredEngineProbeTimeoutMs(connection))) {
|
|
99
119
|
const targets = collectRequiredEngineTargets(plan);
|
|
100
120
|
if (targets.length === 0)
|
|
101
121
|
return [];
|
|
@@ -117,7 +137,7 @@ export async function assertRequiredEnginesReachable(plan, probeReachable = (con
|
|
|
117
137
|
const unreachable = probed.filter((item) => !item.reach.reachable);
|
|
118
138
|
if (unreachable.length > 0) {
|
|
119
139
|
const lines = unreachable.map((item) => ` - ${item.process} (engine "${item.engine}", ${item.connection.endpoint}): ${item.reach.error ?? "did not respond"}`);
|
|
120
|
-
throw new ConfigError(`--require-engines: ${unreachable.length} improve process${unreachable.length === 1 ? "" : "es"} cannot run because ${unreachable.length === 1 ? "its" : "their"} engine completion path is not reachable:\n${lines.join("\n")}`, "LLM_NOT_CONFIGURED");
|
|
140
|
+
throw new ConfigError(`--require-engines: ${unreachable.length} improve process${unreachable.length === 1 ? "" : "es"} cannot run because ${unreachable.length === 1 ? "its" : "their"} engine completion path is not reachable:\n${lines.join("\n")}`, "LLM_NOT_CONFIGURED", "Check that each listed endpoint is up and serves its model. The probe is one short completion, bounded by the engine's timeoutMs (at most two minutes).");
|
|
121
141
|
}
|
|
122
142
|
return probed.map((item) => ({
|
|
123
143
|
process: item.process,
|
|
@@ -82,13 +82,17 @@ export function recordLedgerAttempt(access, inputs) {
|
|
|
82
82
|
export function ledgerKey(source, ref) {
|
|
83
83
|
return `${source}\0${ref}`;
|
|
84
84
|
}
|
|
85
|
+
/** Read state.db through `fn` without creating it: `undefined` when there is none yet. */
|
|
86
|
+
export function readLedgerDb(access, fn) {
|
|
87
|
+
if (!access?.eventsCtx?.db && !fs.existsSync(ledgerDbPath(access) ?? getStateDbPath()))
|
|
88
|
+
return undefined;
|
|
89
|
+
return withLedgerDb(access, fn);
|
|
90
|
+
}
|
|
85
91
|
/** Every ledger row for `sources` in one query; no state.db yet means nothing was attempted. */
|
|
86
92
|
export function loadLedgerSnapshot(access, stashDir, sources) {
|
|
87
93
|
const out = new Map();
|
|
88
|
-
if (!access?.eventsCtx?.db && !fs.existsSync(ledgerDbPath(access) ?? getStateDbPath()))
|
|
89
|
-
return out;
|
|
90
94
|
try {
|
|
91
|
-
|
|
95
|
+
readLedgerDb(access, (db) => {
|
|
92
96
|
for (const row of listImproveLedgerRows(db, stashDir, sources))
|
|
93
97
|
out.set(ledgerKey(row.source, row.ref), row);
|
|
94
98
|
});
|
|
@@ -475,15 +475,16 @@ export async function runMemoryInferenceMaintenancePass(ctx, dbCell, memoryRefsF
|
|
|
475
475
|
export async function runGraphExtractionMaintenancePass(ctx, dbCell, args) {
|
|
476
476
|
const { config, sources, primaryStashDir, resolvedPlan } = ctx;
|
|
477
477
|
const settings = ctx.improveProfile?.processes?.graphExtraction;
|
|
478
|
-
const graphEnabled = resolvedPlan ? true : isProcessEnabled("index", "graph_extraction", config);
|
|
479
478
|
if (settings?.enabled === false) {
|
|
480
479
|
info("[improve] graph extraction skipped (disabled by improve profile)");
|
|
481
480
|
return { durationMs: 0, warnings: [] };
|
|
482
481
|
}
|
|
483
482
|
if (sources.length === 0)
|
|
484
483
|
return { durationMs: 0, warnings: [] };
|
|
485
|
-
|
|
486
|
-
|
|
484
|
+
// `index.graph.enabled: false` turns graph extraction off everywhere,
|
|
485
|
+
// whatever the strategy enables.
|
|
486
|
+
if (!isProcessEnabled("index", "graph_extraction", config)) {
|
|
487
|
+
info("[improve] graph extraction skipped (index.graph.enabled is false)");
|
|
487
488
|
return { durationMs: 0, warnings: [] };
|
|
488
489
|
}
|
|
489
490
|
const fullScan = settings?.fullScan === true;
|
|
@@ -23,7 +23,7 @@ import { ConfigError, rethrowIfTestIsolationError } from "../../core/errors.js";
|
|
|
23
23
|
import { appendEvent, readEvents } from "../../core/events.js";
|
|
24
24
|
import { withStateDb } from "../../core/state-db.js";
|
|
25
25
|
import { info, warn } from "../../core/warn.js";
|
|
26
|
-
import { countUsageEventsByType } from "../../indexer/usage/usage-events.js";
|
|
26
|
+
import { countUsageEventsByType, USAGE_EVENT_RETENTION_DAYS } from "../../indexer/usage/usage-events.js";
|
|
27
27
|
import { getAvailableHarnesses } from "../../integrations/session-logs/index.js";
|
|
28
28
|
import { getZeroResultSearches } from "../../storage/repositories/index-entries-repository.js";
|
|
29
29
|
import { getRetrievalCounts } from "../../storage/repositories/index-utility-repository.js";
|
|
@@ -41,6 +41,7 @@ import { applyMemoryCleanup } from "./memory/memory-improve.js";
|
|
|
41
41
|
import { getAllAssetOutcomes, getAssetOutcome, getOutcomeScoresByRef, OUTCOME_SCORE_MAX, outcomeScoreToSalience, projectAssetOutcome, updateAssetOutcome, } from "./outcome-loop.js";
|
|
42
42
|
import { projectMemoryCleanup, selectEffectiveImproveRefs } from "./planner.js";
|
|
43
43
|
import { DEFAULT_DUE_DAYS, DEFAULT_MAX_PER_RUN, selectProactiveMaintenanceRefs } from "./proactive-maintenance.js";
|
|
44
|
+
import { isInRetrievalScope, loadRetrievalScope } from "./retrieval-scope.js";
|
|
44
45
|
import { buildRankChangeReport, computeSalience, getAllRankScores, getAssetSalience, getLastUseMsByRef, isContentEncodingRow, SALIENCE_NO_OP_DAMPEN_FACTOR, SALIENCE_NO_OP_DAMPEN_THRESHOLD, upsertAssetSalience, } from "./salience.js";
|
|
45
46
|
import { attributeStage, errMessage } from "./stage.js";
|
|
46
47
|
/** The candidate's durable state key (salience, outcome, ledger). */
|
|
@@ -663,12 +664,23 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
|
|
|
663
664
|
const processableRefs = [...partition.eligibleRefs, ...partition.distillOnlyRefs];
|
|
664
665
|
const signalFiltered = processableRefs.filter((c) => snapshot.feedback.get(c.ref)?.hasSignal === true);
|
|
665
666
|
const signalBearingSet = new Set(signalFiltered.map((r) => r.ref));
|
|
666
|
-
|
|
667
|
+
// The fallback lanes (proactive, high salience, forgetting safety) have no
|
|
668
|
+
// usage evidence of their own: they pick only what retrieval returned or new
|
|
669
|
+
// material improve never processed (#986). Evaluated once per candidate.
|
|
670
|
+
const fallbackEligible = postCleanupRefs.filter((c) => !validationFailureRefs.has(c.ref));
|
|
671
|
+
const allowFallbacks = options.requireFeedbackSignal !== true;
|
|
672
|
+
const retrievalScope = scope.mode === "ref" || !allowFallbacks
|
|
673
|
+
? undefined
|
|
674
|
+
: loadRetrievalScope({ eventsCtx, ...(persist ? {} : { readOnly: true }) }, primaryStashDir ?? options.stashDir);
|
|
675
|
+
const unscoped = new Set(fallbackEligible.filter((c) => !isInRetrievalScope(retrievalScope, c.ref, c.filePath)).map((c) => c.ref));
|
|
676
|
+
const noFeedbackPool = dedupeRefs([
|
|
667
677
|
...processableRefs.filter((r) => !signalBearingSet.has(r.ref)),
|
|
668
678
|
...partition.noFeedbackPool,
|
|
669
679
|
]);
|
|
680
|
+
const noFeedbackCandidates = noFeedbackPool.filter((r) => !unscoped.has(r.ref));
|
|
681
|
+
// Only a ref no fallback lane may pick anymore is charged to the retrieval gate.
|
|
682
|
+
const outOfScope = new Set(noFeedbackPool.filter((r) => unscoped.has(r.ref)).map((r) => r.ref));
|
|
670
683
|
const retrieval = fetchRetrievalSignals(options, signalFiltered, noFeedbackCandidates, eventsCtx, persist);
|
|
671
|
-
const allowFallbacks = options.requireFeedbackSignal !== true;
|
|
672
684
|
const proactive = allowFallbacks
|
|
673
685
|
? selectProactiveMaintenanceLane(args, noFeedbackCandidates, snapshot, retrieval, persist)
|
|
674
686
|
: { proactiveRefs: [] };
|
|
@@ -692,10 +704,10 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
|
|
|
692
704
|
sourceByRef.set(r.ref, "scope");
|
|
693
705
|
for (const r of mergedRefs)
|
|
694
706
|
r.eligibilitySource = sourceByRef.get(r.ref) ?? "unknown";
|
|
695
|
-
// Forgetting safety may only reuse this plan's own surviving objects
|
|
696
|
-
// never a ref whose reflect window is still open.
|
|
697
|
-
const
|
|
698
|
-
|
|
707
|
+
// Forgetting safety may only reuse this plan's own surviving objects inside
|
|
708
|
+
// the retrieval scope, and never a ref whose reflect window is still open.
|
|
709
|
+
const forgettingEligible = fallbackEligible.filter((c) => !unscoped.has(c.ref) &&
|
|
710
|
+
!isLedgerBlocked(ledgerRowFor(snapshot.ledger, "reflect", c.ref, c.itemRef), snapshot.nowIso));
|
|
699
711
|
const scored = scoreSalience(args, mergedRefs, snapshot.feedback, retrieval.retrievalCounts, persist);
|
|
700
712
|
mergedRefs = applyForgettingSafety({
|
|
701
713
|
pendingForgettingRefs: scored.pendingForgettingRefs,
|
|
@@ -739,28 +751,46 @@ async function selectLoopCandidates(args, postCleanupRefs, validationFailureRefs
|
|
|
739
751
|
// Skip observability waits until every fallback lane has finalized the
|
|
740
752
|
// survivors, so a rescued ref is never also reported skipped.
|
|
741
753
|
const survivors = new Set(sorted.map((c) => c.ref));
|
|
742
|
-
const
|
|
754
|
+
const skipped = fallbackEligible.filter((c) => !survivors.has(c.ref));
|
|
755
|
+
const retrievalSkipped = skipped.filter((c) => outOfScope.has(c.ref));
|
|
756
|
+
const signalSkipped = skipped.filter((c) => !outOfScope.has(c.ref));
|
|
743
757
|
for (const ref of partition.distillCooledRefs) {
|
|
744
758
|
actions.push({ ref, mode: "distill-skipped", result: { ok: true, reason: "distill signal-delta" } });
|
|
745
759
|
if (persist)
|
|
746
760
|
recordImproveSkip(eventsCtx, ref, { reason: "distill_no_new_signal" });
|
|
747
761
|
}
|
|
748
|
-
for (const candidate of
|
|
762
|
+
for (const candidate of skipped) {
|
|
749
763
|
actions.push({
|
|
750
764
|
ref: candidate.ref,
|
|
751
765
|
mode: "distill-skipped",
|
|
752
|
-
result: {
|
|
766
|
+
result: {
|
|
767
|
+
ok: true,
|
|
768
|
+
reason: outOfScope.has(candidate.ref)
|
|
769
|
+
? "not retrieved inside the usage window"
|
|
770
|
+
: "no new signal since last proposal",
|
|
771
|
+
},
|
|
753
772
|
});
|
|
754
773
|
}
|
|
755
774
|
if (persist && signalSkipped.length > 0) {
|
|
756
775
|
recordImproveSkip(eventsCtx, undefined, { reason: "no_new_signal", count: signalSkipped.length });
|
|
757
776
|
}
|
|
777
|
+
if (persist && retrievalSkipped.length > 0) {
|
|
778
|
+
recordImproveSkip(eventsCtx, undefined, { reason: "not_retrieved", count: retrievalSkipped.length });
|
|
779
|
+
}
|
|
758
780
|
const blocked = signalSkipped.length + partition.distillOnlyRefs.length;
|
|
759
781
|
if (blocked > 0) {
|
|
760
782
|
info(`[improve] ${blocked} of ${partition.preCooldownCount} indexed refs blocked by reflect signal-delta ` +
|
|
761
783
|
`(${signalSkipped.length} fully skipped, ${partition.distillOnlyRefs.length} routed to distill-only)`);
|
|
762
784
|
}
|
|
785
|
+
if (retrievalSkipped.length > 0) {
|
|
786
|
+
info(`[improve] ${retrievalSkipped.length} refs left out: not retrieved in the last ${USAGE_EVENT_RETENTION_DAYS} days, and not new material`);
|
|
787
|
+
}
|
|
763
788
|
const gates = [
|
|
789
|
+
{
|
|
790
|
+
name: "retrieval",
|
|
791
|
+
removed: retrievalSkipped.length,
|
|
792
|
+
reason: `no feedback, and neither returned by search, curate or show in the last ${USAGE_EVENT_RETENTION_DAYS} days nor new material improve never processed`,
|
|
793
|
+
},
|
|
764
794
|
{
|
|
765
795
|
name: "signal",
|
|
766
796
|
removed: signalSkipped.length,
|
|
@@ -43,6 +43,7 @@ import { findAssetFilePath } from "./eligibility.js";
|
|
|
43
43
|
import { resolveImproveLlmExecution } from "./execution.js";
|
|
44
44
|
import { recordLedgerAttempt } from "./ledger.js";
|
|
45
45
|
import { classifyReflectChange, splitFrontmatter } from "./reflect-noise.js";
|
|
46
|
+
import { loadRetrievalQueries, runRetrievalRegressionGate } from "./retrieval-gate.js";
|
|
46
47
|
import { callStage, mintProposal, noticeSet, rejectedProposalContext, runReflectQualityJudge, } from "./stage.js";
|
|
47
48
|
const MAX_FEEDBACK_LINES = 10;
|
|
48
49
|
const MAX_GLOBAL_FEEDBACK_LINES = 20;
|
|
@@ -960,6 +961,24 @@ async function finalizeReflectProposal(args) {
|
|
|
960
961
|
}
|
|
961
962
|
const flagged = Boolean(sanitized.sizeGuardRatio || sanitized.truncationMarkerLeaked);
|
|
962
963
|
const judged = judge.enabled && !flagged;
|
|
964
|
+
/** A judge refused the revision: record it for the ledger's rejection window and stop. */
|
|
965
|
+
const refuse = (detail, metadata, message) => {
|
|
966
|
+
if (options.ref) {
|
|
967
|
+
recordLedgerAttempt({ proposalsCtx: options.ctx, eventsCtx: options.eventsCtx }, {
|
|
968
|
+
stashDir: run.stash,
|
|
969
|
+
ref: options.itemRef ?? options.ref,
|
|
970
|
+
source: "reflect",
|
|
971
|
+
outcome: "quality_rejected",
|
|
972
|
+
detail,
|
|
973
|
+
});
|
|
974
|
+
}
|
|
975
|
+
appendEvent({
|
|
976
|
+
eventType: "reflect_completed",
|
|
977
|
+
ref: payload.ref,
|
|
978
|
+
metadata: { source: "reflect", qualityRejected: true, ...metadata, ...telemetry },
|
|
979
|
+
}, options.eventsCtx);
|
|
980
|
+
return reflectFailure(run, result, "quality_rejected", message, false);
|
|
981
|
+
};
|
|
963
982
|
if (judged) {
|
|
964
983
|
const verdict = await runReflectQualityJudge(run.config, payload.content, assetContent ?? "", feedback, options.chat, {
|
|
965
984
|
runnerSelectionFrozen: true,
|
|
@@ -969,28 +988,33 @@ async function finalizeReflectProposal(args) {
|
|
|
969
988
|
onNotices: run.notices.add,
|
|
970
989
|
});
|
|
971
990
|
if (!verdict.pass) {
|
|
972
|
-
|
|
973
|
-
|
|
974
|
-
|
|
975
|
-
|
|
976
|
-
|
|
977
|
-
|
|
978
|
-
|
|
979
|
-
|
|
980
|
-
|
|
981
|
-
|
|
982
|
-
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
|
|
986
|
-
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
|
|
990
|
-
|
|
991
|
-
|
|
992
|
-
|
|
993
|
-
return
|
|
991
|
+
return refuse(verdict.reason, {
|
|
992
|
+
qualityScore: verdict.score,
|
|
993
|
+
qualityReason: verdict.reason,
|
|
994
|
+
...(verdict.criteria ? { qualityCriteria: verdict.criteria } : {}),
|
|
995
|
+
}, `Reflect proposal quality gate rejected: score=${verdict.score}, reason="${verdict.reason}"`);
|
|
996
|
+
}
|
|
997
|
+
}
|
|
998
|
+
// #722: a rewrite of an existing asset must not grade lower on its own retrieval queries.
|
|
999
|
+
if (judged && judge.runner && assetContent !== undefined) {
|
|
1000
|
+
const retrieval = await runRetrievalRegressionGate({
|
|
1001
|
+
ref: payload.ref,
|
|
1002
|
+
before: assetContent,
|
|
1003
|
+
after: payload.content,
|
|
1004
|
+
queries: loadRetrievalQueries({ proposalsCtx: options.ctx, eventsCtx: options.eventsCtx }, payload.ref),
|
|
1005
|
+
runner: judge.runner,
|
|
1006
|
+
...(options.chat ? { chat: options.chat } : {}),
|
|
1007
|
+
...(Object.hasOwn(options, "timeoutMs") ? { timeoutMs: options.timeoutMs } : {}),
|
|
1008
|
+
...(options.signal ? { signal: options.signal } : {}),
|
|
1009
|
+
onNotices: run.notices.add,
|
|
1010
|
+
});
|
|
1011
|
+
if (!retrieval.pass) {
|
|
1012
|
+
return refuse(retrieval.reason, {
|
|
1013
|
+
retrievalRegression: true,
|
|
1014
|
+
retrievalQueries: retrieval.queries,
|
|
1015
|
+
...(retrieval.oldMean !== undefined ? { retrievalGradeBefore: retrieval.oldMean } : {}),
|
|
1016
|
+
...(retrieval.newMean !== undefined ? { retrievalGradeAfter: retrieval.newMean } : {}),
|
|
1017
|
+
}, `Reflect proposal refused: ${retrieval.reason}`);
|
|
994
1018
|
}
|
|
995
1019
|
}
|
|
996
1020
|
// A lesson reflect wrote is marked so a later reflect on the same skill does
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/**
|
|
5
|
+
* The retrieval regression gate (#722): a reflect rewrite of an existing asset
|
|
6
|
+
* must not grade lower on the queries that actually retrieved it.
|
|
7
|
+
*
|
|
8
|
+
* Measured before it was built: of 60 accepted reflect rewrites judged against
|
|
9
|
+
* their own queries, 14 graded lower (23%, 95% CI 14–35%) and 12 higher. The
|
|
10
|
+
* old and new content are graded one query at a time, blind, with the
|
|
11
|
+
* retrieval-eval judge's prompt (kappa 0.83 against human grades) and its
|
|
12
|
+
* document shape: type, ref, name, description and the first 1,500 characters
|
|
13
|
+
* of the body.
|
|
14
|
+
*/
|
|
15
|
+
import relevanceJudgePrompt from "../../assets/prompts/retrieval-relevance-judge.md" with { type: "text" };
|
|
16
|
+
import { parseFrontmatter } from "../../core/asset/frontmatter.js";
|
|
17
|
+
import { parseRefInput } from "../../core/asset/resolve-ref.js";
|
|
18
|
+
import { nonTaskInput } from "../../core/non-task-input.js";
|
|
19
|
+
import { parseEmbeddedJsonResponse } from "../../core/parse.js";
|
|
20
|
+
import { listRetrievalQueries } from "../../indexer/usage/usage-events.js";
|
|
21
|
+
import { readLedgerDb, stripBundle } from "./ledger.js";
|
|
22
|
+
import { callStage } from "./stage.js";
|
|
23
|
+
/** Queries graded per rewrite, as measured. */
|
|
24
|
+
const MAX_QUERIES = 5;
|
|
25
|
+
/** The retrieval suite treats a longer input as a paste, not a query; the measurement did the same. */
|
|
26
|
+
const MAX_QUERY_CHARS = 2000;
|
|
27
|
+
/** The judge sees this much of the body, as in the retrieval eval. */
|
|
28
|
+
const MAX_DOC_CHARS = 1500;
|
|
29
|
+
const GRADE_SCHEMA = {
|
|
30
|
+
type: "object",
|
|
31
|
+
required: ["grade", "reason"],
|
|
32
|
+
additionalProperties: false,
|
|
33
|
+
properties: { grade: { type: "integer", minimum: 0, maximum: 3 }, reason: { type: "string" } },
|
|
34
|
+
};
|
|
35
|
+
/** Up to five distinct task queries, in the given order, whitespace collapsed. */
|
|
36
|
+
export function usableRetrievalQueries(raw) {
|
|
37
|
+
const out = [];
|
|
38
|
+
for (const text of raw) {
|
|
39
|
+
const query = text.replace(/\s+/g, " ").trim();
|
|
40
|
+
if (!query || query.length > MAX_QUERY_CHARS || nonTaskInput(query) || out.includes(query))
|
|
41
|
+
continue;
|
|
42
|
+
out.push(query);
|
|
43
|
+
if (out.length === MAX_QUERIES)
|
|
44
|
+
break;
|
|
45
|
+
}
|
|
46
|
+
return out;
|
|
47
|
+
}
|
|
48
|
+
/** The asset's own retrieval queries from the usage log (none when state.db cannot be read). */
|
|
49
|
+
export function loadRetrievalQueries(access, ref) {
|
|
50
|
+
try {
|
|
51
|
+
return usableRetrievalQueries(readLedgerDb(access, (db) => listRetrievalQueries(db, stripBundle(ref))) ?? []);
|
|
52
|
+
}
|
|
53
|
+
catch {
|
|
54
|
+
return [];
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
function judgeDocument(ref, raw) {
|
|
58
|
+
const conceptId = stripBundle(ref);
|
|
59
|
+
const parsed = parseRefInput(conceptId);
|
|
60
|
+
const { data, content } = parseFrontmatter(raw);
|
|
61
|
+
const text = content
|
|
62
|
+
.replace(/\n{3,}/g, "\n\n")
|
|
63
|
+
.trim()
|
|
64
|
+
.slice(0, MAX_DOC_CHARS);
|
|
65
|
+
const name = typeof data.name === "string" && data.name ? data.name : parsed.name;
|
|
66
|
+
const description = typeof data.description === "string" ? data.description : "";
|
|
67
|
+
return [
|
|
68
|
+
"Candidate asset:",
|
|
69
|
+
`Type: ${parsed.type}`,
|
|
70
|
+
`Ref: ${conceptId}`,
|
|
71
|
+
`Name: ${name}`,
|
|
72
|
+
`Description: ${description}`,
|
|
73
|
+
"",
|
|
74
|
+
"Content:",
|
|
75
|
+
text,
|
|
76
|
+
].join("\n");
|
|
77
|
+
}
|
|
78
|
+
/**
|
|
79
|
+
* Grade `before` and `after` on each query; refuse the rewrite when the new
|
|
80
|
+
* content's mean grade is lower. Fails closed, like the quality judge: a grade
|
|
81
|
+
* that cannot be obtained refuses the rewrite.
|
|
82
|
+
*/
|
|
83
|
+
export async function runRetrievalRegressionGate(args) {
|
|
84
|
+
if (args.queries.length === 0)
|
|
85
|
+
return { pass: true, queries: 0, reason: "no retrieval queries to compare on" };
|
|
86
|
+
const documents = { old: judgeDocument(args.ref, args.before), new: judgeDocument(args.ref, args.after) };
|
|
87
|
+
const totals = { old: 0, new: 0 };
|
|
88
|
+
for (const query of args.queries) {
|
|
89
|
+
for (const version of ["old", "new"]) {
|
|
90
|
+
const outcome = await callStage({
|
|
91
|
+
feature: "proposal_quality_gate",
|
|
92
|
+
runner: args.runner,
|
|
93
|
+
system: relevanceJudgePrompt.trim(),
|
|
94
|
+
prompt: `Query: ${query}\n\n${documents[version]}`,
|
|
95
|
+
request: {
|
|
96
|
+
enableThinking: false,
|
|
97
|
+
temperature: 0,
|
|
98
|
+
responseSchema: GRADE_SCHEMA,
|
|
99
|
+
...(args.timeoutMs !== undefined ? { timeoutMs: args.timeoutMs } : {}),
|
|
100
|
+
...(args.signal ? { signal: args.signal } : {}),
|
|
101
|
+
...(args.chat ? { chat: args.chat } : {}),
|
|
102
|
+
},
|
|
103
|
+
...(args.onNotices ? { onNotices: args.onNotices } : {}),
|
|
104
|
+
});
|
|
105
|
+
const grade = outcome.ok ? parseEmbeddedJsonResponse(outcome.raw)?.grade : undefined;
|
|
106
|
+
if (typeof grade !== "number" || !Number.isInteger(grade) || grade < 0 || grade > 3) {
|
|
107
|
+
return {
|
|
108
|
+
pass: false,
|
|
109
|
+
queries: args.queries.length,
|
|
110
|
+
reason: `retrieval check could not grade the ${version} content${outcome.ok ? "" : ` (${outcome.reason})`}`,
|
|
111
|
+
};
|
|
112
|
+
}
|
|
113
|
+
totals[version] += grade;
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
const oldMean = totals.old / args.queries.length;
|
|
117
|
+
const newMean = totals.new / args.queries.length;
|
|
118
|
+
const pass = newMean >= oldMean;
|
|
119
|
+
const grades = `${oldMean.toFixed(2)} -> ${newMean.toFixed(2)} over ${args.queries.length} retrieval ${args.queries.length === 1 ? "query" : "queries"}`;
|
|
120
|
+
return {
|
|
121
|
+
pass,
|
|
122
|
+
queries: args.queries.length,
|
|
123
|
+
oldMean,
|
|
124
|
+
newMean,
|
|
125
|
+
reason: pass ? `retrieval grade ${grades}` : `retrieval regression: grade ${grades}`,
|
|
126
|
+
};
|
|
127
|
+
}
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/**
|
|
5
|
+
* What improve may rework (#986): an asset that retrieval returned inside the
|
|
6
|
+
* usage window, or newly captured material no improve stage has processed.
|
|
7
|
+
*
|
|
8
|
+
* Fresh feedback and an explicit `--scope <ref>` are usage evidence of their
|
|
9
|
+
* own, so the signal-delta and scope lanes need no check. The fallback lanes
|
|
10
|
+
* (proactive maintenance, high salience, forgetting safety) and consolidation
|
|
11
|
+
* pick assets without such evidence, so they pick only inside this scope.
|
|
12
|
+
*/
|
|
13
|
+
import fs from "node:fs";
|
|
14
|
+
import { daysToMs } from "../../core/common.js";
|
|
15
|
+
import { warn } from "../../core/warn.js";
|
|
16
|
+
import { listUsedEntryRefs, USAGE_EVENT_RETENTION_DAYS } from "../../indexer/usage/usage-events.js";
|
|
17
|
+
import { listImproveLedgerRows } from "../../storage/repositories/improve-ledger-repository.js";
|
|
18
|
+
import { listProposalRefSources } from "../../storage/repositories/proposals-repository.js";
|
|
19
|
+
import { readLedgerDb, stripBundle } from "./ledger.js";
|
|
20
|
+
/**
|
|
21
|
+
* Ledger and proposal sources that bring material in. Every other source is an
|
|
22
|
+
* improve stage reworking an asset (reflect, distill, consolidate, schema repair).
|
|
23
|
+
*/
|
|
24
|
+
const CAPTURE_SOURCES = new Set(["extract", "propose", "remember", "import"]);
|
|
25
|
+
/**
|
|
26
|
+
* Load the scope from state.db. The window is the usage log's retention: the
|
|
27
|
+
* log keeps nothing older, and anything shorter would drop assets read less
|
|
28
|
+
* often than the window. A `.derived` hit counts for its parent memory, whose
|
|
29
|
+
* facts it carries. `undefined` (every asset eligible, as before #986) only
|
|
30
|
+
* when state.db cannot be read.
|
|
31
|
+
*/
|
|
32
|
+
export function loadRetrievalScope(access, stashDir) {
|
|
33
|
+
const sinceMs = Date.now() - daysToMs(USAGE_EVENT_RETENTION_DAYS);
|
|
34
|
+
const used = new Set();
|
|
35
|
+
const processed = new Set();
|
|
36
|
+
try {
|
|
37
|
+
readLedgerDb(access, (db) => {
|
|
38
|
+
for (const entryRef of listUsedEntryRefs(db, new Date(sinceMs).toISOString())) {
|
|
39
|
+
const conceptId = stripBundle(entryRef);
|
|
40
|
+
used.add(conceptId);
|
|
41
|
+
if (conceptId.endsWith(".derived"))
|
|
42
|
+
used.add(conceptId.slice(0, -".derived".length));
|
|
43
|
+
}
|
|
44
|
+
if (!stashDir)
|
|
45
|
+
return;
|
|
46
|
+
for (const row of [...listImproveLedgerRows(db, stashDir), ...listProposalRefSources(db, stashDir)]) {
|
|
47
|
+
if (!CAPTURE_SOURCES.has(row.source))
|
|
48
|
+
processed.add(stripBundle(row.ref));
|
|
49
|
+
}
|
|
50
|
+
});
|
|
51
|
+
}
|
|
52
|
+
catch (error) {
|
|
53
|
+
warn(`[improve] usage history unreadable, so every asset stays eligible: ${error instanceof Error ? error.message : String(error)}`);
|
|
54
|
+
return undefined;
|
|
55
|
+
}
|
|
56
|
+
return { used, processed, sinceMs };
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* Whether improve may rework `ref` (whose file is `filePath`) under `scope`. An
|
|
60
|
+
* unprocessed asset whose file cannot be read is left to the disk check that
|
|
61
|
+
* follows, which reports it as missing.
|
|
62
|
+
*/
|
|
63
|
+
export function isInRetrievalScope(scope, ref, filePath) {
|
|
64
|
+
if (!scope)
|
|
65
|
+
return true;
|
|
66
|
+
const conceptId = stripBundle(ref);
|
|
67
|
+
if (scope.used.has(conceptId))
|
|
68
|
+
return true;
|
|
69
|
+
if (scope.processed.has(conceptId))
|
|
70
|
+
return false;
|
|
71
|
+
try {
|
|
72
|
+
return filePath === undefined || fs.statSync(filePath).mtimeMs >= scope.sinceMs;
|
|
73
|
+
}
|
|
74
|
+
catch {
|
|
75
|
+
return true;
|
|
76
|
+
}
|
|
77
|
+
}
|
|
@@ -9,15 +9,19 @@
|
|
|
9
9
|
* needed to act (ref, run, parameters, follow-up command).
|
|
10
10
|
*
|
|
11
11
|
* Curation is one search with the fused ranking, the top `limit` hits, and
|
|
12
|
-
* per-hit enrichment (preview, run and parameters,
|
|
13
|
-
* optional reranker reorders the top fused
|
|
12
|
+
* per-hit enrichment (preview, run and parameters, support refs from the
|
|
13
|
+
* hit's declared links). An optional reranker reorders the top fused
|
|
14
|
+
* candidates first.
|
|
14
15
|
*
|
|
15
16
|
* The exported `akmCurate()` API is the single entry point; tests can also
|
|
16
17
|
* drive `curateSearchResults` with a fixture search response.
|
|
17
18
|
*/
|
|
19
|
+
import { parseBundleRef } from "../../core/asset/asset-ref.js";
|
|
20
|
+
import { typeNameFromConceptId } from "../../core/asset/resolve-ref.js";
|
|
18
21
|
import { loadConfig } from "../../core/config/config.js";
|
|
19
22
|
import { rethrowIfTestIsolationError, UsageError } from "../../core/errors.js";
|
|
20
23
|
import { appendEvent } from "../../core/events.js";
|
|
24
|
+
import { nonTaskInput } from "../../core/non-task-input.js";
|
|
21
25
|
import { redactCredentialPatterns } from "../../core/redaction.js";
|
|
22
26
|
import { withStateDbTelemetry } from "../../core/state-db.js";
|
|
23
27
|
import { searchHitContent } from "../../indexer/search/db-search.js";
|
|
@@ -33,8 +37,6 @@ import { akmSearch, parseSearchSource } from "./search.js";
|
|
|
33
37
|
import { akmShowUnified } from "./show.js";
|
|
34
38
|
const DEFAULT_CURATE_LIMIT = 4;
|
|
35
39
|
const MAX_CURATE_SUPPORT_REFS = 2;
|
|
36
|
-
/** The line of `src/assets/stash-skeleton/README.md` that reaches curate verbatim as a query. */
|
|
37
|
-
const STASH_README_LINE = "This is an **AKM stash** — a structured knowledge repository that stores reusable";
|
|
38
40
|
/** Fused candidates the reranker reorders when `search.curateRerank.topN` is unset. */
|
|
39
41
|
const DEFAULT_CURATE_RERANK_TOP_N = 30;
|
|
40
42
|
/** Characters of name, description and content sent to the reranker per candidate. */
|
|
@@ -131,21 +133,6 @@ export async function akmCurate(options) {
|
|
|
131
133
|
}
|
|
132
134
|
return result;
|
|
133
135
|
}
|
|
134
|
-
/**
|
|
135
|
-
* What the (trimmed) curate input is when it is not a task, else undefined.
|
|
136
|
-
* Harness and tool envelopes (`<task-notification>…`, `<system-reminder>…`,
|
|
137
|
-
* `<cross-session-message …>…`) start with a tag and close one, and the stash
|
|
138
|
-
* README line arrives verbatim; on the retrieval suite neither shape occurs in
|
|
139
|
-
* a real query. Length is not a signal: prompts over 2,000 characters found
|
|
140
|
-
* relevant assets at about the rate of shorter long prompts.
|
|
141
|
-
*/
|
|
142
|
-
function nonTaskInput(query) {
|
|
143
|
-
if (query.startsWith("<") && query.includes("</"))
|
|
144
|
-
return "a harness or tool envelope";
|
|
145
|
-
if (query === STASH_README_LINE)
|
|
146
|
-
return "the akm stash README boilerplate";
|
|
147
|
-
return undefined;
|
|
148
|
-
}
|
|
149
136
|
export async function curateSearchResults(query, result, limit, selectedType, eventSource) {
|
|
150
137
|
const allStashHits = result.hits.filter((hit) => hit.type !== "registry");
|
|
151
138
|
const registryHits = result.registryHits ?? [];
|
|
@@ -228,7 +215,7 @@ async function enrichCuratedStashHit(query, hit, selectedRefs, eventSource) {
|
|
|
228
215
|
}
|
|
229
216
|
const description = shown?.description ?? hit.description;
|
|
230
217
|
const preview = buildCuratedPreview(shown, hit);
|
|
231
|
-
const supportRefs = buildCurateSupportRefs(shown?.
|
|
218
|
+
const supportRefs = buildCurateSupportRefs(shown?.links, selectedRefs, hit.ref);
|
|
232
219
|
const item = {
|
|
233
220
|
source: "local",
|
|
234
221
|
type: shown?.type ?? hit.type,
|
|
@@ -340,17 +327,41 @@ async function maybeRerankCuratedStashHits(query, hits) {
|
|
|
340
327
|
function rerankDocumentTexts(hits) {
|
|
341
328
|
return hits.map((hit) => [hit.name, hit.description, searchHitContent(hit)].filter(Boolean).join("\n").slice(0, RERANK_DOCUMENT_CHARS));
|
|
342
329
|
}
|
|
343
|
-
/**
|
|
344
|
-
|
|
330
|
+
/**
|
|
331
|
+
* Up to {@link MAX_CURATE_SUPPORT_REFS} assets the hit's declared links (#935)
|
|
332
|
+
* name, not already selected: what the hit links to first, then what links to
|
|
333
|
+
* it, in the order `akm show` lists them. The LLM entity graph's `related`
|
|
334
|
+
* list no longer feeds them.
|
|
335
|
+
*/
|
|
336
|
+
function buildCurateSupportRefs(links, selectedRefs, ownerRef) {
|
|
345
337
|
const supportRefs = [];
|
|
346
|
-
for (const
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
338
|
+
for (const [part, direction] of [
|
|
339
|
+
["outgoing", "from"],
|
|
340
|
+
["incoming", "to"],
|
|
341
|
+
]) {
|
|
342
|
+
for (const [kind, group] of Object.entries(links?.[part] ?? {})) {
|
|
343
|
+
for (const ref of group.refs) {
|
|
344
|
+
if (ref === ownerRef || selectedRefs.has(ref) || supportRefs.some((existing) => existing.ref === ref))
|
|
345
|
+
continue;
|
|
346
|
+
const type = supportRefType(ref);
|
|
347
|
+
supportRefs.push({
|
|
348
|
+
ref,
|
|
349
|
+
...(type ? { type } : {}),
|
|
350
|
+
reason: `Declared link (${kind}) ${direction} this asset.`,
|
|
351
|
+
});
|
|
352
|
+
if (supportRefs.length >= MAX_CURATE_SUPPORT_REFS)
|
|
353
|
+
return supportRefs;
|
|
354
|
+
}
|
|
355
|
+
}
|
|
354
356
|
}
|
|
355
357
|
return supportRefs;
|
|
356
358
|
}
|
|
359
|
+
/** The asset type a ref's conceptId names (`memories/x` → `memory`), when it names one. */
|
|
360
|
+
function supportRefType(ref) {
|
|
361
|
+
try {
|
|
362
|
+
return typeNameFromConceptId(parseBundleRef(ref).conceptId)?.type;
|
|
363
|
+
}
|
|
364
|
+
catch {
|
|
365
|
+
return undefined;
|
|
366
|
+
}
|
|
367
|
+
}
|