akm-cli 0.9.0-beta.9 → 0.9.0-rc.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +592 -0
- package/README.md +12 -4
- package/dist/akm +38 -0
- package/dist/akm-migrate-storage +38 -0
- package/dist/assets/help/help-improve.md +9 -6
- package/dist/assets/hints/cli-hints-full.md +6 -5
- package/dist/assets/profiles/default.json +9 -4
- package/dist/assets/profiles/frequent.json +1 -1
- package/dist/assets/profiles/memory-focus.json +1 -1
- package/dist/assets/profiles/proactive-maintenance.json +25 -0
- package/dist/assets/profiles/quick.json +1 -1
- package/dist/assets/profiles/recombine-only.json +21 -0
- package/dist/assets/profiles/reflect-distill.json +30 -0
- package/dist/assets/profiles/synthesize.json +15 -0
- package/dist/assets/profiles/thorough.json +1 -1
- package/dist/assets/prompts/consolidate-system.md +23 -0
- package/dist/assets/prompts/contradiction-judge.md +33 -0
- package/dist/assets/prompts/distill-knowledge-system.md +22 -0
- package/dist/assets/prompts/distill-lesson-system.md +36 -0
- package/dist/assets/prompts/extract-session.md +11 -3
- package/dist/assets/prompts/graph-extract-system.md +1 -0
- package/dist/assets/prompts/graph-extract-user-prompt.md +1 -1
- package/dist/assets/prompts/memory-infer-system.md +1 -0
- package/dist/assets/prompts/memory-infer-user.md +5 -0
- package/dist/assets/prompts/metadata-enhance-system.md +1 -0
- package/dist/assets/prompts/procedural-system.md +44 -0
- package/dist/assets/prompts/recombine-system.md +40 -0
- package/dist/assets/prompts/staleness-detect-system.md +6 -0
- package/dist/assets/prompts/validate-summary-judge.md +1 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/agent.md +38 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/command.md +38 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/fact.md +39 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/knowledge.md +40 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/lesson.md +43 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/memory.md +38 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/script.md +43 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/skill.md +40 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/workflow.md +43 -0
- package/dist/assets/templates/html/health.html +281 -111
- package/dist/assets/wiki/ingest-workflow-template.md +45 -16
- package/dist/assets/wiki/schema-template.md +4 -4
- package/dist/cli/clack.js +56 -0
- package/dist/cli/config-migrate.js +7 -1
- package/dist/cli/confirm.js +1 -1
- package/dist/cli/parse-args.js +46 -1
- package/dist/cli/shared.js +28 -0
- package/dist/cli.js +25 -21
- package/dist/commands/agent/agent-dispatch.js +3 -2
- package/dist/commands/agent/agent-support.js +0 -7
- package/dist/commands/agent/contribute-cli.js +26 -7
- package/dist/commands/config-cli.js +26 -13
- package/dist/commands/env/child-env.js +47 -0
- package/dist/commands/env/env-cli.js +220 -227
- package/dist/commands/env/env.js +14 -67
- package/dist/commands/env/secret-cli.js +140 -138
- package/dist/commands/feedback-cli.js +156 -155
- package/dist/commands/graph/graph-cli.js +5 -13
- package/dist/commands/graph/graph.js +3 -3
- package/dist/commands/health/advisories.js +151 -0
- package/dist/commands/health/checks.js +103 -16
- package/dist/commands/health/html-report.js +447 -81
- package/dist/commands/health/improve-metrics.js +771 -0
- package/dist/commands/health/llm-usage.js +65 -0
- package/dist/commands/health/md-report.js +103 -0
- package/dist/commands/health/metrics.js +278 -0
- package/dist/commands/health/stash-exposure.js +46 -0
- package/dist/commands/health/surfaces.js +216 -0
- package/dist/commands/health/task-runs.js +135 -0
- package/dist/commands/health/types.js +26 -0
- package/dist/commands/health/windows.js +195 -0
- package/dist/commands/health.js +91 -1091
- package/dist/commands/improve/anti-collapse.js +170 -0
- package/dist/commands/improve/calibration.js +161 -0
- package/dist/commands/improve/collapse-detector.js +421 -0
- package/dist/commands/improve/consolidate/chunking.js +141 -0
- package/dist/commands/improve/consolidate/eligibility.js +64 -0
- package/dist/commands/improve/consolidate/merge.js +145 -0
- package/dist/commands/improve/consolidate/sanitize.js +231 -0
- package/dist/commands/{lint.js → improve/consolidate/types.js} +1 -1
- package/dist/commands/improve/consolidate.js +1295 -1277
- package/dist/commands/improve/dedup.js +482 -0
- package/dist/commands/improve/distill/content-repair.js +202 -0
- package/dist/commands/improve/distill/promote-memory.js +229 -0
- package/dist/commands/improve/distill/quality-gate.js +236 -0
- package/dist/commands/improve/distill-guards.js +127 -0
- package/dist/commands/improve/distill-promotion-policy.js +826 -167
- package/dist/commands/improve/distill.js +228 -605
- package/dist/commands/improve/eligibility.js +434 -0
- package/dist/commands/improve/encoding-salience.js +205 -0
- package/dist/commands/improve/extract-cli.js +179 -59
- package/dist/commands/improve/extract-prompt.js +54 -3
- package/dist/commands/improve/extract-watch.js +140 -0
- package/dist/commands/improve/extract.js +409 -43
- package/dist/commands/improve/feedback-valence.js +54 -0
- package/dist/commands/improve/hot-probation.js +45 -0
- package/dist/commands/improve/improve-auto-accept.js +157 -10
- package/dist/commands/improve/improve-cli.js +115 -73
- package/dist/commands/improve/improve-profiles.js +28 -8
- package/dist/commands/improve/improve-result-file.js +15 -25
- package/dist/commands/improve/improve-session.js +58 -0
- package/dist/commands/improve/improve.js +485 -2764
- package/dist/commands/improve/locks.js +154 -0
- package/dist/commands/improve/loop-stages.js +1100 -0
- package/dist/commands/improve/memory/memory-belief.js +14 -15
- package/dist/commands/improve/memory/memory-contradiction-detect.js +83 -60
- package/dist/commands/improve/memory/memory-improve.js +27 -27
- package/dist/commands/improve/outcome-loop.js +270 -0
- package/dist/commands/improve/preparation.js +2002 -0
- package/dist/commands/improve/proactive-maintenance.js +37 -35
- package/dist/commands/improve/procedural.js +398 -0
- package/dist/commands/improve/recombine.js +818 -0
- package/dist/commands/improve/reflect-noise.js +0 -0
- package/dist/commands/improve/reflect.js +206 -45
- package/dist/commands/improve/salience.js +455 -0
- package/dist/commands/improve/schema-similarity-gate.js +168 -0
- package/dist/commands/improve/shared.js +51 -0
- package/dist/commands/improve/triage.js +93 -0
- package/dist/commands/lint/agent-linter.js +19 -24
- package/dist/commands/lint/base-linter.js +173 -60
- package/dist/commands/lint/command-linter.js +19 -24
- package/dist/commands/lint/env-key-rules.js +38 -1
- package/dist/commands/lint/fact-linter.js +39 -0
- package/dist/commands/lint/index.js +31 -13
- package/dist/commands/lint/memory-linter.js +1 -1
- package/dist/commands/lint/registry.js +7 -2
- package/dist/commands/lint/task-linter.js +3 -3
- package/dist/commands/lint/workflow-linter.js +26 -1
- package/dist/commands/observability-cli.js +4 -4
- package/dist/commands/proposal/drain-policies.js +13 -4
- package/dist/commands/proposal/drain.js +45 -51
- package/dist/commands/proposal/legacy-import.js +115 -0
- package/dist/commands/proposal/proposal-cli.js +24 -34
- package/dist/commands/proposal/proposal.js +2 -1
- package/dist/commands/proposal/propose.js +8 -3
- package/dist/commands/proposal/repository.js +829 -0
- package/dist/commands/proposal/validators/proposal-quality-validators.js +9 -8
- package/dist/commands/proposal/validators/proposals.js +93 -895
- package/dist/commands/read/curate.js +410 -111
- package/dist/commands/read/knowledge.js +10 -3
- package/dist/commands/read/remember-cli.js +133 -138
- package/dist/commands/read/search-cli.js +15 -8
- package/dist/commands/read/search.js +22 -11
- package/dist/commands/read/show.js +106 -14
- package/dist/commands/registry-cli.js +76 -87
- package/dist/commands/remember.js +11 -12
- package/dist/commands/sources/add-cli.js +91 -95
- package/dist/commands/sources/history.js +1 -1
- package/dist/commands/sources/init.js +66 -18
- package/dist/commands/sources/installed-stashes.js +11 -3
- package/dist/commands/sources/schema-repair.js +44 -46
- package/dist/commands/sources/self-update.js +2 -2
- package/dist/commands/sources/source-add.js +7 -3
- package/dist/commands/sources/sources-cli.js +3 -3
- package/dist/commands/sources/stash-cli.js +19 -39
- package/dist/commands/sources/stash-skeleton.js +57 -8
- package/dist/commands/tasks/default-tasks.js +15 -2
- package/dist/commands/tasks/tasks-cli.js +20 -29
- package/dist/commands/tasks/tasks.js +39 -11
- package/dist/commands/wiki-cli.js +23 -38
- package/dist/commands/workflow-cli.js +15 -1
- package/dist/core/asset/asset-registry.js +3 -1
- package/dist/core/asset/asset-spec.js +21 -4
- package/dist/core/asset/frontmatter.js +188 -167
- package/dist/core/asset/markdown.js +8 -0
- package/dist/core/authoring-rules.js +92 -0
- package/dist/core/common.js +4 -23
- package/dist/core/concurrent.js +10 -1
- package/dist/core/config/config-io.js +10 -1
- package/dist/core/config/config-migration.js +18 -40
- package/dist/core/config/config-schema.js +382 -62
- package/dist/core/config/config-types.js +3 -3
- package/dist/core/config/config.js +67 -22
- package/dist/core/deep-merge.js +38 -0
- package/dist/core/errors.js +1 -0
- package/dist/core/eval/rank-metrics.js +113 -0
- package/dist/core/events.js +4 -7
- package/dist/core/improve-types.js +47 -8
- package/dist/core/logs-db.js +14 -75
- package/dist/core/parse.js +36 -16
- package/dist/core/paths.js +18 -18
- package/dist/core/standards/resolve-standards-context.js +87 -0
- package/dist/core/standards/resolve-stash-standards.js +99 -0
- package/dist/core/standards/resolve-type-conventions.js +66 -0
- package/dist/core/state/migrations.js +770 -0
- package/dist/core/state-db.js +132 -1126
- package/dist/core/structured.js +69 -0
- package/dist/core/time.js +53 -0
- package/dist/core/warn.js +21 -0
- package/dist/core/write-source.js +37 -0
- package/dist/indexer/db/db.js +259 -769
- package/dist/indexer/db/entry-mapper.js +41 -0
- package/dist/indexer/db/graph-db.js +129 -86
- package/dist/indexer/db/llm-cache.js +2 -2
- package/dist/indexer/db/schema.js +516 -0
- package/dist/indexer/ensure-index.js +36 -92
- package/dist/indexer/feedback/utility-policy.js +75 -0
- package/dist/indexer/graph/graph-boost.js +51 -41
- package/dist/indexer/graph/graph-extraction.js +207 -4
- package/dist/indexer/index-writer-lock.js +18 -11
- package/dist/indexer/index-written-assets.js +105 -0
- package/dist/indexer/indexer.js +182 -204
- package/dist/indexer/passes/dir-staleness.js +114 -0
- package/dist/indexer/passes/memory-inference.js +13 -5
- package/dist/indexer/passes/metadata.js +20 -0
- package/dist/indexer/read-preflight.js +23 -0
- package/dist/indexer/search/db-search.js +89 -13
- package/dist/indexer/search/fts-query.js +51 -0
- package/dist/indexer/search/ranking-contributors.js +95 -9
- package/dist/indexer/search/ranking.js +79 -3
- package/dist/indexer/search/search-fields.js +6 -0
- package/dist/indexer/search/search-source.js +32 -21
- package/dist/indexer/search/semantic-status.js +4 -0
- package/dist/indexer/walk/matchers.js +9 -0
- package/dist/indexer/walk/walker.js +21 -13
- package/dist/integrations/agent/builders.js +39 -13
- package/dist/integrations/agent/config.js +20 -59
- package/dist/integrations/agent/detect.js +9 -0
- package/dist/integrations/agent/index.js +3 -19
- package/dist/integrations/agent/model-aliases.js +7 -2
- package/dist/integrations/agent/profiles.js +7 -1
- package/dist/integrations/agent/prompts.js +75 -9
- package/dist/integrations/agent/runner-dispatch.js +59 -0
- package/dist/integrations/agent/runner.js +13 -9
- package/dist/integrations/agent/spawn.js +69 -67
- package/dist/integrations/harnesses/claude/agent-builder.js +1 -1
- package/dist/integrations/harnesses/claude/index.js +2 -0
- package/dist/integrations/harnesses/claude/session-log.js +10 -0
- package/dist/integrations/harnesses/index.js +2 -3
- package/dist/integrations/harnesses/opencode/agent-builder.js +1 -1
- package/dist/integrations/harnesses/opencode/index.js +2 -0
- package/dist/integrations/harnesses/opencode/session-log.js +173 -3
- package/dist/integrations/harnesses/opencode-sdk/index.js +2 -2
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +98 -17
- package/dist/integrations/harnesses/types.js +1 -0
- package/dist/integrations/session-logs/index.js +16 -0
- package/dist/llm/call-ai.js +2 -2
- package/dist/llm/client.js +34 -11
- package/dist/llm/embedder.js +67 -4
- package/dist/llm/embedders/cache.js +3 -1
- package/dist/llm/embedders/deterministic.js +66 -0
- package/dist/llm/embedders/local.js +73 -3
- package/dist/llm/feature-gate.js +16 -15
- package/dist/llm/graph-extract.js +67 -44
- package/dist/llm/memory-infer-impl.js +138 -0
- package/dist/llm/memory-infer.js +1 -127
- package/dist/llm/metadata-enhance.js +44 -31
- package/dist/llm/structured-call.js +49 -0
- package/dist/migrate-storage-node.mjs +8 -0
- package/dist/output/context.js +5 -5
- package/dist/output/renderers.js +85 -14
- package/dist/output/shapes/curate.js +14 -2
- package/dist/output/shapes/helpers.js +0 -3
- package/dist/output/shapes/passthrough.js +2 -1
- package/dist/output/text/helpers.js +29 -1
- package/dist/output/text/workflow.js +1 -0
- package/dist/registry/providers/skills-sh.js +21 -147
- package/dist/registry/providers/static-index.js +15 -157
- package/dist/registry/resolve.js +27 -9
- package/dist/runtime.js +25 -1
- package/dist/scripts/migrate-storage.js +2661 -2369
- package/dist/scripts/migrations/import-fs-improve-runs-to-db.js +883 -596
- package/dist/setup/detect.js +9 -0
- package/dist/setup/legacy-config.js +106 -0
- package/dist/setup/prompt.js +57 -0
- package/dist/setup/providers.js +14 -0
- package/dist/setup/registry-stash-loader.js +12 -0
- package/dist/setup/semantic-assets.js +124 -0
- package/dist/setup/setup.js +52 -1614
- package/dist/setup/steps/connection.js +734 -0
- package/dist/setup/steps/output.js +31 -0
- package/dist/setup/steps/platforms.js +124 -0
- package/dist/setup/steps/semantic.js +27 -0
- package/dist/setup/steps/sources.js +222 -0
- package/dist/setup/steps/stashdir.js +42 -0
- package/dist/setup/steps/tasks.js +152 -0
- package/dist/sources/include.js +6 -2
- package/dist/sources/providers/filesystem.js +0 -1
- package/dist/sources/providers/git-install.js +210 -0
- package/dist/sources/providers/git-provider.js +234 -0
- package/dist/sources/providers/git-stash.js +248 -0
- package/dist/sources/providers/git.js +10 -661
- package/dist/sources/providers/npm.js +2 -6
- package/dist/sources/providers/provider-utils.js +13 -7
- package/dist/sources/providers/sync-from-ref.js +9 -1
- package/dist/sources/providers/website.js +9 -5
- package/dist/sources/website-ingest.js +187 -29
- package/dist/sources/wiki-fetchers/registry.js +53 -0
- package/dist/sources/wiki-fetchers/youtube.js +239 -0
- package/dist/storage/database.js +45 -10
- package/dist/storage/managed-db.js +82 -0
- package/dist/storage/repositories/canaries-repository.js +107 -0
- package/dist/storage/repositories/consolidation-repository.js +38 -0
- package/dist/storage/repositories/embeddings-repository.js +72 -0
- package/dist/storage/repositories/events-repository.js +187 -0
- package/dist/storage/repositories/extract-sessions-repository.js +96 -0
- package/dist/storage/repositories/improve-runs-repository.js +146 -0
- package/dist/storage/repositories/index-db.js +14 -8
- package/dist/storage/repositories/proposals-repository.js +220 -0
- package/dist/storage/repositories/recombine-repository.js +213 -0
- package/dist/storage/repositories/registry-cache.js +93 -0
- package/dist/storage/repositories/registry-index-cache-repository.js +46 -0
- package/dist/storage/repositories/task-history-repository.js +93 -0
- package/dist/storage/sqlite-pragmas.js +146 -0
- package/dist/tasks/backends/cron.js +1 -1
- package/dist/tasks/backends/index.js +9 -0
- package/dist/tasks/backends/launchd.js +1 -1
- package/dist/tasks/backends/schtasks.js +1 -1
- package/dist/tasks/{resolveAkmBin.js → resolve-akm-bin.js} +2 -2
- package/dist/tasks/runner.js +15 -13
- package/dist/text-import-hook.mjs +0 -0
- package/dist/wiki/wiki.js +52 -11
- package/dist/workflows/cli.js +1 -0
- package/dist/workflows/db.js +3 -4
- package/dist/workflows/runtime/runs.js +43 -118
- package/dist/workflows/runtime/workflow-asset-loader.js +125 -0
- package/dist/workflows/validate-summary.js +2 -7
- package/docs/README.md +69 -18
- package/docs/data-and-telemetry.md +5 -4
- package/docs/migration/release-notes/0.7.0.md +1 -1
- package/docs/migration/release-notes/0.9.0.md +39 -0
- package/package.json +10 -10
- package/dist/assets/tasks/core/update-stashes.yml +0 -4
- package/dist/commands/db-cli.js +0 -23
- package/dist/indexer/db/db-backup.js +0 -376
- package/dist/indexer/passes/staleness-detect.js +0 -488
|
@@ -25,7 +25,9 @@ export function getCachedEmbedding(key) {
|
|
|
25
25
|
return cached;
|
|
26
26
|
}
|
|
27
27
|
export function setCachedEmbedding(key, value) {
|
|
28
|
-
//
|
|
28
|
+
// Delete first so an overwrite refreshes LRU recency AND is not counted as a
|
|
29
|
+
// new insert: only a genuinely new key at capacity should evict the oldest.
|
|
30
|
+
embedCache.delete(key);
|
|
29
31
|
if (embedCache.size >= EMBED_CACHE_MAX) {
|
|
30
32
|
const oldest = embedCache.keys().next().value;
|
|
31
33
|
if (oldest !== undefined) {
|
|
@@ -0,0 +1,66 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/** Env var that switches the whole embedding facade into deterministic mode. */
|
|
5
|
+
export const DETERMINISTIC_EMBED_ENV = "AKM_EMBED_DETERMINISTIC";
|
|
6
|
+
/**
|
|
7
|
+
* Vector width. Matches the default local model (`bge-small`, 384 dims) so the
|
|
8
|
+
* index DB's embedding column and sqlite-vec table dimensions line up without
|
|
9
|
+
* any extra config.
|
|
10
|
+
*/
|
|
11
|
+
export const DETERMINISTIC_EMBED_DIM = 384;
|
|
12
|
+
/**
|
|
13
|
+
* Stable model id reported for deterministic mode. Used as the embedding
|
|
14
|
+
* `model_id` and folded into the provider fingerprint so a deterministic index
|
|
15
|
+
* is never confused with a real-model index (and vice versa).
|
|
16
|
+
*/
|
|
17
|
+
export const DETERMINISTIC_EMBED_MODEL_ID = "akm-deterministic-hash-v1";
|
|
18
|
+
/** True when deterministic embedding is enabled via env. */
|
|
19
|
+
export function isDeterministicEmbedEnabled() {
|
|
20
|
+
return process.env[DETERMINISTIC_EMBED_ENV] === "1";
|
|
21
|
+
}
|
|
22
|
+
/** FNV-1a 32-bit hash. Platform- and version-stable. */
|
|
23
|
+
function fnv1a(str) {
|
|
24
|
+
let h = 0x811c9dc5;
|
|
25
|
+
for (let i = 0; i < str.length; i++) {
|
|
26
|
+
h ^= str.charCodeAt(i);
|
|
27
|
+
// 32-bit FNV prime multiply via shifts to stay in uint32.
|
|
28
|
+
h = (h + ((h << 1) + (h << 4) + (h << 7) + (h << 8) + (h << 24))) >>> 0;
|
|
29
|
+
}
|
|
30
|
+
return h >>> 0;
|
|
31
|
+
}
|
|
32
|
+
/** Lowercase, split on non-alphanumeric, drop empties. */
|
|
33
|
+
function tokenize(text) {
|
|
34
|
+
return text
|
|
35
|
+
.toLowerCase()
|
|
36
|
+
.split(/[^a-z0-9]+/)
|
|
37
|
+
.filter((t) => t.length > 0);
|
|
38
|
+
}
|
|
39
|
+
/**
|
|
40
|
+
* Deterministically embed `text` into a unit-length vector of width `dim`
|
|
41
|
+
* using feature hashing. Empty / token-less input returns a fixed unit
|
|
42
|
+
* vector so cosine similarity never sees a zero vector (NaN guard).
|
|
43
|
+
*/
|
|
44
|
+
export function deterministicEmbed(text, dim = DETERMINISTIC_EMBED_DIM) {
|
|
45
|
+
const vec = new Array(dim).fill(0);
|
|
46
|
+
const tokens = tokenize(text);
|
|
47
|
+
for (const tok of tokens) {
|
|
48
|
+
const h = fnv1a(tok);
|
|
49
|
+
const idx = h % dim;
|
|
50
|
+
// Use a higher bit for the sign so it is independent of the bucket index.
|
|
51
|
+
const sign = (h >>> 16) & 1 ? 1 : -1;
|
|
52
|
+
vec[idx] += sign;
|
|
53
|
+
}
|
|
54
|
+
let norm = 0;
|
|
55
|
+
for (const v of vec)
|
|
56
|
+
norm += v * v;
|
|
57
|
+
norm = Math.sqrt(norm);
|
|
58
|
+
if (norm === 0) {
|
|
59
|
+
// No usable tokens — return a fixed, stable unit vector.
|
|
60
|
+
vec[0] = 1;
|
|
61
|
+
return vec;
|
|
62
|
+
}
|
|
63
|
+
for (let i = 0; i < dim; i++)
|
|
64
|
+
vec[i] /= norm;
|
|
65
|
+
return vec;
|
|
66
|
+
}
|
|
@@ -19,8 +19,30 @@ import { getDirname, resolveModule } from "../../runtime.js";
|
|
|
19
19
|
* `all-MiniLM-L6-v2` at the same 384-dimension footprint.
|
|
20
20
|
*/
|
|
21
21
|
export const DEFAULT_LOCAL_MODEL = "Xenova/bge-small-en-v1.5";
|
|
22
|
+
/** Type-guard: true when the value looks like a batch Tensor (has .dims). */
|
|
23
|
+
function isBatchTensor(v) {
|
|
24
|
+
return (v !== null &&
|
|
25
|
+
typeof v === "object" &&
|
|
26
|
+
"data" in v &&
|
|
27
|
+
"dims" in v &&
|
|
28
|
+
Array.isArray(v.dims) &&
|
|
29
|
+
v.dims.length >= 2);
|
|
30
|
+
}
|
|
31
|
+
const realTransformersLoader = () => import("@huggingface/transformers");
|
|
32
|
+
let transformersLoader = realTransformersLoader;
|
|
33
|
+
/** TEST-ONLY. Swap the transformers module loader; pass undefined to restore. */
|
|
34
|
+
export function _setTransformersLoaderForTests(fake) {
|
|
35
|
+
transformersLoader = fake ?? realTransformersLoader;
|
|
36
|
+
}
|
|
22
37
|
const LOCAL_EMBEDDER_DTYPE = "fp32";
|
|
23
38
|
const LOCAL_EMBEDDER_FALLBACK_DTYPE = "auto";
|
|
39
|
+
/**
|
|
40
|
+
* Maximum texts per batch for the local transformers pipeline. The pipeline
|
|
41
|
+
* can run genuine batched inference over a string array; 32 is a safe default
|
|
42
|
+
* that fits well inside most model context budgets while providing 10–50×
|
|
43
|
+
* throughput improvement over one-at-a-time calls on the cold minority.
|
|
44
|
+
*/
|
|
45
|
+
const LOCAL_BATCH_SIZE = 32;
|
|
24
46
|
/**
|
|
25
47
|
* Return the local model name that will be used for embedding.
|
|
26
48
|
* When `overrideModel` is provided it takes precedence; otherwise
|
|
@@ -77,15 +99,63 @@ export class LocalEmbedder {
|
|
|
77
99
|
}
|
|
78
100
|
return this.embedWithModel(text, this.defaultModel);
|
|
79
101
|
}
|
|
102
|
+
/**
|
|
103
|
+
* Embed a batch of texts. Processes in chunks of `LOCAL_BATCH_SIZE` (32) so
|
|
104
|
+
* the transformers pipeline can run genuine batched inference rather than one
|
|
105
|
+
* call per text. Falls back to one-at-a-time if the pipeline does not support
|
|
106
|
+
* array input (older versions of @huggingface/transformers). Each chunk is
|
|
107
|
+
* checked against the AbortSignal between calls.
|
|
108
|
+
*/
|
|
80
109
|
async embedBatch(texts, signal) {
|
|
81
110
|
if (texts.length === 0)
|
|
82
111
|
return [];
|
|
112
|
+
if (signal?.aborted) {
|
|
113
|
+
throw signal.reason instanceof Error ? signal.reason : new Error("embedding interrupted");
|
|
114
|
+
}
|
|
115
|
+
const pipeline = await this.getPipeline(this.defaultModel);
|
|
83
116
|
const results = [];
|
|
84
|
-
for (
|
|
117
|
+
for (let i = 0; i < texts.length; i += LOCAL_BATCH_SIZE) {
|
|
85
118
|
if (signal?.aborted) {
|
|
86
119
|
throw signal.reason instanceof Error ? signal.reason : new Error("embedding interrupted");
|
|
87
120
|
}
|
|
88
|
-
|
|
121
|
+
const chunk = texts.slice(i, i + LOCAL_BATCH_SIZE);
|
|
122
|
+
try {
|
|
123
|
+
// @huggingface/transformers feature-extraction pipeline accepts a
|
|
124
|
+
// string[] and returns a batch Tensor (NOT an Array<{data}>).
|
|
125
|
+
// The Tensor has .data (flat Float32Array, length = batch * dim) and
|
|
126
|
+
// .dims = [batch, dim]. Slice .data into per-row vectors using .dims.
|
|
127
|
+
const batchResult = await pipeline(chunk, {
|
|
128
|
+
pooling: "mean",
|
|
129
|
+
normalize: true,
|
|
130
|
+
});
|
|
131
|
+
if (isBatchTensor(batchResult)) {
|
|
132
|
+
const dim = batchResult.dims[1];
|
|
133
|
+
for (let row = 0; row < chunk.length; row++) {
|
|
134
|
+
results.push(Array.from(batchResult.data.subarray(row * dim, (row + 1) * dim)));
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
else if (Array.isArray(batchResult)) {
|
|
138
|
+
// Older versions of @huggingface/transformers returned Array<{data}>.
|
|
139
|
+
for (const r of batchResult) {
|
|
140
|
+
results.push(Array.from(r.data));
|
|
141
|
+
}
|
|
142
|
+
}
|
|
143
|
+
else {
|
|
144
|
+
// Single-text result returned for a chunk — should not happen for
|
|
145
|
+
// string[] input, but handle defensively.
|
|
146
|
+
throw new Error("unexpected pipeline return shape for batch input");
|
|
147
|
+
}
|
|
148
|
+
}
|
|
149
|
+
catch {
|
|
150
|
+
// Fallback: process one-at-a-time (older pipeline versions or mismatched
|
|
151
|
+
// return type). Fail-open per text: a single failure aborts the chunk.
|
|
152
|
+
for (const text of chunk) {
|
|
153
|
+
if (signal?.aborted) {
|
|
154
|
+
throw signal.reason instanceof Error ? signal.reason : new Error("embedding interrupted");
|
|
155
|
+
}
|
|
156
|
+
results.push(await this.embedWithModel(text, this.defaultModel));
|
|
157
|
+
}
|
|
158
|
+
}
|
|
89
159
|
}
|
|
90
160
|
return results;
|
|
91
161
|
}
|
|
@@ -116,7 +186,7 @@ export class LocalEmbedder {
|
|
|
116
186
|
}
|
|
117
187
|
let pipeline;
|
|
118
188
|
try {
|
|
119
|
-
const mod = await
|
|
189
|
+
const mod = await transformersLoader();
|
|
120
190
|
pipeline = mod.pipeline;
|
|
121
191
|
}
|
|
122
192
|
catch (importError) {
|
package/dist/llm/feature-gate.js
CHANGED
|
@@ -22,18 +22,25 @@ const FEATURE_LOCATION = {
|
|
|
22
22
|
graph_extraction: (cfg) => cfg.profiles?.improve?.default?.processes?.graphExtraction?.enabled ?? true,
|
|
23
23
|
// Legacy default: false
|
|
24
24
|
metadata_enhance: (cfg) => cfg.index?.metadataEnhance?.enabled ?? false,
|
|
25
|
-
//
|
|
26
|
-
|
|
27
|
-
//
|
|
28
|
-
|
|
25
|
+
// Default ON since R3 (docs/design/improve-self-learning-analysis.md G5):
|
|
26
|
+
// distill is a primary acquisition path, so the gate guards minted content by
|
|
27
|
+
// default. The judge fails CLOSED (07 P0-2): no LLM / timeout / parse failure
|
|
28
|
+
// reject the proposal rather than passing it through — an unjudgeable proposal
|
|
29
|
+
// must not slip into the stash. Opt out via
|
|
30
|
+
// profiles.improve.default.processes.distill.qualityGate.enabled: false.
|
|
31
|
+
lesson_quality_gate: (cfg) => cfg.profiles?.improve?.default?.processes?.distill?.qualityGate?.enabled ?? true,
|
|
29
32
|
// Legacy default: false
|
|
30
33
|
proposal_quality_gate: (cfg) => cfg.profiles?.improve?.default?.processes?.reflect?.qualityGate?.enabled ?? false,
|
|
31
34
|
// Legacy default: false
|
|
32
35
|
memory_contradiction_detection: (cfg) => cfg.profiles?.improve?.default?.processes?.consolidate?.contradictionDetection?.enabled ?? false,
|
|
33
|
-
//
|
|
34
|
-
//
|
|
35
|
-
//
|
|
36
|
-
|
|
36
|
+
// Always on at the LLM-wrapper level. Enablement is decided ONCE at the
|
|
37
|
+
// extract entry point (`akmExtract`): the `extract.enabled` process toggle
|
|
38
|
+
// gates extract as a STAGE of `akm improve` (the active improve profile, per
|
|
39
|
+
// #593/#594), while an explicit `akm extract` command always runs. Gating the
|
|
40
|
+
// inner LLM calls on `default.processes.extract.enabled` here was a footgun —
|
|
41
|
+
// dropping extract from the daily improve profile silently disabled the
|
|
42
|
+
// standalone `akm extract` command. (cfg unused — kept for resolver signature.)
|
|
43
|
+
session_extraction: (_cfg) => true,
|
|
37
44
|
};
|
|
38
45
|
/**
|
|
39
46
|
* Pure predicate: is the named feature gate enabled in `config`?
|
|
@@ -89,14 +96,11 @@ export async function tryLlmFeature(feature, config, fn, fallback, opts) {
|
|
|
89
96
|
export function isProcessEnabled(section, processName, config) {
|
|
90
97
|
if (!config)
|
|
91
98
|
return false;
|
|
92
|
-
// index.metadataEnhance
|
|
99
|
+
// index.metadataEnhance is a first-class new-shape entry.
|
|
93
100
|
if (section === "index") {
|
|
94
101
|
if (processName === "metadata_enhance" || processName === "metadataEnhance") {
|
|
95
102
|
return config.index?.metadataEnhance?.enabled ?? true;
|
|
96
103
|
}
|
|
97
|
-
if (processName === "staleness_detection" || processName === "stalenessDetection") {
|
|
98
|
-
return config.index?.stalenessDetection?.enabled ?? false;
|
|
99
|
-
}
|
|
100
104
|
if (processName === "memory_inference" || processName === "memoryInference") {
|
|
101
105
|
return isLlmFeatureEnabled(config, "memory_inference");
|
|
102
106
|
}
|
|
@@ -104,9 +108,6 @@ export function isProcessEnabled(section, processName, config) {
|
|
|
104
108
|
return isLlmFeatureEnabled(config, "graph_extraction");
|
|
105
109
|
}
|
|
106
110
|
}
|
|
107
|
-
if (section === "search" && (processName === "curate_rerank" || processName === "curateRerank")) {
|
|
108
|
-
return config.search?.curateRerank?.enabled ?? false;
|
|
109
|
-
}
|
|
110
111
|
if (section === "improve") {
|
|
111
112
|
const processes = config.profiles?.improve?.default?.processes;
|
|
112
113
|
const entry = processes?.[processName];
|
|
@@ -20,11 +20,13 @@
|
|
|
20
20
|
* the connection via `resolveIndexPassLLM("graph", config)` and pass it
|
|
21
21
|
* straight through.
|
|
22
22
|
*/
|
|
23
|
+
import systemPromptTemplate from "../assets/prompts/graph-extract-system.md" with { type: "text" };
|
|
23
24
|
import userPromptTemplate from "../assets/prompts/graph-extract-user-prompt.md" with { type: "text" };
|
|
24
25
|
import { toErrorMessage } from "../core/common.js";
|
|
25
26
|
import { warn, warnVerbose } from "../core/warn.js";
|
|
26
|
-
import { chatCompletion,
|
|
27
|
+
import { chatCompletion, isContextSizeError, parseEmbeddedJsonResponse } from "./client.js";
|
|
27
28
|
import { tryLlmFeature } from "./feature-gate.js";
|
|
29
|
+
import { callStructured } from "./structured-call.js";
|
|
28
30
|
/**
|
|
29
31
|
* Separator token used between assets in a batch prompt.
|
|
30
32
|
* Chosen to be visually clear and unlikely to appear verbatim in asset bodies.
|
|
@@ -41,30 +43,15 @@ const NON_ARRAY_BATCH_DISABLE_THRESHOLD = 2;
|
|
|
41
43
|
const MAX_ENTITIES_PER_ASSET = 32;
|
|
42
44
|
/** Hard cap on relations returned per asset. */
|
|
43
45
|
const MAX_RELATIONS_PER_ASSET = 32;
|
|
44
|
-
const SYSTEM_PROMPT =
|
|
46
|
+
const SYSTEM_PROMPT = systemPromptTemplate;
|
|
45
47
|
const USER_PROMPT_PREFIX = userPromptTemplate
|
|
46
48
|
.replace("{{MAX_ENTITIES}}", String(MAX_ENTITIES_PER_ASSET))
|
|
47
49
|
.replace("{{MAX_RELATIONS}}", String(MAX_RELATIONS_PER_ASSET));
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
* model prose merely mentioning "context size" / "context length" (e.g. gemma
|
|
54
|
-
* narrating about a document) does not get misclassified as a provider
|
|
55
|
-
* context-limit error (#496).
|
|
56
|
-
*/
|
|
57
|
-
export function isContextSizeError(message) {
|
|
58
|
-
const lower = message.toLowerCase();
|
|
59
|
-
const contextKw = /context (size|length|window)|prompt too long|exceeds.*context/.test(lower);
|
|
60
|
-
if (!contextKw) {
|
|
61
|
-
return false;
|
|
62
|
-
}
|
|
63
|
-
const evidence = /\b\d+\s*(token|tokens|tk)\b/.test(lower) ||
|
|
64
|
-
/max(imum)?\s+(context|token|input)/.test(lower) ||
|
|
65
|
-
/exceeded|over.*limit|too.*long/.test(lower);
|
|
66
|
-
return evidence;
|
|
67
|
-
}
|
|
50
|
+
// `isContextSizeError` is defined in `./client` and re-exported here so the
|
|
51
|
+
// graph extractor and the retry classifier (`isRetryable`) share one
|
|
52
|
+
// definition (#496). Re-exported (not just imported) to preserve existing
|
|
53
|
+
// importers of this module — including its unit test.
|
|
54
|
+
export { isContextSizeError } from "./client.js";
|
|
68
55
|
const GENERIC_ENTITIES = new Set([
|
|
69
56
|
"agent",
|
|
70
57
|
"application",
|
|
@@ -327,7 +314,13 @@ function parseGraphExtraction(raw) {
|
|
|
327
314
|
if (!normalized)
|
|
328
315
|
continue;
|
|
329
316
|
const normalizedKey = normalized.toLowerCase();
|
|
330
|
-
|
|
317
|
+
// Drop generic/empty entities AND raw file/dir paths (anything with a
|
|
318
|
+
// path separator) — the prompt no longer asks for them and isJunkEntity
|
|
319
|
+
// discards them downstream, so emitting them is pure waste/junk (#632).
|
|
320
|
+
if (!/[a-z0-9]/i.test(normalized) ||
|
|
321
|
+
GENERIC_ENTITIES.has(normalizedKey) ||
|
|
322
|
+
normalized.includes("/") ||
|
|
323
|
+
normalized.includes("\\")) {
|
|
331
324
|
filteredGenericEntities += 1;
|
|
332
325
|
continue;
|
|
333
326
|
}
|
|
@@ -432,6 +425,18 @@ function buildBatchSystemPrompt() {
|
|
|
432
425
|
"The array length MUST equal the number of assets provided. " +
|
|
433
426
|
'Use {"entities":[],"relations":[]} for assets with no extractable graph content.');
|
|
434
427
|
}
|
|
428
|
+
/**
|
|
429
|
+
* Hardened system prompt for the single batch retry (#635). Used only after a
|
|
430
|
+
* first response failed array salvage — leans harder on "raw array only" so a
|
|
431
|
+
* model that wrapped the array in prose/fences corrects itself before we pay
|
|
432
|
+
* the per-asset fallback.
|
|
433
|
+
*/
|
|
434
|
+
function buildBatchRetrySystemPrompt() {
|
|
435
|
+
return (`${buildBatchSystemPrompt()} ` +
|
|
436
|
+
"Your previous response could NOT be parsed as a JSON array. " +
|
|
437
|
+
"Respond with ONLY the raw JSON array — start with '[' and end with ']'. " +
|
|
438
|
+
"No prose, no explanation, no markdown code fences, no preamble.");
|
|
439
|
+
}
|
|
435
440
|
function buildBatchUserPrompt(bodies) {
|
|
436
441
|
const count = bodies.length;
|
|
437
442
|
const assetBlocks = bodies.map((body, i) => `${BATCH_ASSET_SEPARATOR} ${i + 1} ===\n${body.trim()}`).join("\n\n");
|
|
@@ -541,7 +546,21 @@ export async function extractGraphFromBodies(llmConfig, bodies, signal, akmConfi
|
|
|
541
546
|
});
|
|
542
547
|
if (!raw)
|
|
543
548
|
return null;
|
|
544
|
-
|
|
549
|
+
// Array-preferring salvage (#635): the batch contract is a top-level
|
|
550
|
+
// JSON array. A leading/example `{…}` object in the response must not
|
|
551
|
+
// mask a valid `[…]` array as a false "non-array" failure.
|
|
552
|
+
let parsed = parseEmbeddedJsonResponse(raw, { expect: "array" });
|
|
553
|
+
if (!Array.isArray(parsed)) {
|
|
554
|
+
// One stricter-reprompt retry before paying the per-asset fallback
|
|
555
|
+
// (#635). Many genuine non-array responses recover when the model is
|
|
556
|
+
// told explicitly to emit only the raw array.
|
|
557
|
+
bumpTelemetry(options.telemetry, "retryAttempts");
|
|
558
|
+
const retryRaw = await chatCompletion(llmConfig, [
|
|
559
|
+
{ role: "system", content: buildBatchRetrySystemPrompt() },
|
|
560
|
+
{ role: "user", content: userPrompt },
|
|
561
|
+
], { temperature: 0, timeoutMs: llmConfig.timeoutMs, signal });
|
|
562
|
+
parsed = retryRaw ? parseEmbeddedJsonResponse(retryRaw, { expect: "array" }) : undefined;
|
|
563
|
+
}
|
|
545
564
|
if (!Array.isArray(parsed)) {
|
|
546
565
|
nonArrayResponse = true;
|
|
547
566
|
bumpTelemetry(options.telemetry, "nonArrayBatchFailures");
|
|
@@ -551,8 +570,9 @@ export async function extractGraphFromBodies(llmConfig, bodies, signal, akmConfi
|
|
|
551
570
|
batchState.batchingDisabled = true;
|
|
552
571
|
}
|
|
553
572
|
}
|
|
554
|
-
warn(`graph extraction (batch): LLM response was not a JSON array for ${nonEmptyBodies.length} asset(s)
|
|
555
|
-
`will fall back per-asset.
|
|
573
|
+
warn(`graph extraction (batch): LLM response was not a JSON array for ${nonEmptyBodies.length} asset(s) ` +
|
|
574
|
+
`even after a stricter retry; will fall back per-asset. ` +
|
|
575
|
+
`promptChars=${userPrompt.length}${formatContextHint(llmConfig)}`);
|
|
556
576
|
return null;
|
|
557
577
|
}
|
|
558
578
|
return parsed;
|
|
@@ -675,17 +695,21 @@ export async function extractGraphFromBody(llmConfig, body, signal, akmConfig, o
|
|
|
675
695
|
return merged;
|
|
676
696
|
}
|
|
677
697
|
const userPrompt = `${USER_PROMPT_PREFIX}${trimmedBody}`;
|
|
678
|
-
return
|
|
679
|
-
|
|
680
|
-
|
|
681
|
-
|
|
682
|
-
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
|
|
687
|
-
|
|
688
|
-
|
|
698
|
+
return callStructured({
|
|
699
|
+
feature: "graph_extraction",
|
|
700
|
+
akmConfig,
|
|
701
|
+
config: llmConfig,
|
|
702
|
+
messages: [
|
|
703
|
+
{ role: "system", content: SYSTEM_PROMPT },
|
|
704
|
+
{ role: "user", content: userPrompt },
|
|
705
|
+
],
|
|
706
|
+
request: {
|
|
707
|
+
temperature: 0.1,
|
|
708
|
+
timeoutMs: llmConfig.timeoutMs,
|
|
709
|
+
signal,
|
|
710
|
+
onRetryAttempt: () => bumpTelemetry(options.telemetry, "retryAttempts"),
|
|
711
|
+
},
|
|
712
|
+
parse: (raw) => {
|
|
689
713
|
if (!raw)
|
|
690
714
|
return empty();
|
|
691
715
|
const parsed = parseEmbeddedJsonResponse(raw);
|
|
@@ -701,16 +725,16 @@ export async function extractGraphFromBody(llmConfig, body, signal, akmConfig, o
|
|
|
701
725
|
if (extraction.status === "failed")
|
|
702
726
|
bumpTelemetry(options.telemetry, "failureCount");
|
|
703
727
|
return extraction;
|
|
704
|
-
}
|
|
705
|
-
|
|
728
|
+
},
|
|
729
|
+
onError: (cls, err) => {
|
|
706
730
|
const errMsg = toErrorMessage(err);
|
|
707
|
-
if (
|
|
731
|
+
if (cls === "context_limit") {
|
|
708
732
|
bumpTelemetry(options.telemetry, "failureCount");
|
|
709
733
|
warn(`graph extraction: context size exceeded for asset; promptChars=${userPrompt.length}${formatContextHint(llmConfig)}. ` +
|
|
710
734
|
`Consider increasing llm.contextLength in config.json.`);
|
|
711
735
|
return empty("context_limit", "failed");
|
|
712
736
|
}
|
|
713
|
-
else if (
|
|
737
|
+
else if (cls === "html") {
|
|
714
738
|
bumpTelemetry(options.telemetry, "htmlErrorCount");
|
|
715
739
|
warn(`graph extraction: provider returned HTML instead of JSON for asset; promptChars=${userPrompt.length}${formatContextHint(llmConfig)}: ${errMsg}`);
|
|
716
740
|
return empty("llm_error", "failed");
|
|
@@ -720,9 +744,8 @@ export async function extractGraphFromBody(llmConfig, body, signal, akmConfig, o
|
|
|
720
744
|
warn(`graph extraction failed for asset; promptChars=${userPrompt.length}${formatContextHint(llmConfig)}: ${errMsg}`);
|
|
721
745
|
return empty("llm_error", "failed");
|
|
722
746
|
}
|
|
723
|
-
}
|
|
724
|
-
|
|
725
|
-
timeoutMs: llmConfig.timeoutMs,
|
|
747
|
+
},
|
|
748
|
+
fallback: empty(),
|
|
726
749
|
onFallback,
|
|
727
750
|
});
|
|
728
751
|
}
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/**
|
|
5
|
+
* LLM helper for the `akm index` memory-inference pass (#201).
|
|
6
|
+
*
|
|
7
|
+
* Compresses a single memory body into one higher-signal derived memory. The
|
|
8
|
+
* pass itself (in `src/indexer/memory-inference.ts`) is responsible for
|
|
9
|
+
* deciding which memories are pending, persisting the derived memory with the
|
|
10
|
+
* correct frontmatter (`inferred: true`, `source: <parent-ref>`), and marking
|
|
11
|
+
* the parent as processed for idempotency.
|
|
12
|
+
*
|
|
13
|
+
* This module is intentionally tiny and stateless so tests can stub it via
|
|
14
|
+
* `mock.module("../src/llm/memory-infer", ...)` without hitting a network.
|
|
15
|
+
*
|
|
16
|
+
* Locked v1 contract (#208): the LLM connection always comes from the
|
|
17
|
+
* shared `akm.llm` block — never from a per-pass override. Callers obtain
|
|
18
|
+
* the connection via `resolveIndexPassLLM("memory", config)` and pass it
|
|
19
|
+
* straight through.
|
|
20
|
+
*/
|
|
21
|
+
import memoryInferSystemPrompt from "../assets/prompts/memory-infer-system.md" with { type: "text" };
|
|
22
|
+
import memoryInferUserPrompt from "../assets/prompts/memory-infer-user.md" with { type: "text" };
|
|
23
|
+
import { toErrorMessage } from "../core/common.js";
|
|
24
|
+
import { warn } from "../core/warn.js";
|
|
25
|
+
import { parseEmbeddedJsonResponse } from "./client.js";
|
|
26
|
+
import { callStructured } from "./structured-call.js";
|
|
27
|
+
/** Hard cap on body chars sent to the model — pragmatic and matches `runLlmEnrich`. */
|
|
28
|
+
const MAX_BODY_CHARS = 4000;
|
|
29
|
+
const SYSTEM_PROMPT = memoryInferSystemPrompt;
|
|
30
|
+
const USER_PROMPT_PREFIX = memoryInferUserPrompt;
|
|
31
|
+
/**
|
|
32
|
+
* Strict JSON Schema for the derived-memory payload. Sent to providers that
|
|
33
|
+
* opt in via `LlmConnectionConfig.supportsJsonSchema = true`; the client
|
|
34
|
+
* silently drops the schema for providers that don't.
|
|
35
|
+
*
|
|
36
|
+
* Extends the responseSchema lift (PR 1, asset-writers-investigation §5) to
|
|
37
|
+
* the memory-inference path. Mirrors the validation gate below
|
|
38
|
+
* (title/description/content + non-empty tags/searchHints) so a
|
|
39
|
+
* schema-compliant response is guaranteed to pass the downstream check
|
|
40
|
+
* — no more "incomplete derived memory payload from LLM; skipping memory"
|
|
41
|
+
* for shape-only failures.
|
|
42
|
+
*/
|
|
43
|
+
const DERIVED_MEMORY_JSON_SCHEMA = {
|
|
44
|
+
type: "object",
|
|
45
|
+
properties: {
|
|
46
|
+
title: { type: "string", minLength: 1 },
|
|
47
|
+
description: { type: "string", minLength: 1 },
|
|
48
|
+
content: { type: "string", minLength: 1 },
|
|
49
|
+
tags: { type: "array", items: { type: "string" }, minItems: 1, maxItems: 8 },
|
|
50
|
+
searchHints: { type: "array", items: { type: "string" }, minItems: 1, maxItems: 6 },
|
|
51
|
+
},
|
|
52
|
+
required: ["title", "description", "content", "tags", "searchHints"],
|
|
53
|
+
additionalProperties: false,
|
|
54
|
+
};
|
|
55
|
+
/**
|
|
56
|
+
* Compress a single memory body into one derived memory via the configured LLM.
|
|
57
|
+
*
|
|
58
|
+
* Returns `undefined` on any failure (timeout, invalid JSON, empty response).
|
|
59
|
+
* Errors are logged via `warn()` but never thrown — a failed split for one memory
|
|
60
|
+
* must not abort the rest of the index pass.
|
|
61
|
+
*
|
|
62
|
+
* Routes through `callStructured({ feature: "memory_inference", ... })` so the
|
|
63
|
+
* feature gate, error classification, and onFallback hook are honoured uniformly
|
|
64
|
+
* (Fix C5).
|
|
65
|
+
*/
|
|
66
|
+
export async function compressMemoryToDerivedMemory(llmConfig, body, signal, akmConfig, onFallback, telemetry, onRetryAttempt) {
|
|
67
|
+
const trimmedBody = body.trim();
|
|
68
|
+
if (!trimmedBody)
|
|
69
|
+
return undefined;
|
|
70
|
+
const userPrompt = `${USER_PROMPT_PREFIX}${trimmedBody.slice(0, MAX_BODY_CHARS)}`;
|
|
71
|
+
// Memory-inference is ALWAYS gated: no `akmConfig` ⇒ gate closed (no chat,
|
|
72
|
+
// `disabled` fallback), never the seam's ungated/propagate path (which is for
|
|
73
|
+
// direct callers like `enhanceMetadata`). This is the gate-closed branch
|
|
74
|
+
// `tryLlmFeature(_, undefined, _)` took before the migration.
|
|
75
|
+
if (!akmConfig) {
|
|
76
|
+
onFallback?.({ feature: "memory_inference", reason: "disabled" });
|
|
77
|
+
return undefined;
|
|
78
|
+
}
|
|
79
|
+
return callStructured({
|
|
80
|
+
feature: "memory_inference",
|
|
81
|
+
akmConfig,
|
|
82
|
+
config: llmConfig,
|
|
83
|
+
messages: [
|
|
84
|
+
{ role: "system", content: SYSTEM_PROMPT },
|
|
85
|
+
{ role: "user", content: userPrompt },
|
|
86
|
+
],
|
|
87
|
+
request: {
|
|
88
|
+
temperature: 0.1,
|
|
89
|
+
timeoutMs: llmConfig.timeoutMs,
|
|
90
|
+
signal,
|
|
91
|
+
responseSchema: DERIVED_MEMORY_JSON_SCHEMA,
|
|
92
|
+
onRetryAttempt,
|
|
93
|
+
},
|
|
94
|
+
parse: (raw) => {
|
|
95
|
+
if (!raw)
|
|
96
|
+
return undefined;
|
|
97
|
+
const parsed = parseEmbeddedJsonResponse(raw);
|
|
98
|
+
if (!parsed) {
|
|
99
|
+
warn("memory inference: invalid JSON response from LLM; skipping memory.");
|
|
100
|
+
return undefined;
|
|
101
|
+
}
|
|
102
|
+
const title = typeof parsed.title === "string" ? parsed.title.trim() : "";
|
|
103
|
+
const description = typeof parsed.description === "string" ? parsed.description.trim() : "";
|
|
104
|
+
const content = typeof parsed.content === "string" ? parsed.content.trim() : "";
|
|
105
|
+
const tags = Array.isArray(parsed.tags)
|
|
106
|
+
? parsed.tags
|
|
107
|
+
.filter((t) => typeof t === "string")
|
|
108
|
+
.map((t) => t.trim())
|
|
109
|
+
.filter(Boolean)
|
|
110
|
+
.slice(0, 8)
|
|
111
|
+
: [];
|
|
112
|
+
const searchHints = Array.isArray(parsed.searchHints)
|
|
113
|
+
? parsed.searchHints
|
|
114
|
+
.filter((h) => typeof h === "string")
|
|
115
|
+
.map((h) => h.trim())
|
|
116
|
+
.filter(Boolean)
|
|
117
|
+
.slice(0, 6)
|
|
118
|
+
: [];
|
|
119
|
+
if (!title || !description || !content || tags.length === 0 || searchHints.length === 0) {
|
|
120
|
+
warn("memory inference: incomplete derived memory payload from LLM; skipping memory.");
|
|
121
|
+
return undefined;
|
|
122
|
+
}
|
|
123
|
+
return { title, description, tags, searchHints, content };
|
|
124
|
+
},
|
|
125
|
+
onError: (cls, err) => {
|
|
126
|
+
if (cls === "html") {
|
|
127
|
+
if (telemetry)
|
|
128
|
+
telemetry.htmlErrorCount = (telemetry.htmlErrorCount ?? 0) + 1;
|
|
129
|
+
warn(`memory inference: provider returned HTML instead of JSON; skipping memory: ${toErrorMessage(err)}`);
|
|
130
|
+
return undefined;
|
|
131
|
+
}
|
|
132
|
+
warn(`memory inference failed: ${toErrorMessage(err)}`);
|
|
133
|
+
return undefined;
|
|
134
|
+
},
|
|
135
|
+
fallback: undefined,
|
|
136
|
+
onFallback,
|
|
137
|
+
});
|
|
138
|
+
}
|