akm-cli 0.9.0-beta.3 → 0.9.0-beta.30

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (107) hide show
  1. package/CHANGELOG.md +600 -0
  2. package/dist/assets/prompts/consolidate-system.md +23 -0
  3. package/dist/assets/prompts/contradiction-judge.md +33 -0
  4. package/dist/assets/prompts/distill-knowledge-system.md +22 -0
  5. package/dist/assets/prompts/distill-lesson-system.md +36 -0
  6. package/dist/assets/prompts/extract-session.md +5 -1
  7. package/dist/assets/prompts/graph-extract-system.md +1 -0
  8. package/dist/assets/prompts/memory-infer-system.md +1 -0
  9. package/dist/assets/prompts/memory-infer-user.md +5 -0
  10. package/dist/assets/prompts/metadata-enhance-system.md +1 -0
  11. package/dist/assets/prompts/procedural-system.md +44 -0
  12. package/dist/assets/prompts/recombine-system.md +40 -0
  13. package/dist/assets/prompts/staleness-detect-system.md +6 -0
  14. package/dist/assets/prompts/validate-summary-judge.md +1 -0
  15. package/dist/assets/templates/html/health.html +281 -111
  16. package/dist/cli.js +14 -3
  17. package/dist/commands/agent/contribute-cli.js +16 -3
  18. package/dist/commands/feedback-cli.js +15 -6
  19. package/dist/commands/graph/graph.js +75 -71
  20. package/dist/commands/health/checks.js +48 -0
  21. package/dist/commands/health/html-report.js +422 -80
  22. package/dist/commands/health.js +381 -9
  23. package/dist/commands/improve/calibration.js +161 -0
  24. package/dist/commands/improve/consolidate.js +634 -111
  25. package/dist/commands/improve/dedup.js +482 -0
  26. package/dist/commands/improve/distill.js +145 -69
  27. package/dist/commands/improve/encoding-salience.js +205 -0
  28. package/dist/commands/improve/extract-cli.js +115 -1
  29. package/dist/commands/improve/extract-prompt.js +33 -2
  30. package/dist/commands/improve/extract-watch.js +140 -0
  31. package/dist/commands/improve/extract.js +244 -35
  32. package/dist/commands/improve/feedback-valence.js +54 -0
  33. package/dist/commands/improve/homeostatic.js +467 -0
  34. package/dist/commands/improve/improve-auto-accept.js +113 -6
  35. package/dist/commands/improve/improve-profiles.js +12 -0
  36. package/dist/commands/improve/improve.js +1974 -614
  37. package/dist/commands/improve/memory/memory-contradiction-detect.js +23 -28
  38. package/dist/commands/improve/outcome-loop.js +256 -0
  39. package/dist/commands/improve/proactive-maintenance.js +87 -0
  40. package/dist/commands/improve/procedural.js +409 -0
  41. package/dist/commands/improve/recombine.js +528 -0
  42. package/dist/commands/improve/reflect.js +26 -1
  43. package/dist/commands/improve/related-sessions.js +120 -0
  44. package/dist/commands/improve/salience.js +386 -0
  45. package/dist/commands/improve/triage.js +95 -0
  46. package/dist/commands/lint/agent-linter.js +19 -24
  47. package/dist/commands/lint/base-linter.js +173 -60
  48. package/dist/commands/lint/command-linter.js +19 -24
  49. package/dist/commands/lint/env-key-rules.js +34 -1
  50. package/dist/commands/lint/fact-linter.js +39 -0
  51. package/dist/commands/lint/index.js +31 -13
  52. package/dist/commands/lint/memory-linter.js +1 -1
  53. package/dist/commands/lint/registry.js +7 -2
  54. package/dist/commands/lint/task-linter.js +3 -3
  55. package/dist/commands/lint/workflow-linter.js +26 -1
  56. package/dist/commands/proposal/proposal.js +5 -0
  57. package/dist/commands/proposal/validators/proposals.js +71 -54
  58. package/dist/commands/read/curate.js +344 -80
  59. package/dist/commands/read/search-cli.js +7 -0
  60. package/dist/commands/read/search.js +1 -0
  61. package/dist/commands/read/show.js +67 -2
  62. package/dist/commands/sources/installed-stashes.js +5 -1
  63. package/dist/commands/sources/stash-cli.js +10 -2
  64. package/dist/core/asset/asset-registry.js +2 -0
  65. package/dist/core/asset/asset-spec.js +14 -0
  66. package/dist/core/asset/frontmatter.js +166 -167
  67. package/dist/core/asset/markdown.js +8 -0
  68. package/dist/core/config/config-schema.js +259 -2
  69. package/dist/core/config/config.js +2 -2
  70. package/dist/core/logs-db.js +4 -3
  71. package/dist/core/paths.js +3 -0
  72. package/dist/core/state-db.js +649 -30
  73. package/dist/indexer/db/db.js +364 -38
  74. package/dist/indexer/db/graph-db.js +129 -86
  75. package/dist/indexer/ensure-index.js +152 -17
  76. package/dist/indexer/graph/graph-boost.js +51 -41
  77. package/dist/indexer/graph/graph-extraction.js +203 -3
  78. package/dist/indexer/index-writer-lock.js +99 -0
  79. package/dist/indexer/indexer.js +114 -111
  80. package/dist/indexer/passes/memory-inference.js +10 -3
  81. package/dist/indexer/passes/staleness-detect.js +2 -5
  82. package/dist/indexer/search/db-search.js +15 -4
  83. package/dist/indexer/search/ranking-contributors.js +22 -0
  84. package/dist/indexer/search/ranking.js +4 -0
  85. package/dist/indexer/walk/matchers.js +9 -0
  86. package/dist/integrations/agent/prompts.js +1 -0
  87. package/dist/integrations/harnesses/claude/session-log.js +11 -1
  88. package/dist/integrations/harnesses/opencode/session-log.js +9 -0
  89. package/dist/integrations/session-logs/index.js +16 -0
  90. package/dist/llm/client.js +23 -4
  91. package/dist/llm/embedder.js +27 -3
  92. package/dist/llm/embedders/local.js +66 -2
  93. package/dist/llm/graph-extract.js +2 -1
  94. package/dist/llm/memory-infer.js +4 -8
  95. package/dist/llm/metadata-enhance.js +9 -1
  96. package/dist/output/renderers.js +73 -1
  97. package/dist/output/shapes/curate.js +14 -2
  98. package/dist/output/text/helpers.js +9 -0
  99. package/dist/runtime.js +25 -1
  100. package/dist/scripts/migrate-storage.js +1242 -594
  101. package/dist/scripts/migrations/import-fs-improve-runs-to-db.js +473 -270
  102. package/dist/sources/providers/tar-utils.js +16 -8
  103. package/dist/storage/sqlite-pragmas.js +146 -0
  104. package/dist/workflows/db.js +3 -4
  105. package/dist/workflows/validate-summary.js +2 -7
  106. package/docs/data-and-telemetry.md +1 -0
  107. package/package.json +9 -6
@@ -0,0 +1,33 @@
1
+ You are evaluating two derived memory entries to determine if they contain
2
+ directly contradictory factual claims about the same subject.
3
+
4
+ Memory A:
5
+ Ref: {{A_REF}}
6
+ Description: {{A_DESCRIPTION}}
7
+ Content:
8
+ ```
9
+ {{A_BODY}}
10
+ ```
11
+
12
+ Memory B:
13
+ Ref: {{B_REF}}
14
+ Description: {{B_DESCRIPTION}}
15
+ Content:
16
+ ```
17
+ {{B_BODY}}
18
+ ```
19
+
20
+ Answer ONLY with valid JSON — no prose, no code fences:
21
+ {"contradicts": true|false, "confidence": 0.0, "reason": "<cite the exact opposing sentence from each memory, or explain why not contradicted>"}
22
+
23
+ A contradiction means the memories make LOGICALLY EXCLUSIVE claims: a practitioner
24
+ cannot follow BOTH simultaneously. The test: cite the exact sentence from Memory A
25
+ and the exact sentence from Memory B that are in direct conflict. If you cannot cite
26
+ specific opposing sentences, return false.
27
+
28
+ Sharing a topic, tool, domain, or workflow stage is NOT a contradiction. Only direct
29
+ factual opposites qualify: opposing recommended commands, opposing boolean flags,
30
+ opposing version numbers, or mutually exclusive instructions.
31
+
32
+ Set confidence ≥ 0.92 only when evidence is unambiguous. Use lower values when
33
+ uncertain — the caller will skip edges below 0.92.
@@ -0,0 +1,22 @@
1
+ You are the akm `distill` distiller.
2
+ Given an asset and recent feedback events about it, produce a concise
3
+ *knowledge* markdown document capturing the durable, reusable facts.
4
+ Prefer stable guidance over narrative recap.
5
+
6
+ YOUR RESPONSE MUST START EXACTLY WITH `---` ON THE VERY FIRST LINE.
7
+ DO NOT output any prose, explanation, or code fences before or after.
8
+
9
+ Required output format:
10
+ ---
11
+ description: <one-line summary of the knowledge asset>
12
+ tags: [<tag1>, <tag2>]
13
+ ---
14
+
15
+ # <Title>
16
+
17
+ <body — structured markdown, durable facts only>
18
+
19
+ RULES:
20
+ - `description` MUST be a non-empty single-line string.
21
+ - Include a meaningful markdown body with a `# Title` heading.
22
+ - Output ONLY the knowledge file. No preamble, no code fences, no trailing prose.
@@ -0,0 +1,36 @@
1
+ You are the akm `distill` distiller.
2
+ Given an asset and recent feedback events about it, produce a single
3
+ concise *lesson* an agent should remember next time it works on this
4
+ asset's domain.
5
+
6
+ YOUR RESPONSE MUST START EXACTLY WITH `---` ON THE VERY FIRST LINE.
7
+ DO NOT output any prose, explanation, or code fences before or after.
8
+
9
+ Required output format — copy this structure exactly:
10
+ ---
11
+ description: <one complete sentence (ending with `.`) summarising what the lesson teaches>
12
+ when_to_use: <one complete sentence describing the concrete trigger condition>
13
+ ---
14
+
15
+ <lesson body — plain markdown, 1–3 short paragraphs of practical guidance>
16
+
17
+ ## description field (MANDATORY)
18
+ - A single complete sentence in present tense, 80-200 chars, NO markdown.
19
+ - Self-contained: a reviewer must understand the lesson from this field alone.
20
+ - DO NOT start with "When ", "If ", or a connector word — that belongs in when_to_use.
21
+ - DO NOT copy a section heading ("Key takeaways", "For example", "Key pitfalls").
22
+ - DO NOT begin with a numbered list marker, code fence, or markdown heading.
23
+
24
+ GOOD: "Always validate ref existence before promoting a memory to knowledge; missing refs surface as silent 404s during accept."
25
+ BAD: "Key pitfalls"
26
+ BAD: "When working with the akm CLI"
27
+ BAD: "For example, you might..."
28
+ BAD: "1. Check the file"
29
+
30
+ RULES:
31
+ - `when_to_use` MUST be a complete sentence describing a concrete trigger. Never write `When working with <asset-name>` — that is circular and useless.
32
+ - `description` and `when_to_use` MUST differ from each other.
33
+ - The lesson body MUST be non-empty markdown prose. Do NOT restate `description:` or `when_to_use:` inside the body (no `**description:** ...` or `**when_to_use:** ...` lines — the frontmatter is the only place those keys belong).
34
+ - Do NOT emit a second `---` fence after the opening frontmatter — there are exactly two `---` lines in the output, both belonging to the single frontmatter block at the top.
35
+ - Do NOT reproduce the source asset verbatim — distil what a caller needs to know.
36
+ - Output ONLY the lesson file. No preamble, no code fences, no trailing prose.
@@ -49,13 +49,17 @@ Respond with EXACTLY one JSON object matching this shape:
49
49
  "when_to_use": "<one sentence 15-400 chars; REQUIRED only when type=lesson>",
50
50
  "body": "<markdown body, 200-3000 chars typical>",
51
51
  "confidence": <number 0.0-1.0>,
52
- "evidence": "<one-line pointer to the moment in the session>"
52
+ "evidence": "<one-line pointer to the moment in the session>",
53
+ "orderedActions": ["<action-1>", "<action-2>", ...],
54
+ "outcomeData": "<one sentence describing the outcome of the action sequence>"
53
55
  }
54
56
  ],
55
57
  "rationale_if_empty": "<one sentence; REQUIRED when candidates is empty>"
56
58
  }
57
59
  ```
58
60
 
61
+ `orderedActions` and `outcomeData` are **optional**. Include them only when the candidate represents a recurring action sequence (e.g. a recovery procedure, a build-fix recipe, a deployment checklist) where preserving the ordered steps adds future value. When present, `outcomeData` is required and must describe what happened when the sequence completed (success or failure). Omit both fields entirely for standalone facts, observations, or lessons that are not action-sequence-shaped.
62
+
59
63
  ## Rules
60
64
 
61
65
  1. **Zero candidates is a valid and frequent answer.** Most sessions yield no new durable insight. When that's the case, return `{"candidates": [], "rationale_if_empty": "..."}` explaining what you saw and why it didn't rise to durable-knowledge level. Do not fabricate.
@@ -0,0 +1 @@
1
+ You extract a knowledge graph from developer notes. Return ONLY valid JSON — no prose, no markdown fences, no preamble.
@@ -0,0 +1 @@
1
+ You compress a developer memory into one high-signal derived memory for later retrieval. Return only valid JSON. No prose outside the JSON object. No markdown fences.
@@ -0,0 +1,5 @@
1
+ Compress the memory below into one derived memory. Output ONLY JSON:
2
+ {"title":"short title string","description":"one sentence summary string","tags":["tag1","tag2"],"searchHints":["search phrase 1","search phrase 2"],"content":"2-3 sentence compressed body preserving key facts verbatim"}
3
+ Rules: be specific, no vague generalizations, preserve key facts (names/versions/paths/config keys verbatim), merge related points, 3-8 tags, 3-6 searchHints. The content field must be a plain string with 2-3 sentences.
4
+
5
+ Memory:
@@ -0,0 +1 @@
1
+ You are a metadata generator for a developer asset registry. Given a script/skill/command/agent entry, generate improved metadata. Respond with ONLY valid JSON, no markdown fencing.
@@ -0,0 +1,44 @@
1
+ You are the akm `procedural` compiler.
2
+
3
+ You are given a RECURRING ordered action sequence — the SAME ordered list of
4
+ steps that an agent has successfully performed across several independent
5
+ sessions. Your job is to turn that bare sequence into a clean, reusable
6
+ WORKFLOW: give the workflow a title and a one-sentence description, and turn each
7
+ ordered action into a named step with clear imperative instructions.
8
+
9
+ You MUST NOT invent new steps, drop steps, merge steps, or reorder them. The
10
+ ordered action list is the source of truth. Return EXACTLY one step per input
11
+ action, in the SAME order.
12
+
13
+ YOUR RESPONSE MUST BE A SINGLE JSON OBJECT AND NOTHING ELSE.
14
+ DO NOT output prose, explanation, or code fences before or after the JSON.
15
+
16
+ When the sequence is a coherent, reusable procedure, return:
17
+ {
18
+ "title": "<short imperative workflow title, no trailing period>",
19
+ "description": "<one complete sentence (ending with a period) stating what the workflow accomplishes>",
20
+ "steps": [
21
+ {
22
+ "title": "<short imperative step title>",
23
+ "instructions": "<one or more imperative sentences telling an agent exactly how to perform this step>",
24
+ "completionCriteria": ["<optional bullet: an observable signal the step is done>"]
25
+ }
26
+ ]
27
+ }
28
+
29
+ When the sequence is NOT a coherent reusable procedure (it is noise, the steps do
30
+ not form a meaningful workflow, or any reusable framing would be a stretch),
31
+ return an explicit null:
32
+ null
33
+
34
+ A justified null is a CORRECT and expected outcome. Do NOT fabricate a hollow
35
+ workflow to avoid returning null.
36
+
37
+ ## Rules
38
+ - `steps` MUST have EXACTLY as many entries as the input action list, in order.
39
+ - Every step MUST have a non-empty `title` and non-empty `instructions`.
40
+ - `completionCriteria` is OPTIONAL; omit it rather than inventing weak criteria.
41
+ - The `description` MUST be a single complete present-tense sentence with NO
42
+ markdown, self-contained enough for a reviewer to understand the workflow.
43
+ - Ground every step in the corresponding input action; do not introduce outside
44
+ facts or tools the action does not mention.
@@ -0,0 +1,40 @@
1
+ You are the akm `recombine` synthesizer.
2
+
3
+ You are given a CLUSTER of related memories — distinct episodes that share a
4
+ topic (a tag or a graph entity) but were recorded independently. Your job is to
5
+ induce ONE cross-episodic generalization: a single durable insight that none of
6
+ the input memories states on its own, but that the cluster as a whole supports.
7
+
8
+ This is hypothesis formation, not summarization. A good generalization explains
9
+ WHY the individual episodes are instances of the same underlying pattern and
10
+ gives an agent a reusable rule for the next, unseen episode.
11
+
12
+ YOUR RESPONSE MUST BE A SINGLE JSON OBJECT AND NOTHING ELSE.
13
+ DO NOT output prose, explanation, or code fences before or after the JSON.
14
+
15
+ When a defensible generalization exists, return:
16
+ {
17
+ "description": "<one complete sentence (ending with a period) stating the generalization>",
18
+ "when_to_use": "<one complete sentence describing the concrete trigger condition>",
19
+ "body": "<1-3 short paragraphs of practical guidance grounded in the cluster>"
20
+ }
21
+
22
+ When NO defensible generalization exists — the memories merely share a keyword,
23
+ or any unifying claim would be a stretch — return an explicit null:
24
+ null
25
+
26
+ A justified null is a CORRECT and expected outcome. Do NOT invent a weak or
27
+ generic generalization to avoid returning null. It is better to propose nothing
28
+ than to propose a hollow over-generalization.
29
+
30
+ ## description field (MANDATORY when not null)
31
+ - A single complete present-tense sentence, NO markdown.
32
+ - Self-contained: a reviewer must understand the hypothesis from this field alone.
33
+ - DO NOT start with "When " or "If " — that belongs in `when_to_use`.
34
+ - DO NOT merely restate one input memory; the value is the CROSS-episode pattern.
35
+
36
+ ## Guardrails
37
+ - Induce exactly ONE generalization for the whole cluster.
38
+ - Ground every claim in the supplied memories; do not introduce outside facts.
39
+ - This is a HYPOTHESIS — it will be re-confirmed across future runs before it is
40
+ ever promoted to a durable lesson. Frame it as a candidate rule, not a verdict.
@@ -0,0 +1,6 @@
1
+ You are a belief-state classifier for a memory store. Given a candidate memory and a list of more-recent similar memories from the same store, decide whether the candidate is still current or has been superseded.
2
+
3
+ Respond on the first line with exactly YES or NO.
4
+ If YES, the second line MUST be of the form `SUPERSEDED_BY: <ref>` where <ref> is the exact ref of the superseding memory from the list provided. Do NOT invent refs.
5
+ If NO, do not include any additional lines.
6
+ No prose, no preamble, no markdown.
@@ -0,0 +1 @@
1
+ You are a strict completion auditor for a software workflow engine. Given a step's completion criteria and a summary of the work an agent claims to have done, judge whether the summary provides concrete evidence that EVERY criterion is satisfied. Be skeptical: vague, hand-wavy, or unsubstantiated claims do NOT satisfy a criterion. Respond with ONLY a JSON object: {"complete": boolean, "missing": string[], "feedback": string}. "missing" lists the exact criteria that are not yet satisfied; "feedback" is a short directive telling the agent what to finish or fix. No prose, no markdown fences.