akm-opencode 0.8.0-rc.8 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,292 @@
1
+ import { inspect } from "node:util"
2
+
3
+ export type RedactionResult = {
4
+ text: string
5
+ redacted: boolean
6
+ categories: string[]
7
+ }
8
+
9
+ export type RedactedObjectResult<T> = {
10
+ value: T
11
+ redacted: boolean
12
+ categories: string[]
13
+ }
14
+
15
+ const SENSITIVE_KEY_RE = /(?:OPENAI_API_KEY|ANTHROPIC_API_KEY|GITHUB_TOKEN|GH_TOKEN|AWS_ACCESS_KEY_ID|AWS_SECRET_ACCESS_KEY|AWS_SESSION_TOKEN|AZURE_[A-Z0-9_]*KEY|SLACK_BOT_TOKEN|SLACK_TOKEN|PASSWORD|PASSWD|SECRET|CONNECTION_STRING|DATABASE_URL|AKM_[A-Z0-9_]*TOKEN)/i
16
+ const SIMPLE_REPLACEMENTS: Array<{ category: string; pattern: RegExp; replacement: string }> = [
17
+ {
18
+ category: "private_key",
19
+ pattern: /-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z ]*PRIVATE KEY-----/g,
20
+ replacement: "[REDACTED:PRIVATE_KEY]",
21
+ },
22
+ {
23
+ category: "bearer_token",
24
+ pattern: /Bearer\s+[A-Za-z0-9._~+/=-]+/gi,
25
+ replacement: "[REDACTED:BEARER_TOKEN]",
26
+ },
27
+ {
28
+ category: "github_token",
29
+ pattern: /\bgh[pousr]_[A-Za-z0-9_]+\b/g,
30
+ replacement: "[REDACTED:GITHUB_TOKEN]",
31
+ },
32
+ {
33
+ category: "github_pat",
34
+ pattern: /\bgithub_pat_[A-Za-z0-9_]+\b/g,
35
+ replacement: "[REDACTED:GITHUB_PAT]",
36
+ },
37
+ {
38
+ category: "openai_key",
39
+ pattern: /\bsk-[A-Za-z0-9_-]+\b/g,
40
+ replacement: "[REDACTED:OPENAI_API_KEY]",
41
+ },
42
+ {
43
+ category: "slack_token",
44
+ pattern: /\bxox[baprs]-[A-Za-z0-9-]+\b/g,
45
+ replacement: "[REDACTED:SLACK_TOKEN]",
46
+ },
47
+ // 0.8.0 release-hardening: AWS access key (AKIA prefix, 16 alphanumerics).
48
+ // IAM user access keys are 20 chars total starting with AKIA; STS temporary
49
+ // creds use ASIA but those are paired with a session token and short-lived,
50
+ // so the AKIA prefix is the high-value leak.
51
+ {
52
+ category: "aws_access_key",
53
+ pattern: /\bAKIA[0-9A-Z]{16}\b/g,
54
+ replacement: "[REDACTED:AWS_ACCESS_KEY_ID]",
55
+ },
56
+ // Google Cloud API key (AIza prefix, 35 base64url-safe chars). Pattern is
57
+ // documented at cloud.google.com/docs/authentication/api-keys.
58
+ {
59
+ category: "google_api_key",
60
+ pattern: /\bAIza[0-9A-Za-z_-]{35}\b/g,
61
+ replacement: "[REDACTED:GOOGLE_API_KEY]",
62
+ },
63
+ // Stripe live-mode secret/publishable keys (sk_live_, pk_live_). Stripe
64
+ // documents these as the high-severity leak surface in their API docs.
65
+ {
66
+ category: "stripe_live_key",
67
+ pattern: /\b(?:sk|pk)_live_[0-9a-zA-Z]+\b/g,
68
+ replacement: "[REDACTED:STRIPE_LIVE_KEY]",
69
+ },
70
+ // Stripe test-mode keys (sk_test_, pk_test_). Lower severity but still
71
+ // sensitive — test keys often grant access to test webhooks and dashboards
72
+ // that mirror production object shapes.
73
+ {
74
+ category: "stripe_test_key",
75
+ pattern: /\b(?:sk|pk)_test_[0-9a-zA-Z]+\b/g,
76
+ replacement: "[REDACTED:STRIPE_TEST_KEY]",
77
+ },
78
+ // Raw `Authorization: <token>` header without the `Bearer` keyword. Some
79
+ // services (legacy AWS, internal APIs, OAuth1 signed requests) send the
80
+ // token directly. The Bearer-form pattern above handles the common case.
81
+ // We require at least 12 chars of token to avoid matching `Authorization: ?`
82
+ // or other very short placeholder values.
83
+ {
84
+ category: "authorization_header",
85
+ pattern: /(Authorization:\s+)(?!Bearer\b)([A-Za-z0-9._~+/=-]{12,})/gi,
86
+ replacement: "$1[REDACTED:AUTHORIZATION]",
87
+ },
88
+ // WS-7b: Database connection strings — redact credentials portion.
89
+ // Matches postgres://, mysql://, mongodb+srv://, mongodb://, redis:// with user:pass@ form.
90
+ {
91
+ category: "connection_string",
92
+ pattern: /((?:postgres|postgresql|mysql|mongodb(?:\+srv)?|redis):\/\/)([^@\s]+@)([^\s"'`,]+)/gi,
93
+ replacement: "$1[REDACTED:CREDENTIALS]@$3",
94
+ },
95
+ // WS-7b: AWS ARN structures with embedded account IDs (12-digit).
96
+ // Matches arn:aws:...:123456789012:... patterns.
97
+ {
98
+ category: "aws_arn",
99
+ pattern: /arn:aws:[a-z0-9_-]+:[a-z0-9-]*:(\d{12}):[^\s"'`,]*/gi,
100
+ replacement: "arn:aws:...[REDACTED:ACCOUNT_ID]:...",
101
+ },
102
+ // WS-7b: JWT-shaped tokens — three base64url segments separated by dots (eyJ prefix).
103
+ {
104
+ category: "jwt_token",
105
+ pattern: /\beyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\b/g,
106
+ replacement: "[REDACTED:JWT_TOKEN]",
107
+ },
108
+ ]
109
+
110
+ /**
111
+ * High-entropy string redaction (WS-7b).
112
+ *
113
+ * Generic strings >= 32 chars matching [A-Za-z0-9+/=_-]+ that look like
114
+ * base64/hex secrets. Disabled by default to avoid false positives (e.g.
115
+ * long identifiers, file paths, etc.). Opt in by setting the env var:
116
+ *
117
+ * AKM_REDACT_HIGH_ENTROPY=1
118
+ *
119
+ * The threshold is 32 chars, configurable via AKM_REDACT_ENTROPY_MIN_LEN.
120
+ */
121
+ const HIGH_ENTROPY_ENABLED = process.env.AKM_REDACT_HIGH_ENTROPY === "1"
122
+ const HIGH_ENTROPY_MIN_LEN = Number.parseInt(process.env.AKM_REDACT_ENTROPY_MIN_LEN ?? "32", 10)
123
+ const HIGH_ENTROPY_RE = HIGH_ENTROPY_ENABLED
124
+ ? new RegExp(`\\b[A-Za-z0-9+/=_-]{${Math.max(32, HIGH_ENTROPY_MIN_LEN)},}\\b`, "g")
125
+ : null
126
+
127
+ /**
128
+ * Opt-in PII redaction patterns. Off by default because financial / phone
129
+ * shapes appear in legitimate logs (timestamps, counters, identifiers) and
130
+ * over-redaction makes session logs unreadable. Enable for hosts that
131
+ * legitimately process PII through the agent loop:
132
+ *
133
+ * AKM_REDACT_PII=1
134
+ *
135
+ * Patterns covered:
136
+ * - Credit-card-shaped 13–19 digit sequences with optional `[- ]` delimiters
137
+ * (Luhn is intentionally NOT verified — we redact aggressively when opt-in)
138
+ * - US Social Security Numbers in `\d{3}-\d{2}-\d{4}` form (delimited only;
139
+ * bare 9-digit numbers are too noisy)
140
+ * - US/international phone numbers in `[\+]?\d{1,3}[- .]?\d{3}[- .]?\d{3}[- .]?\d{4}` form
141
+ */
142
+ const PII_REDACTION_ENABLED = process.env.AKM_REDACT_PII === "1"
143
+ const PII_PATTERNS: Array<{ category: string; pattern: RegExp; replacement: string }> = PII_REDACTION_ENABLED
144
+ ? [
145
+ {
146
+ category: "credit_card",
147
+ pattern: /\b(?:\d[ -]?){12,18}\d\b/g,
148
+ replacement: "[REDACTED:CREDIT_CARD]",
149
+ },
150
+ {
151
+ category: "ssn",
152
+ pattern: /\b\d{3}-\d{2}-\d{4}\b/g,
153
+ replacement: "[REDACTED:SSN]",
154
+ },
155
+ {
156
+ category: "phone",
157
+ pattern: /(?<![\w./-])\+?\d{1,3}[- .]?\(?\d{3}\)?[- .]?\d{3}[- .]?\d{4}(?![\w./-])/g,
158
+ replacement: "[REDACTED:PHONE]",
159
+ },
160
+ ]
161
+ : []
162
+
163
+ function uniq(values: string[]): string[] {
164
+ return [...new Set(values)]
165
+ }
166
+
167
+ function redactAssignments(text: string, categories: string[]): string {
168
+ const envLineRe = /^\s*(?:export\s+)?([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(.+)$/gm
169
+ let next = text.replace(envLineRe, (match, rawKey: string) => {
170
+ const key = String(rawKey)
171
+ if (!SENSITIVE_KEY_RE.test(key)) return match
172
+ categories.push(key.toLowerCase().includes("password") || key.toLowerCase().includes("passwd") ? "password" : "env_secret")
173
+ categories.push("env_output")
174
+ return `${key}: [REDACTED:${key}]`
175
+ })
176
+ const inlineAssignmentRe = /\b([A-Za-z_][A-Za-z0-9_]*)\s*=\s*([^\s"'`,;]+)/g
177
+ next = next.replace(inlineAssignmentRe, (match, rawKey: string) => {
178
+ const key = String(rawKey)
179
+ if (!SENSITIVE_KEY_RE.test(key)) return match
180
+ categories.push(key.toLowerCase().includes("password") || key.toLowerCase().includes("passwd") ? "password" : "env_secret")
181
+ categories.push("env_output")
182
+ return `${key}: [REDACTED:${key}]`
183
+ })
184
+ return next
185
+ }
186
+
187
+ function redactJsonLikePairs(text: string, categories: string[]): string {
188
+ const pairRe = /(["']?)(password|passwd|secret|connectionString|apiKey|token|accessKey|secretKey)(\1\s*[:=]\s*)(["']?)([^\n,}"']+)(["']?)/gi
189
+ return text.replace(pairRe, (_match, q1: string, key: string, sep: string) => {
190
+ const upper = key.replace(/[^A-Za-z0-9]+/g, "_").toUpperCase()
191
+ const category = /connection/i.test(key)
192
+ ? "connection_string"
193
+ : /pass/i.test(key)
194
+ ? "password"
195
+ : /secret/i.test(key)
196
+ ? "secret"
197
+ : "token"
198
+ categories.push(category)
199
+ return `${q1}${key}${q1}${sep}[REDACTED:${upper}]`
200
+ })
201
+ }
202
+
203
+ // Redact `--flag value` / `-f value` CLI argv shapes where the flag name
204
+ // contains a sensitive substring (password, token, secret, api-key, etc.).
205
+ // This catches a class of leaks that the env-assignment and JSON-pair
206
+ // regexes miss because CLI args use a space separator rather than `=` / `:`.
207
+ function redactCliArgPairs(text: string, categories: string[]): string {
208
+ const pairRe = /(--?[A-Za-z0-9_-]*?(?:password|passwd|secret|token|api[_-]?key|access[_-]?key|auth)[A-Za-z0-9_-]*)(\s+)(\S+)/gi
209
+ return text.replace(pairRe, (_match, flag: string, sep: string) => {
210
+ categories.push("cli_arg_secret")
211
+ return `${flag}${sep}[REDACTED:CLI_ARG]`
212
+ })
213
+ }
214
+
215
+ export function redactSecrets(input: string): RedactionResult {
216
+ let text = input
217
+ const categories: string[] = []
218
+
219
+ for (const entry of SIMPLE_REPLACEMENTS) {
220
+ if (!entry.pattern.test(text)) continue
221
+ categories.push(entry.category)
222
+ text = text.replace(entry.pattern, entry.replacement)
223
+ }
224
+
225
+ text = redactAssignments(text, categories)
226
+ text = redactJsonLikePairs(text, categories)
227
+ text = redactCliArgPairs(text, categories)
228
+
229
+ // WS-7b: Optional high-entropy string redaction (opt-in via AKM_REDACT_HIGH_ENTROPY=1).
230
+ if (HIGH_ENTROPY_RE && HIGH_ENTROPY_RE.test(text)) {
231
+ categories.push("high_entropy")
232
+ HIGH_ENTROPY_RE.lastIndex = 0 // reset after .test()
233
+ text = text.replace(HIGH_ENTROPY_RE, "[REDACTED:HIGH_ENTROPY]")
234
+ }
235
+
236
+ // Opt-in PII redaction (AKM_REDACT_PII=1).
237
+ for (const entry of PII_PATTERNS) {
238
+ if (!entry.pattern.test(text)) continue
239
+ categories.push(entry.category)
240
+ entry.pattern.lastIndex = 0
241
+ text = text.replace(entry.pattern, entry.replacement)
242
+ }
243
+
244
+ return {
245
+ text,
246
+ redacted: categories.length > 0,
247
+ categories: uniq(categories),
248
+ }
249
+ }
250
+
251
+ function normalizeObjectInput(value: unknown): unknown {
252
+ if (typeof value === "string") return value
253
+ if (value == null) return value
254
+ if (typeof value === "number" || typeof value === "boolean") return value
255
+ if (Array.isArray(value)) return value.map(normalizeObjectInput)
256
+ if (typeof value === "object") {
257
+ const record: Record<string, unknown> = {}
258
+ for (const [key, entry] of Object.entries(value as Record<string, unknown>)) {
259
+ record[key] = normalizeObjectInput(entry)
260
+ }
261
+ return record
262
+ }
263
+ return inspect(value, { depth: 4, breakLength: 120 })
264
+ }
265
+
266
+ function redactUnknown(value: unknown, categories: string[]): unknown {
267
+ if (typeof value === "string") {
268
+ const redacted = redactSecrets(value)
269
+ categories.push(...redacted.categories)
270
+ return redacted.text
271
+ }
272
+ if (Array.isArray(value)) return value.map((entry) => redactUnknown(entry, categories))
273
+ if (value && typeof value === "object") {
274
+ const output: Record<string, unknown> = {}
275
+ for (const [key, entry] of Object.entries(value as Record<string, unknown>)) {
276
+ output[key] = redactUnknown(entry, categories)
277
+ }
278
+ return output
279
+ }
280
+ return value
281
+ }
282
+
283
+ export function redactObject<T>(input: T): RedactedObjectResult<T> {
284
+ const categories: string[] = []
285
+ const normalized = normalizeObjectInput(input)
286
+ const value = redactUnknown(normalized, categories) as T
287
+ return {
288
+ value,
289
+ redacted: categories.length > 0,
290
+ categories: uniq(categories),
291
+ }
292
+ }
@@ -0,0 +1,261 @@
1
+ /**
2
+ * AKM ref extraction & live-stash validation.
3
+ *
4
+ * Background: session-checkpoint memories captured by the hook embed Bash
5
+ * command bodies verbatim — heredocs, grep patterns, jq queries, JSON
6
+ * payloads. Naive ref extraction (`text.match(REF_PATTERN)`) treats every
7
+ * `<type>:<slug>` token as a real reference, which causes `akm lint` to
8
+ * flag string-literal tokens as `missing-ref`. The next session capture
9
+ * regenerates the same flags — a permanent lint treadmill.
10
+ *
11
+ * The fix (this module): we drop the "guess from context" heuristic and
12
+ * instead **validate every candidate against the live local stash**. A
13
+ * token only graduates from "candidate" to "ref" when the referenced asset
14
+ * actually exists on disk. Anything that doesn't resolve is silently
15
+ * dropped — including all the literal-string false positives.
16
+ *
17
+ * The validation logic mirrors the consumer-side lint walker
18
+ * (`src/commands/lint/base-linter.ts#refExistsInAnyStash`). We deliberately
19
+ * inline a small copy here (rather than spawn a subprocess) so the hook's
20
+ * post-tool path stays cheap (zero subprocess, zero JSON parse). The two
21
+ * resolvers must stay in sync; any new asset type added to lint's
22
+ * `refToRelPath` must be added here too.
23
+ */
24
+
25
+ // CONTRACT: ref-resolver
26
+ // ----------------------------------------------------------------------------
27
+ // The `refExistsInAnyStash` and `refToRelPath` helpers below are
28
+ // contract-locked: a sister copy lives in the akm-core repo at
29
+ // `src/commands/lint/base-linter.ts`. Both implementations resolve the same
30
+ // `<type>:<slug>` -> on-disk-asset question and MUST agree on the set of
31
+ // reachable refs for any given stash layout.
32
+ //
33
+ // The lock is enforced by `tests/ref-resolver-contract.test.ts`, which drives
34
+ // `validateRefCandidates` (the only public entry to this resolver) through a
35
+ // canonical fixture set. The akm-core repo ships an equivalent test at
36
+ // `tests/contracts/ref-resolver-contract.test.ts` that drives ITS copy
37
+ // through the SAME inputs. Any change to the resolver behavior on either
38
+ // side MUST update both contract tests in lockstep, or one will fail.
39
+ //
40
+ // NOTE: this file is the SECOND copy of the resolver. The runtime-shipped
41
+ // copy lives at `claude/shared/ref-extraction.ts` and is imported by the
42
+ // post-tool hook. Both copies must agree with each other AND with the
43
+ // akm-core resolver. The contract test runs against `../shared/ref-extraction`
44
+ // (this file) — the divergence between this file and the runtime copy is
45
+ // tracked separately (regex tightness only, see top-level repo notes).
46
+ // ----------------------------------------------------------------------------
47
+
48
+ import { existsSync, statSync, readdirSync } from "node:fs";
49
+ import path from "node:path";
50
+
51
+ /**
52
+ * Permissive ref regex. Matches `[origin//]type:slug` for the known asset
53
+ * types. Origins (e.g. `local//`, `npm:foo//`) are tolerated but stripped
54
+ * before validation — only `local//` refs are resolvable against the local
55
+ * stash; anything else is dropped because we cannot validate it offline.
56
+ *
57
+ * Kept in sync with the lint walker pattern in
58
+ * `src/commands/lint/base-linter.ts`.
59
+ */
60
+ // Slug body allows the same charset as the lint walker, but the closing
61
+ // character must be alphanumeric, `_`, or `-` — so a trailing `.` or `/`
62
+ // from natural prose (`see memory:rollout-notes.`) does not leak into
63
+ // the captured ref string. Mirrors the consumer-side
64
+ // `src/commands/lint/base-linter.ts` REF_RE, which terminates on a
65
+ // punctuation set including `.` via lookahead. We allow `.` mid-slug
66
+ // (e.g. `env:.env`-style names) by requiring the slug to be at least
67
+ // one character and to *end* on `[A-Za-z0-9_-]`.
68
+ const REF_PATTERN =
69
+ /(?:[A-Za-z0-9@._+/-]+\/\/)?(?:skill|command|agent|knowledge|memory|lesson|script|workflow|task|env|secret|wiki):(?:[A-Za-z0-9._/-]*[A-Za-z0-9_-]|[A-Za-z0-9_-])/g;
70
+
71
+ /**
72
+ * Return every `<type>:<slug>` token in `text` regardless of context.
73
+ * Order matches first-occurrence; duplicates are removed.
74
+ */
75
+ export function extractAllRefs(text: string): string[] {
76
+ if (!text) return [];
77
+ return [...new Set(text.match(REF_PATTERN) ?? [])];
78
+ }
79
+
80
+ // Tokenized whitespace-split fallback used by callers that need a strict
81
+ // "this token, in isolation, is a ref" answer (e.g. PreToolUse non-Bash
82
+ // observation, which inspects a single tool input field rather than a
83
+ // transcript-style body). Kept for backward compatibility with existing
84
+ // callers in `claude/hooks/akm-hook.ts` and the opencode plugin.
85
+ const AKM_REF_STRICT =
86
+ /^(?:[A-Za-z0-9@._+/-]+\/\/)?(?:skill|command|agent|knowledge|memory|script|workflow|task|env|secret|wiki|lesson):[A-Za-z0-9._/\-]+$/;
87
+ const EDGE_PUNCTUATION = new Set([".", ",", ";", ":", "!", "?", "(", ")", "[", "]", "{", "}", "'", "\"", "`"]);
88
+
89
+ function normalizeToken(token: string): string {
90
+ let start = 0;
91
+ let end = token.length;
92
+ while (start < end && EDGE_PUNCTUATION.has(token[start] ?? "")) start += 1;
93
+ while (end > start && EDGE_PUNCTUATION.has(token[end - 1] ?? "")) end -= 1;
94
+ return token.slice(start, end);
95
+ }
96
+
97
+ export function extractAkmRefsFromString(text: string): string[] {
98
+ const refs = new Set<string>();
99
+ for (const token of text.split(/\s+/)) {
100
+ const normalized = normalizeToken(token);
101
+ if (normalized && AKM_REF_STRICT.test(normalized)) refs.add(normalized);
102
+ }
103
+ return [...refs];
104
+ }
105
+
106
+ /**
107
+ * Map ref type → relative path within a stash root.
108
+ * Returns `null` for types that cannot be resolved by direct path
109
+ * (scripts live in nested dirs; remote-only types).
110
+ */
111
+ function refToRelPath(refType: string, refName: string): string | null {
112
+ switch (refType) {
113
+ case "agent":
114
+ return path.join("agents", `${refName}.md`);
115
+ case "command":
116
+ return path.join("commands", `${refName}.md`);
117
+ case "knowledge":
118
+ return path.join("knowledge", `${refName}.md`);
119
+ case "memory":
120
+ return path.join("memories", `${refName}.md`);
121
+ case "script":
122
+ return null; // scripts live in nested dirs — skip
123
+ case "skill":
124
+ return path.join("skills", refName, "SKILL.md");
125
+ case "workflow":
126
+ return path.join("workflows", `${refName}.md`);
127
+ case "lesson":
128
+ return path.join("lessons", `${refName}.md`);
129
+ case "task":
130
+ return path.join("tasks", `${refName}.md`);
131
+ case "wiki":
132
+ return path.join("wikis", `${refName}.md`);
133
+ case "env":
134
+ if (!refName || refName === "default") return path.join("env", ".env");
135
+ return path.join("env", `${refName}.env`);
136
+ case "secret":
137
+ return path.join("secrets", refName);
138
+ default:
139
+ return null;
140
+ }
141
+ }
142
+
143
+ /**
144
+ * True if `<type>:<refName>` resolves to a real file under any provided
145
+ * stash root. Mirrors `refExistsInAnyStash` in `src/commands/lint/base-linter.ts`.
146
+ */
147
+ function refExistsInAnyStash(refType: string, refName: string, stashRoots: readonly string[]): boolean {
148
+ const relPath = refToRelPath(refType, refName);
149
+ if (!relPath) return false;
150
+ for (const root of stashRoots) {
151
+ if (!root) continue;
152
+ const absPath = path.join(root, relPath);
153
+ if (existsSync(absPath)) return true;
154
+ // Multi-file skill layout: directory containing SKILL.md
155
+ const bareDir = absPath.replace(/\.md$/, "");
156
+ try {
157
+ if (existsSync(bareDir) && existsSync(path.join(bareDir, "SKILL.md"))) return true;
158
+ } catch {
159
+ // ignore
160
+ }
161
+ // .derived.md variant for memory refs
162
+ if (refType === "memory") {
163
+ const derivedPath = path.join(root, "memories", `${refName}.derived.md`);
164
+ if (existsSync(derivedPath)) return true;
165
+ }
166
+ // Knowledge subdirectory layout (knowledge/projects/foo/...)
167
+ if (refType === "knowledge") {
168
+ try {
169
+ const knowledgeDir = path.join(root, "knowledge");
170
+ if (existsSync(knowledgeDir) && statSync(knowledgeDir).isDirectory()) {
171
+ for (const entry of readdirSync(knowledgeDir)) {
172
+ const subPath = path.join(knowledgeDir, entry, `${refName}.md`);
173
+ if (existsSync(subPath)) return true;
174
+ }
175
+ }
176
+ } catch {
177
+ // ignore
178
+ }
179
+ }
180
+ // Fallback: refName already encodes the stash-relative path
181
+ const directPath = path.join(root, `${refName}.md`);
182
+ if (existsSync(directPath)) return true;
183
+ const directDir = path.join(root, refName);
184
+ try {
185
+ if (existsSync(directDir) && existsSync(path.join(directDir, "SKILL.md"))) return true;
186
+ } catch {
187
+ // ignore
188
+ }
189
+ }
190
+ return false;
191
+ }
192
+
193
+ /**
194
+ * Drop candidate tokens that obviously cannot resolve:
195
+ * - Embedded shell expansion: `memory:$(cmd)`, `knowledge:${VAR}`
196
+ * - ACP type notation: `agent::Type`
197
+ * - Empty / placeholder slugs: single char, `**`, leading `/`, `~`, `http*`
198
+ * - Remote origins other than `local//` (cannot be validated offline)
199
+ */
200
+ function normalizeCandidate(fullRef: string): { type: string; name: string } | null {
201
+ if (fullRef.includes("$(") || fullRef.includes("${")) return null;
202
+ if (fullRef.includes("::")) return null;
203
+
204
+ let ref = fullRef;
205
+ if (ref.startsWith("local//")) {
206
+ ref = ref.slice("local//".length);
207
+ } else if (ref.includes("//")) {
208
+ return null; // remote origin — cannot validate locally
209
+ }
210
+
211
+ const colonIdx = ref.indexOf(":");
212
+ if (colonIdx === -1) return null;
213
+ const type = ref.slice(0, colonIdx);
214
+ const name = ref.slice(colonIdx + 1);
215
+ if (!name || name.startsWith("/") || name.startsWith("~") || name.startsWith("http")) return null;
216
+ if (name.length <= 1 || name === "**") return null;
217
+ // Slug must not contain shell metacharacters — pipe etc. would be a
218
+ // pasted regex like `memory:foo|knowledge:bar`.
219
+ if (/[|&;<>(){}\[\]"`'\\?*]/.test(name)) return null;
220
+ return { type, name };
221
+ }
222
+
223
+ /**
224
+ * Given an arbitrary block of text and one or more stash roots, return the
225
+ * subset of `<type>:<slug>` tokens that actually resolve to a real asset
226
+ * on disk. String literals (heredocs, grep patterns, jq queries) are
227
+ * silently dropped — they don't exist in the stash and therefore cannot
228
+ * generate `missing-ref` lint flags after the body is captured.
229
+ *
230
+ * The returned list is sorted alphabetically and deduplicated.
231
+ */
232
+ export function validateLiveRefs(text: string, stashRoots: readonly string[]): string[] {
233
+ return validateRefCandidates(extractAllRefs(text), stashRoots);
234
+ }
235
+
236
+ /**
237
+ * Validate a pre-extracted list of candidate refs against one or more
238
+ * stash roots. Returns the subset that resolve to a real on-disk asset,
239
+ * sorted alphabetically and deduplicated. The input is typically produced
240
+ * by `extractAllRefs(...)` on raw command/output text — the producer-side
241
+ * collection step is permissive so candidates can be accumulated across
242
+ * tool invocations and validated once at memory-capture time.
243
+ */
244
+ export function validateRefCandidates(candidates: readonly string[], stashRoots: readonly string[]): string[] {
245
+ if (!candidates || candidates.length === 0) return [];
246
+ const roots = stashRoots.filter(Boolean);
247
+ if (roots.length === 0) return [];
248
+ const seen = new Set<string>();
249
+ const out: string[] = [];
250
+ for (const candidate of candidates) {
251
+ const norm = normalizeCandidate(candidate);
252
+ if (!norm) continue;
253
+ if (!refExistsInAnyStash(norm.type, norm.name, roots)) continue;
254
+ const canonical = `${norm.type}:${norm.name}`;
255
+ if (seen.has(canonical)) continue;
256
+ seen.add(canonical);
257
+ out.push(canonical);
258
+ }
259
+ out.sort((a, b) => (a < b ? -1 : a > b ? 1 : 0));
260
+ return out;
261
+ }
@@ -1,7 +0,0 @@
1
- Distill repeated evidence into a proposed lesson.
2
-
3
- 1. Search or curate for candidate refs.
4
- 2. Show the strongest evidence refs.
5
- 3. Call `akm_help` with `topic: "distill"`.
6
- 4. Run `akm distill <ref>` only after the evidence is clear.
7
- 5. Report the resulting proposal and remind the user that proposed assets are not curated until accepted.
@@ -1,7 +0,0 @@
1
- Reflect on an existing AKM asset after failure or drift.
2
-
3
- 1. Identify the failure evidence and touched refs.
4
- 2. Record negative feedback when justified.
5
- 3. Call `akm_help` with `topic: "reflect"`.
6
- 4. Run `akm reflect <ref> --task "..."`.
7
- 5. List resulting pending proposals and do not accept or reject them without explicit user approval.