akm-opencode 0.8.0-rc1 → 0.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +90 -30
- package/agent/akm-curator.md +4 -4
- package/commands/akm-improve-asset.md +9 -0
- package/commands/akm-propose-asset.md +1 -1
- package/commands/akm-review-proposals.md +2 -1
- package/index.ts +1068 -532
- package/package.json +10 -3
- package/shared/feedback-signals.ts +74 -0
- package/shared/memory-candidates.ts +196 -0
- package/shared/memory-events.ts +122 -0
- package/shared/recall-policy.ts +70 -0
- package/shared/redaction.ts +292 -0
- package/shared/ref-extraction.ts +261 -0
- package/commands/akm-distill-lesson.md +0 -7
- package/commands/akm-reflect-on-failure.md +0 -7
|
@@ -0,0 +1,292 @@
|
|
|
1
|
+
import { inspect } from "node:util"
|
|
2
|
+
|
|
3
|
+
export type RedactionResult = {
|
|
4
|
+
text: string
|
|
5
|
+
redacted: boolean
|
|
6
|
+
categories: string[]
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
export type RedactedObjectResult<T> = {
|
|
10
|
+
value: T
|
|
11
|
+
redacted: boolean
|
|
12
|
+
categories: string[]
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
const SENSITIVE_KEY_RE = /(?:OPENAI_API_KEY|ANTHROPIC_API_KEY|GITHUB_TOKEN|GH_TOKEN|AWS_ACCESS_KEY_ID|AWS_SECRET_ACCESS_KEY|AWS_SESSION_TOKEN|AZURE_[A-Z0-9_]*KEY|SLACK_BOT_TOKEN|SLACK_TOKEN|PASSWORD|PASSWD|SECRET|CONNECTION_STRING|DATABASE_URL|AKM_[A-Z0-9_]*TOKEN)/i
|
|
16
|
+
const SIMPLE_REPLACEMENTS: Array<{ category: string; pattern: RegExp; replacement: string }> = [
|
|
17
|
+
{
|
|
18
|
+
category: "private_key",
|
|
19
|
+
pattern: /-----BEGIN [A-Z ]*PRIVATE KEY-----[\s\S]*?-----END [A-Z ]*PRIVATE KEY-----/g,
|
|
20
|
+
replacement: "[REDACTED:PRIVATE_KEY]",
|
|
21
|
+
},
|
|
22
|
+
{
|
|
23
|
+
category: "bearer_token",
|
|
24
|
+
pattern: /Bearer\s+[A-Za-z0-9._~+/=-]+/gi,
|
|
25
|
+
replacement: "[REDACTED:BEARER_TOKEN]",
|
|
26
|
+
},
|
|
27
|
+
{
|
|
28
|
+
category: "github_token",
|
|
29
|
+
pattern: /\bgh[pousr]_[A-Za-z0-9_]+\b/g,
|
|
30
|
+
replacement: "[REDACTED:GITHUB_TOKEN]",
|
|
31
|
+
},
|
|
32
|
+
{
|
|
33
|
+
category: "github_pat",
|
|
34
|
+
pattern: /\bgithub_pat_[A-Za-z0-9_]+\b/g,
|
|
35
|
+
replacement: "[REDACTED:GITHUB_PAT]",
|
|
36
|
+
},
|
|
37
|
+
{
|
|
38
|
+
category: "openai_key",
|
|
39
|
+
pattern: /\bsk-[A-Za-z0-9_-]+\b/g,
|
|
40
|
+
replacement: "[REDACTED:OPENAI_API_KEY]",
|
|
41
|
+
},
|
|
42
|
+
{
|
|
43
|
+
category: "slack_token",
|
|
44
|
+
pattern: /\bxox[baprs]-[A-Za-z0-9-]+\b/g,
|
|
45
|
+
replacement: "[REDACTED:SLACK_TOKEN]",
|
|
46
|
+
},
|
|
47
|
+
// 0.8.0 release-hardening: AWS access key (AKIA prefix, 16 alphanumerics).
|
|
48
|
+
// IAM user access keys are 20 chars total starting with AKIA; STS temporary
|
|
49
|
+
// creds use ASIA but those are paired with a session token and short-lived,
|
|
50
|
+
// so the AKIA prefix is the high-value leak.
|
|
51
|
+
{
|
|
52
|
+
category: "aws_access_key",
|
|
53
|
+
pattern: /\bAKIA[0-9A-Z]{16}\b/g,
|
|
54
|
+
replacement: "[REDACTED:AWS_ACCESS_KEY_ID]",
|
|
55
|
+
},
|
|
56
|
+
// Google Cloud API key (AIza prefix, 35 base64url-safe chars). Pattern is
|
|
57
|
+
// documented at cloud.google.com/docs/authentication/api-keys.
|
|
58
|
+
{
|
|
59
|
+
category: "google_api_key",
|
|
60
|
+
pattern: /\bAIza[0-9A-Za-z_-]{35}\b/g,
|
|
61
|
+
replacement: "[REDACTED:GOOGLE_API_KEY]",
|
|
62
|
+
},
|
|
63
|
+
// Stripe live-mode secret/publishable keys (sk_live_, pk_live_). Stripe
|
|
64
|
+
// documents these as the high-severity leak surface in their API docs.
|
|
65
|
+
{
|
|
66
|
+
category: "stripe_live_key",
|
|
67
|
+
pattern: /\b(?:sk|pk)_live_[0-9a-zA-Z]+\b/g,
|
|
68
|
+
replacement: "[REDACTED:STRIPE_LIVE_KEY]",
|
|
69
|
+
},
|
|
70
|
+
// Stripe test-mode keys (sk_test_, pk_test_). Lower severity but still
|
|
71
|
+
// sensitive — test keys often grant access to test webhooks and dashboards
|
|
72
|
+
// that mirror production object shapes.
|
|
73
|
+
{
|
|
74
|
+
category: "stripe_test_key",
|
|
75
|
+
pattern: /\b(?:sk|pk)_test_[0-9a-zA-Z]+\b/g,
|
|
76
|
+
replacement: "[REDACTED:STRIPE_TEST_KEY]",
|
|
77
|
+
},
|
|
78
|
+
// Raw `Authorization: <token>` header without the `Bearer` keyword. Some
|
|
79
|
+
// services (legacy AWS, internal APIs, OAuth1 signed requests) send the
|
|
80
|
+
// token directly. The Bearer-form pattern above handles the common case.
|
|
81
|
+
// We require at least 12 chars of token to avoid matching `Authorization: ?`
|
|
82
|
+
// or other very short placeholder values.
|
|
83
|
+
{
|
|
84
|
+
category: "authorization_header",
|
|
85
|
+
pattern: /(Authorization:\s+)(?!Bearer\b)([A-Za-z0-9._~+/=-]{12,})/gi,
|
|
86
|
+
replacement: "$1[REDACTED:AUTHORIZATION]",
|
|
87
|
+
},
|
|
88
|
+
// WS-7b: Database connection strings — redact credentials portion.
|
|
89
|
+
// Matches postgres://, mysql://, mongodb+srv://, mongodb://, redis:// with user:pass@ form.
|
|
90
|
+
{
|
|
91
|
+
category: "connection_string",
|
|
92
|
+
pattern: /((?:postgres|postgresql|mysql|mongodb(?:\+srv)?|redis):\/\/)([^@\s]+@)([^\s"'`,]+)/gi,
|
|
93
|
+
replacement: "$1[REDACTED:CREDENTIALS]@$3",
|
|
94
|
+
},
|
|
95
|
+
// WS-7b: AWS ARN structures with embedded account IDs (12-digit).
|
|
96
|
+
// Matches arn:aws:...:123456789012:... patterns.
|
|
97
|
+
{
|
|
98
|
+
category: "aws_arn",
|
|
99
|
+
pattern: /arn:aws:[a-z0-9_-]+:[a-z0-9-]*:(\d{12}):[^\s"'`,]*/gi,
|
|
100
|
+
replacement: "arn:aws:...[REDACTED:ACCOUNT_ID]:...",
|
|
101
|
+
},
|
|
102
|
+
// WS-7b: JWT-shaped tokens — three base64url segments separated by dots (eyJ prefix).
|
|
103
|
+
{
|
|
104
|
+
category: "jwt_token",
|
|
105
|
+
pattern: /\beyJ[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\.[A-Za-z0-9_-]+\b/g,
|
|
106
|
+
replacement: "[REDACTED:JWT_TOKEN]",
|
|
107
|
+
},
|
|
108
|
+
]
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* High-entropy string redaction (WS-7b).
|
|
112
|
+
*
|
|
113
|
+
* Generic strings >= 32 chars matching [A-Za-z0-9+/=_-]+ that look like
|
|
114
|
+
* base64/hex secrets. Disabled by default to avoid false positives (e.g.
|
|
115
|
+
* long identifiers, file paths, etc.). Opt in by setting the env var:
|
|
116
|
+
*
|
|
117
|
+
* AKM_REDACT_HIGH_ENTROPY=1
|
|
118
|
+
*
|
|
119
|
+
* The threshold is 32 chars, configurable via AKM_REDACT_ENTROPY_MIN_LEN.
|
|
120
|
+
*/
|
|
121
|
+
const HIGH_ENTROPY_ENABLED = process.env.AKM_REDACT_HIGH_ENTROPY === "1"
|
|
122
|
+
const HIGH_ENTROPY_MIN_LEN = Number.parseInt(process.env.AKM_REDACT_ENTROPY_MIN_LEN ?? "32", 10)
|
|
123
|
+
const HIGH_ENTROPY_RE = HIGH_ENTROPY_ENABLED
|
|
124
|
+
? new RegExp(`\\b[A-Za-z0-9+/=_-]{${Math.max(32, HIGH_ENTROPY_MIN_LEN)},}\\b`, "g")
|
|
125
|
+
: null
|
|
126
|
+
|
|
127
|
+
/**
|
|
128
|
+
* Opt-in PII redaction patterns. Off by default because financial / phone
|
|
129
|
+
* shapes appear in legitimate logs (timestamps, counters, identifiers) and
|
|
130
|
+
* over-redaction makes session logs unreadable. Enable for hosts that
|
|
131
|
+
* legitimately process PII through the agent loop:
|
|
132
|
+
*
|
|
133
|
+
* AKM_REDACT_PII=1
|
|
134
|
+
*
|
|
135
|
+
* Patterns covered:
|
|
136
|
+
* - Credit-card-shaped 13–19 digit sequences with optional `[- ]` delimiters
|
|
137
|
+
* (Luhn is intentionally NOT verified — we redact aggressively when opt-in)
|
|
138
|
+
* - US Social Security Numbers in `\d{3}-\d{2}-\d{4}` form (delimited only;
|
|
139
|
+
* bare 9-digit numbers are too noisy)
|
|
140
|
+
* - US/international phone numbers in `[\+]?\d{1,3}[- .]?\d{3}[- .]?\d{3}[- .]?\d{4}` form
|
|
141
|
+
*/
|
|
142
|
+
const PII_REDACTION_ENABLED = process.env.AKM_REDACT_PII === "1"
|
|
143
|
+
const PII_PATTERNS: Array<{ category: string; pattern: RegExp; replacement: string }> = PII_REDACTION_ENABLED
|
|
144
|
+
? [
|
|
145
|
+
{
|
|
146
|
+
category: "credit_card",
|
|
147
|
+
pattern: /\b(?:\d[ -]?){12,18}\d\b/g,
|
|
148
|
+
replacement: "[REDACTED:CREDIT_CARD]",
|
|
149
|
+
},
|
|
150
|
+
{
|
|
151
|
+
category: "ssn",
|
|
152
|
+
pattern: /\b\d{3}-\d{2}-\d{4}\b/g,
|
|
153
|
+
replacement: "[REDACTED:SSN]",
|
|
154
|
+
},
|
|
155
|
+
{
|
|
156
|
+
category: "phone",
|
|
157
|
+
pattern: /(?<![\w./-])\+?\d{1,3}[- .]?\(?\d{3}\)?[- .]?\d{3}[- .]?\d{4}(?![\w./-])/g,
|
|
158
|
+
replacement: "[REDACTED:PHONE]",
|
|
159
|
+
},
|
|
160
|
+
]
|
|
161
|
+
: []
|
|
162
|
+
|
|
163
|
+
function uniq(values: string[]): string[] {
|
|
164
|
+
return [...new Set(values)]
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
function redactAssignments(text: string, categories: string[]): string {
|
|
168
|
+
const envLineRe = /^\s*(?:export\s+)?([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(.+)$/gm
|
|
169
|
+
let next = text.replace(envLineRe, (match, rawKey: string) => {
|
|
170
|
+
const key = String(rawKey)
|
|
171
|
+
if (!SENSITIVE_KEY_RE.test(key)) return match
|
|
172
|
+
categories.push(key.toLowerCase().includes("password") || key.toLowerCase().includes("passwd") ? "password" : "env_secret")
|
|
173
|
+
categories.push("env_output")
|
|
174
|
+
return `${key}: [REDACTED:${key}]`
|
|
175
|
+
})
|
|
176
|
+
const inlineAssignmentRe = /\b([A-Za-z_][A-Za-z0-9_]*)\s*=\s*([^\s"'`,;]+)/g
|
|
177
|
+
next = next.replace(inlineAssignmentRe, (match, rawKey: string) => {
|
|
178
|
+
const key = String(rawKey)
|
|
179
|
+
if (!SENSITIVE_KEY_RE.test(key)) return match
|
|
180
|
+
categories.push(key.toLowerCase().includes("password") || key.toLowerCase().includes("passwd") ? "password" : "env_secret")
|
|
181
|
+
categories.push("env_output")
|
|
182
|
+
return `${key}: [REDACTED:${key}]`
|
|
183
|
+
})
|
|
184
|
+
return next
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
function redactJsonLikePairs(text: string, categories: string[]): string {
|
|
188
|
+
const pairRe = /(["']?)(password|passwd|secret|connectionString|apiKey|token|accessKey|secretKey)(\1\s*[:=]\s*)(["']?)([^\n,}"']+)(["']?)/gi
|
|
189
|
+
return text.replace(pairRe, (_match, q1: string, key: string, sep: string) => {
|
|
190
|
+
const upper = key.replace(/[^A-Za-z0-9]+/g, "_").toUpperCase()
|
|
191
|
+
const category = /connection/i.test(key)
|
|
192
|
+
? "connection_string"
|
|
193
|
+
: /pass/i.test(key)
|
|
194
|
+
? "password"
|
|
195
|
+
: /secret/i.test(key)
|
|
196
|
+
? "secret"
|
|
197
|
+
: "token"
|
|
198
|
+
categories.push(category)
|
|
199
|
+
return `${q1}${key}${q1}${sep}[REDACTED:${upper}]`
|
|
200
|
+
})
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
// Redact `--flag value` / `-f value` CLI argv shapes where the flag name
|
|
204
|
+
// contains a sensitive substring (password, token, secret, api-key, etc.).
|
|
205
|
+
// This catches a class of leaks that the env-assignment and JSON-pair
|
|
206
|
+
// regexes miss because CLI args use a space separator rather than `=` / `:`.
|
|
207
|
+
function redactCliArgPairs(text: string, categories: string[]): string {
|
|
208
|
+
const pairRe = /(--?[A-Za-z0-9_-]*?(?:password|passwd|secret|token|api[_-]?key|access[_-]?key|auth)[A-Za-z0-9_-]*)(\s+)(\S+)/gi
|
|
209
|
+
return text.replace(pairRe, (_match, flag: string, sep: string) => {
|
|
210
|
+
categories.push("cli_arg_secret")
|
|
211
|
+
return `${flag}${sep}[REDACTED:CLI_ARG]`
|
|
212
|
+
})
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
export function redactSecrets(input: string): RedactionResult {
|
|
216
|
+
let text = input
|
|
217
|
+
const categories: string[] = []
|
|
218
|
+
|
|
219
|
+
for (const entry of SIMPLE_REPLACEMENTS) {
|
|
220
|
+
if (!entry.pattern.test(text)) continue
|
|
221
|
+
categories.push(entry.category)
|
|
222
|
+
text = text.replace(entry.pattern, entry.replacement)
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
text = redactAssignments(text, categories)
|
|
226
|
+
text = redactJsonLikePairs(text, categories)
|
|
227
|
+
text = redactCliArgPairs(text, categories)
|
|
228
|
+
|
|
229
|
+
// WS-7b: Optional high-entropy string redaction (opt-in via AKM_REDACT_HIGH_ENTROPY=1).
|
|
230
|
+
if (HIGH_ENTROPY_RE && HIGH_ENTROPY_RE.test(text)) {
|
|
231
|
+
categories.push("high_entropy")
|
|
232
|
+
HIGH_ENTROPY_RE.lastIndex = 0 // reset after .test()
|
|
233
|
+
text = text.replace(HIGH_ENTROPY_RE, "[REDACTED:HIGH_ENTROPY]")
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
// Opt-in PII redaction (AKM_REDACT_PII=1).
|
|
237
|
+
for (const entry of PII_PATTERNS) {
|
|
238
|
+
if (!entry.pattern.test(text)) continue
|
|
239
|
+
categories.push(entry.category)
|
|
240
|
+
entry.pattern.lastIndex = 0
|
|
241
|
+
text = text.replace(entry.pattern, entry.replacement)
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
return {
|
|
245
|
+
text,
|
|
246
|
+
redacted: categories.length > 0,
|
|
247
|
+
categories: uniq(categories),
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
|
|
251
|
+
function normalizeObjectInput(value: unknown): unknown {
|
|
252
|
+
if (typeof value === "string") return value
|
|
253
|
+
if (value == null) return value
|
|
254
|
+
if (typeof value === "number" || typeof value === "boolean") return value
|
|
255
|
+
if (Array.isArray(value)) return value.map(normalizeObjectInput)
|
|
256
|
+
if (typeof value === "object") {
|
|
257
|
+
const record: Record<string, unknown> = {}
|
|
258
|
+
for (const [key, entry] of Object.entries(value as Record<string, unknown>)) {
|
|
259
|
+
record[key] = normalizeObjectInput(entry)
|
|
260
|
+
}
|
|
261
|
+
return record
|
|
262
|
+
}
|
|
263
|
+
return inspect(value, { depth: 4, breakLength: 120 })
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
function redactUnknown(value: unknown, categories: string[]): unknown {
|
|
267
|
+
if (typeof value === "string") {
|
|
268
|
+
const redacted = redactSecrets(value)
|
|
269
|
+
categories.push(...redacted.categories)
|
|
270
|
+
return redacted.text
|
|
271
|
+
}
|
|
272
|
+
if (Array.isArray(value)) return value.map((entry) => redactUnknown(entry, categories))
|
|
273
|
+
if (value && typeof value === "object") {
|
|
274
|
+
const output: Record<string, unknown> = {}
|
|
275
|
+
for (const [key, entry] of Object.entries(value as Record<string, unknown>)) {
|
|
276
|
+
output[key] = redactUnknown(entry, categories)
|
|
277
|
+
}
|
|
278
|
+
return output
|
|
279
|
+
}
|
|
280
|
+
return value
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
export function redactObject<T>(input: T): RedactedObjectResult<T> {
|
|
284
|
+
const categories: string[] = []
|
|
285
|
+
const normalized = normalizeObjectInput(input)
|
|
286
|
+
const value = redactUnknown(normalized, categories) as T
|
|
287
|
+
return {
|
|
288
|
+
value,
|
|
289
|
+
redacted: categories.length > 0,
|
|
290
|
+
categories: uniq(categories),
|
|
291
|
+
}
|
|
292
|
+
}
|
|
@@ -0,0 +1,261 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* AKM ref extraction & live-stash validation.
|
|
3
|
+
*
|
|
4
|
+
* Background: session-checkpoint memories captured by the hook embed Bash
|
|
5
|
+
* command bodies verbatim — heredocs, grep patterns, jq queries, JSON
|
|
6
|
+
* payloads. Naive ref extraction (`text.match(REF_PATTERN)`) treats every
|
|
7
|
+
* `<type>:<slug>` token as a real reference, which causes `akm lint` to
|
|
8
|
+
* flag string-literal tokens as `missing-ref`. The next session capture
|
|
9
|
+
* regenerates the same flags — a permanent lint treadmill.
|
|
10
|
+
*
|
|
11
|
+
* The fix (this module): we drop the "guess from context" heuristic and
|
|
12
|
+
* instead **validate every candidate against the live local stash**. A
|
|
13
|
+
* token only graduates from "candidate" to "ref" when the referenced asset
|
|
14
|
+
* actually exists on disk. Anything that doesn't resolve is silently
|
|
15
|
+
* dropped — including all the literal-string false positives.
|
|
16
|
+
*
|
|
17
|
+
* The validation logic mirrors the consumer-side lint walker
|
|
18
|
+
* (`src/commands/lint/base-linter.ts#refExistsInAnyStash`). We deliberately
|
|
19
|
+
* inline a small copy here (rather than spawn a subprocess) so the hook's
|
|
20
|
+
* post-tool path stays cheap (zero subprocess, zero JSON parse). The two
|
|
21
|
+
* resolvers must stay in sync; any new asset type added to lint's
|
|
22
|
+
* `refToRelPath` must be added here too.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
// CONTRACT: ref-resolver
|
|
26
|
+
// ----------------------------------------------------------------------------
|
|
27
|
+
// The `refExistsInAnyStash` and `refToRelPath` helpers below are
|
|
28
|
+
// contract-locked: a sister copy lives in the akm-core repo at
|
|
29
|
+
// `src/commands/lint/base-linter.ts`. Both implementations resolve the same
|
|
30
|
+
// `<type>:<slug>` -> on-disk-asset question and MUST agree on the set of
|
|
31
|
+
// reachable refs for any given stash layout.
|
|
32
|
+
//
|
|
33
|
+
// The lock is enforced by `tests/ref-resolver-contract.test.ts`, which drives
|
|
34
|
+
// `validateRefCandidates` (the only public entry to this resolver) through a
|
|
35
|
+
// canonical fixture set. The akm-core repo ships an equivalent test at
|
|
36
|
+
// `tests/contracts/ref-resolver-contract.test.ts` that drives ITS copy
|
|
37
|
+
// through the SAME inputs. Any change to the resolver behavior on either
|
|
38
|
+
// side MUST update both contract tests in lockstep, or one will fail.
|
|
39
|
+
//
|
|
40
|
+
// NOTE: this file is the SECOND copy of the resolver. The runtime-shipped
|
|
41
|
+
// copy lives at `claude/shared/ref-extraction.ts` and is imported by the
|
|
42
|
+
// post-tool hook. Both copies must agree with each other AND with the
|
|
43
|
+
// akm-core resolver. The contract test runs against `../shared/ref-extraction`
|
|
44
|
+
// (this file) — the divergence between this file and the runtime copy is
|
|
45
|
+
// tracked separately (regex tightness only, see top-level repo notes).
|
|
46
|
+
// ----------------------------------------------------------------------------
|
|
47
|
+
|
|
48
|
+
import { existsSync, statSync, readdirSync } from "node:fs";
|
|
49
|
+
import path from "node:path";
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Permissive ref regex. Matches `[origin//]type:slug` for the known asset
|
|
53
|
+
* types. Origins (e.g. `local//`, `npm:foo//`) are tolerated but stripped
|
|
54
|
+
* before validation — only `local//` refs are resolvable against the local
|
|
55
|
+
* stash; anything else is dropped because we cannot validate it offline.
|
|
56
|
+
*
|
|
57
|
+
* Kept in sync with the lint walker pattern in
|
|
58
|
+
* `src/commands/lint/base-linter.ts`.
|
|
59
|
+
*/
|
|
60
|
+
// Slug body allows the same charset as the lint walker, but the closing
|
|
61
|
+
// character must be alphanumeric, `_`, or `-` — so a trailing `.` or `/`
|
|
62
|
+
// from natural prose (`see memory:rollout-notes.`) does not leak into
|
|
63
|
+
// the captured ref string. Mirrors the consumer-side
|
|
64
|
+
// `src/commands/lint/base-linter.ts` REF_RE, which terminates on a
|
|
65
|
+
// punctuation set including `.` via lookahead. We allow `.` mid-slug
|
|
66
|
+
// (e.g. `env:.env`-style names) by requiring the slug to be at least
|
|
67
|
+
// one character and to *end* on `[A-Za-z0-9_-]`.
|
|
68
|
+
const REF_PATTERN =
|
|
69
|
+
/(?:[A-Za-z0-9@._+/-]+\/\/)?(?:skill|command|agent|knowledge|memory|lesson|script|workflow|task|env|secret|wiki):(?:[A-Za-z0-9._/-]*[A-Za-z0-9_-]|[A-Za-z0-9_-])/g;
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Return every `<type>:<slug>` token in `text` regardless of context.
|
|
73
|
+
* Order matches first-occurrence; duplicates are removed.
|
|
74
|
+
*/
|
|
75
|
+
export function extractAllRefs(text: string): string[] {
|
|
76
|
+
if (!text) return [];
|
|
77
|
+
return [...new Set(text.match(REF_PATTERN) ?? [])];
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
// Tokenized whitespace-split fallback used by callers that need a strict
|
|
81
|
+
// "this token, in isolation, is a ref" answer (e.g. PreToolUse non-Bash
|
|
82
|
+
// observation, which inspects a single tool input field rather than a
|
|
83
|
+
// transcript-style body). Kept for backward compatibility with existing
|
|
84
|
+
// callers in `claude/hooks/akm-hook.ts` and the opencode plugin.
|
|
85
|
+
const AKM_REF_STRICT =
|
|
86
|
+
/^(?:[A-Za-z0-9@._+/-]+\/\/)?(?:skill|command|agent|knowledge|memory|script|workflow|task|env|secret|wiki|lesson):[A-Za-z0-9._/\-]+$/;
|
|
87
|
+
const EDGE_PUNCTUATION = new Set([".", ",", ";", ":", "!", "?", "(", ")", "[", "]", "{", "}", "'", "\"", "`"]);
|
|
88
|
+
|
|
89
|
+
function normalizeToken(token: string): string {
|
|
90
|
+
let start = 0;
|
|
91
|
+
let end = token.length;
|
|
92
|
+
while (start < end && EDGE_PUNCTUATION.has(token[start] ?? "")) start += 1;
|
|
93
|
+
while (end > start && EDGE_PUNCTUATION.has(token[end - 1] ?? "")) end -= 1;
|
|
94
|
+
return token.slice(start, end);
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
export function extractAkmRefsFromString(text: string): string[] {
|
|
98
|
+
const refs = new Set<string>();
|
|
99
|
+
for (const token of text.split(/\s+/)) {
|
|
100
|
+
const normalized = normalizeToken(token);
|
|
101
|
+
if (normalized && AKM_REF_STRICT.test(normalized)) refs.add(normalized);
|
|
102
|
+
}
|
|
103
|
+
return [...refs];
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
/**
|
|
107
|
+
* Map ref type → relative path within a stash root.
|
|
108
|
+
* Returns `null` for types that cannot be resolved by direct path
|
|
109
|
+
* (scripts live in nested dirs; remote-only types).
|
|
110
|
+
*/
|
|
111
|
+
function refToRelPath(refType: string, refName: string): string | null {
|
|
112
|
+
switch (refType) {
|
|
113
|
+
case "agent":
|
|
114
|
+
return path.join("agents", `${refName}.md`);
|
|
115
|
+
case "command":
|
|
116
|
+
return path.join("commands", `${refName}.md`);
|
|
117
|
+
case "knowledge":
|
|
118
|
+
return path.join("knowledge", `${refName}.md`);
|
|
119
|
+
case "memory":
|
|
120
|
+
return path.join("memories", `${refName}.md`);
|
|
121
|
+
case "script":
|
|
122
|
+
return null; // scripts live in nested dirs — skip
|
|
123
|
+
case "skill":
|
|
124
|
+
return path.join("skills", refName, "SKILL.md");
|
|
125
|
+
case "workflow":
|
|
126
|
+
return path.join("workflows", `${refName}.md`);
|
|
127
|
+
case "lesson":
|
|
128
|
+
return path.join("lessons", `${refName}.md`);
|
|
129
|
+
case "task":
|
|
130
|
+
return path.join("tasks", `${refName}.md`);
|
|
131
|
+
case "wiki":
|
|
132
|
+
return path.join("wikis", `${refName}.md`);
|
|
133
|
+
case "env":
|
|
134
|
+
if (!refName || refName === "default") return path.join("env", ".env");
|
|
135
|
+
return path.join("env", `${refName}.env`);
|
|
136
|
+
case "secret":
|
|
137
|
+
return path.join("secrets", refName);
|
|
138
|
+
default:
|
|
139
|
+
return null;
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
/**
|
|
144
|
+
* True if `<type>:<refName>` resolves to a real file under any provided
|
|
145
|
+
* stash root. Mirrors `refExistsInAnyStash` in `src/commands/lint/base-linter.ts`.
|
|
146
|
+
*/
|
|
147
|
+
function refExistsInAnyStash(refType: string, refName: string, stashRoots: readonly string[]): boolean {
|
|
148
|
+
const relPath = refToRelPath(refType, refName);
|
|
149
|
+
if (!relPath) return false;
|
|
150
|
+
for (const root of stashRoots) {
|
|
151
|
+
if (!root) continue;
|
|
152
|
+
const absPath = path.join(root, relPath);
|
|
153
|
+
if (existsSync(absPath)) return true;
|
|
154
|
+
// Multi-file skill layout: directory containing SKILL.md
|
|
155
|
+
const bareDir = absPath.replace(/\.md$/, "");
|
|
156
|
+
try {
|
|
157
|
+
if (existsSync(bareDir) && existsSync(path.join(bareDir, "SKILL.md"))) return true;
|
|
158
|
+
} catch {
|
|
159
|
+
// ignore
|
|
160
|
+
}
|
|
161
|
+
// .derived.md variant for memory refs
|
|
162
|
+
if (refType === "memory") {
|
|
163
|
+
const derivedPath = path.join(root, "memories", `${refName}.derived.md`);
|
|
164
|
+
if (existsSync(derivedPath)) return true;
|
|
165
|
+
}
|
|
166
|
+
// Knowledge subdirectory layout (knowledge/projects/foo/...)
|
|
167
|
+
if (refType === "knowledge") {
|
|
168
|
+
try {
|
|
169
|
+
const knowledgeDir = path.join(root, "knowledge");
|
|
170
|
+
if (existsSync(knowledgeDir) && statSync(knowledgeDir).isDirectory()) {
|
|
171
|
+
for (const entry of readdirSync(knowledgeDir)) {
|
|
172
|
+
const subPath = path.join(knowledgeDir, entry, `${refName}.md`);
|
|
173
|
+
if (existsSync(subPath)) return true;
|
|
174
|
+
}
|
|
175
|
+
}
|
|
176
|
+
} catch {
|
|
177
|
+
// ignore
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
// Fallback: refName already encodes the stash-relative path
|
|
181
|
+
const directPath = path.join(root, `${refName}.md`);
|
|
182
|
+
if (existsSync(directPath)) return true;
|
|
183
|
+
const directDir = path.join(root, refName);
|
|
184
|
+
try {
|
|
185
|
+
if (existsSync(directDir) && existsSync(path.join(directDir, "SKILL.md"))) return true;
|
|
186
|
+
} catch {
|
|
187
|
+
// ignore
|
|
188
|
+
}
|
|
189
|
+
}
|
|
190
|
+
return false;
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
/**
|
|
194
|
+
* Drop candidate tokens that obviously cannot resolve:
|
|
195
|
+
* - Embedded shell expansion: `memory:$(cmd)`, `knowledge:${VAR}`
|
|
196
|
+
* - ACP type notation: `agent::Type`
|
|
197
|
+
* - Empty / placeholder slugs: single char, `**`, leading `/`, `~`, `http*`
|
|
198
|
+
* - Remote origins other than `local//` (cannot be validated offline)
|
|
199
|
+
*/
|
|
200
|
+
function normalizeCandidate(fullRef: string): { type: string; name: string } | null {
|
|
201
|
+
if (fullRef.includes("$(") || fullRef.includes("${")) return null;
|
|
202
|
+
if (fullRef.includes("::")) return null;
|
|
203
|
+
|
|
204
|
+
let ref = fullRef;
|
|
205
|
+
if (ref.startsWith("local//")) {
|
|
206
|
+
ref = ref.slice("local//".length);
|
|
207
|
+
} else if (ref.includes("//")) {
|
|
208
|
+
return null; // remote origin — cannot validate locally
|
|
209
|
+
}
|
|
210
|
+
|
|
211
|
+
const colonIdx = ref.indexOf(":");
|
|
212
|
+
if (colonIdx === -1) return null;
|
|
213
|
+
const type = ref.slice(0, colonIdx);
|
|
214
|
+
const name = ref.slice(colonIdx + 1);
|
|
215
|
+
if (!name || name.startsWith("/") || name.startsWith("~") || name.startsWith("http")) return null;
|
|
216
|
+
if (name.length <= 1 || name === "**") return null;
|
|
217
|
+
// Slug must not contain shell metacharacters — pipe etc. would be a
|
|
218
|
+
// pasted regex like `memory:foo|knowledge:bar`.
|
|
219
|
+
if (/[|&;<>(){}\[\]"`'\\?*]/.test(name)) return null;
|
|
220
|
+
return { type, name };
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
/**
|
|
224
|
+
* Given an arbitrary block of text and one or more stash roots, return the
|
|
225
|
+
* subset of `<type>:<slug>` tokens that actually resolve to a real asset
|
|
226
|
+
* on disk. String literals (heredocs, grep patterns, jq queries) are
|
|
227
|
+
* silently dropped — they don't exist in the stash and therefore cannot
|
|
228
|
+
* generate `missing-ref` lint flags after the body is captured.
|
|
229
|
+
*
|
|
230
|
+
* The returned list is sorted alphabetically and deduplicated.
|
|
231
|
+
*/
|
|
232
|
+
export function validateLiveRefs(text: string, stashRoots: readonly string[]): string[] {
|
|
233
|
+
return validateRefCandidates(extractAllRefs(text), stashRoots);
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
/**
|
|
237
|
+
* Validate a pre-extracted list of candidate refs against one or more
|
|
238
|
+
* stash roots. Returns the subset that resolve to a real on-disk asset,
|
|
239
|
+
* sorted alphabetically and deduplicated. The input is typically produced
|
|
240
|
+
* by `extractAllRefs(...)` on raw command/output text — the producer-side
|
|
241
|
+
* collection step is permissive so candidates can be accumulated across
|
|
242
|
+
* tool invocations and validated once at memory-capture time.
|
|
243
|
+
*/
|
|
244
|
+
export function validateRefCandidates(candidates: readonly string[], stashRoots: readonly string[]): string[] {
|
|
245
|
+
if (!candidates || candidates.length === 0) return [];
|
|
246
|
+
const roots = stashRoots.filter(Boolean);
|
|
247
|
+
if (roots.length === 0) return [];
|
|
248
|
+
const seen = new Set<string>();
|
|
249
|
+
const out: string[] = [];
|
|
250
|
+
for (const candidate of candidates) {
|
|
251
|
+
const norm = normalizeCandidate(candidate);
|
|
252
|
+
if (!norm) continue;
|
|
253
|
+
if (!refExistsInAnyStash(norm.type, norm.name, roots)) continue;
|
|
254
|
+
const canonical = `${norm.type}:${norm.name}`;
|
|
255
|
+
if (seen.has(canonical)) continue;
|
|
256
|
+
seen.add(canonical);
|
|
257
|
+
out.push(canonical);
|
|
258
|
+
}
|
|
259
|
+
out.sort((a, b) => (a < b ? -1 : a > b ? 1 : 0));
|
|
260
|
+
return out;
|
|
261
|
+
}
|
|
@@ -1,7 +0,0 @@
|
|
|
1
|
-
Distill repeated evidence into a proposed lesson.
|
|
2
|
-
|
|
3
|
-
1. Search or curate for candidate refs.
|
|
4
|
-
2. Show the strongest evidence refs.
|
|
5
|
-
3. Call `akm_help` with `topic: "distill"`.
|
|
6
|
-
4. Run `akm distill <ref>` only after the evidence is clear.
|
|
7
|
-
5. Report the resulting proposal and remind the user that proposed assets are not curated until accepted.
|
|
@@ -1,7 +0,0 @@
|
|
|
1
|
-
Reflect on an existing AKM asset after failure or drift.
|
|
2
|
-
|
|
3
|
-
1. Identify the failure evidence and touched refs.
|
|
4
|
-
2. Record negative feedback when justified.
|
|
5
|
-
3. Call `akm_help` with `topic: "reflect"`.
|
|
6
|
-
4. Run `akm reflect <ref> --task "..."`.
|
|
7
|
-
5. List resulting pending proposals and do not accept or reject them without explicit user approval.
|