@kolisachint/hoocode-agent 0.5.13 → 0.5.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +71 -0
- package/dist/config.d.ts +9 -0
- package/dist/config.d.ts.map +1 -1
- package/dist/config.js +16 -0
- package/dist/config.js.map +1 -1
- package/dist/core/context-files.d.ts +16 -3
- package/dist/core/context-files.d.ts.map +1 -1
- package/dist/core/context-files.js +77 -6
- package/dist/core/context-files.js.map +1 -1
- package/dist/core/learn/digest.d.ts +17 -0
- package/dist/core/learn/digest.d.ts.map +1 -0
- package/dist/core/learn/digest.js +131 -0
- package/dist/core/learn/digest.js.map +1 -0
- package/dist/core/learn/extract.d.ts +178 -0
- package/dist/core/learn/extract.d.ts.map +1 -0
- package/dist/core/learn/extract.js +705 -0
- package/dist/core/learn/extract.js.map +1 -0
- package/dist/core/learn/normalize.d.ts +65 -0
- package/dist/core/learn/normalize.d.ts.map +1 -0
- package/dist/core/learn/normalize.js +245 -0
- package/dist/core/learn/normalize.js.map +1 -0
- package/dist/core/learn/state.d.ts +115 -0
- package/dist/core/learn/state.d.ts.map +1 -0
- package/dist/core/learn/state.js +151 -0
- package/dist/core/learn/state.js.map +1 -0
- package/dist/core/settings-defaults.d.ts +5 -0
- package/dist/core/settings-defaults.d.ts.map +1 -1
- package/dist/core/settings-defaults.js +5 -0
- package/dist/core/settings-defaults.js.map +1 -1
- package/dist/core/settings-manager.d.ts +14 -0
- package/dist/core/settings-manager.d.ts.map +1 -1
- package/dist/core/settings-manager.js +17 -0
- package/dist/core/settings-manager.js.map +1 -1
- package/dist/core/settings-types.d.ts +5 -0
- package/dist/core/settings-types.d.ts.map +1 -1
- package/dist/core/settings-types.js.map +1 -1
- package/dist/extensions/core/hoo-core.d.ts.map +1 -1
- package/dist/extensions/core/hoo-core.js +2 -0
- package/dist/extensions/core/hoo-core.js.map +1 -1
- package/dist/extensions/core/learn.d.ts +22 -0
- package/dist/extensions/core/learn.d.ts.map +1 -0
- package/dist/extensions/core/learn.js +167 -0
- package/dist/extensions/core/learn.js.map +1 -0
- package/dist/modes/interactive/resource-display.d.ts.map +1 -1
- package/dist/modes/interactive/resource-display.js +6 -1
- package/dist/modes/interactive/resource-display.js.map +1 -1
- package/docs/settings.md +31 -0
- package/docs/usage.md +64 -2
- package/examples/extensions/custom-provider-anthropic/package.json +1 -1
- package/examples/extensions/custom-provider-gitlab-duo/package.json +1 -1
- package/examples/extensions/sandbox/package.json +1 -1
- package/examples/extensions/with-deps/package.json +1 -1
- package/package.json +4 -4
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"context-files.js","sourceRoot":"","sources":["../../src/core/context-files.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH,OAAO,EAAE,UAAU,EAAE,YAAY,EAAE,MAAM,SAAS,CAAC;AACnD,OAAO,EAAE,IAAI,EAAE,OAAO,EAAE,MAAM,WAAW,CAAC;AAC1C,OAAO,KAAK,MAAM,OAAO,CAAC;AAgB1B;;;;GAIG;AACH,MAAM,UAAU,kBAAkB,CAAC,KAAyB,EAAE,WAAmB,EAAsB;IACtG,IAAI,CAAC,KAAK,EAAE,CAAC;QACZ,OAAO,SAAS,CAAC;IAClB,CAAC;IAED,IAAI,UAAU,CAAC,KAAK,CAAC,EAAE,CAAC;QACvB,IAAI,CAAC;YACJ,OAAO,YAAY,CAAC,KAAK,EAAE,OAAO,CAAC,CAAC;QACrC,CAAC;QAAC,OAAO,KAAK,EAAE,CAAC;YAChB,OAAO,CAAC,KAAK,CAAC,KAAK,CAAC,MAAM,CAAC,2BAA2B,WAAW,SAAS,KAAK,KAAK,KAAK,EAAE,CAAC,CAAC,CAAC;YAC9F,OAAO,KAAK,CAAC;QACd,CAAC;IACF,CAAC;IAED,OAAO,KAAK,CAAC;AAAA,CACb;AAED,+EAA+E;AAC/E,6EAA6E;AAC7E,iFAAiF;AACjF,+DAA+D;AAC/D,MAAM,uBAAuB,GAAG,CAAC,GAAG,IAAI,CAAC;AACzC,MAAM,sBAAsB,GAAG,EAAE,GAAG,IAAI,CAAC;AAEzC,SAAS,sBAAsB,CAAC,GAAW,EAAoD;IAC9F,MAAM,QAAQ,GAAa,EAAE,CAAC;IAC9B,MAAM,UAAU,GAAG,CAAC,WAAW,EAAE,WAAW,EAAE,WAAW,EAAE,WAAW,CAAC,CAAC;IACxE,KAAK,MAAM,QAAQ,IAAI,UAAU,EAAE,CAAC;QACnC,MAAM,QAAQ,GAAG,IAAI,CAAC,GAAG,EAAE,QAAQ,CAAC,CAAC;QACrC,IAAI,UAAU,CAAC,QAAQ,CAAC,EAAE,CAAC;YAC1B,IAAI,CAAC;gBACJ,IAAI,OAAO,GAAG,YAAY,CAAC,QAAQ,EAAE,OAAO,CAAC,CAAC;gBAC9C,MAAM,KAAK,GAAG,MAAM,CAAC,UAAU,CAAC,OAAO,EAAE,OAAO,CAAC,CAAC;gBAClD,wEAAwE;gBACxE,wEAAwE;gBACxE,IAAI,IAAyB,CAAC;gBAC9B,IAAI,KAAK,GAAG,sBAAsB,EAAE,CAAC;oBACpC,OAAO;wBACN,OAAO,CAAC,KAAK,CAAC,CAAC,EAAE,sBAAsB,CAAC;4BACxC,iCAAiC,sBAAsB,kHAAgH,CAAC;oBACzK,IAAI,GAAG,WAAW,CAAC;gBACpB,CAAC;qBAAM,IAAI,KAAK,GAAG,uBAAuB,EAAE,CAAC;oBAC5C,IAAI,GAAG,OAAO,CAAC;gBAChB,CAAC;gBACD,OAAO,EAAE,IAAI,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,OAAO,EAAE,MAAM,EAAE,IAAI,CAAC,KAAK,CAAC,KAAK,GAAG,CAAC,CAAC,EAAE,IAAI,EAAE,EAAE,QAAQ,EAAE,CAAC;YAC7F,CAAC;YAAC,OAAO,KAAK,EAAE,CAAC;gBAChB,QAAQ,CAAC,IAAI,CAAC,kBAAkB,QAAQ,KAAK,KAAK,EAAE,CAAC,CAAC;YACvD,CAAC;QACF,CAAC;IACF,CAAC;IACD,OAAO,EAAE,IAAI,EAAE,IAAI,EAAE,QAAQ,EAAE,CAAC;AAAA,CAChC;AAED,MAAM,UAAU,uBAAuB,CAAC,OAA0C,EAGhF;IACD,MAAM,WAAW,GAAG,OAAO,CAAC,GAAG,CAAC;IAChC,MAAM,gBAAgB,GAAG,OAAO,CAAC,QAAQ,CAAC;IAE1C,MAAM,YAAY,GAAkB,EAAE,CAAC;IACvC,MAAM,QAAQ,GAAa,EAAE,CAAC;IAC9B,MAAM,SAAS,GAAG,IAAI,GAAG,EAAU,CAAC;IAEpC,MAAM,YAAY,GAAG,sBAAsB,CAAC,gBAAgB,CAAC,CAAC;IAC9D,IAAI,YAAY,CAAC,IAAI,EAAE,CAAC;QACvB,YAAY,CAAC,IAAI,CAAC,YAAY,CAAC,IAAI,CAAC,CAAC;QACrC,SAAS,CAAC,GAAG,CAAC,YAAY,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;IACvC,CAAC;IACD,QAAQ,CAAC,IAAI,CAAC,GAAG,YAAY,CAAC,QAAQ,CAAC,CAAC;IAExC,MAAM,oBAAoB,GAAkB,EAAE,CAAC;IAE/C,IAAI,UAAU,GAAG,WAAW,CAAC;IAC7B,MAAM,IAAI,GAAG,OAAO,CAAC,GAAG,CAAC,CAAC;IAE1B,OAAO,IAAI,EAAE,CAAC;QACb,MAAM,MAAM,GAAG,sBAAsB,CAAC,UAAU,CAAC,CAAC;QAClD,IAAI,MAAM,CAAC,IAAI,IAAI,CAAC,SAAS,CAAC,GAAG,CAAC,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;YACrD,oBAAoB,CAAC,OAAO,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;YAC1C,SAAS,CAAC,GAAG,CAAC,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;QACjC,CAAC;QACD,QAAQ,CAAC,IAAI,CAAC,GAAG,MAAM,CAAC,QAAQ,CAAC,CAAC;QAElC,IAAI,UAAU,KAAK,IAAI;YAAE,MAAM;QAE/B,MAAM,SAAS,GAAG,OAAO,CAAC,UAAU,EAAE,IAAI,CAAC,CAAC;QAC5C,IAAI,SAAS,KAAK,UAAU;YAAE,MAAM;QACpC,UAAU,GAAG,SAAS,CAAC;IACxB,CAAC;IAED,YAAY,CAAC,IAAI,CAAC,GAAG,oBAAoB,CAAC,CAAC;IAE3C,OAAO,EAAE,WAAW,EAAE,YAAY,EAAE,QAAQ,EAAE,CAAC;AAAA,CAC/C","sourcesContent":["/**\n * Context-file and prompt-input loading for the resource loader.\n *\n * Reads AGENTS.md / CLAUDE.md context files from the agent dir and the cwd\n * ancestor chain (warning/truncating oversized ones, since they are injected\n * into the system prompt every turn), and resolves a system-prompt input that\n * may be either an inline string or a file path. Extracted from resource-loader.ts.\n */\n\nimport { existsSync, readFileSync } from \"node:fs\";\nimport { join, resolve } from \"node:path\";\nimport chalk from \"chalk\";\n\n/**\n * A loaded context file plus the recurring-cost metadata the UI surfaces.\n * `tokens`/`size` are optional so callers that synthesize context files (SDK\n * overrides, tests) can keep passing plain `{ path, content }`.\n */\nexport interface ContextFile {\n\tpath: string;\n\tcontent: string;\n\t/** Rough token estimate (bytes / 4) of the content as loaded. */\n\ttokens?: number;\n\t/** Set only past the soft limit; \"truncated\" means the hard limit clipped it. */\n\tsize?: \"large\" | \"truncated\";\n}\n\n/**\n * Resolve a prompt input that is either an inline string or a path to a file.\n * If the input names an existing file, its contents are returned; otherwise the\n * input is treated as the prompt text itself.\n */\nexport function resolvePromptInput(input: string | undefined, description: string): string | undefined {\n\tif (!input) {\n\t\treturn undefined;\n\t}\n\n\tif (existsSync(input)) {\n\t\ttry {\n\t\t\treturn readFileSync(input, \"utf-8\");\n\t\t} catch (error) {\n\t\t\tconsole.error(chalk.yellow(`Warning: Could not read ${description} file ${input}: ${error}`));\n\t\t\treturn input;\n\t\t}\n\t}\n\n\treturn input;\n}\n\n// Context files (AGENTS.md / CLAUDE.md) are injected into the system prompt on\n// every turn, so their size has a recurring cost on every provider. Warn the\n// user past a soft limit (~2k tokens) and truncate at a hard limit (~10k tokens)\n// so a pasted spec can't silently bloat every request forever.\nconst CONTEXT_FILE_WARN_BYTES = 8 * 1024;\nconst CONTEXT_FILE_MAX_BYTES = 40 * 1024;\n\nfunction loadContextFileFromDir(dir: string): { file: ContextFile | null; warnings: string[] } {\n\tconst warnings: string[] = [];\n\tconst candidates = [\"AGENTS.md\", \"AGENTS.MD\", \"CLAUDE.md\", \"CLAUDE.MD\"];\n\tfor (const filename of candidates) {\n\t\tconst filePath = join(dir, filename);\n\t\tif (existsSync(filePath)) {\n\t\t\ttry {\n\t\t\t\tlet content = readFileSync(filePath, \"utf-8\");\n\t\t\t\tconst bytes = Buffer.byteLength(content, \"utf-8\");\n\t\t\t\t// Size is reported structurally (not as a warning string) so the UI can\n\t\t\t\t// annotate the file where it is already listed instead of repeating it.\n\t\t\t\tlet size: ContextFile[\"size\"];\n\t\t\t\tif (bytes > CONTEXT_FILE_MAX_BYTES) {\n\t\t\t\t\tcontent =\n\t\t\t\t\t\tcontent.slice(0, CONTEXT_FILE_MAX_BYTES) +\n\t\t\t\t\t\t`\\n\\n[truncated: file exceeded ${CONTEXT_FILE_MAX_BYTES} bytes (~10k tokens); keep context files brief — large specs belong in linked files, not in the system prompt]`;\n\t\t\t\t\tsize = \"truncated\";\n\t\t\t\t} else if (bytes > CONTEXT_FILE_WARN_BYTES) {\n\t\t\t\t\tsize = \"large\";\n\t\t\t\t}\n\t\t\t\treturn { file: { path: filePath, content, tokens: Math.round(bytes / 4), size }, warnings };\n\t\t\t} catch (error) {\n\t\t\t\twarnings.push(`Could not read ${filePath}: ${error}`);\n\t\t\t}\n\t\t}\n\t}\n\treturn { file: null, warnings };\n}\n\nexport function loadProjectContextFiles(options: { cwd: string; agentDir: string }): {\n\tagentsFiles: ContextFile[];\n\twarnings: string[];\n} {\n\tconst resolvedCwd = options.cwd;\n\tconst resolvedAgentDir = options.agentDir;\n\n\tconst contextFiles: ContextFile[] = [];\n\tconst warnings: string[] = [];\n\tconst seenPaths = new Set<string>();\n\n\tconst globalResult = loadContextFileFromDir(resolvedAgentDir);\n\tif (globalResult.file) {\n\t\tcontextFiles.push(globalResult.file);\n\t\tseenPaths.add(globalResult.file.path);\n\t}\n\twarnings.push(...globalResult.warnings);\n\n\tconst ancestorContextFiles: ContextFile[] = [];\n\n\tlet currentDir = resolvedCwd;\n\tconst root = resolve(\"/\");\n\n\twhile (true) {\n\t\tconst result = loadContextFileFromDir(currentDir);\n\t\tif (result.file && !seenPaths.has(result.file.path)) {\n\t\t\tancestorContextFiles.unshift(result.file);\n\t\t\tseenPaths.add(result.file.path);\n\t\t}\n\t\twarnings.push(...result.warnings);\n\n\t\tif (currentDir === root) break;\n\n\t\tconst parentDir = resolve(currentDir, \"..\");\n\t\tif (parentDir === currentDir) break;\n\t\tcurrentDir = parentDir;\n\t}\n\n\tcontextFiles.push(...ancestorContextFiles);\n\n\treturn { agentsFiles: contextFiles, warnings };\n}\n"]}
|
|
1
|
+
{"version":3,"file":"context-files.js","sourceRoot":"","sources":["../../src/core/context-files.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;GAaG;AAEH,OAAO,EAAE,UAAU,EAAE,YAAY,EAAE,MAAM,SAAS,CAAC;AACnD,OAAO,EAAE,IAAI,EAAE,OAAO,EAAE,MAAM,WAAW,CAAC;AAC1C,OAAO,KAAK,MAAM,OAAO,CAAC;AAC1B,OAAO,EAAE,gBAAgB,EAAE,MAAM,cAAc,CAAC;AAgBhD;;;;GAIG;AACH,MAAM,UAAU,kBAAkB,CAAC,KAAyB,EAAE,WAAmB,EAAsB;IACtG,IAAI,CAAC,KAAK,EAAE,CAAC;QACZ,OAAO,SAAS,CAAC;IAClB,CAAC;IAED,IAAI,UAAU,CAAC,KAAK,CAAC,EAAE,CAAC;QACvB,IAAI,CAAC;YACJ,OAAO,YAAY,CAAC,KAAK,EAAE,OAAO,CAAC,CAAC;QACrC,CAAC;QAAC,OAAO,KAAK,EAAE,CAAC;YAChB,OAAO,CAAC,KAAK,CAAC,KAAK,CAAC,MAAM,CAAC,2BAA2B,WAAW,SAAS,KAAK,KAAK,KAAK,EAAE,CAAC,CAAC,CAAC;YAC9F,OAAO,KAAK,CAAC;QACd,CAAC;IACF,CAAC;IAED,OAAO,KAAK,CAAC;AAAA,CACb;AAED,+EAA+E;AAC/E,6EAA6E;AAC7E,iFAAiF;AACjF,+DAA+D;AAC/D,MAAM,uBAAuB,GAAG,CAAC,GAAG,IAAI,CAAC;AACzC,MAAM,sBAAsB,GAAG,EAAE,GAAG,IAAI,CAAC;AAEzC,4EAA4E;AAC5E,8EAA8E;AAC9E,+EAA+E;AAC/E,8EAA8E;AAC9E,8EAA8E;AAC9E,oBAAoB;AACpB,MAAM,wBAAwB,GAAG,EAAE,GAAG,IAAI,CAAC;AAC3C,MAAM,uBAAuB,GAAG,EAAE,GAAG,IAAI,CAAC;AAC1C,+EAA+E;AAC/E,yEAAyE;AACzE,MAAM,sBAAsB,GAAG,GAAG,CAAC;AAEnC,SAAS,sBAAsB,CAAC,GAAW,EAAoD;IAC9F,MAAM,QAAQ,GAAa,EAAE,CAAC;IAC9B,MAAM,UAAU,GAAG,CAAC,WAAW,EAAE,WAAW,EAAE,WAAW,EAAE,WAAW,CAAC,CAAC;IACxE,KAAK,MAAM,QAAQ,IAAI,UAAU,EAAE,CAAC;QACnC,MAAM,QAAQ,GAAG,IAAI,CAAC,GAAG,EAAE,QAAQ,CAAC,CAAC;QACrC,IAAI,UAAU,CAAC,QAAQ,CAAC,EAAE,CAAC;YAC1B,IAAI,CAAC;gBACJ,IAAI,OAAO,GAAG,YAAY,CAAC,QAAQ,EAAE,OAAO,CAAC,CAAC;gBAC9C,MAAM,KAAK,GAAG,MAAM,CAAC,UAAU,CAAC,OAAO,EAAE,OAAO,CAAC,CAAC;gBAClD,wEAAwE;gBACxE,wEAAwE;gBACxE,IAAI,IAAyB,CAAC;gBAC9B,IAAI,KAAK,GAAG,sBAAsB,EAAE,CAAC;oBACpC,OAAO;wBACN,OAAO,CAAC,KAAK,CAAC,CAAC,EAAE,sBAAsB,CAAC;4BACxC,iCAAiC,sBAAsB,kHAAgH,CAAC;oBACzK,IAAI,GAAG,WAAW,CAAC;gBACpB,CAAC;qBAAM,IAAI,KAAK,GAAG,uBAAuB,EAAE,CAAC;oBAC5C,IAAI,GAAG,OAAO,CAAC;gBAChB,CAAC;gBACD,OAAO,EAAE,IAAI,EAAE,EAAE,IAAI,EAAE,QAAQ,EAAE,OAAO,EAAE,MAAM,EAAE,IAAI,CAAC,KAAK,CAAC,KAAK,GAAG,CAAC,CAAC,EAAE,IAAI,EAAE,EAAE,QAAQ,EAAE,CAAC;YAC7F,CAAC;YAAC,OAAO,KAAK,EAAE,CAAC;gBAChB,QAAQ,CAAC,IAAI,CAAC,kBAAkB,QAAQ,KAAK,KAAK,EAAE,CAAC,CAAC;YACvD,CAAC;QACF,CAAC;IACF,CAAC;IACD,OAAO,EAAE,IAAI,EAAE,IAAI,EAAE,QAAQ,EAAE,CAAC;AAAA,CAChC;AAaD,MAAM,UAAU,uBAAuB,CAAC,OAAuC,EAG7E;IACD,MAAM,WAAW,GAAG,OAAO,CAAC,GAAG,CAAC;IAChC,MAAM,gBAAgB,GAAG,OAAO,CAAC,QAAQ,CAAC;IAC1C,MAAM,qBAAqB,GAAG,OAAO,CAAC,aAAa,IAAI,gBAAgB,EAAE,CAAC;IAE1E,MAAM,YAAY,GAAkB,EAAE,CAAC;IACvC,MAAM,QAAQ,GAAa,EAAE,CAAC;IAC9B,MAAM,SAAS,GAAG,IAAI,GAAG,EAAU,CAAC;IAEpC,6EAA6E;IAC7E,4EAA4E;IAC5E,sCAAsC;IACtC,KAAK,MAAM,GAAG,IAAI,CAAC,qBAAqB,EAAE,gBAAgB,CAAC,EAAE,CAAC;QAC7D,MAAM,MAAM,GAAG,sBAAsB,CAAC,GAAG,CAAC,CAAC;QAC3C,IAAI,MAAM,CAAC,IAAI,IAAI,CAAC,SAAS,CAAC,GAAG,CAAC,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;YACrD,YAAY,CAAC,IAAI,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;YAC/B,SAAS,CAAC,GAAG,CAAC,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;QACjC,CAAC;QACD,QAAQ,CAAC,IAAI,CAAC,GAAG,MAAM,CAAC,QAAQ,CAAC,CAAC;IACnC,CAAC;IAED,MAAM,oBAAoB,GAAkB,EAAE,CAAC;IAE/C,IAAI,UAAU,GAAG,WAAW,CAAC;IAC7B,MAAM,IAAI,GAAG,OAAO,CAAC,GAAG,CAAC,CAAC;IAE1B,OAAO,IAAI,EAAE,CAAC;QACb,MAAM,MAAM,GAAG,sBAAsB,CAAC,UAAU,CAAC,CAAC;QAClD,IAAI,MAAM,CAAC,IAAI,IAAI,CAAC,SAAS,CAAC,GAAG,CAAC,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC;YACrD,oBAAoB,CAAC,OAAO,CAAC,MAAM,CAAC,IAAI,CAAC,CAAC;YAC1C,SAAS,CAAC,GAAG,CAAC,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;QACjC,CAAC;QACD,QAAQ,CAAC,IAAI,CAAC,GAAG,MAAM,CAAC,QAAQ,CAAC,CAAC;QAElC,IAAI,UAAU,KAAK,IAAI;YAAE,MAAM;QAE/B,MAAM,SAAS,GAAG,OAAO,CAAC,UAAU,EAAE,IAAI,CAAC,CAAC;QAC5C,IAAI,SAAS,KAAK,UAAU;YAAE,MAAM;QACpC,UAAU,GAAG,SAAS,CAAC;IACxB,CAAC;IAED,YAAY,CAAC,IAAI,CAAC,GAAG,oBAAoB,CAAC,CAAC;IAE3C,QAAQ,CAAC,IAAI,CAAC,GAAG,kBAAkB,CAAC,YAAY,CAAC,CAAC,CAAC;IAEnD,OAAO,EAAE,WAAW,EAAE,YAAY,EAAE,QAAQ,EAAE,CAAC;AAAA,CAC/C;AAED;;;;;;;;GAQG;AACH,SAAS,kBAAkB,CAAC,KAAoB,EAAY;IAC3D,MAAM,QAAQ,GAAa,EAAE,CAAC;IAC9B,MAAM,KAAK,GAAG,KAAK,CAAC,MAAM,CAAC,CAAC,GAAG,EAAE,IAAI,EAAE,EAAE,CAAC,GAAG,GAAG,MAAM,CAAC,UAAU,CAAC,IAAI,CAAC,OAAO,EAAE,OAAO,CAAC,EAAE,CAAC,CAAC,CAAC;IAC7F,IAAI,KAAK,IAAI,wBAAwB,EAAE,CAAC;QACvC,OAAO,QAAQ,CAAC;IACjB,CAAC;IAED,IAAI,KAAK,IAAI,uBAAuB,EAAE,CAAC;QACtC,wEAAwE;QACxE,0EAA0E;QAC1E,uEAAuE;QACvE,uEAAuE;QACvE,IAAI,KAAK,CAAC,MAAM,GAAG,CAAC;YAAE,OAAO,QAAQ,CAAC;QACtC,QAAQ,CAAC,IAAI,CACZ,wBAAwB,IAAI,CAAC,KAAK,CAAC,KAAK,GAAG,CAAC,GAAG,GAAG,CAAC,GAAG,EAAE,mBAAmB,KAAK,CAAC,MAAM,YAAY;YAClG,4GAA4G;YAC5G,gEAAgE,CACjE,CAAC;QACF,OAAO,QAAQ,CAAC;IACjB,CAAC;IAED,IAAI,SAAS,GAAG,uBAAuB,CAAC;IACxC,KAAK,IAAI,CAAC,GAAG,KAAK,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC,IAAI,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC;QAC5C,MAAM,IAAI,GAAG,KAAK,CAAC,CAAC,CAAE,CAAC;QACvB,MAAM,KAAK,GAAG,MAAM,CAAC,UAAU,CAAC,IAAI,CAAC,OAAO,EAAE,OAAO,CAAC,CAAC;QACvD,IAAI,KAAK,IAAI,SAAS,EAAE,CAAC;YACxB,SAAS,IAAI,KAAK,CAAC;YACnB,SAAS;QACV,CAAC;QAED,MAAM,MAAM,GACX,wCAAwC,uBAAuB,iCAAiC;YAChG,8EAA4E,CAAC;QAC9E,IAAI,CAAC,OAAO;YACX,SAAS,IAAI,sBAAsB,CAAC,CAAC,CAAC,IAAI,CAAC,OAAO,CAAC,KAAK,CAAC,CAAC,EAAE,SAAS,CAAC,GAAG,MAAM,CAAC,CAAC,CAAC,MAAM,CAAC,SAAS,EAAE,CAAC;QACtG,IAAI,CAAC,IAAI,GAAG,WAAW,CAAC;QACxB,IAAI,CAAC,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,MAAM,CAAC,UAAU,CAAC,IAAI,CAAC,OAAO,EAAE,OAAO,CAAC,GAAG,CAAC,CAAC,CAAC;QACvE,SAAS,GAAG,CAAC,CAAC;QACd,QAAQ,CAAC,IAAI,CAAC,WAAW,IAAI,CAAC,IAAI,+CAA6C,CAAC,CAAC;IAClF,CAAC;IAED,OAAO,QAAQ,CAAC;AAAA,CAChB","sourcesContent":["/**\n * Context-file and prompt-input loading for the resource loader.\n *\n * Reads AGENTS.md / CLAUDE.md context files from the user scopes and the cwd\n * ancestor chain (warning/truncating oversized ones, since they are injected\n * into the system prompt every turn), and resolves a system-prompt input that\n * may be either an inline string or a file path. Extracted from resource-loader.ts.\n *\n * Two user scopes are read, least specific first: `~/.agents/AGENTS.md` (the\n * cross-vendor convention, so rules written for one tool are seen by hoocode)\n * and `~/.hoocode/AGENTS.md` (the native home, which therefore wins on\n * conflict). Both are additive — neither shadows the other, and no migration\n * is needed for users who already have the native file.\n */\n\nimport { existsSync, readFileSync } from \"node:fs\";\nimport { join, resolve } from \"node:path\";\nimport chalk from \"chalk\";\nimport { getUserAgentsDir } from \"../config.js\";\n\n/**\n * A loaded context file plus the recurring-cost metadata the UI surfaces.\n * `tokens`/`size` are optional so callers that synthesize context files (SDK\n * overrides, tests) can keep passing plain `{ path, content }`.\n */\nexport interface ContextFile {\n\tpath: string;\n\tcontent: string;\n\t/** Rough token estimate (bytes / 4) of the content as loaded. */\n\ttokens?: number;\n\t/** Set only past the soft limit; \"truncated\" means the hard limit clipped it. */\n\tsize?: \"large\" | \"truncated\";\n}\n\n/**\n * Resolve a prompt input that is either an inline string or a path to a file.\n * If the input names an existing file, its contents are returned; otherwise the\n * input is treated as the prompt text itself.\n */\nexport function resolvePromptInput(input: string | undefined, description: string): string | undefined {\n\tif (!input) {\n\t\treturn undefined;\n\t}\n\n\tif (existsSync(input)) {\n\t\ttry {\n\t\t\treturn readFileSync(input, \"utf-8\");\n\t\t} catch (error) {\n\t\t\tconsole.error(chalk.yellow(`Warning: Could not read ${description} file ${input}: ${error}`));\n\t\t\treturn input;\n\t\t}\n\t}\n\n\treturn input;\n}\n\n// Context files (AGENTS.md / CLAUDE.md) are injected into the system prompt on\n// every turn, so their size has a recurring cost on every provider. Warn the\n// user past a soft limit (~2k tokens) and truncate at a hard limit (~10k tokens)\n// so a pasted spec can't silently bloat every request forever.\nconst CONTEXT_FILE_WARN_BYTES = 8 * 1024;\nconst CONTEXT_FILE_MAX_BYTES = 40 * 1024;\n\n// The per-file limits above cap one file; they say nothing about the total,\n// and the set can now stack two user scopes on top of a whole ancestor chain.\n// So the aggregate gets its own budget, deliberately warn-first: past the soft\n// cap (~6k tokens) the user is told what it costs, and only past the hard cap\n// (~16k tokens) is anything trimmed. Trimming should be the branch that never\n// runs in practice.\nconst CONTEXT_TOTAL_WARN_BYTES = 24 * 1024;\nconst CONTEXT_TOTAL_MAX_BYTES = 64 * 1024;\n// Below this, a trimmed file has no room left to say anything useful, so it is\n// replaced by the notice alone rather than a few words of severed prose.\nconst CONTEXT_TRIM_MIN_BYTES = 512;\n\nfunction loadContextFileFromDir(dir: string): { file: ContextFile | null; warnings: string[] } {\n\tconst warnings: string[] = [];\n\tconst candidates = [\"AGENTS.md\", \"AGENTS.MD\", \"CLAUDE.md\", \"CLAUDE.MD\"];\n\tfor (const filename of candidates) {\n\t\tconst filePath = join(dir, filename);\n\t\tif (existsSync(filePath)) {\n\t\t\ttry {\n\t\t\t\tlet content = readFileSync(filePath, \"utf-8\");\n\t\t\t\tconst bytes = Buffer.byteLength(content, \"utf-8\");\n\t\t\t\t// Size is reported structurally (not as a warning string) so the UI can\n\t\t\t\t// annotate the file where it is already listed instead of repeating it.\n\t\t\t\tlet size: ContextFile[\"size\"];\n\t\t\t\tif (bytes > CONTEXT_FILE_MAX_BYTES) {\n\t\t\t\t\tcontent =\n\t\t\t\t\t\tcontent.slice(0, CONTEXT_FILE_MAX_BYTES) +\n\t\t\t\t\t\t`\\n\\n[truncated: file exceeded ${CONTEXT_FILE_MAX_BYTES} bytes (~10k tokens); keep context files brief — large specs belong in linked files, not in the system prompt]`;\n\t\t\t\t\tsize = \"truncated\";\n\t\t\t\t} else if (bytes > CONTEXT_FILE_WARN_BYTES) {\n\t\t\t\t\tsize = \"large\";\n\t\t\t\t}\n\t\t\t\treturn { file: { path: filePath, content, tokens: Math.round(bytes / 4), size }, warnings };\n\t\t\t} catch (error) {\n\t\t\t\twarnings.push(`Could not read ${filePath}: ${error}`);\n\t\t\t}\n\t\t}\n\t}\n\treturn { file: null, warnings };\n}\n\nexport interface LoadProjectContextFilesOptions {\n\tcwd: string;\n\t/** hoocode's native home, e.g. `~/.hoocode`. */\n\tagentDir: string;\n\t/**\n\t * The cross-vendor user scope, e.g. `~/.agents`. Defaults to `~/.agents`;\n\t * injectable so tests do not read the real home directory.\n\t */\n\tuserAgentsDir?: string;\n}\n\nexport function loadProjectContextFiles(options: LoadProjectContextFilesOptions): {\n\tagentsFiles: ContextFile[];\n\twarnings: string[];\n} {\n\tconst resolvedCwd = options.cwd;\n\tconst resolvedAgentDir = options.agentDir;\n\tconst resolvedUserAgentsDir = options.userAgentsDir ?? getUserAgentsDir();\n\n\tconst contextFiles: ContextFile[] = [];\n\tconst warnings: string[] = [];\n\tconst seenPaths = new Set<string>();\n\n\t// Least specific first: the cross-vendor scope, then the native home. A file\n\t// already seen is never added twice, so pointing both at one directory is a\n\t// no-op rather than a doubled prompt.\n\tfor (const dir of [resolvedUserAgentsDir, resolvedAgentDir]) {\n\t\tconst result = loadContextFileFromDir(dir);\n\t\tif (result.file && !seenPaths.has(result.file.path)) {\n\t\t\tcontextFiles.push(result.file);\n\t\t\tseenPaths.add(result.file.path);\n\t\t}\n\t\twarnings.push(...result.warnings);\n\t}\n\n\tconst ancestorContextFiles: ContextFile[] = [];\n\n\tlet currentDir = resolvedCwd;\n\tconst root = resolve(\"/\");\n\n\twhile (true) {\n\t\tconst result = loadContextFileFromDir(currentDir);\n\t\tif (result.file && !seenPaths.has(result.file.path)) {\n\t\t\tancestorContextFiles.unshift(result.file);\n\t\t\tseenPaths.add(result.file.path);\n\t\t}\n\t\twarnings.push(...result.warnings);\n\n\t\tif (currentDir === root) break;\n\n\t\tconst parentDir = resolve(currentDir, \"..\");\n\t\tif (parentDir === currentDir) break;\n\t\tcurrentDir = parentDir;\n\t}\n\n\tcontextFiles.push(...ancestorContextFiles);\n\n\twarnings.push(...enforceTotalBudget(contextFiles));\n\n\treturn { agentsFiles: contextFiles, warnings };\n}\n\n/**\n * Apply the aggregate budget to an already-ordered context-file list.\n *\n * Mutates entries in place and returns any warnings. The list is ordered least\n * specific first, and specificity is what decides who pays: the walk runs from\n * the end so the file nearest the work keeps its budget, and a user-scope file\n * is trimmed before a repo one. Files are trimmed rather than dropped, so a\n * file that stops fitting still says so in the prompt instead of vanishing.\n */\nfunction enforceTotalBudget(files: ContextFile[]): string[] {\n\tconst warnings: string[] = [];\n\tconst total = files.reduce((sum, file) => sum + Buffer.byteLength(file.content, \"utf-8\"), 0);\n\tif (total <= CONTEXT_TOTAL_WARN_BYTES) {\n\t\treturn warnings;\n\t}\n\n\tif (total <= CONTEXT_TOTAL_MAX_BYTES) {\n\t\t// A single oversized file is already priced by the per-file rule, which\n\t\t// flags it `large` and annotates it where it is listed. Repeating that as\n\t\t// an aggregate warning says nothing new. The aggregate exists to catch\n\t\t// what per-file checks structurally cannot see: the sum across scopes.\n\t\tif (files.length < 2) return warnings;\n\t\twarnings.push(\n\t\t\t`Context files total ~${Math.round(total / 4 / 100) / 10}k tokens across ${files.length} file(s), ` +\n\t\t\t\t`re-sent on every request. Keep rules to one line each, and move long or conditional guidance into a skill ` +\n\t\t\t\t`(loaded on demand) rather than a context file (loaded always).`,\n\t\t);\n\t\treturn warnings;\n\t}\n\n\tlet remaining = CONTEXT_TOTAL_MAX_BYTES;\n\tfor (let i = files.length - 1; i >= 0; i--) {\n\t\tconst file = files[i]!;\n\t\tconst bytes = Buffer.byteLength(file.content, \"utf-8\");\n\t\tif (bytes <= remaining) {\n\t\t\tremaining -= bytes;\n\t\t\tcontinue;\n\t\t}\n\n\t\tconst notice =\n\t\t\t`\\n\\n[trimmed: context files exceeded ${CONTEXT_TOTAL_MAX_BYTES} bytes (~16k tokens) in total; ` +\n\t\t\t`least-specific scopes are trimmed first — move long guidance into a skill]`;\n\t\tfile.content =\n\t\t\tremaining >= CONTEXT_TRIM_MIN_BYTES ? file.content.slice(0, remaining) + notice : notice.trimStart();\n\t\tfile.size = \"truncated\";\n\t\tfile.tokens = Math.round(Buffer.byteLength(file.content, \"utf-8\") / 4);\n\t\tremaining = 0;\n\t\twarnings.push(`Trimmed ${file.path} — context files exceeded the total budget.`);\n\t}\n\n\treturn warnings;\n}\n"]}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Renders the extractor's output into the message `/learn` injects.
|
|
3
|
+
*
|
|
4
|
+
* The digest is evidence plus instructions, and the split matters: the numbers
|
|
5
|
+
* come from {@link extractLearnDigest} and are not negotiable, while everything
|
|
6
|
+
* the model does with them — phrasing, routing, deciding a pattern is not worth
|
|
7
|
+
* a rule — is judgement it has to exercise. Counts are printed on every item
|
|
8
|
+
* because "said in 5 of your last 12 sessions" is a decision the reader can
|
|
9
|
+
* make in one keystroke, where "extracted from your session" is not.
|
|
10
|
+
*/
|
|
11
|
+
import type { LearnDigest } from "./extract.js";
|
|
12
|
+
/** True when there is nothing worth asking the model to look at. */
|
|
13
|
+
export declare function isEmptyDigest(digest: LearnDigest): boolean;
|
|
14
|
+
export declare function renderLearnDigest(digest: LearnDigest, options: {
|
|
15
|
+
userScopePath: string;
|
|
16
|
+
}): string;
|
|
17
|
+
//# sourceMappingURL=digest.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"digest.d.ts","sourceRoot":"","sources":["../../../src/core/learn/digest.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAEH,OAAO,KAAK,EAAE,WAAW,EAAE,MAAM,cAAc,CAAC;AAehD,oEAAoE;AACpE,wBAAgB,aAAa,CAAC,MAAM,EAAE,WAAW,GAAG,OAAO,CAE1D;AAED,wBAAgB,iBAAiB,CAAC,MAAM,EAAE,WAAW,EAAE,OAAO,EAAE;IAAE,aAAa,EAAE,MAAM,CAAA;CAAE,GAAG,MAAM,CA2IjG","sourcesContent":["/**\n * Renders the extractor's output into the message `/learn` injects.\n *\n * The digest is evidence plus instructions, and the split matters: the numbers\n * come from {@link extractLearnDigest} and are not negotiable, while everything\n * the model does with them — phrasing, routing, deciding a pattern is not worth\n * a rule — is judgement it has to exercise. Counts are printed on every item\n * because \"said in 5 of your last 12 sessions\" is a decision the reader can\n * make in one keystroke, where \"extracted from your session\" is not.\n */\n\nimport type { LearnDigest } from \"./extract.js\";\nimport { LEARN_DIGEST_MARKER } from \"./extract.js\";\n\nfunction shortDate(iso: string | undefined): string {\n\tif (!iso) return \"unknown\";\n\tconst date = new Date(iso);\n\treturn Number.isNaN(date.getTime()) ? \"unknown\" : date.toISOString().slice(0, 10);\n}\n\nfunction evidence(count: number, sessions: number, lastSeen: string): string {\n\tconst times = count === 1 ? \"once\" : `${count}x`;\n\tconst where = sessions === 1 ? \"1 session\" : `${sessions} sessions`;\n\treturn `${times} across ${where}, last ${shortDate(lastSeen)}`;\n}\n\n/** True when there is nothing worth asking the model to look at. */\nexport function isEmptyDigest(digest: LearnDigest): boolean {\n\treturn digest.directives.length === 0 && digest.fixes.length === 0 && digest.workflows.length === 0;\n}\n\nexport function renderLearnDigest(digest: LearnDigest, options: { userScopePath: string }): string {\n\tconst lines: string[] = [];\n\n\tlines.push(\n\t\t`${LEARN_DIGEST_MARKER} Mined ${digest.scannedSessions} session(s) in this directory` +\n\t\t\t(digest.skippedSessions > 0 ? ` (${digest.skippedSessions} skipped: out of window or unreadable)` : \"\") +\n\t\t\t(digest.oldestSession ? `, ${shortDate(digest.oldestSession)} to ${shortDate(digest.newestSession)}` : \"\") +\n\t\t\t(digest.suppressed > 0 ? `. ${digest.suppressed} item(s) held back — already shown and unchanged since` : \"\") +\n\t\t\t\".\",\n\t);\n\tlines.push(\"\");\n\tlines.push(\n\t\t\"The counts below are computed from session transcripts on disk, not from this conversation. \" +\n\t\t\t\"Treat them as evidence, not conclusions — your job is to decide what deserves to be written down, \" +\n\t\t\t\"phrase it, and put it in the right place.\",\n\t);\n\tlines.push(\"\");\n\n\t// ── Directives ───────────────────────────────────────────────────────────\n\tif (digest.directives.length > 0) {\n\t\tlines.push(\"## Directives you have repeated\");\n\t\tlines.push(\"\");\n\t\tfor (const cluster of digest.directives) {\n\t\t\tlines.push(`- **${cluster.status}** — \"${cluster.text.replace(/\\s+/g, \" \").trim()}\"`);\n\t\t\tlines.push(` - ${evidence(cluster.count, cluster.sessions, cluster.lastSeen)}`);\n\t\t\tif (cluster.existingRule) {\n\t\t\t\tlines.push(` - already covered by: \"${cluster.existingRule.slice(0, 160)}\"`);\n\t\t\t}\n\t\t\tif (cluster.existingSkill) {\n\t\t\t\tlines.push(` - already covered by the \\`${cluster.existingSkill}\\` skill`);\n\t\t\t}\n\t\t\tif (cluster.previouslyDeclined) {\n\t\t\t\tlines.push(\" - proposed before and not written down — you have already passed on this once\");\n\t\t\t}\n\t\t}\n\t\tlines.push(\"\");\n\t}\n\n\t// ── Fixes ────────────────────────────────────────────────────────────────\n\tif (digest.fixes.length > 0) {\n\t\tlines.push(\"## Failures you resolved\");\n\t\tlines.push(\"\");\n\t\tlines.push(\n\t\t\t\"Each is a command that failed, then later succeeded unchanged after intervening work — \" +\n\t\t\t\t\"so something in between was the fix.\",\n\t\t);\n\t\tlines.push(\"\");\n\t\tfor (const fix of digest.fixes) {\n\t\t\tlines.push(`- \\`${fix.command}\\` — ${evidence(fix.count, fix.sessions, fix.lastSeen)}`);\n\t\t\tlines.push(` - error: ${fix.errorExcerpt}`);\n\t\t\tif (fix.interveningCommands.length > 0) {\n\t\t\t\tlines.push(` - commands in between: ${fix.interveningCommands.map((c) => `\\`${c}\\``).join(\", \")}`);\n\t\t\t}\n\t\t\tif (fix.editedFiles.length > 0) {\n\t\t\t\tlines.push(` - files edited: ${fix.editedFiles.join(\", \")}`);\n\t\t\t}\n\t\t}\n\t\tlines.push(\"\");\n\t}\n\n\t// ── Workflows ────────────────────────────────────────────────────────────\n\tif (digest.workflows.length > 0) {\n\t\tlines.push(\"## Repeated tool sequences\");\n\t\tlines.push(\"\");\n\t\tfor (const workflow of digest.workflows) {\n\t\t\tlines.push(\n\t\t\t\t`- \\`${workflow.steps.join(\" → \")}\\` — ${evidence(workflow.count, workflow.sessions, workflow.lastSeen)}`,\n\t\t\t);\n\t\t}\n\t\tlines.push(\"\");\n\t}\n\n\t// ── Instructions ─────────────────────────────────────────────────────────\n\tlines.push(\"## What to do\");\n\tlines.push(\"\");\n\tlines.push(\"Work through the items above and propose concrete edits. For each one, decide:\");\n\tlines.push(\"\");\n\tlines.push(\n\t\t\"1. **Is it durable?** A rule that will still be true next month belongs somewhere. A one-off preference \" +\n\t\t\t\"about the task you happened to be doing does not. When in doubt, drop it — a wrong rule costs more than \" +\n\t\t\t\"a missing one, because it is paid on every request forever.\",\n\t);\n\tlines.push(\n\t\t\"2. **Rule or skill?** This is the most important call. A context file is loaded on **every** turn; a skill \" +\n\t\t\t\"is loaded **on demand**. So: short, always-true, unconditional → a one-line rule. Long, procedural, \" +\n\t\t\t'or conditional (a sequence of steps, a runbook, anything starting \"when X, do Y\") → a skill, not a rule. ' +\n\t\t\t\"Repeated tool sequences are almost always skills.\",\n\t);\n\tlines.push(\n\t\t`3. **Which scope?** Project-specific (this repo's tests, build, architecture, conventions) → the repo ` +\n\t\t\t`\\`AGENTS.md\\`. Personal habits that travel with you across every repo (style preferences, how you like ` +\n\t\t\t`commits written) → \\`${options.userScopePath}\\`. If it names this repo's files or commands, it is not a ` +\n\t\t\t`user-scope rule.`,\n\t);\n\tlines.push(\n\t\t\"4. **Restated items are rewrites, not additions.** An item marked `restated` is already covered by a rule \" +\n\t\t\t\"that is not working — too vague, buried, or contradicted elsewhere. Rewrite the existing line or delete \" +\n\t\t\t\"it in favour of a sharper one. Do not add a second rule saying the same thing.\",\n\t);\n\tlines.push(\n\t\t\"5. **`has-skill` items are a triggering problem, not a missing rule.** A skill already covers it and you \" +\n\t\t\t\"asked by hand anyway, which usually means the skill's `description` frontmatter does not describe the \" +\n\t\t\t\"situation you were in. Sharpen that description so it matches, rather than adding a rule that duplicates \" +\n\t\t\t\"what the skill already does.\",\n\t);\n\tlines.push(\"\");\n\tlines.push(\"Then, while you have the file open, audit it:\");\n\tlines.push(\"\");\n\tlines.push(\n\t\t\"- **Delete rules that no longer match the code.** Check a sample against the repo before trusting them.\",\n\t);\n\tlines.push(\"- **Delete rules that restate default behaviour.** Guidance the agent already follows is pure cost.\");\n\tlines.push(\n\t\t\"- **Collapse duplicates**, including any rule stated at both repo and user scope — that one is paid twice.\",\n\t);\n\tlines.push(\n\t\t\"- **One line per rule.** No rationale, no examples, no preamble, unless the example *is* the rule. Prose is \" +\n\t\t\t\"the single biggest source of context-file bloat.\",\n\t);\n\tlines.push(\"\");\n\n\tif (digest.agentsFilePath) {\n\t\tlines.push(\n\t\t\t`The repo context file is \\`${digest.agentsFilePath}\\`` +\n\t\t\t\t(digest.agentsFileTokens ? ` (~${digest.agentsFileTokens} tokens, re-sent every request)` : \"\") +\n\t\t\t\t\". Report the token delta of your proposed changes before applying them; a net reduction is a good outcome.\",\n\t\t);\n\t} else {\n\t\tlines.push(\n\t\t\t\"No repo context file exists yet. Create one only if at least one durable project rule survives step 1.\",\n\t\t);\n\t}\n\tlines.push(\"\");\n\tlines.push(\n\t\t\"Show what you propose, then apply it with edits — do not ask a separate approval question first, the edit \" +\n\t\t\t\"prompt is the approval. If nothing here is worth writing down, say so plainly and change nothing.\",\n\t);\n\n\treturn lines.join(\"\\n\");\n}\n"]}
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Renders the extractor's output into the message `/learn` injects.
|
|
3
|
+
*
|
|
4
|
+
* The digest is evidence plus instructions, and the split matters: the numbers
|
|
5
|
+
* come from {@link extractLearnDigest} and are not negotiable, while everything
|
|
6
|
+
* the model does with them — phrasing, routing, deciding a pattern is not worth
|
|
7
|
+
* a rule — is judgement it has to exercise. Counts are printed on every item
|
|
8
|
+
* because "said in 5 of your last 12 sessions" is a decision the reader can
|
|
9
|
+
* make in one keystroke, where "extracted from your session" is not.
|
|
10
|
+
*/
|
|
11
|
+
import { LEARN_DIGEST_MARKER } from "./extract.js";
|
|
12
|
+
function shortDate(iso) {
|
|
13
|
+
if (!iso)
|
|
14
|
+
return "unknown";
|
|
15
|
+
const date = new Date(iso);
|
|
16
|
+
return Number.isNaN(date.getTime()) ? "unknown" : date.toISOString().slice(0, 10);
|
|
17
|
+
}
|
|
18
|
+
function evidence(count, sessions, lastSeen) {
|
|
19
|
+
const times = count === 1 ? "once" : `${count}x`;
|
|
20
|
+
const where = sessions === 1 ? "1 session" : `${sessions} sessions`;
|
|
21
|
+
return `${times} across ${where}, last ${shortDate(lastSeen)}`;
|
|
22
|
+
}
|
|
23
|
+
/** True when there is nothing worth asking the model to look at. */
|
|
24
|
+
export function isEmptyDigest(digest) {
|
|
25
|
+
return digest.directives.length === 0 && digest.fixes.length === 0 && digest.workflows.length === 0;
|
|
26
|
+
}
|
|
27
|
+
export function renderLearnDigest(digest, options) {
|
|
28
|
+
const lines = [];
|
|
29
|
+
lines.push(`${LEARN_DIGEST_MARKER} Mined ${digest.scannedSessions} session(s) in this directory` +
|
|
30
|
+
(digest.skippedSessions > 0 ? ` (${digest.skippedSessions} skipped: out of window or unreadable)` : "") +
|
|
31
|
+
(digest.oldestSession ? `, ${shortDate(digest.oldestSession)} to ${shortDate(digest.newestSession)}` : "") +
|
|
32
|
+
(digest.suppressed > 0 ? `. ${digest.suppressed} item(s) held back — already shown and unchanged since` : "") +
|
|
33
|
+
".");
|
|
34
|
+
lines.push("");
|
|
35
|
+
lines.push("The counts below are computed from session transcripts on disk, not from this conversation. " +
|
|
36
|
+
"Treat them as evidence, not conclusions — your job is to decide what deserves to be written down, " +
|
|
37
|
+
"phrase it, and put it in the right place.");
|
|
38
|
+
lines.push("");
|
|
39
|
+
// ── Directives ───────────────────────────────────────────────────────────
|
|
40
|
+
if (digest.directives.length > 0) {
|
|
41
|
+
lines.push("## Directives you have repeated");
|
|
42
|
+
lines.push("");
|
|
43
|
+
for (const cluster of digest.directives) {
|
|
44
|
+
lines.push(`- **${cluster.status}** — "${cluster.text.replace(/\s+/g, " ").trim()}"`);
|
|
45
|
+
lines.push(` - ${evidence(cluster.count, cluster.sessions, cluster.lastSeen)}`);
|
|
46
|
+
if (cluster.existingRule) {
|
|
47
|
+
lines.push(` - already covered by: "${cluster.existingRule.slice(0, 160)}"`);
|
|
48
|
+
}
|
|
49
|
+
if (cluster.existingSkill) {
|
|
50
|
+
lines.push(` - already covered by the \`${cluster.existingSkill}\` skill`);
|
|
51
|
+
}
|
|
52
|
+
if (cluster.previouslyDeclined) {
|
|
53
|
+
lines.push(" - proposed before and not written down — you have already passed on this once");
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
lines.push("");
|
|
57
|
+
}
|
|
58
|
+
// ── Fixes ────────────────────────────────────────────────────────────────
|
|
59
|
+
if (digest.fixes.length > 0) {
|
|
60
|
+
lines.push("## Failures you resolved");
|
|
61
|
+
lines.push("");
|
|
62
|
+
lines.push("Each is a command that failed, then later succeeded unchanged after intervening work — " +
|
|
63
|
+
"so something in between was the fix.");
|
|
64
|
+
lines.push("");
|
|
65
|
+
for (const fix of digest.fixes) {
|
|
66
|
+
lines.push(`- \`${fix.command}\` — ${evidence(fix.count, fix.sessions, fix.lastSeen)}`);
|
|
67
|
+
lines.push(` - error: ${fix.errorExcerpt}`);
|
|
68
|
+
if (fix.interveningCommands.length > 0) {
|
|
69
|
+
lines.push(` - commands in between: ${fix.interveningCommands.map((c) => `\`${c}\``).join(", ")}`);
|
|
70
|
+
}
|
|
71
|
+
if (fix.editedFiles.length > 0) {
|
|
72
|
+
lines.push(` - files edited: ${fix.editedFiles.join(", ")}`);
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
lines.push("");
|
|
76
|
+
}
|
|
77
|
+
// ── Workflows ────────────────────────────────────────────────────────────
|
|
78
|
+
if (digest.workflows.length > 0) {
|
|
79
|
+
lines.push("## Repeated tool sequences");
|
|
80
|
+
lines.push("");
|
|
81
|
+
for (const workflow of digest.workflows) {
|
|
82
|
+
lines.push(`- \`${workflow.steps.join(" → ")}\` — ${evidence(workflow.count, workflow.sessions, workflow.lastSeen)}`);
|
|
83
|
+
}
|
|
84
|
+
lines.push("");
|
|
85
|
+
}
|
|
86
|
+
// ── Instructions ─────────────────────────────────────────────────────────
|
|
87
|
+
lines.push("## What to do");
|
|
88
|
+
lines.push("");
|
|
89
|
+
lines.push("Work through the items above and propose concrete edits. For each one, decide:");
|
|
90
|
+
lines.push("");
|
|
91
|
+
lines.push("1. **Is it durable?** A rule that will still be true next month belongs somewhere. A one-off preference " +
|
|
92
|
+
"about the task you happened to be doing does not. When in doubt, drop it — a wrong rule costs more than " +
|
|
93
|
+
"a missing one, because it is paid on every request forever.");
|
|
94
|
+
lines.push("2. **Rule or skill?** This is the most important call. A context file is loaded on **every** turn; a skill " +
|
|
95
|
+
"is loaded **on demand**. So: short, always-true, unconditional → a one-line rule. Long, procedural, " +
|
|
96
|
+
'or conditional (a sequence of steps, a runbook, anything starting "when X, do Y") → a skill, not a rule. ' +
|
|
97
|
+
"Repeated tool sequences are almost always skills.");
|
|
98
|
+
lines.push(`3. **Which scope?** Project-specific (this repo's tests, build, architecture, conventions) → the repo ` +
|
|
99
|
+
`\`AGENTS.md\`. Personal habits that travel with you across every repo (style preferences, how you like ` +
|
|
100
|
+
`commits written) → \`${options.userScopePath}\`. If it names this repo's files or commands, it is not a ` +
|
|
101
|
+
`user-scope rule.`);
|
|
102
|
+
lines.push("4. **Restated items are rewrites, not additions.** An item marked `restated` is already covered by a rule " +
|
|
103
|
+
"that is not working — too vague, buried, or contradicted elsewhere. Rewrite the existing line or delete " +
|
|
104
|
+
"it in favour of a sharper one. Do not add a second rule saying the same thing.");
|
|
105
|
+
lines.push("5. **`has-skill` items are a triggering problem, not a missing rule.** A skill already covers it and you " +
|
|
106
|
+
"asked by hand anyway, which usually means the skill's `description` frontmatter does not describe the " +
|
|
107
|
+
"situation you were in. Sharpen that description so it matches, rather than adding a rule that duplicates " +
|
|
108
|
+
"what the skill already does.");
|
|
109
|
+
lines.push("");
|
|
110
|
+
lines.push("Then, while you have the file open, audit it:");
|
|
111
|
+
lines.push("");
|
|
112
|
+
lines.push("- **Delete rules that no longer match the code.** Check a sample against the repo before trusting them.");
|
|
113
|
+
lines.push("- **Delete rules that restate default behaviour.** Guidance the agent already follows is pure cost.");
|
|
114
|
+
lines.push("- **Collapse duplicates**, including any rule stated at both repo and user scope — that one is paid twice.");
|
|
115
|
+
lines.push("- **One line per rule.** No rationale, no examples, no preamble, unless the example *is* the rule. Prose is " +
|
|
116
|
+
"the single biggest source of context-file bloat.");
|
|
117
|
+
lines.push("");
|
|
118
|
+
if (digest.agentsFilePath) {
|
|
119
|
+
lines.push(`The repo context file is \`${digest.agentsFilePath}\`` +
|
|
120
|
+
(digest.agentsFileTokens ? ` (~${digest.agentsFileTokens} tokens, re-sent every request)` : "") +
|
|
121
|
+
". Report the token delta of your proposed changes before applying them; a net reduction is a good outcome.");
|
|
122
|
+
}
|
|
123
|
+
else {
|
|
124
|
+
lines.push("No repo context file exists yet. Create one only if at least one durable project rule survives step 1.");
|
|
125
|
+
}
|
|
126
|
+
lines.push("");
|
|
127
|
+
lines.push("Show what you propose, then apply it with edits — do not ask a separate approval question first, the edit " +
|
|
128
|
+
"prompt is the approval. If nothing here is worth writing down, say so plainly and change nothing.");
|
|
129
|
+
return lines.join("\n");
|
|
130
|
+
}
|
|
131
|
+
//# sourceMappingURL=digest.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"digest.js","sourceRoot":"","sources":["../../../src/core/learn/digest.ts"],"names":[],"mappings":"AAAA;;;;;;;;;GASG;AAGH,OAAO,EAAE,mBAAmB,EAAE,MAAM,cAAc,CAAC;AAEnD,SAAS,SAAS,CAAC,GAAuB,EAAU;IACnD,IAAI,CAAC,GAAG;QAAE,OAAO,SAAS,CAAC;IAC3B,MAAM,IAAI,GAAG,IAAI,IAAI,CAAC,GAAG,CAAC,CAAC;IAC3B,OAAO,MAAM,CAAC,KAAK,CAAC,IAAI,CAAC,OAAO,EAAE,CAAC,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,CAAC,IAAI,CAAC,WAAW,EAAE,CAAC,KAAK,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC;AAAA,CAClF;AAED,SAAS,QAAQ,CAAC,KAAa,EAAE,QAAgB,EAAE,QAAgB,EAAU;IAC5E,MAAM,KAAK,GAAG,KAAK,KAAK,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,GAAG,KAAK,GAAG,CAAC;IACjD,MAAM,KAAK,GAAG,QAAQ,KAAK,CAAC,CAAC,CAAC,CAAC,WAAW,CAAC,CAAC,CAAC,GAAG,QAAQ,WAAW,CAAC;IACpE,OAAO,GAAG,KAAK,WAAW,KAAK,UAAU,SAAS,CAAC,QAAQ,CAAC,EAAE,CAAC;AAAA,CAC/D;AAED,oEAAoE;AACpE,MAAM,UAAU,aAAa,CAAC,MAAmB,EAAW;IAC3D,OAAO,MAAM,CAAC,UAAU,CAAC,MAAM,KAAK,CAAC,IAAI,MAAM,CAAC,KAAK,CAAC,MAAM,KAAK,CAAC,IAAI,MAAM,CAAC,SAAS,CAAC,MAAM,KAAK,CAAC,CAAC;AAAA,CACpG;AAED,MAAM,UAAU,iBAAiB,CAAC,MAAmB,EAAE,OAAkC,EAAU;IAClG,MAAM,KAAK,GAAa,EAAE,CAAC;IAE3B,KAAK,CAAC,IAAI,CACT,GAAG,mBAAmB,UAAU,MAAM,CAAC,eAAe,+BAA+B;QACpF,CAAC,MAAM,CAAC,eAAe,GAAG,CAAC,CAAC,CAAC,CAAC,KAAK,MAAM,CAAC,eAAe,wCAAwC,CAAC,CAAC,CAAC,EAAE,CAAC;QACvG,CAAC,MAAM,CAAC,aAAa,CAAC,CAAC,CAAC,KAAK,SAAS,CAAC,MAAM,CAAC,aAAa,CAAC,OAAO,SAAS,CAAC,MAAM,CAAC,aAAa,CAAC,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC;QAC1G,CAAC,MAAM,CAAC,UAAU,GAAG,CAAC,CAAC,CAAC,CAAC,KAAK,MAAM,CAAC,UAAU,0DAAwD,CAAC,CAAC,CAAC,EAAE,CAAC;QAC7G,GAAG,CACJ,CAAC;IACF,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IACf,KAAK,CAAC,IAAI,CACT,8FAA8F;QAC7F,sGAAoG;QACpG,2CAA2C,CAC5C,CAAC;IACF,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IAEf,sMAA4E;IAC5E,IAAI,MAAM,CAAC,UAAU,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QAClC,KAAK,CAAC,IAAI,CAAC,iCAAiC,CAAC,CAAC;QAC9C,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;QACf,KAAK,MAAM,OAAO,IAAI,MAAM,CAAC,UAAU,EAAE,CAAC;YACzC,KAAK,CAAC,IAAI,CAAC,OAAO,OAAO,CAAC,MAAM,WAAS,OAAO,CAAC,IAAI,CAAC,OAAO,CAAC,MAAM,EAAE,GAAG,CAAC,CAAC,IAAI,EAAE,GAAG,CAAC,CAAC;YACtF,KAAK,CAAC,IAAI,CAAC,OAAO,QAAQ,CAAC,OAAO,CAAC,KAAK,EAAE,OAAO,CAAC,QAAQ,EAAE,OAAO,CAAC,QAAQ,CAAC,EAAE,CAAC,CAAC;YACjF,IAAI,OAAO,CAAC,YAAY,EAAE,CAAC;gBAC1B,KAAK,CAAC,IAAI,CAAC,4BAA4B,OAAO,CAAC,YAAY,CAAC,KAAK,CAAC,CAAC,EAAE,GAAG,CAAC,GAAG,CAAC,CAAC;YAC/E,CAAC;YACD,IAAI,OAAO,CAAC,aAAa,EAAE,CAAC;gBAC3B,KAAK,CAAC,IAAI,CAAC,gCAAgC,OAAO,CAAC,aAAa,UAAU,CAAC,CAAC;YAC7E,CAAC;YACD,IAAI,OAAO,CAAC,kBAAkB,EAAE,CAAC;gBAChC,KAAK,CAAC,IAAI,CAAC,mFAAiF,CAAC,CAAC;YAC/F,CAAC;QACF,CAAC;QACD,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IAChB,CAAC;IAED,gNAA4E;IAC5E,IAAI,MAAM,CAAC,KAAK,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QAC7B,KAAK,CAAC,IAAI,CAAC,0BAA0B,CAAC,CAAC;QACvC,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;QACf,KAAK,CAAC,IAAI,CACT,2FAAyF;YACxF,sCAAsC,CACvC,CAAC;QACF,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;QACf,KAAK,MAAM,GAAG,IAAI,MAAM,CAAC,KAAK,EAAE,CAAC;YAChC,KAAK,CAAC,IAAI,CAAC,OAAO,GAAG,CAAC,OAAO,UAAQ,QAAQ,CAAC,GAAG,CAAC,KAAK,EAAE,GAAG,CAAC,QAAQ,EAAE,GAAG,CAAC,QAAQ,CAAC,EAAE,CAAC,CAAC;YACxF,KAAK,CAAC,IAAI,CAAC,cAAc,GAAG,CAAC,YAAY,EAAE,CAAC,CAAC;YAC7C,IAAI,GAAG,CAAC,mBAAmB,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;gBACxC,KAAK,CAAC,IAAI,CAAC,4BAA4B,GAAG,CAAC,mBAAmB,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,KAAK,CAAC,IAAI,CAAC,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;YACrG,CAAC;YACD,IAAI,GAAG,CAAC,WAAW,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;gBAChC,KAAK,CAAC,IAAI,CAAC,qBAAqB,GAAG,CAAC,WAAW,CAAC,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;YAC/D,CAAC;QACF,CAAC;QACD,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IAChB,CAAC;IAED,wMAA4E;IAC5E,IAAI,MAAM,CAAC,SAAS,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QACjC,KAAK,CAAC,IAAI,CAAC,4BAA4B,CAAC,CAAC;QACzC,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;QACf,KAAK,MAAM,QAAQ,IAAI,MAAM,CAAC,SAAS,EAAE,CAAC;YACzC,KAAK,CAAC,IAAI,CACT,OAAO,QAAQ,CAAC,KAAK,CAAC,IAAI,CAAC,OAAK,CAAC,UAAQ,QAAQ,CAAC,QAAQ,CAAC,KAAK,EAAE,QAAQ,CAAC,QAAQ,EAAE,QAAQ,CAAC,QAAQ,CAAC,EAAE,CACzG,CAAC;QACH,CAAC;QACD,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IAChB,CAAC;IAED,kMAA4E;IAC5E,KAAK,CAAC,IAAI,CAAC,eAAe,CAAC,CAAC;IAC5B,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IACf,KAAK,CAAC,IAAI,CAAC,gFAAgF,CAAC,CAAC;IAC7F,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IACf,KAAK,CAAC,IAAI,CACT,0GAA0G;QACzG,4GAA0G;QAC1G,6DAA6D,CAC9D,CAAC;IACF,KAAK,CAAC,IAAI,CACT,6GAA6G;QAC5G,wGAAsG;QACtG,6GAA2G;QAC3G,mDAAmD,CACpD,CAAC;IACF,KAAK,CAAC,IAAI,CACT,0GAAwG;QACvG,yGAAyG;QACzG,0BAAwB,OAAO,CAAC,aAAa,6DAA6D;QAC1G,kBAAkB,CACnB,CAAC;IACF,KAAK,CAAC,IAAI,CACT,4GAA4G;QAC3G,4GAA0G;QAC1G,gFAAgF,CACjF,CAAC;IACF,KAAK,CAAC,IAAI,CACT,2GAA2G;QAC1G,wGAAwG;QACxG,2GAA2G;QAC3G,8BAA8B,CAC/B,CAAC;IACF,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IACf,KAAK,CAAC,IAAI,CAAC,+CAA+C,CAAC,CAAC;IAC5D,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IACf,KAAK,CAAC,IAAI,CACT,yGAAyG,CACzG,CAAC;IACF,KAAK,CAAC,IAAI,CAAC,qGAAqG,CAAC,CAAC;IAClH,KAAK,CAAC,IAAI,CACT,8GAA4G,CAC5G,CAAC;IACF,KAAK,CAAC,IAAI,CACT,8GAA8G;QAC7G,kDAAkD,CACnD,CAAC;IACF,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IAEf,IAAI,MAAM,CAAC,cAAc,EAAE,CAAC;QAC3B,KAAK,CAAC,IAAI,CACT,8BAA8B,MAAM,CAAC,cAAc,IAAI;YACtD,CAAC,MAAM,CAAC,gBAAgB,CAAC,CAAC,CAAC,MAAM,MAAM,CAAC,gBAAgB,iCAAiC,CAAC,CAAC,CAAC,EAAE,CAAC;YAC/F,4GAA4G,CAC7G,CAAC;IACH,CAAC;SAAM,CAAC;QACP,KAAK,CAAC,IAAI,CACT,wGAAwG,CACxG,CAAC;IACH,CAAC;IACD,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC,CAAC;IACf,KAAK,CAAC,IAAI,CACT,8GAA4G;QAC3G,mGAAmG,CACpG,CAAC;IAEF,OAAO,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC;AAAA,CACxB","sourcesContent":["/**\n * Renders the extractor's output into the message `/learn` injects.\n *\n * The digest is evidence plus instructions, and the split matters: the numbers\n * come from {@link extractLearnDigest} and are not negotiable, while everything\n * the model does with them — phrasing, routing, deciding a pattern is not worth\n * a rule — is judgement it has to exercise. Counts are printed on every item\n * because \"said in 5 of your last 12 sessions\" is a decision the reader can\n * make in one keystroke, where \"extracted from your session\" is not.\n */\n\nimport type { LearnDigest } from \"./extract.js\";\nimport { LEARN_DIGEST_MARKER } from \"./extract.js\";\n\nfunction shortDate(iso: string | undefined): string {\n\tif (!iso) return \"unknown\";\n\tconst date = new Date(iso);\n\treturn Number.isNaN(date.getTime()) ? \"unknown\" : date.toISOString().slice(0, 10);\n}\n\nfunction evidence(count: number, sessions: number, lastSeen: string): string {\n\tconst times = count === 1 ? \"once\" : `${count}x`;\n\tconst where = sessions === 1 ? \"1 session\" : `${sessions} sessions`;\n\treturn `${times} across ${where}, last ${shortDate(lastSeen)}`;\n}\n\n/** True when there is nothing worth asking the model to look at. */\nexport function isEmptyDigest(digest: LearnDigest): boolean {\n\treturn digest.directives.length === 0 && digest.fixes.length === 0 && digest.workflows.length === 0;\n}\n\nexport function renderLearnDigest(digest: LearnDigest, options: { userScopePath: string }): string {\n\tconst lines: string[] = [];\n\n\tlines.push(\n\t\t`${LEARN_DIGEST_MARKER} Mined ${digest.scannedSessions} session(s) in this directory` +\n\t\t\t(digest.skippedSessions > 0 ? ` (${digest.skippedSessions} skipped: out of window or unreadable)` : \"\") +\n\t\t\t(digest.oldestSession ? `, ${shortDate(digest.oldestSession)} to ${shortDate(digest.newestSession)}` : \"\") +\n\t\t\t(digest.suppressed > 0 ? `. ${digest.suppressed} item(s) held back — already shown and unchanged since` : \"\") +\n\t\t\t\".\",\n\t);\n\tlines.push(\"\");\n\tlines.push(\n\t\t\"The counts below are computed from session transcripts on disk, not from this conversation. \" +\n\t\t\t\"Treat them as evidence, not conclusions — your job is to decide what deserves to be written down, \" +\n\t\t\t\"phrase it, and put it in the right place.\",\n\t);\n\tlines.push(\"\");\n\n\t// ── Directives ───────────────────────────────────────────────────────────\n\tif (digest.directives.length > 0) {\n\t\tlines.push(\"## Directives you have repeated\");\n\t\tlines.push(\"\");\n\t\tfor (const cluster of digest.directives) {\n\t\t\tlines.push(`- **${cluster.status}** — \"${cluster.text.replace(/\\s+/g, \" \").trim()}\"`);\n\t\t\tlines.push(` - ${evidence(cluster.count, cluster.sessions, cluster.lastSeen)}`);\n\t\t\tif (cluster.existingRule) {\n\t\t\t\tlines.push(` - already covered by: \"${cluster.existingRule.slice(0, 160)}\"`);\n\t\t\t}\n\t\t\tif (cluster.existingSkill) {\n\t\t\t\tlines.push(` - already covered by the \\`${cluster.existingSkill}\\` skill`);\n\t\t\t}\n\t\t\tif (cluster.previouslyDeclined) {\n\t\t\t\tlines.push(\" - proposed before and not written down — you have already passed on this once\");\n\t\t\t}\n\t\t}\n\t\tlines.push(\"\");\n\t}\n\n\t// ── Fixes ────────────────────────────────────────────────────────────────\n\tif (digest.fixes.length > 0) {\n\t\tlines.push(\"## Failures you resolved\");\n\t\tlines.push(\"\");\n\t\tlines.push(\n\t\t\t\"Each is a command that failed, then later succeeded unchanged after intervening work — \" +\n\t\t\t\t\"so something in between was the fix.\",\n\t\t);\n\t\tlines.push(\"\");\n\t\tfor (const fix of digest.fixes) {\n\t\t\tlines.push(`- \\`${fix.command}\\` — ${evidence(fix.count, fix.sessions, fix.lastSeen)}`);\n\t\t\tlines.push(` - error: ${fix.errorExcerpt}`);\n\t\t\tif (fix.interveningCommands.length > 0) {\n\t\t\t\tlines.push(` - commands in between: ${fix.interveningCommands.map((c) => `\\`${c}\\``).join(\", \")}`);\n\t\t\t}\n\t\t\tif (fix.editedFiles.length > 0) {\n\t\t\t\tlines.push(` - files edited: ${fix.editedFiles.join(\", \")}`);\n\t\t\t}\n\t\t}\n\t\tlines.push(\"\");\n\t}\n\n\t// ── Workflows ────────────────────────────────────────────────────────────\n\tif (digest.workflows.length > 0) {\n\t\tlines.push(\"## Repeated tool sequences\");\n\t\tlines.push(\"\");\n\t\tfor (const workflow of digest.workflows) {\n\t\t\tlines.push(\n\t\t\t\t`- \\`${workflow.steps.join(\" → \")}\\` — ${evidence(workflow.count, workflow.sessions, workflow.lastSeen)}`,\n\t\t\t);\n\t\t}\n\t\tlines.push(\"\");\n\t}\n\n\t// ── Instructions ─────────────────────────────────────────────────────────\n\tlines.push(\"## What to do\");\n\tlines.push(\"\");\n\tlines.push(\"Work through the items above and propose concrete edits. For each one, decide:\");\n\tlines.push(\"\");\n\tlines.push(\n\t\t\"1. **Is it durable?** A rule that will still be true next month belongs somewhere. A one-off preference \" +\n\t\t\t\"about the task you happened to be doing does not. When in doubt, drop it — a wrong rule costs more than \" +\n\t\t\t\"a missing one, because it is paid on every request forever.\",\n\t);\n\tlines.push(\n\t\t\"2. **Rule or skill?** This is the most important call. A context file is loaded on **every** turn; a skill \" +\n\t\t\t\"is loaded **on demand**. So: short, always-true, unconditional → a one-line rule. Long, procedural, \" +\n\t\t\t'or conditional (a sequence of steps, a runbook, anything starting \"when X, do Y\") → a skill, not a rule. ' +\n\t\t\t\"Repeated tool sequences are almost always skills.\",\n\t);\n\tlines.push(\n\t\t`3. **Which scope?** Project-specific (this repo's tests, build, architecture, conventions) → the repo ` +\n\t\t\t`\\`AGENTS.md\\`. Personal habits that travel with you across every repo (style preferences, how you like ` +\n\t\t\t`commits written) → \\`${options.userScopePath}\\`. If it names this repo's files or commands, it is not a ` +\n\t\t\t`user-scope rule.`,\n\t);\n\tlines.push(\n\t\t\"4. **Restated items are rewrites, not additions.** An item marked `restated` is already covered by a rule \" +\n\t\t\t\"that is not working — too vague, buried, or contradicted elsewhere. Rewrite the existing line or delete \" +\n\t\t\t\"it in favour of a sharper one. Do not add a second rule saying the same thing.\",\n\t);\n\tlines.push(\n\t\t\"5. **`has-skill` items are a triggering problem, not a missing rule.** A skill already covers it and you \" +\n\t\t\t\"asked by hand anyway, which usually means the skill's `description` frontmatter does not describe the \" +\n\t\t\t\"situation you were in. Sharpen that description so it matches, rather than adding a rule that duplicates \" +\n\t\t\t\"what the skill already does.\",\n\t);\n\tlines.push(\"\");\n\tlines.push(\"Then, while you have the file open, audit it:\");\n\tlines.push(\"\");\n\tlines.push(\n\t\t\"- **Delete rules that no longer match the code.** Check a sample against the repo before trusting them.\",\n\t);\n\tlines.push(\"- **Delete rules that restate default behaviour.** Guidance the agent already follows is pure cost.\");\n\tlines.push(\n\t\t\"- **Collapse duplicates**, including any rule stated at both repo and user scope — that one is paid twice.\",\n\t);\n\tlines.push(\n\t\t\"- **One line per rule.** No rationale, no examples, no preamble, unless the example *is* the rule. Prose is \" +\n\t\t\t\"the single biggest source of context-file bloat.\",\n\t);\n\tlines.push(\"\");\n\n\tif (digest.agentsFilePath) {\n\t\tlines.push(\n\t\t\t`The repo context file is \\`${digest.agentsFilePath}\\`` +\n\t\t\t\t(digest.agentsFileTokens ? ` (~${digest.agentsFileTokens} tokens, re-sent every request)` : \"\") +\n\t\t\t\t\". Report the token delta of your proposed changes before applying them; a net reduction is a good outcome.\",\n\t\t);\n\t} else {\n\t\tlines.push(\n\t\t\t\"No repo context file exists yet. Create one only if at least one durable project rule survives step 1.\",\n\t\t);\n\t}\n\tlines.push(\"\");\n\tlines.push(\n\t\t\"Show what you propose, then apply it with edits — do not ask a separate approval question first, the edit \" +\n\t\t\t\"prompt is the approval. If nothing here is worth writing down, say so plainly and change nothing.\",\n\t);\n\n\treturn lines.join(\"\\n\");\n}\n"]}
|
|
@@ -0,0 +1,178 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Session mining for `/learn`.
|
|
3
|
+
*
|
|
4
|
+
* Reads session `.jsonl` files straight off disk rather than the live context.
|
|
5
|
+
* That is the whole point: the on-disk transcript is complete even when the
|
|
6
|
+
* in-context one has been compacted away, and it spans every past session
|
|
7
|
+
* instead of only this one. Cross-session repetition is the signal that decides
|
|
8
|
+
* whether something is a durable rule or a one-off, and it is the one thing a
|
|
9
|
+
* prompt reading its own context cannot see.
|
|
10
|
+
*
|
|
11
|
+
* The split of labour is deliberate. This module is entirely deterministic: it
|
|
12
|
+
* parses, filters, normalizes, counts and ranks. Judgement — is this a rule, how
|
|
13
|
+
* should it be phrased, which scope owns it — belongs to the model reading the
|
|
14
|
+
* digest, which is why the output carries evidence (counts, sessions, dates)
|
|
15
|
+
* rather than conclusions.
|
|
16
|
+
*/
|
|
17
|
+
import { type LearnState } from "./state.js";
|
|
18
|
+
/**
|
|
19
|
+
* Prefix on the message `/learn` injects. The digest is persisted like any user
|
|
20
|
+
* turn, so without this marker the next `/learn` would mine its own output and
|
|
21
|
+
* every proposal would compound its own count.
|
|
22
|
+
*/
|
|
23
|
+
export declare const LEARN_DIGEST_MARKER = "[learn-digest]";
|
|
24
|
+
/**
|
|
25
|
+
* Where a repeated directive already lives, if anywhere.
|
|
26
|
+
*
|
|
27
|
+
* A directive covered by a rule and said only once is simply dropped — the rule
|
|
28
|
+
* exists and is working. What survives is one of three cases, and they want
|
|
29
|
+
* different responses:
|
|
30
|
+
*
|
|
31
|
+
* - `new` — not written down anywhere. Propose it.
|
|
32
|
+
* - `restated` — a context-file rule covers it and you said it anyway, so the
|
|
33
|
+
* rule is not working. Rewrite it; do not add a second one.
|
|
34
|
+
* - `has-skill` — a *skill* covers it and you asked by hand anyway, which
|
|
35
|
+
* usually means the skill's `description` is not triggering. Sharpen the
|
|
36
|
+
* description rather than writing a rule that duplicates the skill.
|
|
37
|
+
*/
|
|
38
|
+
export type DirectiveStatus = "new" | "restated" | "has-skill";
|
|
39
|
+
/** Fields every proposable item shares, so suppression can be applied uniformly. */
|
|
40
|
+
interface Proposable {
|
|
41
|
+
/** Stable identity across runs — what the state file remembers. */
|
|
42
|
+
key: string;
|
|
43
|
+
/** Newest occurrence in the window, ISO. */
|
|
44
|
+
lastSeen: string;
|
|
45
|
+
}
|
|
46
|
+
export interface DirectiveCluster extends Proposable {
|
|
47
|
+
/** Representative raw text, the longest seen in the cluster. */
|
|
48
|
+
text: string;
|
|
49
|
+
normalized: string;
|
|
50
|
+
/** Total times said. */
|
|
51
|
+
count: number;
|
|
52
|
+
/** Distinct sessions it was said in — the stronger of the two counts. */
|
|
53
|
+
sessions: number;
|
|
54
|
+
status: DirectiveStatus;
|
|
55
|
+
/** The existing rule line matched, when status is `restated`. */
|
|
56
|
+
existingRule?: string;
|
|
57
|
+
/** The skill that already covers this, when status is `has-skill`. */
|
|
58
|
+
existingSkill?: string;
|
|
59
|
+
/**
|
|
60
|
+
* Shown before and still not written down anywhere — neither as a rule nor as
|
|
61
|
+
* a skill — so you saw this proposal and passed on it. Only meaningful for
|
|
62
|
+
* directives, which are the only items with a real coverage signal.
|
|
63
|
+
*/
|
|
64
|
+
previouslyDeclined: boolean;
|
|
65
|
+
}
|
|
66
|
+
export interface FixCandidate extends Proposable {
|
|
67
|
+
/** Normalized failing command. */
|
|
68
|
+
command: string;
|
|
69
|
+
/** Normalized error signature, the dedupe key. */
|
|
70
|
+
signature: string;
|
|
71
|
+
/** Short raw excerpt, so the model sees the real error text. */
|
|
72
|
+
errorExcerpt: string;
|
|
73
|
+
/** Commands run between the failure and the pass. */
|
|
74
|
+
interveningCommands: string[];
|
|
75
|
+
/** Files edited between the failure and the pass. */
|
|
76
|
+
editedFiles: string[];
|
|
77
|
+
/** Times this signature failed and was resolved across the window. */
|
|
78
|
+
count: number;
|
|
79
|
+
sessions: number;
|
|
80
|
+
}
|
|
81
|
+
export interface WorkflowCandidate extends Proposable {
|
|
82
|
+
/** Tool-call signatures in order. */
|
|
83
|
+
steps: string[];
|
|
84
|
+
count: number;
|
|
85
|
+
sessions: number;
|
|
86
|
+
}
|
|
87
|
+
export interface LearnDigest {
|
|
88
|
+
scannedSessions: number;
|
|
89
|
+
skippedSessions: number;
|
|
90
|
+
oldestSession?: string;
|
|
91
|
+
newestSession?: string;
|
|
92
|
+
agentsFilePath?: string;
|
|
93
|
+
agentsFileTokens?: number;
|
|
94
|
+
directives: DirectiveCluster[];
|
|
95
|
+
fixes: FixCandidate[];
|
|
96
|
+
workflows: WorkflowCandidate[];
|
|
97
|
+
/** Items held back because nothing new has happened since they were last shown. */
|
|
98
|
+
suppressed: number;
|
|
99
|
+
/** Everything this run put on screen, for the caller to persist. */
|
|
100
|
+
surfaced: Array<{
|
|
101
|
+
key: string;
|
|
102
|
+
lastSeen: string;
|
|
103
|
+
covered: boolean;
|
|
104
|
+
}>;
|
|
105
|
+
}
|
|
106
|
+
export interface ExtractOptions {
|
|
107
|
+
cwd: string;
|
|
108
|
+
agentDir: string;
|
|
109
|
+
/** Override the directory scanned. Defaults to the per-cwd session dir. */
|
|
110
|
+
sessionDir?: string;
|
|
111
|
+
maxSessions?: number;
|
|
112
|
+
maxAgeDays?: number;
|
|
113
|
+
/** Occurrences a directive needs before it is proposed. The signal/noise dial. */
|
|
114
|
+
minRepeats?: number;
|
|
115
|
+
/** Non-overlapping repeats a tool sequence needs before it is proposed as a skill. */
|
|
116
|
+
minWorkflowRepeats?: number;
|
|
117
|
+
/** Cap on each list in the digest. */
|
|
118
|
+
maxProposals?: number;
|
|
119
|
+
/**
|
|
120
|
+
* What previous runs already showed. Items with no new occurrences since are
|
|
121
|
+
* held back. Omit (or pass `ignoreState`) to propose everything in the window.
|
|
122
|
+
*/
|
|
123
|
+
state?: LearnState;
|
|
124
|
+
/** Re-propose everything, ignoring what previous runs surfaced (`/learn all`). */
|
|
125
|
+
ignoreState?: boolean;
|
|
126
|
+
/**
|
|
127
|
+
* Skills a directive can already be covered by. Defaults to the ones loaded
|
|
128
|
+
* from disk; injectable so tests do not read the developer's real skills.
|
|
129
|
+
*/
|
|
130
|
+
skills?: Array<{
|
|
131
|
+
name: string;
|
|
132
|
+
description: string;
|
|
133
|
+
}>;
|
|
134
|
+
/** Injectable clock, for tests. */
|
|
135
|
+
now?: Date;
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* Everything a proposal could already have been written into.
|
|
139
|
+
*
|
|
140
|
+
* Built once and shared, because the same question — is this already written
|
|
141
|
+
* down? — is asked while ranking a run *and* afterwards by `/learn stats`,
|
|
142
|
+
* which reconstructs adoption by comparing coverage now against coverage when
|
|
143
|
+
* the item was shown.
|
|
144
|
+
*/
|
|
145
|
+
export interface CoverageIndex {
|
|
146
|
+
/** Candidate rule lines from the repo context file and both user scopes. */
|
|
147
|
+
ruleLines: string[];
|
|
148
|
+
skills: Array<{
|
|
149
|
+
name: string;
|
|
150
|
+
description: string;
|
|
151
|
+
}>;
|
|
152
|
+
}
|
|
153
|
+
export interface CoverageMatch {
|
|
154
|
+
/** The context-file line that covers this, if any. */
|
|
155
|
+
rule?: string;
|
|
156
|
+
/** The skill that covers this, if any. Only set when no rule matched. */
|
|
157
|
+
skill?: string;
|
|
158
|
+
}
|
|
159
|
+
/**
|
|
160
|
+
* Where a piece of text is already written down, if anywhere.
|
|
161
|
+
*
|
|
162
|
+
* A rule wins over a skill when both match: it is the more specific answer, and
|
|
163
|
+
* "rewrite this line" is more actionable than "sharpen a description".
|
|
164
|
+
*/
|
|
165
|
+
export declare function matchCoverage(text: string, index: CoverageIndex): CoverageMatch;
|
|
166
|
+
/** Assemble the coverage index for a directory. */
|
|
167
|
+
export declare function buildCoverageIndex(options: {
|
|
168
|
+
cwd: string;
|
|
169
|
+
agentDir: string;
|
|
170
|
+
skills?: Array<{
|
|
171
|
+
name: string;
|
|
172
|
+
description: string;
|
|
173
|
+
}>;
|
|
174
|
+
}): CoverageIndex;
|
|
175
|
+
/** Mine the recent sessions for this cwd and return the ranked digest. */
|
|
176
|
+
export declare function extractLearnDigest(options: ExtractOptions): LearnDigest;
|
|
177
|
+
export {};
|
|
178
|
+
//# sourceMappingURL=extract.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"extract.d.ts","sourceRoot":"","sources":["../../../src/core/learn/extract.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;GAeG;AAqBH,OAAO,EAAS,KAAK,UAAU,EAAE,MAAM,YAAY,CAAC;AAEpD;;;;GAIG;AACH,eAAO,MAAM,mBAAmB,mBAAmB,CAAC;AAoCpD;;;;;;;;;;;;;GAaG;AACH,MAAM,MAAM,eAAe,GAAG,KAAK,GAAG,UAAU,GAAG,WAAW,CAAC;AAE/D,oFAAoF;AACpF,UAAU,UAAU;IACnB,qEAAmE;IACnE,GAAG,EAAE,MAAM,CAAC;IACZ,4CAA4C;IAC5C,QAAQ,EAAE,MAAM,CAAC;CACjB;AAED,MAAM,WAAW,gBAAiB,SAAQ,UAAU;IACnD,gEAAgE;IAChE,IAAI,EAAE,MAAM,CAAC;IACb,UAAU,EAAE,MAAM,CAAC;IACnB,wBAAwB;IACxB,KAAK,EAAE,MAAM,CAAC;IACd,2EAAyE;IACzE,QAAQ,EAAE,MAAM,CAAC;IACjB,MAAM,EAAE,eAAe,CAAC;IACxB,iEAAiE;IACjE,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,sEAAsE;IACtE,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB;;;;OAIG;IACH,kBAAkB,EAAE,OAAO,CAAC;CAC5B;AAED,MAAM,WAAW,YAAa,SAAQ,UAAU;IAC/C,kCAAkC;IAClC,OAAO,EAAE,MAAM,CAAC;IAChB,kDAAkD;IAClD,SAAS,EAAE,MAAM,CAAC;IAClB,gEAAgE;IAChE,YAAY,EAAE,MAAM,CAAC;IACrB,qDAAqD;IACrD,mBAAmB,EAAE,MAAM,EAAE,CAAC;IAC9B,qDAAqD;IACrD,WAAW,EAAE,MAAM,EAAE,CAAC;IACtB,sEAAsE;IACtE,KAAK,EAAE,MAAM,CAAC;IACd,QAAQ,EAAE,MAAM,CAAC;CACjB;AAED,MAAM,WAAW,iBAAkB,SAAQ,UAAU;IACpD,qCAAqC;IACrC,KAAK,EAAE,MAAM,EAAE,CAAC;IAChB,KAAK,EAAE,MAAM,CAAC;IACd,QAAQ,EAAE,MAAM,CAAC;CACjB;AAED,MAAM,WAAW,WAAW;IAC3B,eAAe,EAAE,MAAM,CAAC;IACxB,eAAe,EAAE,MAAM,CAAC;IACxB,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,gBAAgB,CAAC,EAAE,MAAM,CAAC;IAC1B,UAAU,EAAE,gBAAgB,EAAE,CAAC;IAC/B,KAAK,EAAE,YAAY,EAAE,CAAC;IACtB,SAAS,EAAE,iBAAiB,EAAE,CAAC;IAC/B,mFAAmF;IACnF,UAAU,EAAE,MAAM,CAAC;IACnB,oEAAoE;IACpE,QAAQ,EAAE,KAAK,CAAC;QAAE,GAAG,EAAE,MAAM,CAAC;QAAC,QAAQ,EAAE,MAAM,CAAC;QAAC,OAAO,EAAE,OAAO,CAAA;KAAE,CAAC,CAAC;CACrE;AAED,MAAM,WAAW,cAAc;IAC9B,GAAG,EAAE,MAAM,CAAC;IACZ,QAAQ,EAAE,MAAM,CAAC;IACjB,2EAA2E;IAC3E,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,kFAAkF;IAClF,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,sFAAsF;IACtF,kBAAkB,CAAC,EAAE,MAAM,CAAC;IAC5B,sCAAsC;IACtC,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB;;;OAGG;IACH,KAAK,CAAC,EAAE,UAAU,CAAC;IACnB,kFAAkF;IAClF,WAAW,CAAC,EAAE,OAAO,CAAC;IACtB;;;OAGG;IACH,MAAM,CAAC,EAAE,KAAK,CAAC;QAAE,IAAI,EAAE,MAAM,CAAC;QAAC,WAAW,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC;IACtD,mCAAmC;IACnC,GAAG,CAAC,EAAE,IAAI,CAAC;CACX;AA2jBD;;;;;;;GAOG;AACH,MAAM,WAAW,aAAa;IAC7B,4EAA4E;IAC5E,SAAS,EAAE,MAAM,EAAE,CAAC;IACpB,MAAM,EAAE,KAAK,CAAC;QAAE,IAAI,EAAE,MAAM,CAAC;QAAC,WAAW,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC;CACrD;AAED,MAAM,WAAW,aAAa;IAC7B,sDAAsD;IACtD,IAAI,CAAC,EAAE,MAAM,CAAC;IACd,yEAAyE;IACzE,KAAK,CAAC,EAAE,MAAM,CAAC;CACf;AAED;;;;;GAKG;AACH,wBAAgB,aAAa,CAAC,IAAI,EAAE,MAAM,EAAE,KAAK,EAAE,aAAa,GAAG,aAAa,CA2B/E;AAED,mDAAmD;AACnD,wBAAgB,kBAAkB,CAAC,OAAO,EAAE;IAC3C,GAAG,EAAE,MAAM,CAAC;IACZ,QAAQ,EAAE,MAAM,CAAC;IACjB,MAAM,CAAC,EAAE,KAAK,CAAC;QAAE,IAAI,EAAE,MAAM,CAAC;QAAC,WAAW,EAAE,MAAM,CAAA;KAAE,CAAC,CAAC;CACtD,GAAG,aAAa,CAShB;AA4CD,0EAA0E;AAC1E,wBAAgB,kBAAkB,CAAC,OAAO,EAAE,cAAc,GAAG,WAAW,CA+DvE","sourcesContent":["/**\n * Session mining for `/learn`.\n *\n * Reads session `.jsonl` files straight off disk rather than the live context.\n * That is the whole point: the on-disk transcript is complete even when the\n * in-context one has been compacted away, and it spans every past session\n * instead of only this one. Cross-session repetition is the signal that decides\n * whether something is a durable rule or a one-off, and it is the one thing a\n * prompt reading its own context cannot see.\n *\n * The split of labour is deliberate. This module is entirely deterministic: it\n * parses, filters, normalizes, counts and ranks. Judgement — is this a rule, how\n * should it be phrased, which scope owns it — belongs to the model reading the\n * digest, which is why the output carries evidence (counts, sessions, dates)\n * rather than conclusions.\n */\n\nimport { existsSync, readdirSync, readFileSync, statSync } from \"node:fs\";\nimport { dirname, join, resolve } from \"node:path\";\nimport type { AgentMessage } from \"@kolisachint/hoocode-agent-core\";\nimport type { TextContent, ToolCall } from \"@kolisachint/hoocode-ai\";\nimport { getUserAgentsDir } from \"../../config.js\";\nimport { getDefaultSessionDir } from \"../session-manager.js\";\nimport { loadSkills } from \"../skills.js\";\nimport {\n\tcommandHead,\n\tcontentWords,\n\textractErrorRegion,\n\tisBenignFailure,\n\tisRuleShapedDirective,\n\tisUninformativeFailure,\n\tnormalizeCommand,\n\tnormalizeDirective,\n\tnormalizeErrorSignature,\n\twordOverlap,\n} from \"./normalize.js\";\nimport { judge, type LearnState } from \"./state.js\";\n\n/**\n * Prefix on the message `/learn` injects. The digest is persisted like any user\n * turn, so without this marker the next `/learn` would mine its own output and\n * every proposal would compound its own count.\n */\nexport const LEARN_DIGEST_MARKER = \"[learn-digest]\";\n\n/** Sessions considered, newest first. */\nconst DEFAULT_MAX_SESSIONS = 20;\n/** Sessions older than this are ignored — a pattern that stopped is not a rule. */\nconst DEFAULT_MAX_AGE_DAYS = 30;\n/** Entries parsed per session file, as a guard against pathological transcripts. */\nconst MAX_ENTRIES_PER_SESSION = 8000;\n/** Tool calls per session fed to the workflow detector. */\nconst MAX_TOOL_CALLS_PER_SESSION = 400;\n/** How far forward the fix extractor looks for the same command succeeding. */\nconst FIX_LOOKAHEAD = 40;\n/** Word overlap against an existing rule above which a directive counts as covered. */\nconst COVERED_OVERLAP = 0.6;\n/**\n * The same bar for skills, set higher on purpose.\n *\n * A rule is one line, so overlap against it is a sharp signal. A skill is a name\n * plus a description written to attract matches, which is a far larger haystack\n * — a short directive's words turn up in it by chance much more readily. The\n * higher bar and the truncation below keep \"you already have a skill for this\"\n * from being said on a coincidence.\n */\nconst SKILL_COVERED_OVERLAP = 0.75;\n/** Description characters considered. The opening says what a skill does; the rest is trigger bait. */\nconst SKILL_DESCRIPTION_CHARS = 300;\n/** Directives must reach this many occurrences to be reported at all. */\nconst DEFAULT_MIN_DIRECTIVE_COUNT = 2;\n/** Tool sequence lengths considered as workflow candidates. */\nconst WORKFLOW_MIN_LEN = 3;\nconst WORKFLOW_MAX_LEN = 5;\n/** Repeats before a tool sequence is worth proposing as a skill. */\nconst DEFAULT_MIN_WORKFLOW_COUNT = 3;\n/** Cap on each list in the digest, so the model's budget goes to the top signals. */\nconst DEFAULT_MAX_PER_CATEGORY = 8;\n\n/**\n * Where a repeated directive already lives, if anywhere.\n *\n * A directive covered by a rule and said only once is simply dropped — the rule\n * exists and is working. What survives is one of three cases, and they want\n * different responses:\n *\n * - `new` — not written down anywhere. Propose it.\n * - `restated` — a context-file rule covers it and you said it anyway, so the\n * rule is not working. Rewrite it; do not add a second one.\n * - `has-skill` — a *skill* covers it and you asked by hand anyway, which\n * usually means the skill's `description` is not triggering. Sharpen the\n * description rather than writing a rule that duplicates the skill.\n */\nexport type DirectiveStatus = \"new\" | \"restated\" | \"has-skill\";\n\n/** Fields every proposable item shares, so suppression can be applied uniformly. */\ninterface Proposable {\n\t/** Stable identity across runs — what the state file remembers. */\n\tkey: string;\n\t/** Newest occurrence in the window, ISO. */\n\tlastSeen: string;\n}\n\nexport interface DirectiveCluster extends Proposable {\n\t/** Representative raw text, the longest seen in the cluster. */\n\ttext: string;\n\tnormalized: string;\n\t/** Total times said. */\n\tcount: number;\n\t/** Distinct sessions it was said in — the stronger of the two counts. */\n\tsessions: number;\n\tstatus: DirectiveStatus;\n\t/** The existing rule line matched, when status is `restated`. */\n\texistingRule?: string;\n\t/** The skill that already covers this, when status is `has-skill`. */\n\texistingSkill?: string;\n\t/**\n\t * Shown before and still not written down anywhere — neither as a rule nor as\n\t * a skill — so you saw this proposal and passed on it. Only meaningful for\n\t * directives, which are the only items with a real coverage signal.\n\t */\n\tpreviouslyDeclined: boolean;\n}\n\nexport interface FixCandidate extends Proposable {\n\t/** Normalized failing command. */\n\tcommand: string;\n\t/** Normalized error signature, the dedupe key. */\n\tsignature: string;\n\t/** Short raw excerpt, so the model sees the real error text. */\n\terrorExcerpt: string;\n\t/** Commands run between the failure and the pass. */\n\tinterveningCommands: string[];\n\t/** Files edited between the failure and the pass. */\n\teditedFiles: string[];\n\t/** Times this signature failed and was resolved across the window. */\n\tcount: number;\n\tsessions: number;\n}\n\nexport interface WorkflowCandidate extends Proposable {\n\t/** Tool-call signatures in order. */\n\tsteps: string[];\n\tcount: number;\n\tsessions: number;\n}\n\nexport interface LearnDigest {\n\tscannedSessions: number;\n\tskippedSessions: number;\n\toldestSession?: string;\n\tnewestSession?: string;\n\tagentsFilePath?: string;\n\tagentsFileTokens?: number;\n\tdirectives: DirectiveCluster[];\n\tfixes: FixCandidate[];\n\tworkflows: WorkflowCandidate[];\n\t/** Items held back because nothing new has happened since they were last shown. */\n\tsuppressed: number;\n\t/** Everything this run put on screen, for the caller to persist. */\n\tsurfaced: Array<{ key: string; lastSeen: string; covered: boolean }>;\n}\n\nexport interface ExtractOptions {\n\tcwd: string;\n\tagentDir: string;\n\t/** Override the directory scanned. Defaults to the per-cwd session dir. */\n\tsessionDir?: string;\n\tmaxSessions?: number;\n\tmaxAgeDays?: number;\n\t/** Occurrences a directive needs before it is proposed. The signal/noise dial. */\n\tminRepeats?: number;\n\t/** Non-overlapping repeats a tool sequence needs before it is proposed as a skill. */\n\tminWorkflowRepeats?: number;\n\t/** Cap on each list in the digest. */\n\tmaxProposals?: number;\n\t/**\n\t * What previous runs already showed. Items with no new occurrences since are\n\t * held back. Omit (or pass `ignoreState`) to propose everything in the window.\n\t */\n\tstate?: LearnState;\n\t/** Re-propose everything, ignoring what previous runs surfaced (`/learn all`). */\n\tignoreState?: boolean;\n\t/**\n\t * Skills a directive can already be covered by. Defaults to the ones loaded\n\t * from disk; injectable so tests do not read the developer's real skills.\n\t */\n\tskills?: Array<{ name: string; description: string }>;\n\t/** Injectable clock, for tests. */\n\tnow?: Date;\n}\n\ninterface SessionHeaderLike {\n\ttype: \"session\";\n\tid?: string;\n\ttimestamp?: string;\n\tcwd?: string;\n}\n\ninterface EntryLike {\n\ttype: string;\n\tid?: string;\n\tparentId?: string | null;\n\ttimestamp?: string;\n\tmessage?: AgentMessage;\n}\n\n/** One session, reduced to the branch that was actually taken. */\ninterface ParsedSession {\n\tfile: string;\n\tid: string;\n\ttimestamp: string;\n\tentries: EntryLike[];\n}\n\nfunction textOf(content: unknown): string {\n\tif (typeof content === \"string\") return content;\n\tif (!Array.isArray(content)) return \"\";\n\treturn content\n\t\t.map((block) =>\n\t\t\tblock && typeof block === \"object\" && (block as TextContent).type === \"text\"\n\t\t\t\t? ((block as TextContent).text ?? \"\")\n\t\t\t\t: \"\",\n\t\t)\n\t\t.join(\"\\n\")\n\t\t.trim();\n}\n\nfunction isToolCall(block: unknown): block is ToolCall {\n\treturn !!block && typeof block === \"object\" && (block as ToolCall).type === \"toolCall\";\n}\n\n/**\n * Reduce a session's raw entries to the branch that was actually taken.\n *\n * Session files are trees — forks and clones append entries that were never\n * part of the same conversation. Walking parent links back from the last entry\n * keeps the extractor from stitching a \"fix\" out of two turns that never\n * happened in sequence. Sessions written before entry ids existed are flat, and\n * for those file order *is* the branch.\n */\nfunction activeBranch(entries: EntryLike[]): EntryLike[] {\n\tconst withIds = entries.filter((e) => typeof e.id === \"string\");\n\tif (withIds.length === 0) return entries;\n\n\tconst byId = new Map<string, EntryLike>();\n\tfor (const entry of withIds) byId.set(entry.id as string, entry);\n\n\tconst branch: EntryLike[] = [];\n\tconst seen = new Set<string>();\n\tlet cursor: EntryLike | undefined = withIds[withIds.length - 1];\n\twhile (cursor?.id && !seen.has(cursor.id)) {\n\t\tseen.add(cursor.id);\n\t\tbranch.push(cursor);\n\t\tcursor = cursor.parentId ? byId.get(cursor.parentId) : undefined;\n\t}\n\treturn branch.reverse();\n}\n\nfunction parseSessionFile(file: string, cwd: string): ParsedSession | undefined {\n\tlet raw: string;\n\ttry {\n\t\traw = readFileSync(file, \"utf-8\");\n\t} catch {\n\t\treturn undefined;\n\t}\n\n\tconst lines = raw.split(\"\\n\");\n\tlet header: SessionHeaderLike | undefined;\n\tconst entries: EntryLike[] = [];\n\tfor (const line of lines) {\n\t\tif (!line.trim()) continue;\n\t\tif (entries.length >= MAX_ENTRIES_PER_SESSION) break;\n\t\tlet parsed: EntryLike | SessionHeaderLike;\n\t\ttry {\n\t\t\tparsed = JSON.parse(line);\n\t\t} catch {\n\t\t\t// A partially-flushed final line is normal for a live session.\n\t\t\tcontinue;\n\t\t}\n\t\tif (parsed.type === \"session\") {\n\t\t\theader ??= parsed as SessionHeaderLike;\n\t\t\tcontinue;\n\t\t}\n\t\tentries.push(parsed as EntryLike);\n\t}\n\n\t// An explicit `--session` path can put a session for another directory in\n\t// this directory, so trust the header over the file's location.\n\tif (header?.cwd && resolve(header.cwd) !== resolve(cwd)) return undefined;\n\tif (entries.length === 0) return undefined;\n\n\treturn {\n\t\tfile,\n\t\tid: header?.id ?? file,\n\t\ttimestamp: header?.timestamp ?? statSync(file).mtime.toISOString(),\n\t\tentries: activeBranch(entries),\n\t};\n}\n\nfunction listSessions(options: ExtractOptions): { sessions: ParsedSession[]; skipped: number } {\n\tconst dir = options.sessionDir ?? getDefaultSessionDir(options.cwd, options.agentDir);\n\tif (!existsSync(dir)) return { sessions: [], skipped: 0 };\n\n\tconst maxSessions = options.maxSessions ?? DEFAULT_MAX_SESSIONS;\n\tconst maxAgeDays = options.maxAgeDays ?? DEFAULT_MAX_AGE_DAYS;\n\tconst now = options.now ?? new Date();\n\tconst cutoff = now.getTime() - maxAgeDays * 24 * 60 * 60 * 1000;\n\n\tlet files: string[];\n\ttry {\n\t\tfiles = readdirSync(dir)\n\t\t\t.filter((f) => f.endsWith(\".jsonl\"))\n\t\t\t.map((f) => join(dir, f));\n\t} catch {\n\t\treturn { sessions: [], skipped: 0 };\n\t}\n\n\tconst dated = files\n\t\t.map((file) => {\n\t\t\ttry {\n\t\t\t\treturn { file, mtime: statSync(file).mtime.getTime() };\n\t\t\t} catch {\n\t\t\t\treturn undefined;\n\t\t\t}\n\t\t})\n\t\t.filter((f): f is { file: string; mtime: number } => !!f)\n\t\t.sort((a, b) => b.mtime - a.mtime);\n\n\tconst sessions: ParsedSession[] = [];\n\tlet skipped = 0;\n\tfor (const { file, mtime } of dated) {\n\t\tif (sessions.length >= maxSessions) {\n\t\t\tskipped++;\n\t\t\tcontinue;\n\t\t}\n\t\tif (mtime < cutoff) {\n\t\t\tskipped++;\n\t\t\tcontinue;\n\t\t}\n\t\tconst parsed = parseSessionFile(file, options.cwd);\n\t\tif (parsed) sessions.push(parsed);\n\t\telse skipped++;\n\t}\n\treturn { sessions, skipped };\n}\n\n/**\n * Hold back items already shown that have not recurred since, then cap the rest.\n *\n * Order matters: suppression runs *before* the cap, or an item you already\n * decided on would occupy one of the few slots the digest has and push a live\n * signal off the list.\n */\nfunction applySuppression<T extends Proposable>(\n\titems: T[],\n\tstate: LearnState | undefined,\n\tmaxProposals: number,\n\tcovered: (item: T) => boolean,\n\tonDeclined?: (item: T) => void,\n): { kept: T[]; suppressed: number } {\n\tif (!state) return { kept: items.slice(0, maxProposals), suppressed: 0 };\n\n\tconst kept: T[] = [];\n\tlet suppressed = 0;\n\tfor (const item of items) {\n\t\tconst verdict = judge(state, { key: item.key, lastSeen: item.lastSeen, covered: covered(item) });\n\t\tif (verdict.suppressed) {\n\t\t\tsuppressed++;\n\t\t\tcontinue;\n\t\t}\n\t\tif (verdict.previouslyDeclined) onDeclined?.(item);\n\t\tkept.push(item);\n\t}\n\treturn { kept: kept.slice(0, maxProposals), suppressed };\n}\n\n/** Nearest AGENTS.md walking up from cwd, so proposals can be checked against it. */\nfunction findAgentsFile(cwd: string): string | undefined {\n\tlet dir = resolve(cwd);\n\twhile (true) {\n\t\tfor (const name of [\"AGENTS.md\", \"AGENTS.MD\", \"CLAUDE.md\", \"CLAUDE.MD\"]) {\n\t\t\tconst candidate = join(dir, name);\n\t\t\tif (existsSync(candidate)) return candidate;\n\t\t}\n\t\tconst parent = dirname(dir);\n\t\tif (parent === dir) return undefined;\n\t\tdir = parent;\n\t}\n}\n\ninterface ToolEvent {\n\tname: string;\n\targs: Record<string, any>;\n\t/** Set once the matching result is seen. */\n\tisError?: boolean;\n\toutput?: string;\n}\n\n/** Pair tool calls with their results along one branch, in call order. */\nfunction toolEvents(entries: EntryLike[]): ToolEvent[] {\n\tconst byCallId = new Map<string, ToolEvent>();\n\tconst ordered: ToolEvent[] = [];\n\n\tfor (const entry of entries) {\n\t\tconst message = entry.type === \"message\" ? entry.message : undefined;\n\t\tif (!message) continue;\n\t\tif (message.role === \"assistant\") {\n\t\t\tfor (const block of (message.content ?? []) as unknown[]) {\n\t\t\t\tif (!isToolCall(block)) continue;\n\t\t\t\tconst event: ToolEvent = { name: block.name, args: block.arguments ?? {} };\n\t\t\t\tbyCallId.set(block.id, event);\n\t\t\t\tordered.push(event);\n\t\t\t}\n\t\t} else if (message.role === \"toolResult\") {\n\t\t\tconst event = byCallId.get(message.toolCallId);\n\t\t\tif (!event) continue;\n\t\t\tevent.isError = message.isError;\n\t\t\tevent.output = textOf(message.content);\n\t\t}\n\t}\n\treturn ordered;\n}\n\n/** User turns worth mining, in order, with the digest's own output excluded. */\nfunction userDirectives(entries: EntryLike[]): string[] {\n\tconst out: string[] = [];\n\tfor (const entry of entries) {\n\t\tconst message = entry.type === \"message\" ? entry.message : undefined;\n\t\tif (!message || message.role !== \"user\") continue;\n\t\tconst text = textOf(message.content);\n\t\tif (!text || text.startsWith(LEARN_DIGEST_MARKER)) continue;\n\t\tif (!isRuleShapedDirective(text)) continue;\n\t\tout.push(text.trim());\n\t}\n\treturn out;\n}\n\nfunction clusterDirectives(\n\tperSession: Array<{ session: ParsedSession; directives: string[] }>,\n\tcoverage: CoverageIndex,\n\tminRepeats: number,\n): DirectiveCluster[] {\n\tinterface Acc {\n\t\ttext: string;\n\t\tnormalized: string;\n\t\tcount: number;\n\t\tsessions: Set<string>;\n\t\tlastSeen: string;\n\t}\n\tconst acc = new Map<string, Acc>();\n\n\tfor (const { session, directives } of perSession) {\n\t\tfor (const text of directives) {\n\t\t\tconst normalized = normalizeDirective(text);\n\t\t\tif (!normalized) continue;\n\t\t\tconst existing = acc.get(normalized);\n\t\t\tif (existing) {\n\t\t\t\texisting.count++;\n\t\t\t\texisting.sessions.add(session.id);\n\t\t\t\tif (session.timestamp > existing.lastSeen) existing.lastSeen = session.timestamp;\n\t\t\t\tif (text.length > existing.text.length) existing.text = text;\n\t\t\t} else {\n\t\t\t\tacc.set(normalized, {\n\t\t\t\t\ttext,\n\t\t\t\t\tnormalized,\n\t\t\t\t\tcount: 1,\n\t\t\t\t\tsessions: new Set([session.id]),\n\t\t\t\t\tlastSeen: session.timestamp,\n\t\t\t\t});\n\t\t\t}\n\t\t}\n\t}\n\n\tconst clusters: DirectiveCluster[] = [];\n\tfor (const entry of acc.values()) {\n\t\tif (entry.count < minRepeats) continue;\n\n\t\t// Everything reaching here cleared the repeat threshold. Suppression handles\n\t\t// the case that used to make these labels lie — a proposal accepted from a\n\t\t// previous run coming back as \"not working\" when nothing had happened\n\t\t// since. By the time an item survives that filter, a match genuinely means\n\t\t// you repeated yourself after the rule or skill already existed.\n\t\tconst match = matchCoverage(entry.text, coverage);\n\t\tclusters.push({\n\t\t\tkey: `directive:${entry.normalized}`,\n\t\t\ttext: entry.text,\n\t\t\tnormalized: entry.normalized,\n\t\t\tcount: entry.count,\n\t\t\tsessions: entry.sessions.size,\n\t\t\tlastSeen: entry.lastSeen,\n\t\t\tstatus: match.rule ? \"restated\" : match.skill ? \"has-skill\" : \"new\",\n\t\t\texistingRule: match.rule,\n\t\t\texistingSkill: match.skill,\n\t\t\tpreviouslyDeclined: false,\n\t\t});\n\t}\n\n\treturn clusters.sort((a, b) => b.sessions - a.sessions || b.count - a.count || a.text.localeCompare(b.text));\n}\n\n/** Files a mutating tool touched, for the resolution summary. */\nfunction editedFile(event: ToolEvent): string | undefined {\n\tif (![\"edit\", \"write\", \"multi_edit\", \"apply_patch\"].includes(event.name)) return undefined;\n\tconst path = event.args?.path ?? event.args?.file_path ?? event.args?.filePath;\n\treturn typeof path === \"string\" ? path : undefined;\n}\n\nfunction extractFixes(perSession: Array<{ session: ParsedSession; events: ToolEvent[] }>): FixCandidate[] {\n\tinterface Acc {\n\t\tcandidate: FixCandidate;\n\t\tsessions: Set<string>;\n\t}\n\tconst acc = new Map<string, Acc>();\n\n\tfor (const { session, events } of perSession) {\n\t\tfor (let i = 0; i < events.length; i++) {\n\t\t\tconst failure = events[i]!;\n\t\t\tif (failure.name !== \"bash\" || !failure.isError) continue;\n\t\t\tconst command = typeof failure.args?.command === \"string\" ? failure.args.command : \"\";\n\t\t\tif (!command || isBenignFailure(command)) continue;\n\n\t\t\tconst normalized = normalizeCommand(command);\n\t\t\tconst interveningCommands: string[] = [];\n\t\t\tconst editedFiles: string[] = [];\n\t\t\tlet resolved = false;\n\n\t\t\tfor (let j = i + 1; j < Math.min(events.length, i + 1 + FIX_LOOKAHEAD); j++) {\n\t\t\t\tconst next = events[j]!;\n\t\t\t\tconst file = editedFile(next);\n\t\t\t\tif (file) editedFiles.push(file);\n\n\t\t\t\tif (next.name !== \"bash\") continue;\n\t\t\t\tconst nextCommand = typeof next.args?.command === \"string\" ? next.args.command : \"\";\n\t\t\t\tif (!nextCommand) continue;\n\n\t\t\t\t// The same command later succeeding is the only evidence that the\n\t\t\t\t// problem was actually fixed. A *different* command passing says\n\t\t\t\t// nothing, and neither does the model moving on.\n\t\t\t\tif (normalizeCommand(nextCommand) === normalized && !next.isError) {\n\t\t\t\t\tresolved = true;\n\t\t\t\t\tbreak;\n\t\t\t\t}\n\t\t\t\tinterveningCommands.push(nextCommand.trim());\n\t\t\t}\n\n\t\t\tif (!resolved) continue;\n\n\t\t\tconst output = failure.output ?? \"\";\n\t\t\t// An abort is the user changing their mind, not a problem that was\n\t\t\t// solved, and empty output carries nothing to sign or show.\n\t\t\tif (isUninformativeFailure(output)) continue;\n\n\t\t\t// Sign the error region, not the whole output: build tools lead with an\n\t\t\t// identical banner, so signing everything makes unrelated failures of\n\t\t\t// the same command collide on their shared preamble.\n\t\t\tconst errorRegion = extractErrorRegion(output);\n\t\t\tconst signature = normalizeErrorSignature(errorRegion);\n\t\t\tif (!signature) continue;\n\n\t\t\tconst key = `${normalized}\u0000${signature}`;\n\t\t\tconst existing = acc.get(key);\n\t\t\tif (existing) {\n\t\t\t\texisting.candidate.count++;\n\t\t\t\texisting.sessions.add(session.id);\n\t\t\t\tif (session.timestamp > existing.candidate.lastSeen) existing.candidate.lastSeen = session.timestamp;\n\t\t\t} else {\n\t\t\t\tacc.set(key, {\n\t\t\t\t\tsessions: new Set([session.id]),\n\t\t\t\t\tcandidate: {\n\t\t\t\t\t\tkey: `fix:${key}`,\n\t\t\t\t\t\tcommand: normalized,\n\t\t\t\t\t\tsignature,\n\t\t\t\t\t\terrorExcerpt: errorRegion.replace(/\\s+/g, \" \").trim().slice(0, 240),\n\t\t\t\t\t\tinterveningCommands: [...new Set(interveningCommands)].slice(0, 5),\n\t\t\t\t\t\teditedFiles: [...new Set(editedFiles)].slice(0, 5),\n\t\t\t\t\t\tcount: 1,\n\t\t\t\t\t\tsessions: 1,\n\t\t\t\t\t\tlastSeen: session.timestamp,\n\t\t\t\t\t},\n\t\t\t\t});\n\t\t\t}\n\t\t}\n\t}\n\n\tconst out: FixCandidate[] = [];\n\tfor (const { candidate, sessions } of acc.values()) {\n\t\tcandidate.sessions = sessions.size;\n\t\tout.push(candidate);\n\t}\n\treturn out.sort((a, b) => b.count - a.count || b.sessions - a.sessions || a.signature.localeCompare(b.signature));\n}\n\n/** A tool call reduced to a comparable step: the tool, plus what a bash call runs. */\nfunction stepSignature(event: ToolEvent): string {\n\tif (event.name === \"bash\") {\n\t\tconst command = typeof event.args?.command === \"string\" ? event.args.command : \"\";\n\t\tconst head = commandHead(command);\n\t\treturn head ? `bash:${head}` : \"bash\";\n\t}\n\treturn event.name;\n}\n\n/**\n * Commands that are how an agent looks around rather than what the user was\n * doing. A sequence built only from these plus file edits describes \"coding\",\n * not a workflow, and no useful skill has ever come out of one.\n */\nconst PLUMBING_COMMANDS = new Set([\n\t\"cd\",\n\t\"ls\",\n\t\"pwd\",\n\t\"cat\",\n\t\"head\",\n\t\"tail\",\n\t\"wc\",\n\t\"echo\",\n\t\"which\",\n\t\"find\",\n\t\"fd\",\n\t\"grep\",\n\t\"rg\",\n\t\"sed\",\n\t\"awk\",\n\t\"git status\",\n\t\"git diff\",\n\t\"git log\",\n\t\"git show\",\n]);\n\n/**\n * Whether a sequence is a procedure rather than the rhythm of editing code.\n *\n * Two distinct doing-commands is the bar, and it was set by looking at real\n * transcripts. One command is not enough: the edit/test loop\n * (`edit → edit → bash:npm run`) satisfies it, and because a sliding window\n * over a long alternating run produces every rotation of that cycle, it alone\n * filled all eight slots with `edit → npm run → edit`, `npm run → edit → edit`\n * and so on — one habit described eight ways.\n *\n * A procedure worth a skill chains *different* actions: test then commit then\n * push, build then tag then publish. Requiring two distinct ones keeps those and\n * drops the rhythm. The cost is real — a genuine one-command routine with setup\n * is missed — and that is the intended trade, since a missed skill costs nothing\n * while a digest full of noise costs the reader's attention every run.\n */\nfunction isProcedure(steps: string[]): boolean {\n\tconst commands = new Set<string>();\n\tfor (const step of steps) {\n\t\tif (!step.startsWith(\"bash:\")) continue;\n\t\tconst head = step.slice(\"bash:\".length);\n\t\tif (PLUMBING_COMMANDS.has(head) || PLUMBING_COMMANDS.has(head.split(\" \")[0] ?? \"\")) continue;\n\t\tcommands.add(head);\n\t}\n\treturn commands.size >= 2;\n}\n\n/** True when `needle` appears as a contiguous run inside `haystack`. */\nfunction containsSequence(haystack: string[], needle: string[]): boolean {\n\tif (needle.length > haystack.length) return false;\n\tfor (let i = 0; i + needle.length <= haystack.length; i++) {\n\t\tif (needle.every((step, offset) => haystack[i + offset] === step)) return true;\n\t}\n\treturn false;\n}\n\nfunction extractWorkflows(\n\tperSession: Array<{ session: ParsedSession; events: ToolEvent[] }>,\n\tminRepeats: number,\n): WorkflowCandidate[] {\n\tinterface Acc {\n\t\tsteps: string[];\n\t\tcount: number;\n\t\tsessions: Set<string>;\n\t\tlastSeen: string;\n\t}\n\tconst acc = new Map<string, Acc>();\n\n\tfor (const { session, events } of perSession) {\n\t\tconst steps = events.slice(0, MAX_TOOL_CALLS_PER_SESSION).map(stepSignature);\n\n\t\tfor (let len = WORKFLOW_MIN_LEN; len <= WORKFLOW_MAX_LEN; len++) {\n\t\t\t// Collect every position first, then count greedily without overlap.\n\t\t\t// Counting each sliding position separately treats one long stretch of\n\t\t\t// edit/read churn as dozens of repeats: an `edit > read > edit` run of\n\t\t\t// length 12 scores 10 occurrences when it is really one stretch of work.\n\t\t\tconst positions = new Map<string, number[]>();\n\t\t\tfor (let i = 0; i + len <= steps.length; i++) {\n\t\t\t\tconst window = steps.slice(i, i + len);\n\t\t\t\t// A run of one repeated tool is a loop, not a workflow.\n\t\t\t\tif (new Set(window).size < 2) continue;\n\t\t\t\tif (!isProcedure(window)) continue;\n\t\t\t\tconst key = window.join(\" > \");\n\t\t\t\tconst list = positions.get(key);\n\t\t\t\tif (list) list.push(i);\n\t\t\t\telse positions.set(key, [i]);\n\t\t\t}\n\n\t\t\tfor (const [key, occurrences] of positions) {\n\t\t\t\tlet count = 0;\n\t\t\t\tlet nextFree = -1;\n\t\t\t\tfor (const start of occurrences) {\n\t\t\t\t\tif (start < nextFree) continue;\n\t\t\t\t\tcount++;\n\t\t\t\t\tnextFree = start + len;\n\t\t\t\t}\n\n\t\t\t\tconst existing = acc.get(key);\n\t\t\t\tif (existing) {\n\t\t\t\t\texisting.count += count;\n\t\t\t\t\texisting.sessions.add(session.id);\n\t\t\t\t\tif (session.timestamp > existing.lastSeen) existing.lastSeen = session.timestamp;\n\t\t\t\t} else {\n\t\t\t\t\tacc.set(key, {\n\t\t\t\t\t\tsteps: key.split(\" > \"),\n\t\t\t\t\t\tcount,\n\t\t\t\t\t\tsessions: new Set([session.id]),\n\t\t\t\t\t\tlastSeen: session.timestamp,\n\t\t\t\t\t});\n\t\t\t\t}\n\t\t\t}\n\t\t}\n\t}\n\n\tconst ranked = [...acc.values()]\n\t\t.filter((entry) => entry.count >= minRepeats)\n\t\t.map((entry) => ({\n\t\t\tkey: `workflow:${entry.steps.join(\" > \")}`,\n\t\t\tsteps: entry.steps,\n\t\t\tcount: entry.count,\n\t\t\tsessions: entry.sessions.size,\n\t\t\tlastSeen: entry.lastSeen,\n\t\t}))\n\t\t// Sessions first, matching directives: a sequence seen in three sessions is\n\t\t// a workflow, while one repeated ten times in a single session is usually\n\t\t// just the shape of that one task.\n\t\t.sort(\n\t\t\t(a, b) =>\n\t\t\t\tb.sessions - a.sessions ||\n\t\t\t\tb.count - a.count ||\n\t\t\t\tb.steps.length - a.steps.length ||\n\t\t\t\ta.steps.join().localeCompare(b.steps.join()),\n\t\t);\n\n\t// Every n-gram overlaps its own extensions and prefixes, so without this the\n\t// list is one workflow described five slightly different ways. The test runs\n\t// both directions on purpose: a shorter sequence always outranks the longer\n\t// one containing it (it occurs at least as often), so checking only\n\t// shorter-inside-kept would never fire. Keep the best-ranked member of each\n\t// family and drop the rest.\n\tconst distinct: typeof ranked = [];\n\tfor (const candidate of ranked) {\n\t\tconst overlapsKept = distinct.some(\n\t\t\t(kept) => containsSequence(kept.steps, candidate.steps) || containsSequence(candidate.steps, kept.steps),\n\t\t);\n\t\tif (overlapsKept) continue;\n\t\tdistinct.push(candidate);\n\t}\n\treturn distinct;\n}\n\n/**\n * Everything a proposal could already have been written into.\n *\n * Built once and shared, because the same question — is this already written\n * down? — is asked while ranking a run *and* afterwards by `/learn stats`,\n * which reconstructs adoption by comparing coverage now against coverage when\n * the item was shown.\n */\nexport interface CoverageIndex {\n\t/** Candidate rule lines from the repo context file and both user scopes. */\n\truleLines: string[];\n\tskills: Array<{ name: string; description: string }>;\n}\n\nexport interface CoverageMatch {\n\t/** The context-file line that covers this, if any. */\n\trule?: string;\n\t/** The skill that covers this, if any. Only set when no rule matched. */\n\tskill?: string;\n}\n\n/**\n * Where a piece of text is already written down, if anywhere.\n *\n * A rule wins over a skill when both match: it is the more specific answer, and\n * \"rewrite this line\" is more actionable than \"sharpen a description\".\n */\nexport function matchCoverage(text: string, index: CoverageIndex): CoverageMatch {\n\tconst words = contentWords(text);\n\n\tlet bestLine: string | undefined;\n\tlet bestOverlap = 0;\n\tfor (const line of index.ruleLines) {\n\t\tconst overlap = wordOverlap(words, line);\n\t\tif (overlap > bestOverlap) {\n\t\t\tbestOverlap = overlap;\n\t\t\tbestLine = line;\n\t\t}\n\t}\n\tif (bestOverlap >= COVERED_OVERLAP) return { rule: bestLine };\n\n\tlet bestSkill: string | undefined;\n\tlet bestSkillOverlap = 0;\n\tfor (const skill of index.skills) {\n\t\tconst haystack = `${skill.name} ${skill.description.slice(0, SKILL_DESCRIPTION_CHARS)}`;\n\t\tconst overlap = wordOverlap(words, haystack);\n\t\tif (overlap > bestSkillOverlap) {\n\t\t\tbestSkillOverlap = overlap;\n\t\t\tbestSkill = skill.name;\n\t\t}\n\t}\n\tif (bestSkillOverlap >= SKILL_COVERED_OVERLAP) return { skill: bestSkill };\n\n\treturn {};\n}\n\n/** Assemble the coverage index for a directory. */\nexport function buildCoverageIndex(options: {\n\tcwd: string;\n\tagentDir: string;\n\tskills?: Array<{ name: string; description: string }>;\n}): CoverageIndex {\n\tconst corpus = coverageCorpus(options.agentDir, findAgentsFile(options.cwd));\n\treturn {\n\t\truleLines: corpus\n\t\t\t.split(\"\\n\")\n\t\t\t.map((line) => line.trim())\n\t\t\t.filter((line) => line.length > 0 && !line.startsWith(\"#\")),\n\t\tskills: options.skills ?? loadSkillIndex(options.cwd, options.agentDir),\n\t};\n}\n\n/**\n * Text a proposal is checked against to decide whether it is already written\n * down — the nearest repo context file plus both user scopes.\n *\n * All three matter for suppression, because `/learn` can route a rule to the\n * user scope. Checking only the repo file would report a rule you accepted into\n * `~/.agents/AGENTS.md` as declined.\n */\nfunction coverageCorpus(agentDir: string, repoFile: string | undefined): string {\n\tconst parts: string[] = [];\n\tfor (const candidate of [repoFile, join(getUserAgentsDir(), \"AGENTS.md\"), join(agentDir, \"AGENTS.md\")]) {\n\t\tif (!candidate || !existsSync(candidate)) continue;\n\t\ttry {\n\t\t\tparts.push(readFileSync(candidate, \"utf-8\"));\n\t\t} catch {\n\t\t\t// Unreadable context file: treat as absent rather than failing the run.\n\t\t}\n\t}\n\treturn parts.join(\"\\n\");\n}\n\n/**\n * Skills a proposal could already have become.\n *\n * `/learn` routes long or conditional guidance to a skill rather than a rule, so\n * without this a proposal you adopted *as a skill* would read as declined —\n * looking only at context files sees an unchanged `AGENTS.md` and concludes you\n * passed. Reuses the real loader rather than a second SKILL.md scanner so the\n * set of locations cannot drift from what the session actually loads.\n */\nfunction loadSkillIndex(cwd: string, agentDir: string): Array<{ name: string; description: string }> {\n\ttry {\n\t\treturn loadSkills({ cwd, agentDir, skillPaths: [], includeDefaults: true }).skills.map((skill) => ({\n\t\t\tname: skill.name,\n\t\t\tdescription: skill.description ?? \"\",\n\t\t}));\n\t} catch {\n\t\t// Skills are an enrichment here, not the point of the command.\n\t\treturn [];\n\t}\n}\n\n/** Mine the recent sessions for this cwd and return the ranked digest. */\nexport function extractLearnDigest(options: ExtractOptions): LearnDigest {\n\tconst { sessions, skipped } = listSessions(options);\n\n\tconst agentsFilePath = findAgentsFile(options.cwd);\n\tlet agentsContent: string | undefined;\n\tif (agentsFilePath) {\n\t\ttry {\n\t\t\tagentsContent = readFileSync(agentsFilePath, \"utf-8\");\n\t\t} catch {\n\t\t\tagentsContent = undefined;\n\t\t}\n\t}\n\tconst coverage = buildCoverageIndex({ cwd: options.cwd, agentDir: options.agentDir, skills: options.skills });\n\n\tconst withDirectives = sessions.map((session) => ({ session, directives: userDirectives(session.entries) }));\n\tconst withEvents = sessions.map((session) => ({ session, events: toolEvents(session.entries) }));\n\n\tconst timestamps = sessions.map((s) => s.timestamp).sort();\n\tconst state = options.ignoreState ? undefined : options.state;\n\n\t// Directives carry a real coverage signal — is this written down as a rule or\n\t// a skill right now? — which is what separates an adopted proposal from a\n\t// declined one. Fixes and workflows do not: a fix may have become a rule, a\n\t// skill, or a habit, and which one is not recoverable here, so they get\n\t// suppression only and are never labelled declined.\n\tconst maxProposals = options.maxProposals ?? DEFAULT_MAX_PER_CATEGORY;\n\tconst directives = applySuppression(\n\t\tclusterDirectives(withDirectives, coverage, options.minRepeats ?? DEFAULT_MIN_DIRECTIVE_COUNT),\n\t\tstate,\n\t\tmaxProposals,\n\t\t(item) => item.status !== \"new\",\n\t\t(item) => {\n\t\t\titem.previouslyDeclined = true;\n\t\t},\n\t);\n\tconst fixes = applySuppression(extractFixes(withEvents), state, maxProposals, () => false);\n\tconst workflows = applySuppression(\n\t\textractWorkflows(withEvents, options.minWorkflowRepeats ?? DEFAULT_MIN_WORKFLOW_COUNT),\n\t\tstate,\n\t\tmaxProposals,\n\t\t() => false,\n\t);\n\n\tconst surfaced = [\n\t\t...directives.kept.map((d) => ({ key: d.key, lastSeen: d.lastSeen, covered: d.status !== \"new\" })),\n\t\t...fixes.kept.map((f) => ({ key: f.key, lastSeen: f.lastSeen, covered: false })),\n\t\t...workflows.kept.map((w) => ({ key: w.key, lastSeen: w.lastSeen, covered: false })),\n\t];\n\n\treturn {\n\t\tscannedSessions: sessions.length,\n\t\tskippedSessions: skipped,\n\t\toldestSession: timestamps[0],\n\t\tnewestSession: timestamps[timestamps.length - 1],\n\t\tagentsFilePath,\n\t\tagentsFileTokens:\n\t\t\tagentsContent === undefined ? undefined : Math.round(Buffer.byteLength(agentsContent, \"utf-8\") / 4),\n\t\tdirectives: directives.kept,\n\t\tfixes: fixes.kept,\n\t\tworkflows: workflows.kept,\n\t\tsuppressed: directives.suppressed + fixes.suppressed + workflows.suppressed,\n\t\tsurfaced,\n\t};\n}\n"]}
|