@caupulican/pi-adaptative 0.81.5 → 0.81.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +30 -0
- package/dist/core/agent-session.d.ts +8 -0
- package/dist/core/agent-session.d.ts.map +1 -1
- package/dist/core/agent-session.js +125 -66
- package/dist/core/agent-session.js.map +1 -1
- package/dist/core/background-lane-controller.d.ts +1 -0
- package/dist/core/background-lane-controller.d.ts.map +1 -1
- package/dist/core/background-lane-controller.js +10 -3
- package/dist/core/background-lane-controller.js.map +1 -1
- package/dist/core/bash-execution-controller.d.ts +1 -1
- package/dist/core/bash-execution-controller.d.ts.map +1 -1
- package/dist/core/bash-execution-controller.js +9 -6
- package/dist/core/bash-execution-controller.js.map +1 -1
- package/dist/core/billing-failover-controller.d.ts +2 -2
- package/dist/core/billing-failover-controller.d.ts.map +1 -1
- package/dist/core/billing-failover-controller.js +2 -8
- package/dist/core/billing-failover-controller.js.map +1 -1
- package/dist/core/compaction-support.d.ts +3 -0
- package/dist/core/compaction-support.d.ts.map +1 -1
- package/dist/core/compaction-support.js +18 -4
- package/dist/core/compaction-support.js.map +1 -1
- package/dist/core/context-pipeline.d.ts +1 -0
- package/dist/core/context-pipeline.d.ts.map +1 -1
- package/dist/core/context-pipeline.js +12 -0
- package/dist/core/context-pipeline.js.map +1 -1
- package/dist/core/exec.d.ts.map +1 -1
- package/dist/core/exec.js +7 -2
- package/dist/core/exec.js.map +1 -1
- package/dist/core/extensions/loader.d.ts.map +1 -1
- package/dist/core/extensions/loader.js +33 -9
- package/dist/core/extensions/loader.js.map +1 -1
- package/dist/core/extensions/runner.d.ts.map +1 -1
- package/dist/core/extensions/runner.js +19 -5
- package/dist/core/extensions/runner.js.map +1 -1
- package/dist/core/failure-corpus.d.ts.map +1 -1
- package/dist/core/failure-corpus.js +1 -1
- package/dist/core/failure-corpus.js.map +1 -1
- package/dist/core/learning/reflection-engine.d.ts.map +1 -1
- package/dist/core/learning/reflection-engine.js +3 -1
- package/dist/core/learning/reflection-engine.js.map +1 -1
- package/dist/core/models/local-registration.d.ts +1 -0
- package/dist/core/models/local-registration.d.ts.map +1 -1
- package/dist/core/models/local-registration.js +3 -2
- package/dist/core/models/local-registration.js.map +1 -1
- package/dist/core/research/model-fitness.d.ts +16 -0
- package/dist/core/research/model-fitness.d.ts.map +1 -1
- package/dist/core/research/model-fitness.js +74 -1
- package/dist/core/research/model-fitness.js.map +1 -1
- package/dist/core/scout-controller.d.ts +5 -0
- package/dist/core/scout-controller.d.ts.map +1 -1
- package/dist/core/scout-controller.js +4 -1
- package/dist/core/scout-controller.js.map +1 -1
- package/dist/core/session-tree-navigator.d.ts.map +1 -1
- package/dist/core/session-tree-navigator.js +1 -0
- package/dist/core/session-tree-navigator.js.map +1 -1
- package/dist/core/settings-manager.d.ts.map +1 -1
- package/dist/core/settings-manager.js +1 -1
- package/dist/core/settings-manager.js.map +1 -1
- package/dist/core/trust-manager.d.ts.map +1 -1
- package/dist/core/trust-manager.js +6 -3
- package/dist/core/trust-manager.js.map +1 -1
- package/dist/migrations.d.ts +1 -0
- package/dist/migrations.d.ts.map +1 -1
- package/dist/migrations.js +14 -2
- package/dist/migrations.js.map +1 -1
- package/dist/modes/interactive/auto-learn-controller.d.ts.map +1 -1
- package/dist/modes/interactive/auto-learn-controller.js +14 -6
- package/dist/modes/interactive/auto-learn-controller.js.map +1 -1
- package/dist/modes/interactive/components/session-selector.d.ts +1 -0
- package/dist/modes/interactive/components/session-selector.d.ts.map +1 -1
- package/dist/modes/interactive/components/session-selector.js +11 -4
- package/dist/modes/interactive/components/session-selector.js.map +1 -1
- package/dist/modes/interactive/interactive-mode.d.ts +2 -0
- package/dist/modes/interactive/interactive-mode.d.ts.map +1 -1
- package/dist/modes/interactive/interactive-mode.js +107 -91
- package/dist/modes/interactive/interactive-mode.js.map +1 -1
- package/dist/modes/interactive/session-flow-commands.d.ts.map +1 -1
- package/dist/modes/interactive/session-flow-commands.js +10 -7
- package/dist/modes/interactive/session-flow-commands.js.map +1 -1
- package/dist/utils/tools-manager.d.ts +1 -0
- package/dist/utils/tools-manager.d.ts.map +1 -1
- package/dist/utils/tools-manager.js +14 -1
- package/dist/utils/tools-manager.js.map +1 -1
- package/docs/bug-ledger.md +45 -44
- package/examples/extensions/custom-provider-anthropic/package-lock.json +2 -2
- package/examples/extensions/custom-provider-anthropic/package.json +1 -1
- package/examples/extensions/custom-provider-gitlab-duo/package.json +1 -1
- package/examples/extensions/sandbox/package-lock.json +2 -2
- package/examples/extensions/sandbox/package.json +1 -1
- package/examples/extensions/with-deps/package-lock.json +2 -2
- package/examples/extensions/with-deps/package.json +1 -1
- package/npm-shrinkwrap.json +12 -12
- package/package.json +4 -4
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"failure-corpus.d.ts","sourceRoot":"","sources":["../../src/core/failure-corpus.ts"],"names":[],"mappings":"AAEA,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,2BAA2B,CAAC;AAEjE,MAAM,WAAW,mBAAmB;IACnC,EAAE,EAAE,MAAM,CAAC;IACX,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,MAAM,EAAE,eAAe,CAAC,QAAQ,CAAC,CAAC;IAClC,SAAS,EAAE,OAAO,CAAC;IACnB,OAAO,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,WAAW,kBAAkB;IAClC,KAAK,EAAE,MAAM,CAAC;IACd,OAAO,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,WAAW,eAAe;IAC/B,UAAU,CAAC,IAAI,EAAE,MAAM,GAAG,OAAO,CAAC;IAClC,SAAS,CAAC,IAAI,EAAE,MAAM,EAAE,OAAO,EAAE;QAAE,SAAS,EAAE,IAAI,CAAA;KAAE,GAAG,IAAI,CAAC;IAC5D,cAAc,CAAC,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,QAAQ,EAAE,OAAO,GAAG,IAAI,CAAC;IACpE,YAAY,CAAC,IAAI,EAAE,MAAM,EAAE,QAAQ,EAAE,OAAO,GAAG,MAAM,CAAC;IACtD,aAAa,CAAC,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,QAAQ,EAAE,OAAO,GAAG,IAAI,CAAC;IACnE,QAAQ,CAAC,IAAI,EAAE,MAAM,GAAG;QAAE,IAAI,EAAE,MAAM,CAAA;KAAE,CAAC;CACzC;AAOD,qBAAa,qBAAqB;IACjC,OAAO,CAAC,KAAK,CAAK;IAClB,OAAO,CAAC,OAAO,CAAK;IACpB,OAAO,CAAC,QAAQ,CAAC,QAAQ,CAAS;IAClC,OAAO,CAAC,QAAQ,CAAC,EAAE,CAAkB;IACrC,OAAO,CAAC,QAAQ,CAAC,GAAG,CAAa;IACjC,OAAO,CAAC,QAAQ,CAAC,KAAK,CAA4B;IAElD,YAAY,IAAI,EAAE;QAAE,QAAQ,EAAE,MAAM,CAAC;QAAC,EAAE,CAAC,EAAE,eAAe,CAAC;QAAC,GAAG,CAAC,EAAE,MAAM,IAAI,CAAC;QAAC,KAAK,CAAC,EAAE,CAAC,OAAO,EAAE,MAAM,KAAK,IAAI,CAAA;KAAE,EAKhH;IAED,MAAM,CAAC,IAAI,EAAE;QAAE,QAAQ,CAAC,EAAE,MAAM,CAAC;QAAC,OAAO,CAAC,EAAE,MAAM,CAAC;QAAC,OAAO,EAAE,MAAM,CAAC;QAAC,UAAU,EAAE,eAAe,CAAA;KAAE,GAAG,IAAI,CAkBxG;IAED,KAAK,IAAI,kBAAkB,CAE1B;IAED,OAAO,CAAC,cAAc;CAStB;AAED,wBAAgB,aAAa,CAAC,OAAO,EAAE,MAAM,GAAG,MAAM,CAKrD","sourcesContent":["import { appendFileSync, existsSync, mkdirSync, readFileSync, statSync, writeFileSync } from \"node:fs\";\nimport { dirname } from \"node:path\";\nimport type { ClassifiedError } from \"@caupulican/pi-agent-core\";\n\nexport interface FailureCorpusRecord {\n\tts: string;\n\tprovider?: string;\n\tmodelId?: string;\n\treason: ClassifiedError[\"reason\"];\n\tretryable: boolean;\n\tmessage: string;\n}\n\nexport interface FailureCorpusStats {\n\ttotal: number;\n\tunknown: number;\n}\n\nexport interface FailureCorpusFs {\n\texistsSync(path: string): boolean;\n\tmkdirSync(path: string, options: { recursive: true }): void;\n\tappendFileSync(path: string, data: string, encoding: \"utf-8\"): void;\n\treadFileSync(path: string, encoding: \"utf-8\"): string;\n\twriteFileSync(path: string, data: string, encoding: \"utf-8\"): void;\n\tstatSync(path: string): { size: number };\n}\n\nconst DEFAULT_FS: FailureCorpusFs = { existsSync, mkdirSync, appendFileSync, readFileSync, writeFileSync, statSync };\nconst MAX_MESSAGE_CHARS = 500;\nconst MAX_FILE_BYTES = 512 * 1024;\nconst MAX_ROTATED_RECORDS = 1000;\n\nexport class FailureCorpusRecorder {\n\tprivate total = 0;\n\tprivate unknown = 0;\n\tprivate readonly filePath: string;\n\tprivate readonly fs: FailureCorpusFs;\n\tprivate readonly now: () => Date;\n\tprivate readonly debug: (message: string) => void;\n\n\tconstructor(args: { filePath: string; fs?: FailureCorpusFs; now?: () => Date; debug?: (message: string) => void }) {\n\t\tthis.filePath = args.filePath;\n\t\tthis.fs = args.fs ?? DEFAULT_FS;\n\t\tthis.now = args.now ?? (() => new Date());\n\t\tthis.debug = args.debug ?? (() => {});\n\t}\n\n\trecord(args: { provider?: string; modelId?: string; message: string; classified: ClassifiedError }): void {\n\t\tthis.total += 1;\n\t\tif (args.classified.reason === \"unknown\") this.unknown += 1;\n\t\ttry {\n\t\t\tthis.fs.mkdirSync(dirname(this.filePath), { recursive: true });\n\t\t\tconst record: FailureCorpusRecord = {\n\t\t\t\tts: this.now().toISOString(),\n\t\t\t\tprovider: args.provider,\n\t\t\t\tmodelId: args.modelId,\n\t\t\t\treason: args.classified.reason,\n\t\t\t\tretryable: args.classified.retryable,\n\t\t\t\tmessage: redactSecrets(args.message).slice(0, MAX_MESSAGE_CHARS),\n\t\t\t};\n\t\t\tthis.fs.appendFileSync(this.filePath, `${JSON.stringify(record)}\\n`, \"utf-8\");\n\t\t\tthis.rotateIfNeeded();\n\t\t} catch (error) {\n\t\t\tthis.debug(`failure corpus write skipped: ${error instanceof Error ? error.message : String(error)}`);\n\t\t}\n\t}\n\n\tstats(): FailureCorpusStats {\n\t\treturn { total: this.total, unknown: this.unknown };\n\t}\n\n\tprivate rotateIfNeeded(): void {\n\t\tif (!this.fs.existsSync(this.filePath) || this.fs.statSync(this.filePath).size <= MAX_FILE_BYTES) return;\n\t\tconst lines = this.fs\n\t\t\t.readFileSync(this.filePath, \"utf-8\")\n\t\t\t.split(\"\\n\")\n\t\t\t.filter((line) => line.trim().length > 0)\n\t\t\t.slice(-MAX_ROTATED_RECORDS);\n\t\tthis.fs.writeFileSync(this.filePath, `${lines.join(\"\\n\")}\\n`, \"utf-8\");\n\t}\n}\n\nexport function redactSecrets(message: string): string {\n\treturn message\n\t\t.replace(/sk-[A-Za-z0-9]{8,}/g, \"[redacted]\")\n\t\t.replace(/Bearer\\s+[A-Za-z0-9._~+/=-]+/gi, \"Bearer [redacted]\")\n\t\t.replace(/[A-Za-z0-9+/]{40,}={0,2}/g, \"[redacted]\");\n}\n"]}
|
|
1
|
+
{"version":3,"file":"failure-corpus.d.ts","sourceRoot":"","sources":["../../src/core/failure-corpus.ts"],"names":[],"mappings":"AAEA,OAAO,KAAK,EAAE,eAAe,EAAE,MAAM,2BAA2B,CAAC;AAEjE,MAAM,WAAW,mBAAmB;IACnC,EAAE,EAAE,MAAM,CAAC;IACX,QAAQ,CAAC,EAAE,MAAM,CAAC;IAClB,OAAO,CAAC,EAAE,MAAM,CAAC;IACjB,MAAM,EAAE,eAAe,CAAC,QAAQ,CAAC,CAAC;IAClC,SAAS,EAAE,OAAO,CAAC;IACnB,OAAO,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,WAAW,kBAAkB;IAClC,KAAK,EAAE,MAAM,CAAC;IACd,OAAO,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,WAAW,eAAe;IAC/B,UAAU,CAAC,IAAI,EAAE,MAAM,GAAG,OAAO,CAAC;IAClC,SAAS,CAAC,IAAI,EAAE,MAAM,EAAE,OAAO,EAAE;QAAE,SAAS,EAAE,IAAI,CAAA;KAAE,GAAG,IAAI,CAAC;IAC5D,cAAc,CAAC,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,QAAQ,EAAE,OAAO,GAAG,IAAI,CAAC;IACpE,YAAY,CAAC,IAAI,EAAE,MAAM,EAAE,QAAQ,EAAE,OAAO,GAAG,MAAM,CAAC;IACtD,aAAa,CAAC,IAAI,EAAE,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,QAAQ,EAAE,OAAO,GAAG,IAAI,CAAC;IACnE,QAAQ,CAAC,IAAI,EAAE,MAAM,GAAG;QAAE,IAAI,EAAE,MAAM,CAAA;KAAE,CAAC;CACzC;AAOD,qBAAa,qBAAqB;IACjC,OAAO,CAAC,KAAK,CAAK;IAClB,OAAO,CAAC,OAAO,CAAK;IACpB,OAAO,CAAC,QAAQ,CAAC,QAAQ,CAAS;IAClC,OAAO,CAAC,QAAQ,CAAC,EAAE,CAAkB;IACrC,OAAO,CAAC,QAAQ,CAAC,GAAG,CAAa;IACjC,OAAO,CAAC,QAAQ,CAAC,KAAK,CAA4B;IAElD,YAAY,IAAI,EAAE;QAAE,QAAQ,EAAE,MAAM,CAAC;QAAC,EAAE,CAAC,EAAE,eAAe,CAAC;QAAC,GAAG,CAAC,EAAE,MAAM,IAAI,CAAC;QAAC,KAAK,CAAC,EAAE,CAAC,OAAO,EAAE,MAAM,KAAK,IAAI,CAAA;KAAE,EAKhH;IAED,MAAM,CAAC,IAAI,EAAE;QAAE,QAAQ,CAAC,EAAE,MAAM,CAAC;QAAC,OAAO,CAAC,EAAE,MAAM,CAAC;QAAC,OAAO,EAAE,MAAM,CAAC;QAAC,UAAU,EAAE,eAAe,CAAA;KAAE,GAAG,IAAI,CAkBxG;IAED,KAAK,IAAI,kBAAkB,CAE1B;IAED,OAAO,CAAC,cAAc;CAStB;AAED,wBAAgB,aAAa,CAAC,OAAO,EAAE,MAAM,GAAG,MAAM,CAKrD","sourcesContent":["import { appendFileSync, existsSync, mkdirSync, readFileSync, statSync, writeFileSync } from \"node:fs\";\nimport { dirname } from \"node:path\";\nimport type { ClassifiedError } from \"@caupulican/pi-agent-core\";\n\nexport interface FailureCorpusRecord {\n\tts: string;\n\tprovider?: string;\n\tmodelId?: string;\n\treason: ClassifiedError[\"reason\"];\n\tretryable: boolean;\n\tmessage: string;\n}\n\nexport interface FailureCorpusStats {\n\ttotal: number;\n\tunknown: number;\n}\n\nexport interface FailureCorpusFs {\n\texistsSync(path: string): boolean;\n\tmkdirSync(path: string, options: { recursive: true }): void;\n\tappendFileSync(path: string, data: string, encoding: \"utf-8\"): void;\n\treadFileSync(path: string, encoding: \"utf-8\"): string;\n\twriteFileSync(path: string, data: string, encoding: \"utf-8\"): void;\n\tstatSync(path: string): { size: number };\n}\n\nconst DEFAULT_FS: FailureCorpusFs = { existsSync, mkdirSync, appendFileSync, readFileSync, writeFileSync, statSync };\nconst MAX_MESSAGE_CHARS = 500;\nconst MAX_FILE_BYTES = 512 * 1024;\nconst MAX_ROTATED_RECORDS = 1000;\n\nexport class FailureCorpusRecorder {\n\tprivate total = 0;\n\tprivate unknown = 0;\n\tprivate readonly filePath: string;\n\tprivate readonly fs: FailureCorpusFs;\n\tprivate readonly now: () => Date;\n\tprivate readonly debug: (message: string) => void;\n\n\tconstructor(args: { filePath: string; fs?: FailureCorpusFs; now?: () => Date; debug?: (message: string) => void }) {\n\t\tthis.filePath = args.filePath;\n\t\tthis.fs = args.fs ?? DEFAULT_FS;\n\t\tthis.now = args.now ?? (() => new Date());\n\t\tthis.debug = args.debug ?? (() => {});\n\t}\n\n\trecord(args: { provider?: string; modelId?: string; message: string; classified: ClassifiedError }): void {\n\t\tthis.total += 1;\n\t\tif (args.classified.reason === \"unknown\") this.unknown += 1;\n\t\ttry {\n\t\t\tthis.fs.mkdirSync(dirname(this.filePath), { recursive: true });\n\t\t\tconst record: FailureCorpusRecord = {\n\t\t\t\tts: this.now().toISOString(),\n\t\t\t\tprovider: args.provider,\n\t\t\t\tmodelId: args.modelId,\n\t\t\t\treason: args.classified.reason,\n\t\t\t\tretryable: args.classified.retryable,\n\t\t\t\tmessage: redactSecrets(args.message).slice(0, MAX_MESSAGE_CHARS),\n\t\t\t};\n\t\t\tthis.fs.appendFileSync(this.filePath, `${JSON.stringify(record)}\\n`, \"utf-8\");\n\t\t\tthis.rotateIfNeeded();\n\t\t} catch (error) {\n\t\t\tthis.debug(`failure corpus write skipped: ${error instanceof Error ? error.message : String(error)}`);\n\t\t}\n\t}\n\n\tstats(): FailureCorpusStats {\n\t\treturn { total: this.total, unknown: this.unknown };\n\t}\n\n\tprivate rotateIfNeeded(): void {\n\t\tif (!this.fs.existsSync(this.filePath) || this.fs.statSync(this.filePath).size <= MAX_FILE_BYTES) return;\n\t\tconst lines = this.fs\n\t\t\t.readFileSync(this.filePath, \"utf-8\")\n\t\t\t.split(\"\\n\")\n\t\t\t.filter((line) => line.trim().length > 0)\n\t\t\t.slice(-MAX_ROTATED_RECORDS);\n\t\tthis.fs.writeFileSync(this.filePath, `${lines.join(\"\\n\")}\\n`, \"utf-8\");\n\t}\n}\n\nexport function redactSecrets(message: string): string {\n\treturn message\n\t\t.replace(/sk-(?:proj-|ant-)?[A-Za-z0-9._-]{8,}/g, \"[redacted]\")\n\t\t.replace(/Bearer\\s+[A-Za-z0-9._~+/=-]+/gi, \"Bearer [redacted]\")\n\t\t.replace(/[A-Za-z0-9+/]{40,}={0,2}/g, \"[redacted]\");\n}\n"]}
|
|
@@ -54,7 +54,7 @@ export class FailureCorpusRecorder {
|
|
|
54
54
|
}
|
|
55
55
|
export function redactSecrets(message) {
|
|
56
56
|
return message
|
|
57
|
-
.replace(/sk-[A-Za-z0-9]{8,}/g, "[redacted]")
|
|
57
|
+
.replace(/sk-(?:proj-|ant-)?[A-Za-z0-9._-]{8,}/g, "[redacted]")
|
|
58
58
|
.replace(/Bearer\s+[A-Za-z0-9._~+/=-]+/gi, "Bearer [redacted]")
|
|
59
59
|
.replace(/[A-Za-z0-9+/]{40,}={0,2}/g, "[redacted]");
|
|
60
60
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"failure-corpus.js","sourceRoot":"","sources":["../../src/core/failure-corpus.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,cAAc,EAAE,UAAU,EAAE,SAAS,EAAE,YAAY,EAAE,QAAQ,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AACvG,OAAO,EAAE,OAAO,EAAE,MAAM,WAAW,CAAC;AA0BpC,MAAM,UAAU,GAAoB,EAAE,UAAU,EAAE,SAAS,EAAE,cAAc,EAAE,YAAY,EAAE,aAAa,EAAE,QAAQ,EAAE,CAAC;AACrH,MAAM,iBAAiB,GAAG,GAAG,CAAC;AAC9B,MAAM,cAAc,GAAG,GAAG,GAAG,IAAI,CAAC;AAClC,MAAM,mBAAmB,GAAG,IAAI,CAAC;AAEjC,MAAM,OAAO,qBAAqB;IACzB,KAAK,GAAG,CAAC,CAAC;IACV,OAAO,GAAG,CAAC,CAAC;IACH,QAAQ,CAAS;IACjB,EAAE,CAAkB;IACpB,GAAG,CAAa;IAChB,KAAK,CAA4B;IAElD,YAAY,IAAqG,EAAE;QAClH,IAAI,CAAC,QAAQ,GAAG,IAAI,CAAC,QAAQ,CAAC;QAC9B,IAAI,CAAC,EAAE,GAAG,IAAI,CAAC,EAAE,IAAI,UAAU,CAAC;QAChC,IAAI,CAAC,GAAG,GAAG,IAAI,CAAC,GAAG,IAAI,CAAC,GAAG,EAAE,CAAC,IAAI,IAAI,EAAE,CAAC,CAAC;QAC1C,IAAI,CAAC,KAAK,GAAG,IAAI,CAAC,KAAK,IAAI,CAAC,GAAG,EAAE,CAAC,EAAC,CAAC,CAAC,CAAC;IAAA,CACtC;IAED,MAAM,CAAC,IAA2F,EAAQ;QACzG,IAAI,CAAC,KAAK,IAAI,CAAC,CAAC;QAChB,IAAI,IAAI,CAAC,UAAU,CAAC,MAAM,KAAK,SAAS;YAAE,IAAI,CAAC,OAAO,IAAI,CAAC,CAAC;QAC5D,IAAI,CAAC;YACJ,IAAI,CAAC,EAAE,CAAC,SAAS,CAAC,OAAO,CAAC,IAAI,CAAC,QAAQ,CAAC,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC,CAAC;YAC/D,MAAM,MAAM,GAAwB;gBACnC,EAAE,EAAE,IAAI,CAAC,GAAG,EAAE,CAAC,WAAW,EAAE;gBAC5B,QAAQ,EAAE,IAAI,CAAC,QAAQ;gBACvB,OAAO,EAAE,IAAI,CAAC,OAAO;gBACrB,MAAM,EAAE,IAAI,CAAC,UAAU,CAAC,MAAM;gBAC9B,SAAS,EAAE,IAAI,CAAC,UAAU,CAAC,SAAS;gBACpC,OAAO,EAAE,aAAa,CAAC,IAAI,CAAC,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC,EAAE,iBAAiB,CAAC;aAChE,CAAC;YACF,IAAI,CAAC,EAAE,CAAC,cAAc,CAAC,IAAI,CAAC,QAAQ,EAAE,GAAG,IAAI,CAAC,SAAS,CAAC,MAAM,CAAC,IAAI,EAAE,OAAO,CAAC,CAAC;YAC9E,IAAI,CAAC,cAAc,EAAE,CAAC;QACvB,CAAC;QAAC,OAAO,KAAK,EAAE,CAAC;YAChB,IAAI,CAAC,KAAK,CAAC,iCAAiC,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC;QACvG,CAAC;IAAA,CACD;IAED,KAAK,GAAuB;QAC3B,OAAO,EAAE,KAAK,EAAE,IAAI,CAAC,KAAK,EAAE,OAAO,EAAE,IAAI,CAAC,OAAO,EAAE,CAAC;IAAA,CACpD;IAEO,cAAc,GAAS;QAC9B,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,UAAU,CAAC,IAAI,CAAC,QAAQ,CAAC,IAAI,IAAI,CAAC,EAAE,CAAC,QAAQ,CAAC,IAAI,CAAC,QAAQ,CAAC,CAAC,IAAI,IAAI,cAAc;YAAE,OAAO;QACzG,MAAM,KAAK,GAAG,IAAI,CAAC,EAAE;aACnB,YAAY,CAAC,IAAI,CAAC,QAAQ,EAAE,OAAO,CAAC;aACpC,KAAK,CAAC,IAAI,CAAC;aACX,MAAM,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,CAAC;aACxC,KAAK,CAAC,CAAC,mBAAmB,CAAC,CAAC;QAC9B,IAAI,CAAC,EAAE,CAAC,aAAa,CAAC,IAAI,CAAC,QAAQ,EAAE,GAAG,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,EAAE,OAAO,CAAC,CAAC;IAAA,CACvE;CACD;AAED,MAAM,UAAU,aAAa,CAAC,OAAe,EAAU;IACtD,OAAO,OAAO;SACZ,OAAO,CAAC,
|
|
1
|
+
{"version":3,"file":"failure-corpus.js","sourceRoot":"","sources":["../../src/core/failure-corpus.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,cAAc,EAAE,UAAU,EAAE,SAAS,EAAE,YAAY,EAAE,QAAQ,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AACvG,OAAO,EAAE,OAAO,EAAE,MAAM,WAAW,CAAC;AA0BpC,MAAM,UAAU,GAAoB,EAAE,UAAU,EAAE,SAAS,EAAE,cAAc,EAAE,YAAY,EAAE,aAAa,EAAE,QAAQ,EAAE,CAAC;AACrH,MAAM,iBAAiB,GAAG,GAAG,CAAC;AAC9B,MAAM,cAAc,GAAG,GAAG,GAAG,IAAI,CAAC;AAClC,MAAM,mBAAmB,GAAG,IAAI,CAAC;AAEjC,MAAM,OAAO,qBAAqB;IACzB,KAAK,GAAG,CAAC,CAAC;IACV,OAAO,GAAG,CAAC,CAAC;IACH,QAAQ,CAAS;IACjB,EAAE,CAAkB;IACpB,GAAG,CAAa;IAChB,KAAK,CAA4B;IAElD,YAAY,IAAqG,EAAE;QAClH,IAAI,CAAC,QAAQ,GAAG,IAAI,CAAC,QAAQ,CAAC;QAC9B,IAAI,CAAC,EAAE,GAAG,IAAI,CAAC,EAAE,IAAI,UAAU,CAAC;QAChC,IAAI,CAAC,GAAG,GAAG,IAAI,CAAC,GAAG,IAAI,CAAC,GAAG,EAAE,CAAC,IAAI,IAAI,EAAE,CAAC,CAAC;QAC1C,IAAI,CAAC,KAAK,GAAG,IAAI,CAAC,KAAK,IAAI,CAAC,GAAG,EAAE,CAAC,EAAC,CAAC,CAAC,CAAC;IAAA,CACtC;IAED,MAAM,CAAC,IAA2F,EAAQ;QACzG,IAAI,CAAC,KAAK,IAAI,CAAC,CAAC;QAChB,IAAI,IAAI,CAAC,UAAU,CAAC,MAAM,KAAK,SAAS;YAAE,IAAI,CAAC,OAAO,IAAI,CAAC,CAAC;QAC5D,IAAI,CAAC;YACJ,IAAI,CAAC,EAAE,CAAC,SAAS,CAAC,OAAO,CAAC,IAAI,CAAC,QAAQ,CAAC,EAAE,EAAE,SAAS,EAAE,IAAI,EAAE,CAAC,CAAC;YAC/D,MAAM,MAAM,GAAwB;gBACnC,EAAE,EAAE,IAAI,CAAC,GAAG,EAAE,CAAC,WAAW,EAAE;gBAC5B,QAAQ,EAAE,IAAI,CAAC,QAAQ;gBACvB,OAAO,EAAE,IAAI,CAAC,OAAO;gBACrB,MAAM,EAAE,IAAI,CAAC,UAAU,CAAC,MAAM;gBAC9B,SAAS,EAAE,IAAI,CAAC,UAAU,CAAC,SAAS;gBACpC,OAAO,EAAE,aAAa,CAAC,IAAI,CAAC,OAAO,CAAC,CAAC,KAAK,CAAC,CAAC,EAAE,iBAAiB,CAAC;aAChE,CAAC;YACF,IAAI,CAAC,EAAE,CAAC,cAAc,CAAC,IAAI,CAAC,QAAQ,EAAE,GAAG,IAAI,CAAC,SAAS,CAAC,MAAM,CAAC,IAAI,EAAE,OAAO,CAAC,CAAC;YAC9E,IAAI,CAAC,cAAc,EAAE,CAAC;QACvB,CAAC;QAAC,OAAO,KAAK,EAAE,CAAC;YAChB,IAAI,CAAC,KAAK,CAAC,iCAAiC,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC;QACvG,CAAC;IAAA,CACD;IAED,KAAK,GAAuB;QAC3B,OAAO,EAAE,KAAK,EAAE,IAAI,CAAC,KAAK,EAAE,OAAO,EAAE,IAAI,CAAC,OAAO,EAAE,CAAC;IAAA,CACpD;IAEO,cAAc,GAAS;QAC9B,IAAI,CAAC,IAAI,CAAC,EAAE,CAAC,UAAU,CAAC,IAAI,CAAC,QAAQ,CAAC,IAAI,IAAI,CAAC,EAAE,CAAC,QAAQ,CAAC,IAAI,CAAC,QAAQ,CAAC,CAAC,IAAI,IAAI,cAAc;YAAE,OAAO;QACzG,MAAM,KAAK,GAAG,IAAI,CAAC,EAAE;aACnB,YAAY,CAAC,IAAI,CAAC,QAAQ,EAAE,OAAO,CAAC;aACpC,KAAK,CAAC,IAAI,CAAC;aACX,MAAM,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,IAAI,CAAC,IAAI,EAAE,CAAC,MAAM,GAAG,CAAC,CAAC;aACxC,KAAK,CAAC,CAAC,mBAAmB,CAAC,CAAC;QAC9B,IAAI,CAAC,EAAE,CAAC,aAAa,CAAC,IAAI,CAAC,QAAQ,EAAE,GAAG,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,EAAE,OAAO,CAAC,CAAC;IAAA,CACvE;CACD;AAED,MAAM,UAAU,aAAa,CAAC,OAAe,EAAU;IACtD,OAAO,OAAO;SACZ,OAAO,CAAC,uCAAuC,EAAE,YAAY,CAAC;SAC9D,OAAO,CAAC,gCAAgC,EAAE,mBAAmB,CAAC;SAC9D,OAAO,CAAC,2BAA2B,EAAE,YAAY,CAAC,CAAC;AAAA,CACrD","sourcesContent":["import { appendFileSync, existsSync, mkdirSync, readFileSync, statSync, writeFileSync } from \"node:fs\";\nimport { dirname } from \"node:path\";\nimport type { ClassifiedError } from \"@caupulican/pi-agent-core\";\n\nexport interface FailureCorpusRecord {\n\tts: string;\n\tprovider?: string;\n\tmodelId?: string;\n\treason: ClassifiedError[\"reason\"];\n\tretryable: boolean;\n\tmessage: string;\n}\n\nexport interface FailureCorpusStats {\n\ttotal: number;\n\tunknown: number;\n}\n\nexport interface FailureCorpusFs {\n\texistsSync(path: string): boolean;\n\tmkdirSync(path: string, options: { recursive: true }): void;\n\tappendFileSync(path: string, data: string, encoding: \"utf-8\"): void;\n\treadFileSync(path: string, encoding: \"utf-8\"): string;\n\twriteFileSync(path: string, data: string, encoding: \"utf-8\"): void;\n\tstatSync(path: string): { size: number };\n}\n\nconst DEFAULT_FS: FailureCorpusFs = { existsSync, mkdirSync, appendFileSync, readFileSync, writeFileSync, statSync };\nconst MAX_MESSAGE_CHARS = 500;\nconst MAX_FILE_BYTES = 512 * 1024;\nconst MAX_ROTATED_RECORDS = 1000;\n\nexport class FailureCorpusRecorder {\n\tprivate total = 0;\n\tprivate unknown = 0;\n\tprivate readonly filePath: string;\n\tprivate readonly fs: FailureCorpusFs;\n\tprivate readonly now: () => Date;\n\tprivate readonly debug: (message: string) => void;\n\n\tconstructor(args: { filePath: string; fs?: FailureCorpusFs; now?: () => Date; debug?: (message: string) => void }) {\n\t\tthis.filePath = args.filePath;\n\t\tthis.fs = args.fs ?? DEFAULT_FS;\n\t\tthis.now = args.now ?? (() => new Date());\n\t\tthis.debug = args.debug ?? (() => {});\n\t}\n\n\trecord(args: { provider?: string; modelId?: string; message: string; classified: ClassifiedError }): void {\n\t\tthis.total += 1;\n\t\tif (args.classified.reason === \"unknown\") this.unknown += 1;\n\t\ttry {\n\t\t\tthis.fs.mkdirSync(dirname(this.filePath), { recursive: true });\n\t\t\tconst record: FailureCorpusRecord = {\n\t\t\t\tts: this.now().toISOString(),\n\t\t\t\tprovider: args.provider,\n\t\t\t\tmodelId: args.modelId,\n\t\t\t\treason: args.classified.reason,\n\t\t\t\tretryable: args.classified.retryable,\n\t\t\t\tmessage: redactSecrets(args.message).slice(0, MAX_MESSAGE_CHARS),\n\t\t\t};\n\t\t\tthis.fs.appendFileSync(this.filePath, `${JSON.stringify(record)}\\n`, \"utf-8\");\n\t\t\tthis.rotateIfNeeded();\n\t\t} catch (error) {\n\t\t\tthis.debug(`failure corpus write skipped: ${error instanceof Error ? error.message : String(error)}`);\n\t\t}\n\t}\n\n\tstats(): FailureCorpusStats {\n\t\treturn { total: this.total, unknown: this.unknown };\n\t}\n\n\tprivate rotateIfNeeded(): void {\n\t\tif (!this.fs.existsSync(this.filePath) || this.fs.statSync(this.filePath).size <= MAX_FILE_BYTES) return;\n\t\tconst lines = this.fs\n\t\t\t.readFileSync(this.filePath, \"utf-8\")\n\t\t\t.split(\"\\n\")\n\t\t\t.filter((line) => line.trim().length > 0)\n\t\t\t.slice(-MAX_ROTATED_RECORDS);\n\t\tthis.fs.writeFileSync(this.filePath, `${lines.join(\"\\n\")}\\n`, \"utf-8\");\n\t}\n}\n\nexport function redactSecrets(message: string): string {\n\treturn message\n\t\t.replace(/sk-(?:proj-|ant-)?[A-Za-z0-9._-]{8,}/g, \"[redacted]\")\n\t\t.replace(/Bearer\\s+[A-Za-z0-9._~+/=-]+/gi, \"Bearer [redacted]\")\n\t\t.replace(/[A-Za-z0-9+/]{40,}={0,2}/g, \"[redacted]\");\n}\n"]}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"reflection-engine.d.ts","sourceRoot":"","sources":["../../../src/core/learning/reflection-engine.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,KAAK,EAAE,MAAM,mBAAmB,CAAC;AAE/C,MAAM,MAAM,UAAU,GAAG,MAAM,GAAG,SAAS,GAAG,SAAS,GAAG,OAAO,GAAG,MAAM,CAAC;AAE3E,MAAM,WAAW,wBAAwB;IACxC,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,EAAE,KAAK,CAAC;IACb,UAAU,EAAE,UAAU,CAAC;CACvB;AAED,MAAM,MAAM,iBAAiB,GAAG,SAAS,GAAG,YAAY,GAAG,aAAa,GAAG,MAAM,CAAC;AAElF,MAAM,WAAW,aAAa;IAC7B,OAAO,EAAE,iBAAiB,CAAC;IAC3B,aAAa,EAAE,MAAM,CAAC;IACtB,aAAa,EAAE,OAAO,CAAC;IACvB,kBAAkB,EAAE,MAAM,CAAC;IAC3B,YAAY,EAAE,MAAM,CAAC;CACrB;AAED,MAAM,WAAW,UAAU;IAC1B,GAAG,EAAE,MAAM,GAAG,SAAS,CAAC;IACxB,MAAM,EAAE,MAAM,CAAC;IACf,WAAW,EAAE,MAAM,CAAC;CACpB;AAED;;;GAGG;AACH,wBAAgB,YAAY,CAAC,OAAO,EAAE,aAAa,GAAG,UAAU,CAyB/D;AAED,MAAM,WAAW,eAAe;IAC/B,cAAc,EAAE,MAAM,CAAC;IACvB,cAAc,EAAE,MAAM,CAAC;IACvB,IAAI,EAAE,UAAU,CAAC;IAEjB,QAAQ,EAAE,CAAC,YAAY,EAAE,MAAM,EAAE,UAAU,EAAE,MAAM,KAAK,OAAO,CAAC,wBAAwB,CAAC,CAAC;CAC1F;AAED,MAAM,MAAM,eAAe,GACxB;IAAE,IAAI,EAAE,YAAY,CAAC;IAAC,OAAO,EAAE,QAAQ,GAAG,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GAChE;IAAE,IAAI,EAAE,gBAAgB,CAAC;IAAC,MAAM,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GACxD;IAAE,IAAI,EAAE,eAAe,CAAC;IAAC,MAAM,EAAE,MAAM,CAAA;CAAE,GAEzC;IAAE,IAAI,EAAE,eAAe,CAAC;IAAC,IAAI,EAAE,MAAM,CAAC;IAAC,WAAW,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,CAAC;AAE9E,MAAM,WAAW,gBAAgB;IAChC,MAAM,EAAE,eAAe,EAAE,CAAC;IAC1B,KAAK,EAAE,KAAK,CAAC;IACb,SAAS,EAAE,MAAM,CAAC;CAClB;AAED;;;;;GAKG;AACH,eAAO,MAAM,wBAAwB,s4DAqBpC,CAAC;AAEF,qBAAa,gBAAgB;IAC5B;;;;OAIG;IACG,OAAO,CAAC,KAAK,EAAE,eAAe,GAAG,OAAO,CAAC,gBAAgB,CAAC,CA+E/D;CACD","sourcesContent":["import type { Usage } from \"@caupulican/pi-ai\";\n\nexport type StopReason = \"stop\" | \"toolUse\" | \"aborted\" | \"error\" | string;\n\nexport interface IsolatedCompletionResult {\n\ttext: string;\n\tusage: Usage;\n\tstopReason: StopReason;\n}\n\nexport type ReflectionTrigger = \"complex\" | \"corrective\" | \"session-end\" | \"none\";\n\nexport interface DemandSignals {\n\ttrigger: ReflectionTrigger;\n\ttoolCallCount: number;\n\thadCorrection: boolean;\n\tcontextHeadroomPct: number; // 0..100\n\tusefulLately: number; // 0..1 rolling score\n}\n\nexport interface DemandPlan {\n\tact: \"skip\" | \"reflect\";\n\treason: string;\n\ttokenBudget: number;\n}\n\n/**\n * Pure zero-I/O heuristic to decide whether the current turn justifies a reflection run\n * and determine the token budget under the cheap-tool net-negative doctrine.\n */\nexport function decideDemand(signals: DemandSignals): DemandPlan {\n\tif (signals.trigger === \"none\") {\n\t\treturn { act: \"skip\", reason: \"No trigger detected\", tokenBudget: 0 };\n\t}\n\tif (signals.contextHeadroomPct < 10) {\n\t\treturn { act: \"skip\", reason: \"Context headroom is critically low (< 10%)\", tokenBudget: 0 };\n\t}\n\n\t// Dynamic token budget based on headroom (keep reflection bounded between 500 and 1500 tokens)\n\tconst baseBudget = 1000;\n\tconst tokenBudget = Math.max(500, Math.min(1500, Math.round(baseBudget * (signals.contextHeadroomPct / 100))));\n\n\tif (signals.hadCorrection) {\n\t\treturn { act: \"reflect\", reason: \"Correction detected in the turn\", tokenBudget };\n\t}\n\tif (signals.trigger === \"session-end\") {\n\t\treturn { act: \"reflect\", reason: \"Session end reflection triggered\", tokenBudget };\n\t}\n\tif (signals.trigger === \"complex\") {\n\t\tif (signals.toolCallCount >= 3) {\n\t\t\treturn { act: \"reflect\", reason: `Complex turn with ${signals.toolCallCount} tool calls`, tokenBudget };\n\t\t}\n\t}\n\n\treturn { act: \"skip\", reason: \"Signals do not justify reflection overhead\", tokenBudget: 0 };\n}\n\nexport interface ReflectionInput {\n\trecentTurnText: string; // host serializes the just-finished turn\n\texistingMemory: string; // current MEMORY.md + USER.md snapshot\n\tplan: DemandPlan;\n\t// host-injected isolated completion function:\n\tcomplete: (systemPrompt: string, userPrompt: string) => Promise<IsolatedCompletionResult>;\n}\n\nexport type ReflectionWrite =\n\t| { kind: \"memory_add\"; section: \"MEMORY\" | \"USER\"; text: string }\n\t| { kind: \"memory_replace\"; target: string; text: string }\n\t| { kind: \"memory_remove\"; target: string }\n\t// R7 memory-to-behavior: promote a recurring procedural workflow into an executable skill.\n\t| { kind: \"promote_skill\"; name: string; description: string; body: string };\n\nexport interface ReflectionResult {\n\twrites: ReflectionWrite[];\n\tusage: Usage;\n\trationale: string;\n}\n\n/**\n * STATIC reflection system prompt (Hermes-parity #33). It is byte-identical across every reflection\n * pass — the variable parts (existing memory snapshot + the turn transcript) live in the USER prompt —\n * so the provider prompt-cache reuses this prefix instead of re-billing it each pass (cost guard).\n * Do NOT interpolate per-call data into this constant or caching breaks.\n */\nexport const REFLECTION_SYSTEM_PROMPT = `You are a reflection engine. Your job is to analyze the recent conversation turn, compare it against the agent's existing memory, and decide if any memory updates are needed.\n\nMemory guidelines:\n- \"MEMORY\" is for project facts, configuration, repeatable workflows, and coding findings.\n- \"USER\" is for user preferences, patterns, and style specifications.\n- Avoid duplicate facts. If the fact is already represented, do not add it.\n- CONFRONT existing memory: if the new turn contradicts or updates an existing fact, use \"memory_replace\" or \"memory_remove\" to supersede the old fact rather than blindly appending.\n- Keep memories short, factual, and direct. No fluff.\n- Do NOT capture transient/environment-specific noise: tool/network failures, one-off errors, or a single narrative event. Persist only durable facts and preferences.\n- PROMOTE to behavior: if the turn established a REPEATABLE, multi-step PROCEDURE/workflow (not a one-off fact) that should govern a future class of tasks, emit a \"promote_skill\" instead of (or in addition to) a memory fact. Only promote a genuinely reusable procedure — never a single fact, a one-off narrative, or environment-specific noise. Prefer a memory fact when unsure.\n\nYou must output your analysis and writes in the following JSON format inside a \\`\\`\\`json\\`\\`\\` code fence:\n{\n \"rationale\": \"Explanation of your reasoning\",\n \"writes\": [\n { \"kind\": \"memory_add\", \"section\": \"MEMORY\" | \"USER\", \"text\": \"New direct fact to append\" },\n { \"kind\": \"memory_replace\", \"target\": \"Exact text substring to replace\", \"text\": \"New replacement text\" },\n { \"kind\": \"memory_remove\", \"target\": \"Exact text substring to remove\" },\n { \"kind\": \"promote_skill\", \"name\": \"kebab-case-skill-name\", \"description\": \"one line of when to use it\", \"body\": \"Markdown: the step-by-step procedure\" }\n ]\n}\n`;\n\nexport class ReflectionEngine {\n\t/**\n\t * Build the reflection prompt, call the injected isolated complete(),\n\t * parse the response, confront existing memory, and return memory writes.\n\t * Zero direct I/O.\n\t */\n\tasync reflect(input: ReflectionInput): Promise<ReflectionResult> {\n\t\tconst systemPrompt = REFLECTION_SYSTEM_PROMPT;\n\n\t\t// Variable inputs go in the USER prompt so the system prefix above stays cache-stable (#33).\n\t\tconst userPrompt = `Existing Memory snapshot:\n${input.existingMemory}\n\nRecent turn transcript:\n${input.recentTurnText}\n\nAnalyze this turn against the existing memory and output your memory updates.`;\n\n\t\ttry {\n\t\t\tconst compResult = await input.complete(systemPrompt, userPrompt);\n\t\t\tconst text = compResult.text;\n\n\t\t\tconst jsonMatch = text.match(/```json\\s*([\\s\\S]*?)\\s*```/) || text.match(/{[\\s\\S]*}/);\n\t\t\tif (!jsonMatch) {\n\t\t\t\treturn {\n\t\t\t\t\twrites: [],\n\t\t\t\t\tusage: compResult.usage,\n\t\t\t\t\trationale: `Failed to locate JSON response. Raw text:\\n${text}`,\n\t\t\t\t};\n\t\t\t}\n\n\t\t\tconst parsed = JSON.parse(jsonMatch[1] || jsonMatch[0]);\n\t\t\tconst rationale = parsed.rationale || \"\";\n\t\t\tconst writes: ReflectionWrite[] = [];\n\n\t\t\tif (Array.isArray(parsed.writes)) {\n\t\t\t\tfor (const w of parsed.writes) {\n\t\t\t\t\tif (w && typeof w === \"object\") {\n\t\t\t\t\t\tif (\n\t\t\t\t\t\t\tw.kind === \"memory_add\" &&\n\t\t\t\t\t\t\t(w.section === \"MEMORY\" || w.section === \"USER\") &&\n\t\t\t\t\t\t\ttypeof w.text === \"string\"\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\twrites.push({ kind: \"memory_add\", section: w.section, text: w.text });\n\t\t\t\t\t\t} else if (\n\t\t\t\t\t\t\tw.kind === \"memory_replace\" &&\n\t\t\t\t\t\t\ttypeof w.target === \"string\" &&\n\t\t\t\t\t\t\ttypeof w.text === \"string\"\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\twrites.push({ kind: \"memory_replace\", target: w.target, text: w.text });\n\t\t\t\t\t\t} else if (w.kind === \"memory_remove\" && typeof w.target === \"string\") {\n\t\t\t\t\t\t\twrites.push({ kind: \"memory_remove\", target: w.target });\n\t\t\t\t\t\t} else if (\n\t\t\t\t\t\t\tw.kind === \"promote_skill\" &&\n\t\t\t\t\t\t\ttypeof w.name === \"string\" &&\n\t\t\t\t\t\t\ttypeof w.description === \"string\" &&\n\t\t\t\t\t\t\ttypeof w.body === \"string\"\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\twrites.push({ kind: \"promote_skill\", name: w.name, description: w.description, body: w.body });\n\t\t\t\t\t\t}\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}\n\n\t\t\treturn {\n\t\t\t\twrites,\n\t\t\t\tusage: compResult.usage,\n\t\t\t\trationale,\n\t\t\t};\n\t\t} catch (err) {\n\t\t\t// Zeroed/fallback usage representation\n\t\t\tconst emptyUsage: Usage = {\n\t\t\t\tinput: 0,\n\t\t\t\toutput: 0,\n\t\t\t\tcacheRead: 0,\n\t\t\t\tcacheWrite: 0,\n\t\t\t\ttotalTokens: 0,\n\t\t\t\tcost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },\n\t\t\t};\n\t\t\treturn {\n\t\t\t\twrites: [],\n\t\t\t\tusage: emptyUsage,\n\t\t\t\trationale: `Error during reflection: ${String(err)}`,\n\t\t\t};\n\t\t}\n\t}\n}\n"]}
|
|
1
|
+
{"version":3,"file":"reflection-engine.d.ts","sourceRoot":"","sources":["../../../src/core/learning/reflection-engine.ts"],"names":[],"mappings":"AAAA,OAAO,KAAK,EAAE,KAAK,EAAE,MAAM,mBAAmB,CAAC;AAE/C,MAAM,MAAM,UAAU,GAAG,MAAM,GAAG,SAAS,GAAG,SAAS,GAAG,OAAO,GAAG,MAAM,CAAC;AAE3E,MAAM,WAAW,wBAAwB;IACxC,IAAI,EAAE,MAAM,CAAC;IACb,KAAK,EAAE,KAAK,CAAC;IACb,UAAU,EAAE,UAAU,CAAC;CACvB;AAED,MAAM,MAAM,iBAAiB,GAAG,SAAS,GAAG,YAAY,GAAG,aAAa,GAAG,MAAM,CAAC;AAElF,MAAM,WAAW,aAAa;IAC7B,OAAO,EAAE,iBAAiB,CAAC;IAC3B,aAAa,EAAE,MAAM,CAAC;IACtB,aAAa,EAAE,OAAO,CAAC;IACvB,kBAAkB,EAAE,MAAM,CAAC;IAC3B,YAAY,EAAE,MAAM,CAAC;CACrB;AAED,MAAM,WAAW,UAAU;IAC1B,GAAG,EAAE,MAAM,GAAG,SAAS,CAAC;IACxB,MAAM,EAAE,MAAM,CAAC;IACf,WAAW,EAAE,MAAM,CAAC;CACpB;AAED;;;GAGG;AACH,wBAAgB,YAAY,CAAC,OAAO,EAAE,aAAa,GAAG,UAAU,CAyB/D;AAED,MAAM,WAAW,eAAe;IAC/B,cAAc,EAAE,MAAM,CAAC;IACvB,cAAc,EAAE,MAAM,CAAC;IACvB,IAAI,EAAE,UAAU,CAAC;IAEjB,QAAQ,EAAE,CAAC,YAAY,EAAE,MAAM,EAAE,UAAU,EAAE,MAAM,KAAK,OAAO,CAAC,wBAAwB,CAAC,CAAC;CAC1F;AAED,MAAM,MAAM,eAAe,GACxB;IAAE,IAAI,EAAE,YAAY,CAAC;IAAC,OAAO,EAAE,QAAQ,GAAG,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GAChE;IAAE,IAAI,EAAE,gBAAgB,CAAC;IAAC,MAAM,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,GACxD;IAAE,IAAI,EAAE,eAAe,CAAC;IAAC,MAAM,EAAE,MAAM,CAAA;CAAE,GAEzC;IAAE,IAAI,EAAE,eAAe,CAAC;IAAC,IAAI,EAAE,MAAM,CAAC;IAAC,WAAW,EAAE,MAAM,CAAC;IAAC,IAAI,EAAE,MAAM,CAAA;CAAE,CAAC;AAE9E,MAAM,WAAW,gBAAgB;IAChC,MAAM,EAAE,eAAe,EAAE,CAAC;IAC1B,KAAK,EAAE,KAAK,CAAC;IACb,SAAS,EAAE,MAAM,CAAC;CAClB;AAED;;;;;GAKG;AACH,eAAO,MAAM,wBAAwB,s4DAqBpC,CAAC;AAEF,qBAAa,gBAAgB;IAC5B;;;;OAIG;IACG,OAAO,CAAC,KAAK,EAAE,eAAe,GAAG,OAAO,CAAC,gBAAgB,CAAC,CAiF/D;CACD","sourcesContent":["import type { Usage } from \"@caupulican/pi-ai\";\n\nexport type StopReason = \"stop\" | \"toolUse\" | \"aborted\" | \"error\" | string;\n\nexport interface IsolatedCompletionResult {\n\ttext: string;\n\tusage: Usage;\n\tstopReason: StopReason;\n}\n\nexport type ReflectionTrigger = \"complex\" | \"corrective\" | \"session-end\" | \"none\";\n\nexport interface DemandSignals {\n\ttrigger: ReflectionTrigger;\n\ttoolCallCount: number;\n\thadCorrection: boolean;\n\tcontextHeadroomPct: number; // 0..100\n\tusefulLately: number; // 0..1 rolling score\n}\n\nexport interface DemandPlan {\n\tact: \"skip\" | \"reflect\";\n\treason: string;\n\ttokenBudget: number;\n}\n\n/**\n * Pure zero-I/O heuristic to decide whether the current turn justifies a reflection run\n * and determine the token budget under the cheap-tool net-negative doctrine.\n */\nexport function decideDemand(signals: DemandSignals): DemandPlan {\n\tif (signals.trigger === \"none\") {\n\t\treturn { act: \"skip\", reason: \"No trigger detected\", tokenBudget: 0 };\n\t}\n\tif (signals.contextHeadroomPct < 10) {\n\t\treturn { act: \"skip\", reason: \"Context headroom is critically low (< 10%)\", tokenBudget: 0 };\n\t}\n\n\t// Dynamic token budget based on headroom (keep reflection bounded between 500 and 1500 tokens)\n\tconst baseBudget = 1000;\n\tconst tokenBudget = Math.max(500, Math.min(1500, Math.round(baseBudget * (signals.contextHeadroomPct / 100))));\n\n\tif (signals.hadCorrection) {\n\t\treturn { act: \"reflect\", reason: \"Correction detected in the turn\", tokenBudget };\n\t}\n\tif (signals.trigger === \"session-end\") {\n\t\treturn { act: \"reflect\", reason: \"Session end reflection triggered\", tokenBudget };\n\t}\n\tif (signals.trigger === \"complex\") {\n\t\tif (signals.toolCallCount >= 3) {\n\t\t\treturn { act: \"reflect\", reason: `Complex turn with ${signals.toolCallCount} tool calls`, tokenBudget };\n\t\t}\n\t}\n\n\treturn { act: \"skip\", reason: \"Signals do not justify reflection overhead\", tokenBudget: 0 };\n}\n\nexport interface ReflectionInput {\n\trecentTurnText: string; // host serializes the just-finished turn\n\texistingMemory: string; // current MEMORY.md + USER.md snapshot\n\tplan: DemandPlan;\n\t// host-injected isolated completion function:\n\tcomplete: (systemPrompt: string, userPrompt: string) => Promise<IsolatedCompletionResult>;\n}\n\nexport type ReflectionWrite =\n\t| { kind: \"memory_add\"; section: \"MEMORY\" | \"USER\"; text: string }\n\t| { kind: \"memory_replace\"; target: string; text: string }\n\t| { kind: \"memory_remove\"; target: string }\n\t// R7 memory-to-behavior: promote a recurring procedural workflow into an executable skill.\n\t| { kind: \"promote_skill\"; name: string; description: string; body: string };\n\nexport interface ReflectionResult {\n\twrites: ReflectionWrite[];\n\tusage: Usage;\n\trationale: string;\n}\n\n/**\n * STATIC reflection system prompt (Hermes-parity #33). It is byte-identical across every reflection\n * pass — the variable parts (existing memory snapshot + the turn transcript) live in the USER prompt —\n * so the provider prompt-cache reuses this prefix instead of re-billing it each pass (cost guard).\n * Do NOT interpolate per-call data into this constant or caching breaks.\n */\nexport const REFLECTION_SYSTEM_PROMPT = `You are a reflection engine. Your job is to analyze the recent conversation turn, compare it against the agent's existing memory, and decide if any memory updates are needed.\n\nMemory guidelines:\n- \"MEMORY\" is for project facts, configuration, repeatable workflows, and coding findings.\n- \"USER\" is for user preferences, patterns, and style specifications.\n- Avoid duplicate facts. If the fact is already represented, do not add it.\n- CONFRONT existing memory: if the new turn contradicts or updates an existing fact, use \"memory_replace\" or \"memory_remove\" to supersede the old fact rather than blindly appending.\n- Keep memories short, factual, and direct. No fluff.\n- Do NOT capture transient/environment-specific noise: tool/network failures, one-off errors, or a single narrative event. Persist only durable facts and preferences.\n- PROMOTE to behavior: if the turn established a REPEATABLE, multi-step PROCEDURE/workflow (not a one-off fact) that should govern a future class of tasks, emit a \"promote_skill\" instead of (or in addition to) a memory fact. Only promote a genuinely reusable procedure — never a single fact, a one-off narrative, or environment-specific noise. Prefer a memory fact when unsure.\n\nYou must output your analysis and writes in the following JSON format inside a \\`\\`\\`json\\`\\`\\` code fence:\n{\n \"rationale\": \"Explanation of your reasoning\",\n \"writes\": [\n { \"kind\": \"memory_add\", \"section\": \"MEMORY\" | \"USER\", \"text\": \"New direct fact to append\" },\n { \"kind\": \"memory_replace\", \"target\": \"Exact text substring to replace\", \"text\": \"New replacement text\" },\n { \"kind\": \"memory_remove\", \"target\": \"Exact text substring to remove\" },\n { \"kind\": \"promote_skill\", \"name\": \"kebab-case-skill-name\", \"description\": \"one line of when to use it\", \"body\": \"Markdown: the step-by-step procedure\" }\n ]\n}\n`;\n\nexport class ReflectionEngine {\n\t/**\n\t * Build the reflection prompt, call the injected isolated complete(),\n\t * parse the response, confront existing memory, and return memory writes.\n\t * Zero direct I/O.\n\t */\n\tasync reflect(input: ReflectionInput): Promise<ReflectionResult> {\n\t\tconst systemPrompt = REFLECTION_SYSTEM_PROMPT;\n\n\t\t// Variable inputs go in the USER prompt so the system prefix above stays cache-stable (#33).\n\t\tconst userPrompt = `Existing Memory snapshot:\n${input.existingMemory}\n\nRecent turn transcript:\n${input.recentTurnText}\n\nAnalyze this turn against the existing memory and output your memory updates.`;\n\n\t\tlet usage: Usage | undefined;\n\t\ttry {\n\t\t\tconst compResult = await input.complete(systemPrompt, userPrompt);\n\t\t\tusage = compResult.usage;\n\t\t\tconst text = compResult.text;\n\n\t\t\tconst jsonMatch = text.match(/```json\\s*([\\s\\S]*?)\\s*```/) || text.match(/{[\\s\\S]*}/);\n\t\t\tif (!jsonMatch) {\n\t\t\t\treturn {\n\t\t\t\t\twrites: [],\n\t\t\t\t\tusage: compResult.usage,\n\t\t\t\t\trationale: `Failed to locate JSON response. Raw text:\\n${text}`,\n\t\t\t\t};\n\t\t\t}\n\n\t\t\tconst parsed = JSON.parse(jsonMatch[1] || jsonMatch[0]);\n\t\t\tconst rationale = parsed.rationale || \"\";\n\t\t\tconst writes: ReflectionWrite[] = [];\n\n\t\t\tif (Array.isArray(parsed.writes)) {\n\t\t\t\tfor (const w of parsed.writes) {\n\t\t\t\t\tif (w && typeof w === \"object\") {\n\t\t\t\t\t\tif (\n\t\t\t\t\t\t\tw.kind === \"memory_add\" &&\n\t\t\t\t\t\t\t(w.section === \"MEMORY\" || w.section === \"USER\") &&\n\t\t\t\t\t\t\ttypeof w.text === \"string\"\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\twrites.push({ kind: \"memory_add\", section: w.section, text: w.text });\n\t\t\t\t\t\t} else if (\n\t\t\t\t\t\t\tw.kind === \"memory_replace\" &&\n\t\t\t\t\t\t\ttypeof w.target === \"string\" &&\n\t\t\t\t\t\t\ttypeof w.text === \"string\"\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\twrites.push({ kind: \"memory_replace\", target: w.target, text: w.text });\n\t\t\t\t\t\t} else if (w.kind === \"memory_remove\" && typeof w.target === \"string\") {\n\t\t\t\t\t\t\twrites.push({ kind: \"memory_remove\", target: w.target });\n\t\t\t\t\t\t} else if (\n\t\t\t\t\t\t\tw.kind === \"promote_skill\" &&\n\t\t\t\t\t\t\ttypeof w.name === \"string\" &&\n\t\t\t\t\t\t\ttypeof w.description === \"string\" &&\n\t\t\t\t\t\t\ttypeof w.body === \"string\"\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\twrites.push({ kind: \"promote_skill\", name: w.name, description: w.description, body: w.body });\n\t\t\t\t\t\t}\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}\n\n\t\t\treturn {\n\t\t\t\twrites,\n\t\t\t\tusage: compResult.usage,\n\t\t\t\trationale,\n\t\t\t};\n\t\t} catch (err) {\n\t\t\t// Zeroed/fallback usage representation\n\t\t\tconst emptyUsage: Usage = {\n\t\t\t\tinput: 0,\n\t\t\t\toutput: 0,\n\t\t\t\tcacheRead: 0,\n\t\t\t\tcacheWrite: 0,\n\t\t\t\ttotalTokens: 0,\n\t\t\t\tcost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },\n\t\t\t};\n\t\t\treturn {\n\t\t\t\twrites: [],\n\t\t\t\tusage: usage ?? emptyUsage,\n\t\t\t\trationale: `Error during reflection: ${String(err)}`,\n\t\t\t};\n\t\t}\n\t}\n}\n"]}
|
|
@@ -69,8 +69,10 @@ Recent turn transcript:
|
|
|
69
69
|
${input.recentTurnText}
|
|
70
70
|
|
|
71
71
|
Analyze this turn against the existing memory and output your memory updates.`;
|
|
72
|
+
let usage;
|
|
72
73
|
try {
|
|
73
74
|
const compResult = await input.complete(systemPrompt, userPrompt);
|
|
75
|
+
usage = compResult.usage;
|
|
74
76
|
const text = compResult.text;
|
|
75
77
|
const jsonMatch = text.match(/```json\s*([\s\S]*?)\s*```/) || text.match(/{[\s\S]*}/);
|
|
76
78
|
if (!jsonMatch) {
|
|
@@ -126,7 +128,7 @@ Analyze this turn against the existing memory and output your memory updates.`;
|
|
|
126
128
|
};
|
|
127
129
|
return {
|
|
128
130
|
writes: [],
|
|
129
|
-
usage: emptyUsage,
|
|
131
|
+
usage: usage ?? emptyUsage,
|
|
130
132
|
rationale: `Error during reflection: ${String(err)}`,
|
|
131
133
|
};
|
|
132
134
|
}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"reflection-engine.js","sourceRoot":"","sources":["../../../src/core/learning/reflection-engine.ts"],"names":[],"mappings":"AA0BA;;;GAGG;AACH,MAAM,UAAU,YAAY,CAAC,OAAsB,EAAc;IAChE,IAAI,OAAO,CAAC,OAAO,KAAK,MAAM,EAAE,CAAC;QAChC,OAAO,EAAE,GAAG,EAAE,MAAM,EAAE,MAAM,EAAE,qBAAqB,EAAE,WAAW,EAAE,CAAC,EAAE,CAAC;IACvE,CAAC;IACD,IAAI,OAAO,CAAC,kBAAkB,GAAG,EAAE,EAAE,CAAC;QACrC,OAAO,EAAE,GAAG,EAAE,MAAM,EAAE,MAAM,EAAE,4CAA4C,EAAE,WAAW,EAAE,CAAC,EAAE,CAAC;IAC9F,CAAC;IAED,+FAA+F;IAC/F,MAAM,UAAU,GAAG,IAAI,CAAC;IACxB,MAAM,WAAW,GAAG,IAAI,CAAC,GAAG,CAAC,GAAG,EAAE,IAAI,CAAC,GAAG,CAAC,IAAI,EAAE,IAAI,CAAC,KAAK,CAAC,UAAU,GAAG,CAAC,OAAO,CAAC,kBAAkB,GAAG,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC;IAE/G,IAAI,OAAO,CAAC,aAAa,EAAE,CAAC;QAC3B,OAAO,EAAE,GAAG,EAAE,SAAS,EAAE,MAAM,EAAE,iCAAiC,EAAE,WAAW,EAAE,CAAC;IACnF,CAAC;IACD,IAAI,OAAO,CAAC,OAAO,KAAK,aAAa,EAAE,CAAC;QACvC,OAAO,EAAE,GAAG,EAAE,SAAS,EAAE,MAAM,EAAE,kCAAkC,EAAE,WAAW,EAAE,CAAC;IACpF,CAAC;IACD,IAAI,OAAO,CAAC,OAAO,KAAK,SAAS,EAAE,CAAC;QACnC,IAAI,OAAO,CAAC,aAAa,IAAI,CAAC,EAAE,CAAC;YAChC,OAAO,EAAE,GAAG,EAAE,SAAS,EAAE,MAAM,EAAE,qBAAqB,OAAO,CAAC,aAAa,aAAa,EAAE,WAAW,EAAE,CAAC;QACzG,CAAC;IACF,CAAC;IAED,OAAO,EAAE,GAAG,EAAE,MAAM,EAAE,MAAM,EAAE,4CAA4C,EAAE,WAAW,EAAE,CAAC,EAAE,CAAC;AAAA,CAC7F;AAuBD;;;;;GAKG;AACH,MAAM,CAAC,MAAM,wBAAwB,GAAG;;;;;;;;;;;;;;;;;;;;;CAqBvC,CAAC;AAEF,MAAM,OAAO,gBAAgB;IAC5B;;;;OAIG;IACH,KAAK,CAAC,OAAO,CAAC,KAAsB,EAA6B;QAChE,MAAM,YAAY,GAAG,wBAAwB,CAAC;QAE9C,6FAA6F;QAC7F,MAAM,UAAU,GAAG;EACnB,KAAK,CAAC,cAAc;;;EAGpB,KAAK,CAAC,cAAc;;8EAEwD,CAAC;QAE7E,IAAI,CAAC;YACJ,MAAM,UAAU,GAAG,MAAM,KAAK,CAAC,QAAQ,CAAC,YAAY,EAAE,UAAU,CAAC,CAAC;YAClE,MAAM,IAAI,GAAG,UAAU,CAAC,IAAI,CAAC;YAE7B,MAAM,SAAS,GAAG,IAAI,CAAC,KAAK,CAAC,4BAA4B,CAAC,IAAI,IAAI,CAAC,KAAK,CAAC,WAAW,CAAC,CAAC;YACtF,IAAI,CAAC,SAAS,EAAE,CAAC;gBAChB,OAAO;oBACN,MAAM,EAAE,EAAE;oBACV,KAAK,EAAE,UAAU,CAAC,KAAK;oBACvB,SAAS,EAAE,8CAA8C,IAAI,EAAE;iBAC/D,CAAC;YACH,CAAC;YAED,MAAM,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,SAAS,CAAC,CAAC,CAAC,IAAI,SAAS,CAAC,CAAC,CAAC,CAAC,CAAC;YACxD,MAAM,SAAS,GAAG,MAAM,CAAC,SAAS,IAAI,EAAE,CAAC;YACzC,MAAM,MAAM,GAAsB,EAAE,CAAC;YAErC,IAAI,KAAK,CAAC,OAAO,CAAC,MAAM,CAAC,MAAM,CAAC,EAAE,CAAC;gBAClC,KAAK,MAAM,CAAC,IAAI,MAAM,CAAC,MAAM,EAAE,CAAC;oBAC/B,IAAI,CAAC,IAAI,OAAO,CAAC,KAAK,QAAQ,EAAE,CAAC;wBAChC,IACC,CAAC,CAAC,IAAI,KAAK,YAAY;4BACvB,CAAC,CAAC,CAAC,OAAO,KAAK,QAAQ,IAAI,CAAC,CAAC,OAAO,KAAK,MAAM,CAAC;4BAChD,OAAO,CAAC,CAAC,IAAI,KAAK,QAAQ,EACzB,CAAC;4BACF,MAAM,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,YAAY,EAAE,OAAO,EAAE,CAAC,CAAC,OAAO,EAAE,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC;wBACvE,CAAC;6BAAM,IACN,CAAC,CAAC,IAAI,KAAK,gBAAgB;4BAC3B,OAAO,CAAC,CAAC,MAAM,KAAK,QAAQ;4BAC5B,OAAO,CAAC,CAAC,IAAI,KAAK,QAAQ,EACzB,CAAC;4BACF,MAAM,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,gBAAgB,EAAE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC;wBACzE,CAAC;6BAAM,IAAI,CAAC,CAAC,IAAI,KAAK,eAAe,IAAI,OAAO,CAAC,CAAC,MAAM,KAAK,QAAQ,EAAE,CAAC;4BACvE,MAAM,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,eAAe,EAAE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,CAAC;wBAC1D,CAAC;6BAAM,IACN,CAAC,CAAC,IAAI,KAAK,eAAe;4BAC1B,OAAO,CAAC,CAAC,IAAI,KAAK,QAAQ;4BAC1B,OAAO,CAAC,CAAC,WAAW,KAAK,QAAQ;4BACjC,OAAO,CAAC,CAAC,IAAI,KAAK,QAAQ,EACzB,CAAC;4BACF,MAAM,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,eAAe,EAAE,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,WAAW,EAAE,CAAC,CAAC,WAAW,EAAE,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC;wBAChG,CAAC;oBACF,CAAC;gBACF,CAAC;YACF,CAAC;YAED,OAAO;gBACN,MAAM;gBACN,KAAK,EAAE,UAAU,CAAC,KAAK;gBACvB,SAAS;aACT,CAAC;QACH,CAAC;QAAC,OAAO,GAAG,EAAE,CAAC;YACd,uCAAuC;YACvC,MAAM,UAAU,GAAU;gBACzB,KAAK,EAAE,CAAC;gBACR,MAAM,EAAE,CAAC;gBACT,SAAS,EAAE,CAAC;gBACZ,UAAU,EAAE,CAAC;gBACb,WAAW,EAAE,CAAC;gBACd,IAAI,EAAE,EAAE,KAAK,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,SAAS,EAAE,CAAC,EAAE,UAAU,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE;aACpE,CAAC;YACF,OAAO;gBACN,MAAM,EAAE,EAAE;gBACV,KAAK,EAAE,UAAU;gBACjB,SAAS,EAAE,4BAA4B,MAAM,CAAC,GAAG,CAAC,EAAE;aACpD,CAAC;QACH,CAAC;IAAA,CACD;CACD","sourcesContent":["import type { Usage } from \"@caupulican/pi-ai\";\n\nexport type StopReason = \"stop\" | \"toolUse\" | \"aborted\" | \"error\" | string;\n\nexport interface IsolatedCompletionResult {\n\ttext: string;\n\tusage: Usage;\n\tstopReason: StopReason;\n}\n\nexport type ReflectionTrigger = \"complex\" | \"corrective\" | \"session-end\" | \"none\";\n\nexport interface DemandSignals {\n\ttrigger: ReflectionTrigger;\n\ttoolCallCount: number;\n\thadCorrection: boolean;\n\tcontextHeadroomPct: number; // 0..100\n\tusefulLately: number; // 0..1 rolling score\n}\n\nexport interface DemandPlan {\n\tact: \"skip\" | \"reflect\";\n\treason: string;\n\ttokenBudget: number;\n}\n\n/**\n * Pure zero-I/O heuristic to decide whether the current turn justifies a reflection run\n * and determine the token budget under the cheap-tool net-negative doctrine.\n */\nexport function decideDemand(signals: DemandSignals): DemandPlan {\n\tif (signals.trigger === \"none\") {\n\t\treturn { act: \"skip\", reason: \"No trigger detected\", tokenBudget: 0 };\n\t}\n\tif (signals.contextHeadroomPct < 10) {\n\t\treturn { act: \"skip\", reason: \"Context headroom is critically low (< 10%)\", tokenBudget: 0 };\n\t}\n\n\t// Dynamic token budget based on headroom (keep reflection bounded between 500 and 1500 tokens)\n\tconst baseBudget = 1000;\n\tconst tokenBudget = Math.max(500, Math.min(1500, Math.round(baseBudget * (signals.contextHeadroomPct / 100))));\n\n\tif (signals.hadCorrection) {\n\t\treturn { act: \"reflect\", reason: \"Correction detected in the turn\", tokenBudget };\n\t}\n\tif (signals.trigger === \"session-end\") {\n\t\treturn { act: \"reflect\", reason: \"Session end reflection triggered\", tokenBudget };\n\t}\n\tif (signals.trigger === \"complex\") {\n\t\tif (signals.toolCallCount >= 3) {\n\t\t\treturn { act: \"reflect\", reason: `Complex turn with ${signals.toolCallCount} tool calls`, tokenBudget };\n\t\t}\n\t}\n\n\treturn { act: \"skip\", reason: \"Signals do not justify reflection overhead\", tokenBudget: 0 };\n}\n\nexport interface ReflectionInput {\n\trecentTurnText: string; // host serializes the just-finished turn\n\texistingMemory: string; // current MEMORY.md + USER.md snapshot\n\tplan: DemandPlan;\n\t// host-injected isolated completion function:\n\tcomplete: (systemPrompt: string, userPrompt: string) => Promise<IsolatedCompletionResult>;\n}\n\nexport type ReflectionWrite =\n\t| { kind: \"memory_add\"; section: \"MEMORY\" | \"USER\"; text: string }\n\t| { kind: \"memory_replace\"; target: string; text: string }\n\t| { kind: \"memory_remove\"; target: string }\n\t// R7 memory-to-behavior: promote a recurring procedural workflow into an executable skill.\n\t| { kind: \"promote_skill\"; name: string; description: string; body: string };\n\nexport interface ReflectionResult {\n\twrites: ReflectionWrite[];\n\tusage: Usage;\n\trationale: string;\n}\n\n/**\n * STATIC reflection system prompt (Hermes-parity #33). It is byte-identical across every reflection\n * pass — the variable parts (existing memory snapshot + the turn transcript) live in the USER prompt —\n * so the provider prompt-cache reuses this prefix instead of re-billing it each pass (cost guard).\n * Do NOT interpolate per-call data into this constant or caching breaks.\n */\nexport const REFLECTION_SYSTEM_PROMPT = `You are a reflection engine. Your job is to analyze the recent conversation turn, compare it against the agent's existing memory, and decide if any memory updates are needed.\n\nMemory guidelines:\n- \"MEMORY\" is for project facts, configuration, repeatable workflows, and coding findings.\n- \"USER\" is for user preferences, patterns, and style specifications.\n- Avoid duplicate facts. If the fact is already represented, do not add it.\n- CONFRONT existing memory: if the new turn contradicts or updates an existing fact, use \"memory_replace\" or \"memory_remove\" to supersede the old fact rather than blindly appending.\n- Keep memories short, factual, and direct. No fluff.\n- Do NOT capture transient/environment-specific noise: tool/network failures, one-off errors, or a single narrative event. Persist only durable facts and preferences.\n- PROMOTE to behavior: if the turn established a REPEATABLE, multi-step PROCEDURE/workflow (not a one-off fact) that should govern a future class of tasks, emit a \"promote_skill\" instead of (or in addition to) a memory fact. Only promote a genuinely reusable procedure — never a single fact, a one-off narrative, or environment-specific noise. Prefer a memory fact when unsure.\n\nYou must output your analysis and writes in the following JSON format inside a \\`\\`\\`json\\`\\`\\` code fence:\n{\n \"rationale\": \"Explanation of your reasoning\",\n \"writes\": [\n { \"kind\": \"memory_add\", \"section\": \"MEMORY\" | \"USER\", \"text\": \"New direct fact to append\" },\n { \"kind\": \"memory_replace\", \"target\": \"Exact text substring to replace\", \"text\": \"New replacement text\" },\n { \"kind\": \"memory_remove\", \"target\": \"Exact text substring to remove\" },\n { \"kind\": \"promote_skill\", \"name\": \"kebab-case-skill-name\", \"description\": \"one line of when to use it\", \"body\": \"Markdown: the step-by-step procedure\" }\n ]\n}\n`;\n\nexport class ReflectionEngine {\n\t/**\n\t * Build the reflection prompt, call the injected isolated complete(),\n\t * parse the response, confront existing memory, and return memory writes.\n\t * Zero direct I/O.\n\t */\n\tasync reflect(input: ReflectionInput): Promise<ReflectionResult> {\n\t\tconst systemPrompt = REFLECTION_SYSTEM_PROMPT;\n\n\t\t// Variable inputs go in the USER prompt so the system prefix above stays cache-stable (#33).\n\t\tconst userPrompt = `Existing Memory snapshot:\n${input.existingMemory}\n\nRecent turn transcript:\n${input.recentTurnText}\n\nAnalyze this turn against the existing memory and output your memory updates.`;\n\n\t\ttry {\n\t\t\tconst compResult = await input.complete(systemPrompt, userPrompt);\n\t\t\tconst text = compResult.text;\n\n\t\t\tconst jsonMatch = text.match(/```json\\s*([\\s\\S]*?)\\s*```/) || text.match(/{[\\s\\S]*}/);\n\t\t\tif (!jsonMatch) {\n\t\t\t\treturn {\n\t\t\t\t\twrites: [],\n\t\t\t\t\tusage: compResult.usage,\n\t\t\t\t\trationale: `Failed to locate JSON response. Raw text:\\n${text}`,\n\t\t\t\t};\n\t\t\t}\n\n\t\t\tconst parsed = JSON.parse(jsonMatch[1] || jsonMatch[0]);\n\t\t\tconst rationale = parsed.rationale || \"\";\n\t\t\tconst writes: ReflectionWrite[] = [];\n\n\t\t\tif (Array.isArray(parsed.writes)) {\n\t\t\t\tfor (const w of parsed.writes) {\n\t\t\t\t\tif (w && typeof w === \"object\") {\n\t\t\t\t\t\tif (\n\t\t\t\t\t\t\tw.kind === \"memory_add\" &&\n\t\t\t\t\t\t\t(w.section === \"MEMORY\" || w.section === \"USER\") &&\n\t\t\t\t\t\t\ttypeof w.text === \"string\"\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\twrites.push({ kind: \"memory_add\", section: w.section, text: w.text });\n\t\t\t\t\t\t} else if (\n\t\t\t\t\t\t\tw.kind === \"memory_replace\" &&\n\t\t\t\t\t\t\ttypeof w.target === \"string\" &&\n\t\t\t\t\t\t\ttypeof w.text === \"string\"\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\twrites.push({ kind: \"memory_replace\", target: w.target, text: w.text });\n\t\t\t\t\t\t} else if (w.kind === \"memory_remove\" && typeof w.target === \"string\") {\n\t\t\t\t\t\t\twrites.push({ kind: \"memory_remove\", target: w.target });\n\t\t\t\t\t\t} else if (\n\t\t\t\t\t\t\tw.kind === \"promote_skill\" &&\n\t\t\t\t\t\t\ttypeof w.name === \"string\" &&\n\t\t\t\t\t\t\ttypeof w.description === \"string\" &&\n\t\t\t\t\t\t\ttypeof w.body === \"string\"\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\twrites.push({ kind: \"promote_skill\", name: w.name, description: w.description, body: w.body });\n\t\t\t\t\t\t}\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}\n\n\t\t\treturn {\n\t\t\t\twrites,\n\t\t\t\tusage: compResult.usage,\n\t\t\t\trationale,\n\t\t\t};\n\t\t} catch (err) {\n\t\t\t// Zeroed/fallback usage representation\n\t\t\tconst emptyUsage: Usage = {\n\t\t\t\tinput: 0,\n\t\t\t\toutput: 0,\n\t\t\t\tcacheRead: 0,\n\t\t\t\tcacheWrite: 0,\n\t\t\t\ttotalTokens: 0,\n\t\t\t\tcost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },\n\t\t\t};\n\t\t\treturn {\n\t\t\t\twrites: [],\n\t\t\t\tusage: emptyUsage,\n\t\t\t\trationale: `Error during reflection: ${String(err)}`,\n\t\t\t};\n\t\t}\n\t}\n}\n"]}
|
|
1
|
+
{"version":3,"file":"reflection-engine.js","sourceRoot":"","sources":["../../../src/core/learning/reflection-engine.ts"],"names":[],"mappings":"AA0BA;;;GAGG;AACH,MAAM,UAAU,YAAY,CAAC,OAAsB,EAAc;IAChE,IAAI,OAAO,CAAC,OAAO,KAAK,MAAM,EAAE,CAAC;QAChC,OAAO,EAAE,GAAG,EAAE,MAAM,EAAE,MAAM,EAAE,qBAAqB,EAAE,WAAW,EAAE,CAAC,EAAE,CAAC;IACvE,CAAC;IACD,IAAI,OAAO,CAAC,kBAAkB,GAAG,EAAE,EAAE,CAAC;QACrC,OAAO,EAAE,GAAG,EAAE,MAAM,EAAE,MAAM,EAAE,4CAA4C,EAAE,WAAW,EAAE,CAAC,EAAE,CAAC;IAC9F,CAAC;IAED,+FAA+F;IAC/F,MAAM,UAAU,GAAG,IAAI,CAAC;IACxB,MAAM,WAAW,GAAG,IAAI,CAAC,GAAG,CAAC,GAAG,EAAE,IAAI,CAAC,GAAG,CAAC,IAAI,EAAE,IAAI,CAAC,KAAK,CAAC,UAAU,GAAG,CAAC,OAAO,CAAC,kBAAkB,GAAG,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC;IAE/G,IAAI,OAAO,CAAC,aAAa,EAAE,CAAC;QAC3B,OAAO,EAAE,GAAG,EAAE,SAAS,EAAE,MAAM,EAAE,iCAAiC,EAAE,WAAW,EAAE,CAAC;IACnF,CAAC;IACD,IAAI,OAAO,CAAC,OAAO,KAAK,aAAa,EAAE,CAAC;QACvC,OAAO,EAAE,GAAG,EAAE,SAAS,EAAE,MAAM,EAAE,kCAAkC,EAAE,WAAW,EAAE,CAAC;IACpF,CAAC;IACD,IAAI,OAAO,CAAC,OAAO,KAAK,SAAS,EAAE,CAAC;QACnC,IAAI,OAAO,CAAC,aAAa,IAAI,CAAC,EAAE,CAAC;YAChC,OAAO,EAAE,GAAG,EAAE,SAAS,EAAE,MAAM,EAAE,qBAAqB,OAAO,CAAC,aAAa,aAAa,EAAE,WAAW,EAAE,CAAC;QACzG,CAAC;IACF,CAAC;IAED,OAAO,EAAE,GAAG,EAAE,MAAM,EAAE,MAAM,EAAE,4CAA4C,EAAE,WAAW,EAAE,CAAC,EAAE,CAAC;AAAA,CAC7F;AAuBD;;;;;GAKG;AACH,MAAM,CAAC,MAAM,wBAAwB,GAAG;;;;;;;;;;;;;;;;;;;;;CAqBvC,CAAC;AAEF,MAAM,OAAO,gBAAgB;IAC5B;;;;OAIG;IACH,KAAK,CAAC,OAAO,CAAC,KAAsB,EAA6B;QAChE,MAAM,YAAY,GAAG,wBAAwB,CAAC;QAE9C,6FAA6F;QAC7F,MAAM,UAAU,GAAG;EACnB,KAAK,CAAC,cAAc;;;EAGpB,KAAK,CAAC,cAAc;;8EAEwD,CAAC;QAE7E,IAAI,KAAwB,CAAC;QAC7B,IAAI,CAAC;YACJ,MAAM,UAAU,GAAG,MAAM,KAAK,CAAC,QAAQ,CAAC,YAAY,EAAE,UAAU,CAAC,CAAC;YAClE,KAAK,GAAG,UAAU,CAAC,KAAK,CAAC;YACzB,MAAM,IAAI,GAAG,UAAU,CAAC,IAAI,CAAC;YAE7B,MAAM,SAAS,GAAG,IAAI,CAAC,KAAK,CAAC,4BAA4B,CAAC,IAAI,IAAI,CAAC,KAAK,CAAC,WAAW,CAAC,CAAC;YACtF,IAAI,CAAC,SAAS,EAAE,CAAC;gBAChB,OAAO;oBACN,MAAM,EAAE,EAAE;oBACV,KAAK,EAAE,UAAU,CAAC,KAAK;oBACvB,SAAS,EAAE,8CAA8C,IAAI,EAAE;iBAC/D,CAAC;YACH,CAAC;YAED,MAAM,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,SAAS,CAAC,CAAC,CAAC,IAAI,SAAS,CAAC,CAAC,CAAC,CAAC,CAAC;YACxD,MAAM,SAAS,GAAG,MAAM,CAAC,SAAS,IAAI,EAAE,CAAC;YACzC,MAAM,MAAM,GAAsB,EAAE,CAAC;YAErC,IAAI,KAAK,CAAC,OAAO,CAAC,MAAM,CAAC,MAAM,CAAC,EAAE,CAAC;gBAClC,KAAK,MAAM,CAAC,IAAI,MAAM,CAAC,MAAM,EAAE,CAAC;oBAC/B,IAAI,CAAC,IAAI,OAAO,CAAC,KAAK,QAAQ,EAAE,CAAC;wBAChC,IACC,CAAC,CAAC,IAAI,KAAK,YAAY;4BACvB,CAAC,CAAC,CAAC,OAAO,KAAK,QAAQ,IAAI,CAAC,CAAC,OAAO,KAAK,MAAM,CAAC;4BAChD,OAAO,CAAC,CAAC,IAAI,KAAK,QAAQ,EACzB,CAAC;4BACF,MAAM,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,YAAY,EAAE,OAAO,EAAE,CAAC,CAAC,OAAO,EAAE,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC;wBACvE,CAAC;6BAAM,IACN,CAAC,CAAC,IAAI,KAAK,gBAAgB;4BAC3B,OAAO,CAAC,CAAC,MAAM,KAAK,QAAQ;4BAC5B,OAAO,CAAC,CAAC,IAAI,KAAK,QAAQ,EACzB,CAAC;4BACF,MAAM,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,gBAAgB,EAAE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC;wBACzE,CAAC;6BAAM,IAAI,CAAC,CAAC,IAAI,KAAK,eAAe,IAAI,OAAO,CAAC,CAAC,MAAM,KAAK,QAAQ,EAAE,CAAC;4BACvE,MAAM,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,eAAe,EAAE,MAAM,EAAE,CAAC,CAAC,MAAM,EAAE,CAAC,CAAC;wBAC1D,CAAC;6BAAM,IACN,CAAC,CAAC,IAAI,KAAK,eAAe;4BAC1B,OAAO,CAAC,CAAC,IAAI,KAAK,QAAQ;4BAC1B,OAAO,CAAC,CAAC,WAAW,KAAK,QAAQ;4BACjC,OAAO,CAAC,CAAC,IAAI,KAAK,QAAQ,EACzB,CAAC;4BACF,MAAM,CAAC,IAAI,CAAC,EAAE,IAAI,EAAE,eAAe,EAAE,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,WAAW,EAAE,CAAC,CAAC,WAAW,EAAE,IAAI,EAAE,CAAC,CAAC,IAAI,EAAE,CAAC,CAAC;wBAChG,CAAC;oBACF,CAAC;gBACF,CAAC;YACF,CAAC;YAED,OAAO;gBACN,MAAM;gBACN,KAAK,EAAE,UAAU,CAAC,KAAK;gBACvB,SAAS;aACT,CAAC;QACH,CAAC;QAAC,OAAO,GAAG,EAAE,CAAC;YACd,uCAAuC;YACvC,MAAM,UAAU,GAAU;gBACzB,KAAK,EAAE,CAAC;gBACR,MAAM,EAAE,CAAC;gBACT,SAAS,EAAE,CAAC;gBACZ,UAAU,EAAE,CAAC;gBACb,WAAW,EAAE,CAAC;gBACd,IAAI,EAAE,EAAE,KAAK,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,SAAS,EAAE,CAAC,EAAE,UAAU,EAAE,CAAC,EAAE,KAAK,EAAE,CAAC,EAAE;aACpE,CAAC;YACF,OAAO;gBACN,MAAM,EAAE,EAAE;gBACV,KAAK,EAAE,KAAK,IAAI,UAAU;gBAC1B,SAAS,EAAE,4BAA4B,MAAM,CAAC,GAAG,CAAC,EAAE;aACpD,CAAC;QACH,CAAC;IAAA,CACD;CACD","sourcesContent":["import type { Usage } from \"@caupulican/pi-ai\";\n\nexport type StopReason = \"stop\" | \"toolUse\" | \"aborted\" | \"error\" | string;\n\nexport interface IsolatedCompletionResult {\n\ttext: string;\n\tusage: Usage;\n\tstopReason: StopReason;\n}\n\nexport type ReflectionTrigger = \"complex\" | \"corrective\" | \"session-end\" | \"none\";\n\nexport interface DemandSignals {\n\ttrigger: ReflectionTrigger;\n\ttoolCallCount: number;\n\thadCorrection: boolean;\n\tcontextHeadroomPct: number; // 0..100\n\tusefulLately: number; // 0..1 rolling score\n}\n\nexport interface DemandPlan {\n\tact: \"skip\" | \"reflect\";\n\treason: string;\n\ttokenBudget: number;\n}\n\n/**\n * Pure zero-I/O heuristic to decide whether the current turn justifies a reflection run\n * and determine the token budget under the cheap-tool net-negative doctrine.\n */\nexport function decideDemand(signals: DemandSignals): DemandPlan {\n\tif (signals.trigger === \"none\") {\n\t\treturn { act: \"skip\", reason: \"No trigger detected\", tokenBudget: 0 };\n\t}\n\tif (signals.contextHeadroomPct < 10) {\n\t\treturn { act: \"skip\", reason: \"Context headroom is critically low (< 10%)\", tokenBudget: 0 };\n\t}\n\n\t// Dynamic token budget based on headroom (keep reflection bounded between 500 and 1500 tokens)\n\tconst baseBudget = 1000;\n\tconst tokenBudget = Math.max(500, Math.min(1500, Math.round(baseBudget * (signals.contextHeadroomPct / 100))));\n\n\tif (signals.hadCorrection) {\n\t\treturn { act: \"reflect\", reason: \"Correction detected in the turn\", tokenBudget };\n\t}\n\tif (signals.trigger === \"session-end\") {\n\t\treturn { act: \"reflect\", reason: \"Session end reflection triggered\", tokenBudget };\n\t}\n\tif (signals.trigger === \"complex\") {\n\t\tif (signals.toolCallCount >= 3) {\n\t\t\treturn { act: \"reflect\", reason: `Complex turn with ${signals.toolCallCount} tool calls`, tokenBudget };\n\t\t}\n\t}\n\n\treturn { act: \"skip\", reason: \"Signals do not justify reflection overhead\", tokenBudget: 0 };\n}\n\nexport interface ReflectionInput {\n\trecentTurnText: string; // host serializes the just-finished turn\n\texistingMemory: string; // current MEMORY.md + USER.md snapshot\n\tplan: DemandPlan;\n\t// host-injected isolated completion function:\n\tcomplete: (systemPrompt: string, userPrompt: string) => Promise<IsolatedCompletionResult>;\n}\n\nexport type ReflectionWrite =\n\t| { kind: \"memory_add\"; section: \"MEMORY\" | \"USER\"; text: string }\n\t| { kind: \"memory_replace\"; target: string; text: string }\n\t| { kind: \"memory_remove\"; target: string }\n\t// R7 memory-to-behavior: promote a recurring procedural workflow into an executable skill.\n\t| { kind: \"promote_skill\"; name: string; description: string; body: string };\n\nexport interface ReflectionResult {\n\twrites: ReflectionWrite[];\n\tusage: Usage;\n\trationale: string;\n}\n\n/**\n * STATIC reflection system prompt (Hermes-parity #33). It is byte-identical across every reflection\n * pass — the variable parts (existing memory snapshot + the turn transcript) live in the USER prompt —\n * so the provider prompt-cache reuses this prefix instead of re-billing it each pass (cost guard).\n * Do NOT interpolate per-call data into this constant or caching breaks.\n */\nexport const REFLECTION_SYSTEM_PROMPT = `You are a reflection engine. Your job is to analyze the recent conversation turn, compare it against the agent's existing memory, and decide if any memory updates are needed.\n\nMemory guidelines:\n- \"MEMORY\" is for project facts, configuration, repeatable workflows, and coding findings.\n- \"USER\" is for user preferences, patterns, and style specifications.\n- Avoid duplicate facts. If the fact is already represented, do not add it.\n- CONFRONT existing memory: if the new turn contradicts or updates an existing fact, use \"memory_replace\" or \"memory_remove\" to supersede the old fact rather than blindly appending.\n- Keep memories short, factual, and direct. No fluff.\n- Do NOT capture transient/environment-specific noise: tool/network failures, one-off errors, or a single narrative event. Persist only durable facts and preferences.\n- PROMOTE to behavior: if the turn established a REPEATABLE, multi-step PROCEDURE/workflow (not a one-off fact) that should govern a future class of tasks, emit a \"promote_skill\" instead of (or in addition to) a memory fact. Only promote a genuinely reusable procedure — never a single fact, a one-off narrative, or environment-specific noise. Prefer a memory fact when unsure.\n\nYou must output your analysis and writes in the following JSON format inside a \\`\\`\\`json\\`\\`\\` code fence:\n{\n \"rationale\": \"Explanation of your reasoning\",\n \"writes\": [\n { \"kind\": \"memory_add\", \"section\": \"MEMORY\" | \"USER\", \"text\": \"New direct fact to append\" },\n { \"kind\": \"memory_replace\", \"target\": \"Exact text substring to replace\", \"text\": \"New replacement text\" },\n { \"kind\": \"memory_remove\", \"target\": \"Exact text substring to remove\" },\n { \"kind\": \"promote_skill\", \"name\": \"kebab-case-skill-name\", \"description\": \"one line of when to use it\", \"body\": \"Markdown: the step-by-step procedure\" }\n ]\n}\n`;\n\nexport class ReflectionEngine {\n\t/**\n\t * Build the reflection prompt, call the injected isolated complete(),\n\t * parse the response, confront existing memory, and return memory writes.\n\t * Zero direct I/O.\n\t */\n\tasync reflect(input: ReflectionInput): Promise<ReflectionResult> {\n\t\tconst systemPrompt = REFLECTION_SYSTEM_PROMPT;\n\n\t\t// Variable inputs go in the USER prompt so the system prefix above stays cache-stable (#33).\n\t\tconst userPrompt = `Existing Memory snapshot:\n${input.existingMemory}\n\nRecent turn transcript:\n${input.recentTurnText}\n\nAnalyze this turn against the existing memory and output your memory updates.`;\n\n\t\tlet usage: Usage | undefined;\n\t\ttry {\n\t\t\tconst compResult = await input.complete(systemPrompt, userPrompt);\n\t\t\tusage = compResult.usage;\n\t\t\tconst text = compResult.text;\n\n\t\t\tconst jsonMatch = text.match(/```json\\s*([\\s\\S]*?)\\s*```/) || text.match(/{[\\s\\S]*}/);\n\t\t\tif (!jsonMatch) {\n\t\t\t\treturn {\n\t\t\t\t\twrites: [],\n\t\t\t\t\tusage: compResult.usage,\n\t\t\t\t\trationale: `Failed to locate JSON response. Raw text:\\n${text}`,\n\t\t\t\t};\n\t\t\t}\n\n\t\t\tconst parsed = JSON.parse(jsonMatch[1] || jsonMatch[0]);\n\t\t\tconst rationale = parsed.rationale || \"\";\n\t\t\tconst writes: ReflectionWrite[] = [];\n\n\t\t\tif (Array.isArray(parsed.writes)) {\n\t\t\t\tfor (const w of parsed.writes) {\n\t\t\t\t\tif (w && typeof w === \"object\") {\n\t\t\t\t\t\tif (\n\t\t\t\t\t\t\tw.kind === \"memory_add\" &&\n\t\t\t\t\t\t\t(w.section === \"MEMORY\" || w.section === \"USER\") &&\n\t\t\t\t\t\t\ttypeof w.text === \"string\"\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\twrites.push({ kind: \"memory_add\", section: w.section, text: w.text });\n\t\t\t\t\t\t} else if (\n\t\t\t\t\t\t\tw.kind === \"memory_replace\" &&\n\t\t\t\t\t\t\ttypeof w.target === \"string\" &&\n\t\t\t\t\t\t\ttypeof w.text === \"string\"\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\twrites.push({ kind: \"memory_replace\", target: w.target, text: w.text });\n\t\t\t\t\t\t} else if (w.kind === \"memory_remove\" && typeof w.target === \"string\") {\n\t\t\t\t\t\t\twrites.push({ kind: \"memory_remove\", target: w.target });\n\t\t\t\t\t\t} else if (\n\t\t\t\t\t\t\tw.kind === \"promote_skill\" &&\n\t\t\t\t\t\t\ttypeof w.name === \"string\" &&\n\t\t\t\t\t\t\ttypeof w.description === \"string\" &&\n\t\t\t\t\t\t\ttypeof w.body === \"string\"\n\t\t\t\t\t\t) {\n\t\t\t\t\t\t\twrites.push({ kind: \"promote_skill\", name: w.name, description: w.description, body: w.body });\n\t\t\t\t\t\t}\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}\n\n\t\t\treturn {\n\t\t\t\twrites,\n\t\t\t\tusage: compResult.usage,\n\t\t\t\trationale,\n\t\t\t};\n\t\t} catch (err) {\n\t\t\t// Zeroed/fallback usage representation\n\t\t\tconst emptyUsage: Usage = {\n\t\t\t\tinput: 0,\n\t\t\t\toutput: 0,\n\t\t\t\tcacheRead: 0,\n\t\t\t\tcacheWrite: 0,\n\t\t\t\ttotalTokens: 0,\n\t\t\t\tcost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },\n\t\t\t};\n\t\t\treturn {\n\t\t\t\twrites: [],\n\t\t\t\tusage: usage ?? emptyUsage,\n\t\t\t\trationale: `Error during reflection: ${String(err)}`,\n\t\t\t};\n\t\t}\n\t}\n}\n"]}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"local-registration.d.ts","sourceRoot":"","sources":["../../../src/core/models/local-registration.ts"],"names":[],"mappings":"
|
|
1
|
+
{"version":3,"file":"local-registration.d.ts","sourceRoot":"","sources":["../../../src/core/models/local-registration.ts"],"names":[],"mappings":"AAwCA,MAAM,WAAW,uBAAuB;IACvC,EAAE,EAAE,OAAO,CAAC;IACZ,cAAc,EAAE,MAAM,CAAC;IACvB,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,uFAAuF;IACvF,aAAa,CAAC,EAAE,MAAM,CAAC;CACvB;AAED,2FAA2F;AAC3F,eAAO,MAAM,eAAe,WAAW,CAAC;AA2BxC,wBAAgB,kBAAkB,CAAC,IAAI,EAAE;IACxC,QAAQ,EAAE,MAAM,CAAC;IACjB,GAAG,EAAE,MAAM,CAAC;IACZ,OAAO,EAAE,MAAM,CAAC;IAChB,aAAa,CAAC,EAAE,MAAM,CAAC;IACvB,mBAAmB,CAAC,EAAE,MAAM,CAAC;CAC7B,GAAG,uBAAuB,CAoC1B;AAED,wBAAgB,oBAAoB,CAAC,IAAI,EAAE;IAAE,QAAQ,EAAE,MAAM,CAAC;IAAC,GAAG,EAAE,MAAM,CAAA;CAAE,GAAG,uBAAuB,CAgBrG","sourcesContent":["import { existsSync, readFileSync, writeFileSync } from \"node:fs\";\nimport { join } from \"node:path\";\n\n/**\n * Persistent registration for pulled local models: merges an \"ollama\" provider entry into the\n * user's `<agentDir>/models.json` — the exact file ModelRegistry loads at startup — so a pulled\n * model resolves as `ollama/<ref>` immediately AND across sessions (usable as session model,\n * lane model, judge, or curator).\n *\n * Non-destructive contract: the file is parsed with STRICT JSON first; a file that only parses\n * with comments/relaxed syntax is the user's hand-authored config and is never rewritten — the\n * caller gets `manualSnippet` to show instead.\n */\n\ninterface ModelsJsonModel {\n\tid: string;\n\tname?: string;\n\tcontextWindow?: number;\n\t/** Measured by the local capacity probe; compaction uses min(contextWindow, servedContextWindow). */\n\tservedContextWindow?: number;\n\tmaxTokens?: number;\n\treasoning?: boolean;\n\tinput?: string[];\n\tcost?: { input: number; output: number; cacheRead: number; cacheWrite: number };\n}\n\ninterface ModelsJson {\n\tproviders: Record<\n\t\tstring,\n\t\t{\n\t\t\tbaseUrl?: string;\n\t\t\tapi?: string;\n\t\t\tapiKey?: string;\n\t\t\tmodels?: ModelsJsonModel[];\n\t\t\t[key: string]: unknown;\n\t\t}\n\t>;\n\t[key: string]: unknown;\n}\n\nexport interface LocalRegistrationResult {\n\tok: boolean;\n\tmodelsJsonPath: string;\n\treason?: string;\n\t/** When the file cannot be safely rewritten: the entry the user should add by hand. */\n\tmanualSnippet?: string;\n}\n\n/** Provider name pi registers pulled local models under (see registerLocalModel below). */\nexport const OLLAMA_PROVIDER = \"ollama\";\n\nfunction localModelEntry(ref: string, contextWindow: number, servedContextWindow?: number): ModelsJsonModel {\n\treturn {\n\t\tid: ref,\n\t\tname: ref,\n\t\tcontextWindow,\n\t\t...(servedContextWindow !== undefined ? { servedContextWindow } : {}),\n\t\tmaxTokens: 2048,\n\t\treasoning: false,\n\t\tinput: [\"text\"],\n\t\tcost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },\n\t};\n}\n\nfunction loadStrict(path: string): { json?: ModelsJson; reason?: string } {\n\tif (!existsSync(path)) return { json: { providers: {} } };\n\ttry {\n\t\tconst parsed = JSON.parse(readFileSync(path, \"utf-8\")) as ModelsJson;\n\t\tif (!parsed || typeof parsed !== \"object\") return { reason: \"models.json is not a JSON object\" };\n\t\tparsed.providers = parsed.providers ?? {};\n\t\treturn { json: parsed };\n\t} catch {\n\t\treturn { reason: \"models.json uses comments/relaxed JSON — pi will not rewrite a hand-authored file\" };\n\t}\n}\n\nexport function registerLocalModel(args: {\n\tagentDir: string;\n\tref: string;\n\tbaseUrl: string;\n\tcontextWindow?: number;\n\tservedContextWindow?: number;\n}): LocalRegistrationResult {\n\tconst modelsJsonPath = join(args.agentDir, \"models.json\");\n\tconst contextWindow = args.contextWindow ?? 8192;\n\tconst entry = localModelEntry(args.ref, contextWindow, args.servedContextWindow);\n\tconst providerBase = {\n\t\tbaseUrl: `${args.baseUrl.replace(/\\/$/, \"\")}/v1`,\n\t\tapi: \"openai-completions\",\n\t\tapiKey: \"ollama\",\n\t};\n\tconst { json, reason } = loadStrict(modelsJsonPath);\n\tif (!json) {\n\t\treturn {\n\t\t\tok: false,\n\t\t\tmodelsJsonPath,\n\t\t\treason,\n\t\t\tmanualSnippet: JSON.stringify(\n\t\t\t\t{ providers: { [OLLAMA_PROVIDER]: { ...providerBase, models: [entry] } } },\n\t\t\t\tnull,\n\t\t\t\t\"\\t\",\n\t\t\t),\n\t\t};\n\t}\n\tjson.providers[OLLAMA_PROVIDER] ??= { ...providerBase, models: [] };\n\tconst provider = json.providers[OLLAMA_PROVIDER];\n\tprovider.baseUrl ??= providerBase.baseUrl;\n\tprovider.api ??= providerBase.api;\n\tprovider.apiKey ??= providerBase.apiKey;\n\tprovider.models ??= [];\n\tconst existing = provider.models.findIndex((model) => model.id === args.ref);\n\tif (existing >= 0) {\n\t\tprovider.models[existing] = { ...provider.models[existing], ...entry };\n\t} else {\n\t\tprovider.models.push(entry);\n\t}\n\twriteFileSync(modelsJsonPath, `${JSON.stringify(json, null, \"\\t\")}\\n`, \"utf-8\");\n\treturn { ok: true, modelsJsonPath };\n}\n\nexport function unregisterLocalModel(args: { agentDir: string; ref: string }): LocalRegistrationResult {\n\tconst modelsJsonPath = join(args.agentDir, \"models.json\");\n\tconst { json, reason } = loadStrict(modelsJsonPath);\n\tif (!json) return { ok: false, modelsJsonPath, reason };\n\tconst provider = json.providers[OLLAMA_PROVIDER];\n\tif (!provider?.models) return { ok: true, modelsJsonPath };\n\tconst before = provider.models.length;\n\tprovider.models = provider.models.filter((model) => model.id !== args.ref);\n\tif (provider.models.length === before) return { ok: true, modelsJsonPath };\n\t// Drop the whole provider entry when its last pi-registered model goes (leave user fields alone\n\t// if they added any models themselves — only an all-pi-managed empty list is removed).\n\tif (provider.models.length === 0) {\n\t\tdelete json.providers[OLLAMA_PROVIDER];\n\t}\n\twriteFileSync(modelsJsonPath, `${JSON.stringify(json, null, \"\\t\")}\\n`, \"utf-8\");\n\treturn { ok: true, modelsJsonPath };\n}\n"]}
|
|
@@ -2,11 +2,12 @@ import { existsSync, readFileSync, writeFileSync } from "node:fs";
|
|
|
2
2
|
import { join } from "node:path";
|
|
3
3
|
/** Provider name pi registers pulled local models under (see registerLocalModel below). */
|
|
4
4
|
export const OLLAMA_PROVIDER = "ollama";
|
|
5
|
-
function localModelEntry(ref, contextWindow) {
|
|
5
|
+
function localModelEntry(ref, contextWindow, servedContextWindow) {
|
|
6
6
|
return {
|
|
7
7
|
id: ref,
|
|
8
8
|
name: ref,
|
|
9
9
|
contextWindow,
|
|
10
|
+
...(servedContextWindow !== undefined ? { servedContextWindow } : {}),
|
|
10
11
|
maxTokens: 2048,
|
|
11
12
|
reasoning: false,
|
|
12
13
|
input: ["text"],
|
|
@@ -30,7 +31,7 @@ function loadStrict(path) {
|
|
|
30
31
|
export function registerLocalModel(args) {
|
|
31
32
|
const modelsJsonPath = join(args.agentDir, "models.json");
|
|
32
33
|
const contextWindow = args.contextWindow ?? 8192;
|
|
33
|
-
const entry = localModelEntry(args.ref, contextWindow);
|
|
34
|
+
const entry = localModelEntry(args.ref, contextWindow, args.servedContextWindow);
|
|
34
35
|
const providerBase = {
|
|
35
36
|
baseUrl: `${args.baseUrl.replace(/\/$/, "")}/v1`,
|
|
36
37
|
api: "openai-completions",
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"local-registration.js","sourceRoot":"","sources":["../../../src/core/models/local-registration.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,UAAU,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AAClE,OAAO,EAAE,IAAI,EAAE,MAAM,WAAW,CAAC;
|
|
1
|
+
{"version":3,"file":"local-registration.js","sourceRoot":"","sources":["../../../src/core/models/local-registration.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,UAAU,EAAE,YAAY,EAAE,aAAa,EAAE,MAAM,SAAS,CAAC;AAClE,OAAO,EAAE,IAAI,EAAE,MAAM,WAAW,CAAC;AA+CjC,2FAA2F;AAC3F,MAAM,CAAC,MAAM,eAAe,GAAG,QAAQ,CAAC;AAExC,SAAS,eAAe,CAAC,GAAW,EAAE,aAAqB,EAAE,mBAA4B,EAAmB;IAC3G,OAAO;QACN,EAAE,EAAE,GAAG;QACP,IAAI,EAAE,GAAG;QACT,aAAa;QACb,GAAG,CAAC,mBAAmB,KAAK,SAAS,CAAC,CAAC,CAAC,EAAE,mBAAmB,EAAE,CAAC,CAAC,CAAC,EAAE,CAAC;QACrE,SAAS,EAAE,IAAI;QACf,SAAS,EAAE,KAAK;QAChB,KAAK,EAAE,CAAC,MAAM,CAAC;QACf,IAAI,EAAE,EAAE,KAAK,EAAE,CAAC,EAAE,MAAM,EAAE,CAAC,EAAE,SAAS,EAAE,CAAC,EAAE,UAAU,EAAE,CAAC,EAAE;KAC1D,CAAC;AAAA,CACF;AAED,SAAS,UAAU,CAAC,IAAY,EAA0C;IACzE,IAAI,CAAC,UAAU,CAAC,IAAI,CAAC;QAAE,OAAO,EAAE,IAAI,EAAE,EAAE,SAAS,EAAE,EAAE,EAAE,EAAE,CAAC;IAC1D,IAAI,CAAC;QACJ,MAAM,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,YAAY,CAAC,IAAI,EAAE,OAAO,CAAC,CAAe,CAAC;QACrE,IAAI,CAAC,MAAM,IAAI,OAAO,MAAM,KAAK,QAAQ;YAAE,OAAO,EAAE,MAAM,EAAE,kCAAkC,EAAE,CAAC;QACjG,MAAM,CAAC,SAAS,GAAG,MAAM,CAAC,SAAS,IAAI,EAAE,CAAC;QAC1C,OAAO,EAAE,IAAI,EAAE,MAAM,EAAE,CAAC;IACzB,CAAC;IAAC,MAAM,CAAC;QACR,OAAO,EAAE,MAAM,EAAE,qFAAmF,EAAE,CAAC;IACxG,CAAC;AAAA,CACD;AAED,MAAM,UAAU,kBAAkB,CAAC,IAMlC,EAA2B;IAC3B,MAAM,cAAc,GAAG,IAAI,CAAC,IAAI,CAAC,QAAQ,EAAE,aAAa,CAAC,CAAC;IAC1D,MAAM,aAAa,GAAG,IAAI,CAAC,aAAa,IAAI,IAAI,CAAC;IACjD,MAAM,KAAK,GAAG,eAAe,CAAC,IAAI,CAAC,GAAG,EAAE,aAAa,EAAE,IAAI,CAAC,mBAAmB,CAAC,CAAC;IACjF,MAAM,YAAY,GAAG;QACpB,OAAO,EAAE,GAAG,IAAI,CAAC,OAAO,CAAC,OAAO,CAAC,KAAK,EAAE,EAAE,CAAC,KAAK;QAChD,GAAG,EAAE,oBAAoB;QACzB,MAAM,EAAE,QAAQ;KAChB,CAAC;IACF,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,GAAG,UAAU,CAAC,cAAc,CAAC,CAAC;IACpD,IAAI,CAAC,IAAI,EAAE,CAAC;QACX,OAAO;YACN,EAAE,EAAE,KAAK;YACT,cAAc;YACd,MAAM;YACN,aAAa,EAAE,IAAI,CAAC,SAAS,CAC5B,EAAE,SAAS,EAAE,EAAE,CAAC,eAAe,CAAC,EAAE,EAAE,GAAG,YAAY,EAAE,MAAM,EAAE,CAAC,KAAK,CAAC,EAAE,EAAE,EAAE,EAC1E,IAAI,EACJ,IAAI,CACJ;SACD,CAAC;IACH,CAAC;IACD,IAAI,CAAC,SAAS,CAAC,eAAe,CAAC,KAAK,EAAE,GAAG,YAAY,EAAE,MAAM,EAAE,EAAE,EAAE,CAAC;IACpE,MAAM,QAAQ,GAAG,IAAI,CAAC,SAAS,CAAC,eAAe,CAAC,CAAC;IACjD,QAAQ,CAAC,OAAO,KAAK,YAAY,CAAC,OAAO,CAAC;IAC1C,QAAQ,CAAC,GAAG,KAAK,YAAY,CAAC,GAAG,CAAC;IAClC,QAAQ,CAAC,MAAM,KAAK,YAAY,CAAC,MAAM,CAAC;IACxC,QAAQ,CAAC,MAAM,KAAK,EAAE,CAAC;IACvB,MAAM,QAAQ,GAAG,QAAQ,CAAC,MAAM,CAAC,SAAS,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,KAAK,CAAC,EAAE,KAAK,IAAI,CAAC,GAAG,CAAC,CAAC;IAC7E,IAAI,QAAQ,IAAI,CAAC,EAAE,CAAC;QACnB,QAAQ,CAAC,MAAM,CAAC,QAAQ,CAAC,GAAG,EAAE,GAAG,QAAQ,CAAC,MAAM,CAAC,QAAQ,CAAC,EAAE,GAAG,KAAK,EAAE,CAAC;IACxE,CAAC;SAAM,CAAC;QACP,QAAQ,CAAC,MAAM,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC;IAC7B,CAAC;IACD,aAAa,CAAC,cAAc,EAAE,GAAG,IAAI,CAAC,SAAS,CAAC,IAAI,EAAE,IAAI,EAAE,IAAI,CAAC,IAAI,EAAE,OAAO,CAAC,CAAC;IAChF,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,cAAc,EAAE,CAAC;AAAA,CACpC;AAED,MAAM,UAAU,oBAAoB,CAAC,IAAuC,EAA2B;IACtG,MAAM,cAAc,GAAG,IAAI,CAAC,IAAI,CAAC,QAAQ,EAAE,aAAa,CAAC,CAAC;IAC1D,MAAM,EAAE,IAAI,EAAE,MAAM,EAAE,GAAG,UAAU,CAAC,cAAc,CAAC,CAAC;IACpD,IAAI,CAAC,IAAI;QAAE,OAAO,EAAE,EAAE,EAAE,KAAK,EAAE,cAAc,EAAE,MAAM,EAAE,CAAC;IACxD,MAAM,QAAQ,GAAG,IAAI,CAAC,SAAS,CAAC,eAAe,CAAC,CAAC;IACjD,IAAI,CAAC,QAAQ,EAAE,MAAM;QAAE,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,cAAc,EAAE,CAAC;IAC3D,MAAM,MAAM,GAAG,QAAQ,CAAC,MAAM,CAAC,MAAM,CAAC;IACtC,QAAQ,CAAC,MAAM,GAAG,QAAQ,CAAC,MAAM,CAAC,MAAM,CAAC,CAAC,KAAK,EAAE,EAAE,CAAC,KAAK,CAAC,EAAE,KAAK,IAAI,CAAC,GAAG,CAAC,CAAC;IAC3E,IAAI,QAAQ,CAAC,MAAM,CAAC,MAAM,KAAK,MAAM;QAAE,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,cAAc,EAAE,CAAC;IAC3E,gGAAgG;IAChG,yFAAuF;IACvF,IAAI,QAAQ,CAAC,MAAM,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QAClC,OAAO,IAAI,CAAC,SAAS,CAAC,eAAe,CAAC,CAAC;IACxC,CAAC;IACD,aAAa,CAAC,cAAc,EAAE,GAAG,IAAI,CAAC,SAAS,CAAC,IAAI,EAAE,IAAI,EAAE,IAAI,CAAC,IAAI,EAAE,OAAO,CAAC,CAAC;IAChF,OAAO,EAAE,EAAE,EAAE,IAAI,EAAE,cAAc,EAAE,CAAC;AAAA,CACpC","sourcesContent":["import { existsSync, readFileSync, writeFileSync } from \"node:fs\";\nimport { join } from \"node:path\";\n\n/**\n * Persistent registration for pulled local models: merges an \"ollama\" provider entry into the\n * user's `<agentDir>/models.json` — the exact file ModelRegistry loads at startup — so a pulled\n * model resolves as `ollama/<ref>` immediately AND across sessions (usable as session model,\n * lane model, judge, or curator).\n *\n * Non-destructive contract: the file is parsed with STRICT JSON first; a file that only parses\n * with comments/relaxed syntax is the user's hand-authored config and is never rewritten — the\n * caller gets `manualSnippet` to show instead.\n */\n\ninterface ModelsJsonModel {\n\tid: string;\n\tname?: string;\n\tcontextWindow?: number;\n\t/** Measured by the local capacity probe; compaction uses min(contextWindow, servedContextWindow). */\n\tservedContextWindow?: number;\n\tmaxTokens?: number;\n\treasoning?: boolean;\n\tinput?: string[];\n\tcost?: { input: number; output: number; cacheRead: number; cacheWrite: number };\n}\n\ninterface ModelsJson {\n\tproviders: Record<\n\t\tstring,\n\t\t{\n\t\t\tbaseUrl?: string;\n\t\t\tapi?: string;\n\t\t\tapiKey?: string;\n\t\t\tmodels?: ModelsJsonModel[];\n\t\t\t[key: string]: unknown;\n\t\t}\n\t>;\n\t[key: string]: unknown;\n}\n\nexport interface LocalRegistrationResult {\n\tok: boolean;\n\tmodelsJsonPath: string;\n\treason?: string;\n\t/** When the file cannot be safely rewritten: the entry the user should add by hand. */\n\tmanualSnippet?: string;\n}\n\n/** Provider name pi registers pulled local models under (see registerLocalModel below). */\nexport const OLLAMA_PROVIDER = \"ollama\";\n\nfunction localModelEntry(ref: string, contextWindow: number, servedContextWindow?: number): ModelsJsonModel {\n\treturn {\n\t\tid: ref,\n\t\tname: ref,\n\t\tcontextWindow,\n\t\t...(servedContextWindow !== undefined ? { servedContextWindow } : {}),\n\t\tmaxTokens: 2048,\n\t\treasoning: false,\n\t\tinput: [\"text\"],\n\t\tcost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },\n\t};\n}\n\nfunction loadStrict(path: string): { json?: ModelsJson; reason?: string } {\n\tif (!existsSync(path)) return { json: { providers: {} } };\n\ttry {\n\t\tconst parsed = JSON.parse(readFileSync(path, \"utf-8\")) as ModelsJson;\n\t\tif (!parsed || typeof parsed !== \"object\") return { reason: \"models.json is not a JSON object\" };\n\t\tparsed.providers = parsed.providers ?? {};\n\t\treturn { json: parsed };\n\t} catch {\n\t\treturn { reason: \"models.json uses comments/relaxed JSON — pi will not rewrite a hand-authored file\" };\n\t}\n}\n\nexport function registerLocalModel(args: {\n\tagentDir: string;\n\tref: string;\n\tbaseUrl: string;\n\tcontextWindow?: number;\n\tservedContextWindow?: number;\n}): LocalRegistrationResult {\n\tconst modelsJsonPath = join(args.agentDir, \"models.json\");\n\tconst contextWindow = args.contextWindow ?? 8192;\n\tconst entry = localModelEntry(args.ref, contextWindow, args.servedContextWindow);\n\tconst providerBase = {\n\t\tbaseUrl: `${args.baseUrl.replace(/\\/$/, \"\")}/v1`,\n\t\tapi: \"openai-completions\",\n\t\tapiKey: \"ollama\",\n\t};\n\tconst { json, reason } = loadStrict(modelsJsonPath);\n\tif (!json) {\n\t\treturn {\n\t\t\tok: false,\n\t\t\tmodelsJsonPath,\n\t\t\treason,\n\t\t\tmanualSnippet: JSON.stringify(\n\t\t\t\t{ providers: { [OLLAMA_PROVIDER]: { ...providerBase, models: [entry] } } },\n\t\t\t\tnull,\n\t\t\t\t\"\\t\",\n\t\t\t),\n\t\t};\n\t}\n\tjson.providers[OLLAMA_PROVIDER] ??= { ...providerBase, models: [] };\n\tconst provider = json.providers[OLLAMA_PROVIDER];\n\tprovider.baseUrl ??= providerBase.baseUrl;\n\tprovider.api ??= providerBase.api;\n\tprovider.apiKey ??= providerBase.apiKey;\n\tprovider.models ??= [];\n\tconst existing = provider.models.findIndex((model) => model.id === args.ref);\n\tif (existing >= 0) {\n\t\tprovider.models[existing] = { ...provider.models[existing], ...entry };\n\t} else {\n\t\tprovider.models.push(entry);\n\t}\n\twriteFileSync(modelsJsonPath, `${JSON.stringify(json, null, \"\\t\")}\\n`, \"utf-8\");\n\treturn { ok: true, modelsJsonPath };\n}\n\nexport function unregisterLocalModel(args: { agentDir: string; ref: string }): LocalRegistrationResult {\n\tconst modelsJsonPath = join(args.agentDir, \"models.json\");\n\tconst { json, reason } = loadStrict(modelsJsonPath);\n\tif (!json) return { ok: false, modelsJsonPath, reason };\n\tconst provider = json.providers[OLLAMA_PROVIDER];\n\tif (!provider?.models) return { ok: true, modelsJsonPath };\n\tconst before = provider.models.length;\n\tprovider.models = provider.models.filter((model) => model.id !== args.ref);\n\tif (provider.models.length === before) return { ok: true, modelsJsonPath };\n\t// Drop the whole provider entry when its last pi-registered model goes (leave user fields alone\n\t// if they added any models themselves — only an all-pi-managed empty list is removed).\n\tif (provider.models.length === 0) {\n\t\tdelete json.providers[OLLAMA_PROVIDER];\n\t}\n\twriteFileSync(modelsJsonPath, `${JSON.stringify(json, null, \"\\t\")}\\n`, \"utf-8\");\n\treturn { ok: true, modelsJsonPath };\n}\n"]}
|
|
@@ -26,6 +26,11 @@ export interface JudgeFitnessPrompt {
|
|
|
26
26
|
}
|
|
27
27
|
/** Default judge probe set: three planning-shaped prompts, three trivial lookups. */
|
|
28
28
|
export declare const DEFAULT_JUDGE_FITNESS_PROMPTS: readonly JudgeFitnessPrompt[];
|
|
29
|
+
export interface CapacityProbeOptions {
|
|
30
|
+
registeredContextWindow: number;
|
|
31
|
+
/** Smallest candidate window to try before declaring the capacity unknown. Default 1024. */
|
|
32
|
+
minContextWindow?: number;
|
|
33
|
+
}
|
|
29
34
|
export interface ModelFitnessOptions {
|
|
30
35
|
complete: FitnessComplete;
|
|
31
36
|
/** Trials per lane surface. Default 3. */
|
|
@@ -33,6 +38,8 @@ export interface ModelFitnessOptions {
|
|
|
33
38
|
/** Wall-clock budget per call in ms. Default 120000. */
|
|
34
39
|
maxWallClockMs?: number;
|
|
35
40
|
judgePrompts?: readonly JudgeFitnessPrompt[];
|
|
41
|
+
/** Optional local-model capacity lane: measures the actually served context window. */
|
|
42
|
+
capacityProbe?: CapacityProbeOptions;
|
|
36
43
|
signal?: AbortSignal;
|
|
37
44
|
/** Injected clock for latency measurement (test seam). Defaults to Date.now. */
|
|
38
45
|
now?: () => number;
|
|
@@ -57,6 +64,12 @@ export interface JudgeFitnessScore {
|
|
|
57
64
|
/** Mean output tokens/second across the judge calls; undefined when not reported. */
|
|
58
65
|
tokensPerSecond?: number;
|
|
59
66
|
}
|
|
67
|
+
export interface CapacityFitnessScore {
|
|
68
|
+
registeredContextWindow: number;
|
|
69
|
+
servedContextWindow: number;
|
|
70
|
+
outcomes: string[];
|
|
71
|
+
meanMs: number;
|
|
72
|
+
}
|
|
60
73
|
export interface ModelFitnessReport {
|
|
61
74
|
trials: number;
|
|
62
75
|
/** Aggregate output tokens/second across ALL probe calls (the headline speed number). */
|
|
@@ -70,11 +83,14 @@ export interface ModelFitnessReport {
|
|
|
70
83
|
toolCall: LaneFitnessScore;
|
|
71
84
|
/** Curator surface: can the model digest a context chunk to strict JSON WITHOUT losing key facts? */
|
|
72
85
|
digest: LaneFitnessScore;
|
|
86
|
+
/** Local-model capacity lane: measured context window the server actually serves. */
|
|
87
|
+
capacity?: CapacityFitnessScore;
|
|
73
88
|
totalCostUsd: number;
|
|
74
89
|
}
|
|
75
90
|
/** Static prompts for the heavy-lifter surfaces (stable for provider prompt caching). */
|
|
76
91
|
export declare const SEARCH_PROBE_SYSTEM_PROMPT: string;
|
|
77
92
|
export declare const TOOL_CALL_PROBE_SYSTEM_PROMPT: string;
|
|
93
|
+
export declare const CAPACITY_PROBE_SYSTEM_PROMPT: string;
|
|
78
94
|
export { CURATION_DIGEST_SYSTEM_PROMPT as DIGEST_PROBE_SYSTEM_PROMPT } from "../context/brain-curator.ts";
|
|
79
95
|
export declare function runModelFitnessProbe(options: ModelFitnessOptions): Promise<ModelFitnessReport>;
|
|
80
96
|
/**
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"model-fitness.d.ts","sourceRoot":"","sources":["../../../src/core/research/model-fitness.ts"],"names":[],"mappings":"AAOA;;;;;;GAMG;AAEH,MAAM,WAAW,iBAAiB;IACjC,IAAI,EAAE,MAAM,CAAC;IACb,OAAO,EAAE,MAAM,CAAC;IAChB,UAAU,EAAE,MAAM,CAAC;IACnB,iGAAiG;IACjG,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,kGAAkG;IAClG,MAAM,CAAC,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,MAAM,eAAe,GAAG,CAAC,IAAI,EAAE;IACpC,YAAY,EAAE,MAAM,CAAC;IACrB,UAAU,EAAE,MAAM,CAAC;IACnB,MAAM,CAAC,EAAE,WAAW,CAAC;CACrB,KAAK,OAAO,CAAC,iBAAiB,CAAC,CAAC;AAEjC,MAAM,WAAW,kBAAkB;IAClC,MAAM,EAAE,MAAM,CAAC;IACf,0EAA0E;IAC1E,QAAQ,EAAE,OAAO,CAAC;CAClB;AAED,qFAAqF;AACrF,eAAO,MAAM,6BAA6B,EAAE,SAAS,kBAAkB,EAOtE,CAAC;AAEF,MAAM,WAAW,mBAAmB;IACnC,QAAQ,EAAE,eAAe,CAAC;IAC1B,0CAA0C;IAC1C,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,wDAAwD;IACxD,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,YAAY,CAAC,EAAE,SAAS,kBAAkB,EAAE,CAAC;IAC7C,MAAM,CAAC,EAAE,WAAW,CAAC;IACrB,gFAAgF;IAChF,GAAG,CAAC,EAAE,MAAM,MAAM,CAAC;CACnB;AAED,MAAM,WAAW,gBAAgB;IAChC,SAAS,EAAE,MAAM,CAAC;IAClB,KAAK,EAAE,MAAM,CAAC;IACd,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB,MAAM,EAAE,MAAM,CAAC;IACf,yFAAyF;IACzF,eAAe,CAAC,EAAE,MAAM,CAAC;CACzB;AAED,MAAM,WAAW,iBAAiB;IACjC,MAAM,EAAE,MAAM,CAAC;IACf,gBAAgB,EAAE,MAAM,CAAC;IACzB,aAAa,EAAE,MAAM,CAAC;IACtB,YAAY,EAAE,MAAM,CAAC;IACrB,YAAY,EAAE,MAAM,CAAC;IACrB,KAAK,EAAE,MAAM,CAAC;IACd,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB,MAAM,EAAE,MAAM,CAAC;IACf,qFAAqF;IACrF,eAAe,CAAC,EAAE,MAAM,CAAC;CACzB;AAED,MAAM,WAAW,kBAAkB;IAClC,MAAM,EAAE,MAAM,CAAC;IACf,yFAAyF;IACzF,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,QAAQ,EAAE,gBAAgB,CAAC;IAC3B,MAAM,EAAE,gBAAgB,CAAC;IACzB,KAAK,EAAE,iBAAiB,CAAC;IACzB,8EAA8E;IAC9E,MAAM,EAAE,gBAAgB,CAAC;IACzB,yFAAyF;IACzF,QAAQ,EAAE,gBAAgB,CAAC;IAC3B,qGAAqG;IACrG,MAAM,EAAE,gBAAgB,CAAC;IACzB,YAAY,EAAE,MAAM,CAAC;CACrB;AAED,yFAAyF;AACzF,eAAO,MAAM,0BAA0B,QAK3B,CAAC;AAEb,eAAO,MAAM,6BAA6B,QAK9B,CAAC;AASb,OAAO,EAAE,6BAA6B,IAAI,0BAA0B,EAAE,MAAM,6BAA6B,CAAC;AA6G1G,wBAAsB,oBAAoB,CAAC,OAAO,EAAE,mBAAmB,GAAG,OAAO,CAAC,kBAAkB,CAAC,CAkKpG;AAED;;;;;;;;;;GAUG;AACH,wBAAgB,gBAAgB,CAAC,MAAM,EAAE,kBAAkB,GAAG,OAAO,CAQpE;AAED,kGAAkG;AAClG,wBAAgB,wBAAwB,CAAC,KAAK,EAAE,MAAM,EAAE,MAAM,EAAE,kBAAkB,GAAG,MAAM,CAiB1F","sourcesContent":["import { runBoundedCompletion } from \"../autonomy/bounded-completion.ts\";\nimport type { CapabilityEnvelope } from \"../autonomy/contracts.ts\";\nimport { CURATION_DIGEST_SYSTEM_PROMPT } from \"../context/brain-curator.ts\";\nimport { runWorker } from \"../delegation/worker-runner.ts\";\nimport { runRouteJudge } from \"../model-router/route-judge.ts\";\nimport { runResearch } from \"./research-runner.ts\";\n\n/**\n * Model fitness probe: measures whether a candidate model can actually drive the harness's\n * subagent contracts — the research lane, the scout-worker lane, and the routing judge — by\n * running each real runner against the model and scoring parse/success rates plus judge\n * discrimination. Provider-free: the completion executor is injected, so this works against any\n * registered model (local Ollama models included) and against faux providers in tests.\n */\n\nexport interface FitnessCompletion {\n\ttext: string;\n\tcostUsd: number;\n\tstopReason: string;\n\t/** Output tokens generated (for tok/s). Optional: providers that don't report it are skipped. */\n\toutputTokens?: number;\n\t/** Pure generation time in ms (e.g. Ollama eval_duration). Falls back to wall-clock if absent. */\n\tevalMs?: number;\n}\n\nexport type FitnessComplete = (args: {\n\tsystemPrompt: string;\n\tuserPrompt: string;\n\tsignal?: AbortSignal;\n}) => Promise<FitnessCompletion>;\n\nexport interface JudgeFitnessPrompt {\n\tprompt: string;\n\t/** True when the prompt is planning-shaped and must never route cheap. */\n\tplanning: boolean;\n}\n\n/** Default judge probe set: three planning-shaped prompts, three trivial lookups. */\nexport const DEFAULT_JUDGE_FITNESS_PROMPTS: readonly JudgeFitnessPrompt[] = [\n\t{ prompt: \"how should we plan the migration of the session storage layer?\", planning: true },\n\t{ prompt: \"design an approach for splitting the settings manager\", planning: true },\n\t{ prompt: \"draft a roadmap for the autonomy rework\", planning: true },\n\t{ prompt: \"what does the resolvePath function return?\", planning: false },\n\t{ prompt: \"list the files in the delegation module\", planning: false },\n\t{ prompt: \"why is this test flaky?\", planning: false },\n];\n\nexport interface ModelFitnessOptions {\n\tcomplete: FitnessComplete;\n\t/** Trials per lane surface. Default 3. */\n\ttrials?: number;\n\t/** Wall-clock budget per call in ms. Default 120000. */\n\tmaxWallClockMs?: number;\n\tjudgePrompts?: readonly JudgeFitnessPrompt[];\n\tsignal?: AbortSignal;\n\t/** Injected clock for latency measurement (test seam). Defaults to Date.now. */\n\tnow?: () => number;\n}\n\nexport interface LaneFitnessScore {\n\tsucceeded: number;\n\ttotal: number;\n\toutcomes: string[];\n\tmeanMs: number;\n\t/** Mean output tokens/second across the surface's calls; undefined when not reported. */\n\ttokensPerSecond?: number;\n}\n\nexport interface JudgeFitnessScore {\n\tparsed: number;\n\tplanningElevated: number;\n\tplanningTotal: number;\n\ttrivialCheap: number;\n\ttrivialTotal: number;\n\ttotal: number;\n\toutcomes: string[];\n\tmeanMs: number;\n\t/** Mean output tokens/second across the judge calls; undefined when not reported. */\n\ttokensPerSecond?: number;\n}\n\nexport interface ModelFitnessReport {\n\ttrials: number;\n\t/** Aggregate output tokens/second across ALL probe calls (the headline speed number). */\n\ttokensPerSecond?: number;\n\tresearch: LaneFitnessScore;\n\tworker: LaneFitnessScore;\n\tjudge: JudgeFitnessScore;\n\t/** Heavy-lifter surface: can the model formulate a structured search plan? */\n\tsearch: LaneFitnessScore;\n\t/** Heavy-lifter surface: can the model emit a well-formed tool call against a schema? */\n\ttoolCall: LaneFitnessScore;\n\t/** Curator surface: can the model digest a context chunk to strict JSON WITHOUT losing key facts? */\n\tdigest: LaneFitnessScore;\n\ttotalCostUsd: number;\n}\n\n/** Static prompts for the heavy-lifter surfaces (stable for provider prompt caching). */\nexport const SEARCH_PROBE_SYSTEM_PROMPT = [\n\t\"You plan code searches for a coding agent. You never answer the question yourself.\",\n\t\"Given a question about a codebase, respond with STRICT JSON only - no prose:\",\n\t'{\"queries\":[{\"pattern\":\"<regex or literal to grep>\",\"glob\":\"<file glob like **/*.ts>\"}]}',\n\t\"Return 1 to 4 queries, most specific first.\",\n].join(\"\\n\");\n\nexport const TOOL_CALL_PROBE_SYSTEM_PROMPT = [\n\t\"You operate tools for a coding agent. You have exactly one tool:\",\n\t\"grep(pattern: string, path: string) - search files under a path for a pattern.\",\n\t\"Respond to every task with STRICT JSON only - no prose:\",\n\t'{\"tool\":\"grep\",\"arguments\":{\"pattern\":\"<pattern>\",\"path\":\"<path>\"}}',\n].join(\"\\n\");\n\nconst SEARCH_PROBE_TASKS: readonly string[] = [\n\t\"Where is the retry/backoff logic for HTTP requests implemented?\",\n\t\"Which files define the settings for background research?\",\n\t\"Find where session entries of type custom are appended.\",\n];\n\n// The probe measures the REAL curation contract — same prompt the BrainCurator ships.\nexport { CURATION_DIGEST_SYSTEM_PROMPT as DIGEST_PROBE_SYSTEM_PROMPT } from \"../context/brain-curator.ts\";\n\n/**\n * Digest probe chunks each carry a NONCE identifier that cannot be guessed from the\n * instructions: acceptance requires the digest to RETAIN the nonce verbatim, so the score\n * measures extraction fidelity, not narration (a model cannot pass by paraphrasing).\n */\nconst DIGEST_PROBE_TASKS: readonly { chunk: string; nonce: string }[] = [\n\t{\n\t\tnonce: \"retryWithJitter_zx41\",\n\t\tchunk: [\n\t\t\t\"grep results for 'retry' under src/http:\",\n\t\t\t\"src/http/client.ts:88: export function retryWithJitter_zx41(fn, attempts = 3) {\",\n\t\t\t\"src/http/client.ts:112: // exponential backoff capped at 30s\",\n\t\t\t\"src/http/pool.ts:41: client.retry = false\",\n\t\t].join(\"\\n\"),\n\t},\n\t{\n\t\tnonce: \"ERR_QM_7734\",\n\t\tchunk: [\n\t\t\t\"$ npm run migrate\",\n\t\t\t\"migrating 14 files...\",\n\t\t\t\"error ERR_QM_7734: column 'owner_id' missing on table sessions (migration 0009)\",\n\t\t\t\"exit code 1\",\n\t\t].join(\"\\n\"),\n\t},\n\t{\n\t\tnonce: \"v3.9.2-hotfix.1\",\n\t\tchunk: [\n\t\t\t\"read package.json (34 lines):\",\n\t\t\t' \"name\": \"acme-billing\",',\n\t\t\t' \"version\": \"v3.9.2-hotfix.1\",',\n\t\t\t' \"engines\": { \"node\": \">=22\" },',\n\t\t].join(\"\\n\"),\n\t},\n];\n\nfunction parseDigest(text: string, nonce: string): boolean {\n\tconst parsed = extractJsonObject(text);\n\tif (!parsed) return false;\n\tconst digest = (parsed as { digest?: unknown }).digest;\n\tif (typeof digest !== \"string\") return false;\n\tconst trimmed = digest.trim();\n\t// Bounded and faithful: short enough to be a stub annotation, still carrying the nonce fact.\n\treturn trimmed.length > 0 && trimmed.length <= 240 && trimmed.includes(nonce);\n}\n\nconst TOOL_CALL_PROBE_TASKS: readonly string[] = [\n\t\"Find usages of the function resolveCliModel under src/.\",\n\t\"Search for the string 'budget_exhausted' in the core directory.\",\n\t\"Locate where LaneTracker is instantiated under src/core.\",\n];\n\nfunction parseSearchPlan(text: string): boolean {\n\tconst parsed = extractJsonObject(text);\n\tif (!parsed) return false;\n\tconst queries = (parsed as { queries?: unknown }).queries;\n\tif (!Array.isArray(queries) || queries.length === 0 || queries.length > 8) return false;\n\treturn queries.every(\n\t\t(query) =>\n\t\t\tquery &&\n\t\t\ttypeof query === \"object\" &&\n\t\t\ttypeof (query as { pattern?: unknown }).pattern === \"string\" &&\n\t\t\t(query as { pattern: string }).pattern.trim().length > 0,\n\t);\n}\n\nfunction parseToolCall(text: string): boolean {\n\tconst parsed = extractJsonObject(text);\n\tif (!parsed) return false;\n\tconst record = parsed as { tool?: unknown; arguments?: unknown };\n\tif (record.tool !== \"grep\") return false;\n\tconst args = record.arguments;\n\tif (!args || typeof args !== \"object\" || Array.isArray(args)) return false;\n\tconst pattern = (args as { pattern?: unknown }).pattern;\n\tconst path = (args as { path?: unknown }).path;\n\treturn (\n\t\ttypeof pattern === \"string\" && pattern.trim().length > 0 && typeof path === \"string\" && path.trim().length > 0\n\t);\n}\n\nfunction extractJsonObject(text: string): unknown | undefined {\n\tconst trimmed = text.trim();\n\tconst candidates: string[] = [trimmed];\n\tconst fenced = /```(?:json)?\\s*([\\s\\S]*?)```/.exec(trimmed);\n\tif (fenced?.[1]) candidates.push(fenced[1].trim());\n\tconst start = trimmed.indexOf(\"{\");\n\tconst end = trimmed.lastIndexOf(\"}\");\n\tif (start >= 0 && end > start) candidates.push(trimmed.slice(start, end + 1));\n\tfor (const candidate of candidates) {\n\t\ttry {\n\t\t\tconst parsed = JSON.parse(candidate);\n\t\t\tif (parsed && typeof parsed === \"object\" && !Array.isArray(parsed)) return parsed;\n\t\t} catch {\n\t\t\t// try next candidate\n\t\t}\n\t}\n\treturn undefined;\n}\n\nfunction fitnessEnvelope(): CapabilityEnvelope {\n\treturn {\n\t\tid: \"model-fitness-probe\",\n\t\tcapabilities: [\"research\", \"read_files\", \"memory_read\"],\n\t\tmaxEstimatedUsd: 1,\n\t\tcreatedAt: new Date().toISOString(),\n\t};\n}\n\nexport async function runModelFitnessProbe(options: ModelFitnessOptions): Promise<ModelFitnessReport> {\n\tconst trials = Math.max(1, Math.min(options.trials ?? 3, 20));\n\tconst maxWallClockMs = options.maxWallClockMs ?? 120_000;\n\tconst judgePrompts = options.judgePrompts ?? DEFAULT_JUDGE_FITNESS_PROMPTS;\n\tconst now = options.now ?? Date.now;\n\tlet totalCostUsd = 0;\n\n\t// Token-speed instrumentation: the lane runners' own contracts carry text/cost only, so the\n\t// completer is wrapped once here and generation stats are accumulated per surface.\n\tconst overallSpeed = { tokens: 0, evalMs: 0 };\n\tlet surfaceSpeed = { tokens: 0, evalMs: 0 };\n\tconst complete: FitnessComplete = async (args) => {\n\t\tconst completion = await options.complete(args);\n\t\tconst tokens = completion.outputTokens ?? 0;\n\t\tconst evalMs = completion.evalMs ?? 0;\n\t\tif (tokens > 0 && evalMs > 0) {\n\t\t\tsurfaceSpeed.tokens += tokens;\n\t\t\tsurfaceSpeed.evalMs += evalMs;\n\t\t\toverallSpeed.tokens += tokens;\n\t\t\toverallSpeed.evalMs += evalMs;\n\t\t}\n\t\treturn completion;\n\t};\n\tconst takeSurfaceSpeed = (): number | undefined => {\n\t\tconst speed =\n\t\t\tsurfaceSpeed.evalMs > 0 ? Math.round((surfaceSpeed.tokens / surfaceSpeed.evalMs) * 1000) : undefined;\n\t\tsurfaceSpeed = { tokens: 0, evalMs: 0 };\n\t\treturn speed;\n\t};\n\n\tconst research: LaneFitnessScore = { succeeded: 0, total: trials, outcomes: [], meanMs: 0 };\n\tfor (let i = 0; i < trials; i++) {\n\t\tconst started = now();\n\t\tconst result = await runResearch({\n\t\t\tquery: `fitness:probe requirements:req-${i}`,\n\t\t\tcontext: [\n\t\t\t\t\"Goal: add a retry helper to the HTTP client module\",\n\t\t\t\t\"Open requirements:\",\n\t\t\t\t\"- Find what retry/backoff conventions the codebase already uses\",\n\t\t\t\t\"- Identify which call sites would adopt the helper\",\n\t\t\t].join(\"\\n\"),\n\t\t\tenvelope: fitnessEnvelope(),\n\t\t\tmaxUsd: 1,\n\t\t\tmaxSources: 8,\n\t\t\tmaxFindings: 5,\n\t\t\tmaxWallClockMs,\n\t\t\tsignal: options.signal,\n\t\t\tcomplete,\n\t\t});\n\t\tresearch.meanMs += now() - started;\n\t\ttotalCostUsd += result.costUsd;\n\t\tif (result.status === \"succeeded\") research.succeeded++;\n\t\tresearch.outcomes.push(`${result.status}/${result.reasonCode}`);\n\t}\n\tresearch.meanMs = Math.round(research.meanMs / trials);\n\tresearch.tokensPerSecond = takeSurfaceSpeed();\n\n\tconst worker: LaneFitnessScore = { succeeded: 0, total: trials, outcomes: [], meanMs: 0 };\n\tfor (let i = 0; i < trials; i++) {\n\t\tconst started = now();\n\t\tconst outcome = await runWorker({\n\t\t\trequest: {\n\t\t\t\tid: `fitness-worker-${i}`,\n\t\t\t\tinstructions:\n\t\t\t\t\t\"Summarize in two sentences what a capability envelope is: a declared set of allowed tools, paths, and capability names that bounds what a delegated worker may do.\",\n\t\t\t\troute: { tier: \"cheap\", risk: \"read-only\", confidence: 1, reasonCode: \"fitness_probe\", reasons: [] },\n\t\t\t\tenvelope: { id: `fitness-env-${i}`, capabilities: [\"read_files\"], maxEstimatedUsd: 1 },\n\t\t\t\tmaxEstimatedUsd: 1,\n\t\t\t},\n\t\t\tmaxUsd: 1,\n\t\t\tmaxWallClockMs,\n\t\t\tusageReportId: `fitness:${i}`,\n\t\t\tsignal: options.signal,\n\t\t\tcomplete,\n\t\t});\n\t\tworker.meanMs += now() - started;\n\t\ttotalCostUsd += outcome.costUsd;\n\t\tif (outcome.result.status === \"completed\" && outcome.accepted) worker.succeeded++;\n\t\tworker.outcomes.push(`${outcome.result.status}/${outcome.reasonCode}`);\n\t}\n\tworker.meanMs = Math.round(worker.meanMs / trials);\n\tworker.tokensPerSecond = takeSurfaceSpeed();\n\n\tconst judge: JudgeFitnessScore = {\n\t\tparsed: 0,\n\t\tplanningElevated: 0,\n\t\tplanningTotal: judgePrompts.filter((entry) => entry.planning).length,\n\t\ttrivialCheap: 0,\n\t\ttrivialTotal: judgePrompts.filter((entry) => !entry.planning).length,\n\t\ttotal: judgePrompts.length,\n\t\toutcomes: [],\n\t\tmeanMs: 0,\n\t};\n\tfor (const entry of judgePrompts) {\n\t\tconst started = now();\n\t\tconst result = await runRouteJudge({\n\t\t\tprompt: entry.prompt,\n\t\t\tbaseline: { tier: \"cheap\", risk: \"read-only\", confidence: 0.5, reasonCode: \"fitness_probe\", reasons: [] },\n\t\t\tmaxWallClockMs,\n\t\t\tsignal: options.signal,\n\t\t\tcomplete,\n\t\t});\n\t\tjudge.meanMs += now() - started;\n\t\ttotalCostUsd += result.costUsd;\n\t\tconst tier = result.decision.tier;\n\t\tif (result.verdict) {\n\t\t\tjudge.parsed++;\n\t\t\t// A useful judge must both keep planning off the cheap tier AND actually send trivial\n\t\t\t// prompts there — all-medium verdicts are safe but save nothing.\n\t\t\tif (entry.planning && tier !== \"cheap\") judge.planningElevated++;\n\t\t\tif (!entry.planning && tier === \"cheap\") judge.trivialCheap++;\n\t\t}\n\t\tjudge.outcomes.push(\n\t\t\t`\"${entry.prompt.slice(0, 40)}\" -> ${tier}${result.fallbackReason ? ` (${result.fallbackReason})` : \"\"}`,\n\t\t);\n\t}\n\tjudge.meanMs = judgePrompts.length > 0 ? Math.round(judge.meanMs / judgePrompts.length) : 0;\n\tjudge.tokensPerSecond = takeSurfaceSpeed();\n\n\tconst probeSurface = async (\n\t\tsystemPrompt: string,\n\t\ttasks: readonly string[],\n\t\taccepts: (text: string, taskIndex: number) => boolean,\n\t): Promise<LaneFitnessScore> => {\n\t\tconst score: LaneFitnessScore = { succeeded: 0, total: tasks.length, outcomes: [], meanMs: 0 };\n\t\tfor (const [taskIndex, task] of tasks.entries()) {\n\t\t\tconst started = now();\n\t\t\t// Same wall-clock envelope as the lane surfaces — a hung model must not hang the probe.\n\t\t\tconst bounded = await runBoundedCompletion({\n\t\t\t\tmaxWallClockMs,\n\t\t\t\tsignal: options.signal,\n\t\t\t\texecute: (signal) => complete({ systemPrompt, userPrompt: task, signal }),\n\t\t\t});\n\t\t\tif (bounded.completion) totalCostUsd += bounded.completion.costUsd;\n\t\t\tif (bounded.failure || !bounded.completion) {\n\t\t\t\tscore.outcomes.push(bounded.failure ? bounded.failure.status : \"completion_error\");\n\t\t\t} else {\n\t\t\t\tconst ok = accepts(bounded.completion.text, taskIndex);\n\t\t\t\tif (ok) score.succeeded++;\n\t\t\t\tscore.outcomes.push(ok ? \"ok\" : \"unparseable_output\");\n\t\t\t}\n\t\t\tscore.meanMs += now() - started;\n\t\t}\n\t\tscore.meanMs = tasks.length > 0 ? Math.round(score.meanMs / tasks.length) : 0;\n\t\treturn score;\n\t};\n\n\tconst search = await probeSurface(SEARCH_PROBE_SYSTEM_PROMPT, SEARCH_PROBE_TASKS, parseSearchPlan);\n\tsearch.tokensPerSecond = takeSurfaceSpeed();\n\tconst toolCall = await probeSurface(TOOL_CALL_PROBE_SYSTEM_PROMPT, TOOL_CALL_PROBE_TASKS, parseToolCall);\n\ttoolCall.tokensPerSecond = takeSurfaceSpeed();\n\tconst digest = await probeSurface(\n\t\tCURATION_DIGEST_SYSTEM_PROMPT,\n\t\tDIGEST_PROBE_TASKS.map((task) => task.chunk),\n\t\t(text, taskIndex) => parseDigest(text, DIGEST_PROBE_TASKS[taskIndex]!.nonce),\n\t);\n\tdigest.tokensPerSecond = takeSurfaceSpeed();\n\n\tconst tokensPerSecond =\n\t\toverallSpeed.evalMs > 0 ? Math.round((overallSpeed.tokens / overallSpeed.evalMs) * 1000) : undefined;\n\n\treturn { trials, tokensPerSecond, research, worker, judge, search, toolCall, digest, totalCostUsd };\n}\n\n/**\n * Pure verdict: true when the probe found ZERO successes on every LANE surface it actually graded\n * AND the judge (if it ran) also failed. A lane/judge with total 0 (i.e. never run) carries no\n * evidence and is excluded from the lane check — but at least one lane must actually have been\n * graded for an all-failed verdict at all: `gradedLanes.every(...)` is vacuously true over an\n * empty array, so a report where only the judge ran (every research/worker/search/toolCall/digest\n * lane is ungraded) is excluded explicitly rather than misread as \"all lanes failed\" on zero lane\n * evidence. An empty/degenerate report (nothing graded at all, lanes AND judge) is likewise never\n * mistaken for a failed one. This is the gate adoption flows must consult before assigning a role —\n * see `isProbeAllFailed` callers in interactive-mode.ts and agent-session.ts.\n */\nexport function isProbeAllFailed(report: ModelFitnessReport): boolean {\n\tconst lanes = [report.research, report.worker, report.search, report.toolCall, report.digest];\n\tconst gradedLanes = lanes.filter((lane) => lane.total > 0);\n\tconst judgeGraded = report.judge.total > 0;\n\tif (gradedLanes.length === 0 && !judgeGraded) return false;\n\tconst lanesAllFailed = gradedLanes.length > 0 && gradedLanes.every((lane) => lane.succeeded === 0);\n\tconst judgeFailed = !judgeGraded || report.judge.parsed === 0;\n\treturn lanesAllFailed && judgeFailed;\n}\n\n/** Compact human-readable report for tool output / interactive display. Bounded, no raw dumps. */\nexport function formatModelFitnessReport(model: string, report: ModelFitnessReport): string {\n\tconst speed = (tokensPerSecond: number | undefined) =>\n\t\ttokensPerSecond !== undefined ? `, ~${tokensPerSecond} tok/s` : \"\";\n\tconst lines = [\n\t\t`Model fitness: ${model} (${report.trials} trials/lane${speed(report.tokensPerSecond)})`,\n\t\t`- research lane: ${report.research.succeeded}/${report.research.total} succeeded, mean ${report.research.meanMs}ms${speed(report.research.tokensPerSecond)} [${report.research.outcomes.join(\", \")}]`,\n\t\t`- worker lane: ${report.worker.succeeded}/${report.worker.total} completed+accepted, mean ${report.worker.meanMs}ms${speed(report.worker.tokensPerSecond)} [${report.worker.outcomes.join(\", \")}]`,\n\t\t`- search plans: ${report.search.succeeded}/${report.search.total} well-formed, mean ${report.search.meanMs}ms${speed(report.search.tokensPerSecond)}`,\n\t\t`- tool calls: ${report.toolCall.succeeded}/${report.toolCall.total} well-formed, mean ${report.toolCall.meanMs}ms${speed(report.toolCall.tokensPerSecond)}`,\n\t\t`- digests: ${report.digest.succeeded}/${report.digest.total} faithful, mean ${report.digest.meanMs}ms${speed(report.digest.tokensPerSecond)}`,\n\t\t`- route judge: parsed ${report.judge.parsed}/${report.judge.total}, planning-elevated ${report.judge.planningElevated}/${report.judge.planningTotal}, trivial-cheap ${report.judge.trivialCheap}/${report.judge.trivialTotal}, mean ${report.judge.meanMs}ms${speed(report.judge.tokensPerSecond)}`,\n\t\t...report.judge.outcomes.map((outcome) => ` ${outcome}`),\n\t];\n\tif (report.totalCostUsd > 0) {\n\t\tlines.push(`- probe cost: $${report.totalCostUsd.toFixed(4)}`);\n\t}\n\treturn lines.join(\"\\n\");\n}\n"]}
|
|
1
|
+
{"version":3,"file":"model-fitness.d.ts","sourceRoot":"","sources":["../../../src/core/research/model-fitness.ts"],"names":[],"mappings":"AAOA;;;;;;GAMG;AAEH,MAAM,WAAW,iBAAiB;IACjC,IAAI,EAAE,MAAM,CAAC;IACb,OAAO,EAAE,MAAM,CAAC;IAChB,UAAU,EAAE,MAAM,CAAC;IACnB,iGAAiG;IACjG,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,kGAAkG;IAClG,MAAM,CAAC,EAAE,MAAM,CAAC;CAChB;AAED,MAAM,MAAM,eAAe,GAAG,CAAC,IAAI,EAAE;IACpC,YAAY,EAAE,MAAM,CAAC;IACrB,UAAU,EAAE,MAAM,CAAC;IACnB,MAAM,CAAC,EAAE,WAAW,CAAC;CACrB,KAAK,OAAO,CAAC,iBAAiB,CAAC,CAAC;AAEjC,MAAM,WAAW,kBAAkB;IAClC,MAAM,EAAE,MAAM,CAAC;IACf,0EAA0E;IAC1E,QAAQ,EAAE,OAAO,CAAC;CAClB;AAED,qFAAqF;AACrF,eAAO,MAAM,6BAA6B,EAAE,SAAS,kBAAkB,EAOtE,CAAC;AAEF,MAAM,WAAW,oBAAoB;IACpC,uBAAuB,EAAE,MAAM,CAAC;IAChC,4FAA4F;IAC5F,gBAAgB,CAAC,EAAE,MAAM,CAAC;CAC1B;AAED,MAAM,WAAW,mBAAmB;IACnC,QAAQ,EAAE,eAAe,CAAC;IAC1B,0CAA0C;IAC1C,MAAM,CAAC,EAAE,MAAM,CAAC;IAChB,wDAAwD;IACxD,cAAc,CAAC,EAAE,MAAM,CAAC;IACxB,YAAY,CAAC,EAAE,SAAS,kBAAkB,EAAE,CAAC;IAC7C,uFAAuF;IACvF,aAAa,CAAC,EAAE,oBAAoB,CAAC;IACrC,MAAM,CAAC,EAAE,WAAW,CAAC;IACrB,gFAAgF;IAChF,GAAG,CAAC,EAAE,MAAM,MAAM,CAAC;CACnB;AAED,MAAM,WAAW,gBAAgB;IAChC,SAAS,EAAE,MAAM,CAAC;IAClB,KAAK,EAAE,MAAM,CAAC;IACd,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB,MAAM,EAAE,MAAM,CAAC;IACf,yFAAyF;IACzF,eAAe,CAAC,EAAE,MAAM,CAAC;CACzB;AAED,MAAM,WAAW,iBAAiB;IACjC,MAAM,EAAE,MAAM,CAAC;IACf,gBAAgB,EAAE,MAAM,CAAC;IACzB,aAAa,EAAE,MAAM,CAAC;IACtB,YAAY,EAAE,MAAM,CAAC;IACrB,YAAY,EAAE,MAAM,CAAC;IACrB,KAAK,EAAE,MAAM,CAAC;IACd,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB,MAAM,EAAE,MAAM,CAAC;IACf,qFAAqF;IACrF,eAAe,CAAC,EAAE,MAAM,CAAC;CACzB;AAED,MAAM,WAAW,oBAAoB;IACpC,uBAAuB,EAAE,MAAM,CAAC;IAChC,mBAAmB,EAAE,MAAM,CAAC;IAC5B,QAAQ,EAAE,MAAM,EAAE,CAAC;IACnB,MAAM,EAAE,MAAM,CAAC;CACf;AAED,MAAM,WAAW,kBAAkB;IAClC,MAAM,EAAE,MAAM,CAAC;IACf,yFAAyF;IACzF,eAAe,CAAC,EAAE,MAAM,CAAC;IACzB,QAAQ,EAAE,gBAAgB,CAAC;IAC3B,MAAM,EAAE,gBAAgB,CAAC;IACzB,KAAK,EAAE,iBAAiB,CAAC;IACzB,8EAA8E;IAC9E,MAAM,EAAE,gBAAgB,CAAC;IACzB,yFAAyF;IACzF,QAAQ,EAAE,gBAAgB,CAAC;IAC3B,qGAAqG;IACrG,MAAM,EAAE,gBAAgB,CAAC;IACzB,qFAAqF;IACrF,QAAQ,CAAC,EAAE,oBAAoB,CAAC;IAChC,YAAY,EAAE,MAAM,CAAC;CACrB;AAED,yFAAyF;AACzF,eAAO,MAAM,0BAA0B,QAK3B,CAAC;AAEb,eAAO,MAAM,6BAA6B,QAK9B,CAAC;AAEb,eAAO,MAAM,4BAA4B,QAI7B,CAAC;AASb,OAAO,EAAE,6BAA6B,IAAI,0BAA0B,EAAE,MAAM,6BAA6B,CAAC;AA0K1G,wBAAsB,oBAAoB,CAAC,OAAO,EAAE,mBAAmB,GAAG,OAAO,CAAC,kBAAkB,CAAC,CA+KpG;AAED;;;;;;;;;;GAUG;AACH,wBAAgB,gBAAgB,CAAC,MAAM,EAAE,kBAAkB,GAAG,OAAO,CAQpE;AAED,kGAAkG;AAClG,wBAAgB,wBAAwB,CAAC,KAAK,EAAE,MAAM,EAAE,MAAM,EAAE,kBAAkB,GAAG,MAAM,CAsB1F","sourcesContent":["import { runBoundedCompletion } from \"../autonomy/bounded-completion.ts\";\nimport type { CapabilityEnvelope } from \"../autonomy/contracts.ts\";\nimport { CURATION_DIGEST_SYSTEM_PROMPT } from \"../context/brain-curator.ts\";\nimport { runWorker } from \"../delegation/worker-runner.ts\";\nimport { runRouteJudge } from \"../model-router/route-judge.ts\";\nimport { runResearch } from \"./research-runner.ts\";\n\n/**\n * Model fitness probe: measures whether a candidate model can actually drive the harness's\n * subagent contracts — the research lane, the scout-worker lane, and the routing judge — by\n * running each real runner against the model and scoring parse/success rates plus judge\n * discrimination. Provider-free: the completion executor is injected, so this works against any\n * registered model (local Ollama models included) and against faux providers in tests.\n */\n\nexport interface FitnessCompletion {\n\ttext: string;\n\tcostUsd: number;\n\tstopReason: string;\n\t/** Output tokens generated (for tok/s). Optional: providers that don't report it are skipped. */\n\toutputTokens?: number;\n\t/** Pure generation time in ms (e.g. Ollama eval_duration). Falls back to wall-clock if absent. */\n\tevalMs?: number;\n}\n\nexport type FitnessComplete = (args: {\n\tsystemPrompt: string;\n\tuserPrompt: string;\n\tsignal?: AbortSignal;\n}) => Promise<FitnessCompletion>;\n\nexport interface JudgeFitnessPrompt {\n\tprompt: string;\n\t/** True when the prompt is planning-shaped and must never route cheap. */\n\tplanning: boolean;\n}\n\n/** Default judge probe set: three planning-shaped prompts, three trivial lookups. */\nexport const DEFAULT_JUDGE_FITNESS_PROMPTS: readonly JudgeFitnessPrompt[] = [\n\t{ prompt: \"how should we plan the migration of the session storage layer?\", planning: true },\n\t{ prompt: \"design an approach for splitting the settings manager\", planning: true },\n\t{ prompt: \"draft a roadmap for the autonomy rework\", planning: true },\n\t{ prompt: \"what does the resolvePath function return?\", planning: false },\n\t{ prompt: \"list the files in the delegation module\", planning: false },\n\t{ prompt: \"why is this test flaky?\", planning: false },\n];\n\nexport interface CapacityProbeOptions {\n\tregisteredContextWindow: number;\n\t/** Smallest candidate window to try before declaring the capacity unknown. Default 1024. */\n\tminContextWindow?: number;\n}\n\nexport interface ModelFitnessOptions {\n\tcomplete: FitnessComplete;\n\t/** Trials per lane surface. Default 3. */\n\ttrials?: number;\n\t/** Wall-clock budget per call in ms. Default 120000. */\n\tmaxWallClockMs?: number;\n\tjudgePrompts?: readonly JudgeFitnessPrompt[];\n\t/** Optional local-model capacity lane: measures the actually served context window. */\n\tcapacityProbe?: CapacityProbeOptions;\n\tsignal?: AbortSignal;\n\t/** Injected clock for latency measurement (test seam). Defaults to Date.now. */\n\tnow?: () => number;\n}\n\nexport interface LaneFitnessScore {\n\tsucceeded: number;\n\ttotal: number;\n\toutcomes: string[];\n\tmeanMs: number;\n\t/** Mean output tokens/second across the surface's calls; undefined when not reported. */\n\ttokensPerSecond?: number;\n}\n\nexport interface JudgeFitnessScore {\n\tparsed: number;\n\tplanningElevated: number;\n\tplanningTotal: number;\n\ttrivialCheap: number;\n\ttrivialTotal: number;\n\ttotal: number;\n\toutcomes: string[];\n\tmeanMs: number;\n\t/** Mean output tokens/second across the judge calls; undefined when not reported. */\n\ttokensPerSecond?: number;\n}\n\nexport interface CapacityFitnessScore {\n\tregisteredContextWindow: number;\n\tservedContextWindow: number;\n\toutcomes: string[];\n\tmeanMs: number;\n}\n\nexport interface ModelFitnessReport {\n\ttrials: number;\n\t/** Aggregate output tokens/second across ALL probe calls (the headline speed number). */\n\ttokensPerSecond?: number;\n\tresearch: LaneFitnessScore;\n\tworker: LaneFitnessScore;\n\tjudge: JudgeFitnessScore;\n\t/** Heavy-lifter surface: can the model formulate a structured search plan? */\n\tsearch: LaneFitnessScore;\n\t/** Heavy-lifter surface: can the model emit a well-formed tool call against a schema? */\n\ttoolCall: LaneFitnessScore;\n\t/** Curator surface: can the model digest a context chunk to strict JSON WITHOUT losing key facts? */\n\tdigest: LaneFitnessScore;\n\t/** Local-model capacity lane: measured context window the server actually serves. */\n\tcapacity?: CapacityFitnessScore;\n\ttotalCostUsd: number;\n}\n\n/** Static prompts for the heavy-lifter surfaces (stable for provider prompt caching). */\nexport const SEARCH_PROBE_SYSTEM_PROMPT = [\n\t\"You plan code searches for a coding agent. You never answer the question yourself.\",\n\t\"Given a question about a codebase, respond with STRICT JSON only - no prose:\",\n\t'{\"queries\":[{\"pattern\":\"<regex or literal to grep>\",\"glob\":\"<file glob like **/*.ts>\"}]}',\n\t\"Return 1 to 4 queries, most specific first.\",\n].join(\"\\n\");\n\nexport const TOOL_CALL_PROBE_SYSTEM_PROMPT = [\n\t\"You operate tools for a coding agent. You have exactly one tool:\",\n\t\"grep(pattern: string, path: string) - search files under a path for a pattern.\",\n\t\"Respond to every task with STRICT JSON only - no prose:\",\n\t'{\"tool\":\"grep\",\"arguments\":{\"pattern\":\"<pattern>\",\"path\":\"<path>\"}}',\n].join(\"\\n\");\n\nexport const CAPACITY_PROBE_SYSTEM_PROMPT = [\n\t\"You are a context-window capacity probe for a local model server.\",\n\t\"Find the unique NEEDLE token in the user's text and echo that token only.\",\n\t\"Do not summarize, explain, or add punctuation.\",\n].join(\"\\n\");\n\nconst SEARCH_PROBE_TASKS: readonly string[] = [\n\t\"Where is the retry/backoff logic for HTTP requests implemented?\",\n\t\"Which files define the settings for background research?\",\n\t\"Find where session entries of type custom are appended.\",\n];\n\n// The probe measures the REAL curation contract — same prompt the BrainCurator ships.\nexport { CURATION_DIGEST_SYSTEM_PROMPT as DIGEST_PROBE_SYSTEM_PROMPT } from \"../context/brain-curator.ts\";\n\n/**\n * Digest probe chunks each carry a NONCE identifier that cannot be guessed from the\n * instructions: acceptance requires the digest to RETAIN the nonce verbatim, so the score\n * measures extraction fidelity, not narration (a model cannot pass by paraphrasing).\n */\nconst DIGEST_PROBE_TASKS: readonly { chunk: string; nonce: string }[] = [\n\t{\n\t\tnonce: \"retryWithJitter_zx41\",\n\t\tchunk: [\n\t\t\t\"grep results for 'retry' under src/http:\",\n\t\t\t\"src/http/client.ts:88: export function retryWithJitter_zx41(fn, attempts = 3) {\",\n\t\t\t\"src/http/client.ts:112: // exponential backoff capped at 30s\",\n\t\t\t\"src/http/pool.ts:41: client.retry = false\",\n\t\t].join(\"\\n\"),\n\t},\n\t{\n\t\tnonce: \"ERR_QM_7734\",\n\t\tchunk: [\n\t\t\t\"$ npm run migrate\",\n\t\t\t\"migrating 14 files...\",\n\t\t\t\"error ERR_QM_7734: column 'owner_id' missing on table sessions (migration 0009)\",\n\t\t\t\"exit code 1\",\n\t\t].join(\"\\n\"),\n\t},\n\t{\n\t\tnonce: \"v3.9.2-hotfix.1\",\n\t\tchunk: [\n\t\t\t\"read package.json (34 lines):\",\n\t\t\t' \"name\": \"acme-billing\",',\n\t\t\t' \"version\": \"v3.9.2-hotfix.1\",',\n\t\t\t' \"engines\": { \"node\": \">=22\" },',\n\t\t].join(\"\\n\"),\n\t},\n];\n\nfunction parseDigest(text: string, nonce: string): boolean {\n\tconst parsed = extractJsonObject(text);\n\tif (!parsed) return false;\n\tconst digest = (parsed as { digest?: unknown }).digest;\n\tif (typeof digest !== \"string\") return false;\n\tconst trimmed = digest.trim();\n\t// Bounded and faithful: short enough to be a stub annotation, still carrying the nonce fact.\n\treturn trimmed.length > 0 && trimmed.length <= 240 && trimmed.includes(nonce);\n}\n\nconst TOOL_CALL_PROBE_TASKS: readonly string[] = [\n\t\"Find usages of the function resolveCliModel under src/.\",\n\t\"Search for the string 'budget_exhausted' in the core directory.\",\n\t\"Locate where LaneTracker is instantiated under src/core.\",\n];\n\nfunction parseSearchPlan(text: string): boolean {\n\tconst parsed = extractJsonObject(text);\n\tif (!parsed) return false;\n\tconst queries = (parsed as { queries?: unknown }).queries;\n\tif (!Array.isArray(queries) || queries.length === 0 || queries.length > 8) return false;\n\treturn queries.every(\n\t\t(query) =>\n\t\t\tquery &&\n\t\t\ttypeof query === \"object\" &&\n\t\t\ttypeof (query as { pattern?: unknown }).pattern === \"string\" &&\n\t\t\t(query as { pattern: string }).pattern.trim().length > 0,\n\t);\n}\n\nfunction parseToolCall(text: string): boolean {\n\tconst parsed = extractJsonObject(text);\n\tif (!parsed) return false;\n\tconst record = parsed as { tool?: unknown; arguments?: unknown };\n\tif (record.tool !== \"grep\") return false;\n\tconst args = record.arguments;\n\tif (!args || typeof args !== \"object\" || Array.isArray(args)) return false;\n\tconst pattern = (args as { pattern?: unknown }).pattern;\n\tconst path = (args as { path?: unknown }).path;\n\treturn (\n\t\ttypeof pattern === \"string\" && pattern.trim().length > 0 && typeof path === \"string\" && path.trim().length > 0\n\t);\n}\n\nfunction extractJsonObject(text: string): unknown | undefined {\n\tconst trimmed = text.trim();\n\tconst candidates: string[] = [trimmed];\n\tconst fenced = /```(?:json)?\\s*([\\s\\S]*?)```/.exec(trimmed);\n\tif (fenced?.[1]) candidates.push(fenced[1].trim());\n\tconst start = trimmed.indexOf(\"{\");\n\tconst end = trimmed.lastIndexOf(\"}\");\n\tif (start >= 0 && end > start) candidates.push(trimmed.slice(start, end + 1));\n\tfor (const candidate of candidates) {\n\t\ttry {\n\t\t\tconst parsed = JSON.parse(candidate);\n\t\t\tif (parsed && typeof parsed === \"object\" && !Array.isArray(parsed)) return parsed;\n\t\t} catch {\n\t\t\t// try next candidate\n\t\t}\n\t}\n\treturn undefined;\n}\n\nfunction fitnessEnvelope(): CapabilityEnvelope {\n\treturn {\n\t\tid: \"model-fitness-probe\",\n\t\tcapabilities: [\"research\", \"read_files\", \"memory_read\"],\n\t\tmaxEstimatedUsd: 1,\n\t\tcreatedAt: new Date().toISOString(),\n\t};\n}\n\nfunction buildCapacityProbePrompt(targetTokens: number, startNeedle: string, endNeedle: string): string {\n\tconst targetChars = Math.max(\n\t\tstartNeedle.length + endNeedle.length + 64,\n\t\ttargetTokens * 4 - CAPACITY_PROBE_SYSTEM_PROMPT.length,\n\t);\n\tconst prefix = `CAPACITY PROBE START\\n${startNeedle}\\n`;\n\tconst suffix = `\\n${endNeedle}\\nCAPACITY PROBE END`;\n\tconst fillerLength = Math.max(0, targetChars - prefix.length - suffix.length);\n\treturn `${prefix}${\"x \".repeat(Math.ceil(fillerLength / 2)).slice(0, fillerLength)}${suffix}`;\n}\n\nasync function measureServedContextWindow(args: {\n\tprobe: CapacityProbeOptions;\n\tcomplete: FitnessComplete;\n\tmaxWallClockMs: number;\n\tsignal?: AbortSignal;\n\tnow: () => number;\n}): Promise<{ score: CapacityFitnessScore; costUsd: number }> {\n\tconst registered = Math.max(1, Math.floor(args.probe.registeredContextWindow));\n\tconst minWindow = Math.max(1, Math.floor(args.probe.minContextWindow ?? 1024));\n\tconst nonce = `${registered.toString(16).toUpperCase()}_${minWindow.toString(16).toUpperCase()}`;\n\tconst startNeedle = `NEEDLE_START_${nonce}`;\n\tconst endNeedle = `NEEDLE_END_${nonce}`;\n\tconst score: CapacityFitnessScore = {\n\t\tregisteredContextWindow: registered,\n\t\tservedContextWindow: 0,\n\t\toutcomes: [],\n\t\tmeanMs: 0,\n\t};\n\tlet candidate = registered;\n\tlet costUsd = 0;\n\tlet calls = 0;\n\twhile (candidate >= minWindow) {\n\t\tconst started = args.now();\n\t\tcalls++;\n\t\tconst bounded = await runBoundedCompletion({\n\t\t\tmaxWallClockMs: args.maxWallClockMs,\n\t\t\tsignal: args.signal,\n\t\t\texecute: (signal) =>\n\t\t\t\targs.complete({\n\t\t\t\t\tsystemPrompt: CAPACITY_PROBE_SYSTEM_PROMPT,\n\t\t\t\t\tuserPrompt: buildCapacityProbePrompt(candidate, startNeedle, endNeedle),\n\t\t\t\t\tsignal,\n\t\t\t\t}),\n\t\t});\n\t\tscore.meanMs += args.now() - started;\n\t\tif (bounded.completion) costUsd += bounded.completion.costUsd;\n\t\tconst output = bounded.completion?.text.trim() ?? \"\";\n\t\tconst recalled = !bounded.failure && output.includes(startNeedle) && output.includes(endNeedle);\n\t\tscore.outcomes.push(`${candidate}:${recalled ? \"ok\" : (bounded.failure?.status ?? \"miss\")}`);\n\t\tif (recalled) {\n\t\t\tscore.servedContextWindow = candidate;\n\t\t\tbreak;\n\t\t}\n\t\tcandidate = Math.floor(candidate / 2);\n\t}\n\tif (score.servedContextWindow === 0) score.servedContextWindow = minWindow;\n\tscore.meanMs = calls > 0 ? Math.round(score.meanMs / calls) : 0;\n\treturn { score, costUsd };\n}\n\nexport async function runModelFitnessProbe(options: ModelFitnessOptions): Promise<ModelFitnessReport> {\n\tconst trials = Math.max(1, Math.min(options.trials ?? 3, 20));\n\tconst maxWallClockMs = options.maxWallClockMs ?? 120_000;\n\tconst judgePrompts = options.judgePrompts ?? DEFAULT_JUDGE_FITNESS_PROMPTS;\n\tconst now = options.now ?? Date.now;\n\tlet totalCostUsd = 0;\n\n\t// Token-speed instrumentation: the lane runners' own contracts carry text/cost only, so the\n\t// completer is wrapped once here and generation stats are accumulated per surface.\n\tconst overallSpeed = { tokens: 0, evalMs: 0 };\n\tlet surfaceSpeed = { tokens: 0, evalMs: 0 };\n\tconst complete: FitnessComplete = async (args) => {\n\t\tconst completion = await options.complete(args);\n\t\tconst tokens = completion.outputTokens ?? 0;\n\t\tconst evalMs = completion.evalMs ?? 0;\n\t\tif (tokens > 0 && evalMs > 0) {\n\t\t\tsurfaceSpeed.tokens += tokens;\n\t\t\tsurfaceSpeed.evalMs += evalMs;\n\t\t\toverallSpeed.tokens += tokens;\n\t\t\toverallSpeed.evalMs += evalMs;\n\t\t}\n\t\treturn completion;\n\t};\n\tconst takeSurfaceSpeed = (): number | undefined => {\n\t\tconst speed =\n\t\t\tsurfaceSpeed.evalMs > 0 ? Math.round((surfaceSpeed.tokens / surfaceSpeed.evalMs) * 1000) : undefined;\n\t\tsurfaceSpeed = { tokens: 0, evalMs: 0 };\n\t\treturn speed;\n\t};\n\n\tconst research: LaneFitnessScore = { succeeded: 0, total: trials, outcomes: [], meanMs: 0 };\n\tfor (let i = 0; i < trials; i++) {\n\t\tconst started = now();\n\t\tconst result = await runResearch({\n\t\t\tquery: `fitness:probe requirements:req-${i}`,\n\t\t\tcontext: [\n\t\t\t\t\"Goal: add a retry helper to the HTTP client module\",\n\t\t\t\t\"Open requirements:\",\n\t\t\t\t\"- Find what retry/backoff conventions the codebase already uses\",\n\t\t\t\t\"- Identify which call sites would adopt the helper\",\n\t\t\t].join(\"\\n\"),\n\t\t\tenvelope: fitnessEnvelope(),\n\t\t\tmaxUsd: 1,\n\t\t\tmaxSources: 8,\n\t\t\tmaxFindings: 5,\n\t\t\tmaxWallClockMs,\n\t\t\tsignal: options.signal,\n\t\t\tcomplete,\n\t\t});\n\t\tresearch.meanMs += now() - started;\n\t\ttotalCostUsd += result.costUsd;\n\t\tif (result.status === \"succeeded\") research.succeeded++;\n\t\tresearch.outcomes.push(`${result.status}/${result.reasonCode}`);\n\t}\n\tresearch.meanMs = Math.round(research.meanMs / trials);\n\tresearch.tokensPerSecond = takeSurfaceSpeed();\n\n\tconst worker: LaneFitnessScore = { succeeded: 0, total: trials, outcomes: [], meanMs: 0 };\n\tfor (let i = 0; i < trials; i++) {\n\t\tconst started = now();\n\t\tconst outcome = await runWorker({\n\t\t\trequest: {\n\t\t\t\tid: `fitness-worker-${i}`,\n\t\t\t\tinstructions:\n\t\t\t\t\t\"Summarize in two sentences what a capability envelope is: a declared set of allowed tools, paths, and capability names that bounds what a delegated worker may do.\",\n\t\t\t\troute: { tier: \"cheap\", risk: \"read-only\", confidence: 1, reasonCode: \"fitness_probe\", reasons: [] },\n\t\t\t\tenvelope: { id: `fitness-env-${i}`, capabilities: [\"read_files\"], maxEstimatedUsd: 1 },\n\t\t\t\tmaxEstimatedUsd: 1,\n\t\t\t},\n\t\t\tmaxUsd: 1,\n\t\t\tmaxWallClockMs,\n\t\t\tusageReportId: `fitness:${i}`,\n\t\t\tsignal: options.signal,\n\t\t\tcomplete,\n\t\t});\n\t\tworker.meanMs += now() - started;\n\t\ttotalCostUsd += outcome.costUsd;\n\t\tif (outcome.result.status === \"completed\" && outcome.accepted) worker.succeeded++;\n\t\tworker.outcomes.push(`${outcome.result.status}/${outcome.reasonCode}`);\n\t}\n\tworker.meanMs = Math.round(worker.meanMs / trials);\n\tworker.tokensPerSecond = takeSurfaceSpeed();\n\n\tconst judge: JudgeFitnessScore = {\n\t\tparsed: 0,\n\t\tplanningElevated: 0,\n\t\tplanningTotal: judgePrompts.filter((entry) => entry.planning).length,\n\t\ttrivialCheap: 0,\n\t\ttrivialTotal: judgePrompts.filter((entry) => !entry.planning).length,\n\t\ttotal: judgePrompts.length,\n\t\toutcomes: [],\n\t\tmeanMs: 0,\n\t};\n\tfor (const entry of judgePrompts) {\n\t\tconst started = now();\n\t\tconst result = await runRouteJudge({\n\t\t\tprompt: entry.prompt,\n\t\t\tbaseline: { tier: \"cheap\", risk: \"read-only\", confidence: 0.5, reasonCode: \"fitness_probe\", reasons: [] },\n\t\t\tmaxWallClockMs,\n\t\t\tsignal: options.signal,\n\t\t\tcomplete,\n\t\t});\n\t\tjudge.meanMs += now() - started;\n\t\ttotalCostUsd += result.costUsd;\n\t\tconst tier = result.decision.tier;\n\t\tif (result.verdict) {\n\t\t\tjudge.parsed++;\n\t\t\t// A useful judge must both keep planning off the cheap tier AND actually send trivial\n\t\t\t// prompts there — all-medium verdicts are safe but save nothing.\n\t\t\tif (entry.planning && tier !== \"cheap\") judge.planningElevated++;\n\t\t\tif (!entry.planning && tier === \"cheap\") judge.trivialCheap++;\n\t\t}\n\t\tjudge.outcomes.push(\n\t\t\t`\"${entry.prompt.slice(0, 40)}\" -> ${tier}${result.fallbackReason ? ` (${result.fallbackReason})` : \"\"}`,\n\t\t);\n\t}\n\tjudge.meanMs = judgePrompts.length > 0 ? Math.round(judge.meanMs / judgePrompts.length) : 0;\n\tjudge.tokensPerSecond = takeSurfaceSpeed();\n\n\tconst probeSurface = async (\n\t\tsystemPrompt: string,\n\t\ttasks: readonly string[],\n\t\taccepts: (text: string, taskIndex: number) => boolean,\n\t): Promise<LaneFitnessScore> => {\n\t\tconst score: LaneFitnessScore = { succeeded: 0, total: tasks.length, outcomes: [], meanMs: 0 };\n\t\tfor (const [taskIndex, task] of tasks.entries()) {\n\t\t\tconst started = now();\n\t\t\t// Same wall-clock envelope as the lane surfaces — a hung model must not hang the probe.\n\t\t\tconst bounded = await runBoundedCompletion({\n\t\t\t\tmaxWallClockMs,\n\t\t\t\tsignal: options.signal,\n\t\t\t\texecute: (signal) => complete({ systemPrompt, userPrompt: task, signal }),\n\t\t\t});\n\t\t\tif (bounded.completion) totalCostUsd += bounded.completion.costUsd;\n\t\t\tif (bounded.failure || !bounded.completion) {\n\t\t\t\tscore.outcomes.push(bounded.failure ? bounded.failure.status : \"completion_error\");\n\t\t\t} else {\n\t\t\t\tconst ok = accepts(bounded.completion.text, taskIndex);\n\t\t\t\tif (ok) score.succeeded++;\n\t\t\t\tscore.outcomes.push(ok ? \"ok\" : \"unparseable_output\");\n\t\t\t}\n\t\t\tscore.meanMs += now() - started;\n\t\t}\n\t\tscore.meanMs = tasks.length > 0 ? Math.round(score.meanMs / tasks.length) : 0;\n\t\treturn score;\n\t};\n\n\tconst search = await probeSurface(SEARCH_PROBE_SYSTEM_PROMPT, SEARCH_PROBE_TASKS, parseSearchPlan);\n\tsearch.tokensPerSecond = takeSurfaceSpeed();\n\tconst toolCall = await probeSurface(TOOL_CALL_PROBE_SYSTEM_PROMPT, TOOL_CALL_PROBE_TASKS, parseToolCall);\n\ttoolCall.tokensPerSecond = takeSurfaceSpeed();\n\tconst digest = await probeSurface(\n\t\tCURATION_DIGEST_SYSTEM_PROMPT,\n\t\tDIGEST_PROBE_TASKS.map((task) => task.chunk),\n\t\t(text, taskIndex) => parseDigest(text, DIGEST_PROBE_TASKS[taskIndex]!.nonce),\n\t);\n\tdigest.tokensPerSecond = takeSurfaceSpeed();\n\n\tlet capacity: CapacityFitnessScore | undefined;\n\tif (options.capacityProbe) {\n\t\tconst result = await measureServedContextWindow({\n\t\t\tprobe: options.capacityProbe,\n\t\t\tcomplete,\n\t\t\tmaxWallClockMs,\n\t\t\tsignal: options.signal,\n\t\t\tnow,\n\t\t});\n\t\tcapacity = result.score;\n\t\ttotalCostUsd += result.costUsd;\n\t}\n\n\tconst tokensPerSecond =\n\t\toverallSpeed.evalMs > 0 ? Math.round((overallSpeed.tokens / overallSpeed.evalMs) * 1000) : undefined;\n\n\treturn { trials, tokensPerSecond, research, worker, judge, search, toolCall, digest, capacity, totalCostUsd };\n}\n\n/**\n * Pure verdict: true when the probe found ZERO successes on every LANE surface it actually graded\n * AND the judge (if it ran) also failed. A lane/judge with total 0 (i.e. never run) carries no\n * evidence and is excluded from the lane check — but at least one lane must actually have been\n * graded for an all-failed verdict at all: `gradedLanes.every(...)` is vacuously true over an\n * empty array, so a report where only the judge ran (every research/worker/search/toolCall/digest\n * lane is ungraded) is excluded explicitly rather than misread as \"all lanes failed\" on zero lane\n * evidence. An empty/degenerate report (nothing graded at all, lanes AND judge) is likewise never\n * mistaken for a failed one. This is the gate adoption flows must consult before assigning a role —\n * see `isProbeAllFailed` callers in interactive-mode.ts and agent-session.ts.\n */\nexport function isProbeAllFailed(report: ModelFitnessReport): boolean {\n\tconst lanes = [report.research, report.worker, report.search, report.toolCall, report.digest];\n\tconst gradedLanes = lanes.filter((lane) => lane.total > 0);\n\tconst judgeGraded = report.judge.total > 0;\n\tif (gradedLanes.length === 0 && !judgeGraded) return false;\n\tconst lanesAllFailed = gradedLanes.length > 0 && gradedLanes.every((lane) => lane.succeeded === 0);\n\tconst judgeFailed = !judgeGraded || report.judge.parsed === 0;\n\treturn lanesAllFailed && judgeFailed;\n}\n\n/** Compact human-readable report for tool output / interactive display. Bounded, no raw dumps. */\nexport function formatModelFitnessReport(model: string, report: ModelFitnessReport): string {\n\tconst speed = (tokensPerSecond: number | undefined) =>\n\t\ttokensPerSecond !== undefined ? `, ~${tokensPerSecond} tok/s` : \"\";\n\tconst lines = [\n\t\t`Model fitness: ${model} (${report.trials} trials/lane${speed(report.tokensPerSecond)})`,\n\t\t`- research lane: ${report.research.succeeded}/${report.research.total} succeeded, mean ${report.research.meanMs}ms${speed(report.research.tokensPerSecond)} [${report.research.outcomes.join(\", \")}]`,\n\t\t`- worker lane: ${report.worker.succeeded}/${report.worker.total} completed+accepted, mean ${report.worker.meanMs}ms${speed(report.worker.tokensPerSecond)} [${report.worker.outcomes.join(\", \")}]`,\n\t\t`- search plans: ${report.search.succeeded}/${report.search.total} well-formed, mean ${report.search.meanMs}ms${speed(report.search.tokensPerSecond)}`,\n\t\t`- tool calls: ${report.toolCall.succeeded}/${report.toolCall.total} well-formed, mean ${report.toolCall.meanMs}ms${speed(report.toolCall.tokensPerSecond)}`,\n\t\t`- digests: ${report.digest.succeeded}/${report.digest.total} faithful, mean ${report.digest.meanMs}ms${speed(report.digest.tokensPerSecond)}`,\n\t\t...(report.capacity\n\t\t\t? [\n\t\t\t\t\t`- capacity: served window ${report.capacity.servedContextWindow}/${report.capacity.registeredContextWindow} tokens, mean ${report.capacity.meanMs}ms`,\n\t\t\t\t]\n\t\t\t: []),\n\t\t`- route judge: parsed ${report.judge.parsed}/${report.judge.total}, planning-elevated ${report.judge.planningElevated}/${report.judge.planningTotal}, trivial-cheap ${report.judge.trivialCheap}/${report.judge.trivialTotal}, mean ${report.judge.meanMs}ms${speed(report.judge.tokensPerSecond)}`,\n\t\t...report.judge.outcomes.map((outcome) => ` ${outcome}`),\n\t];\n\tif (report.totalCostUsd > 0) {\n\t\tlines.push(`- probe cost: $${report.totalCostUsd.toFixed(4)}`);\n\t}\n\treturn lines.join(\"\\n\");\n}\n"]}
|
|
@@ -25,6 +25,11 @@ export const TOOL_CALL_PROBE_SYSTEM_PROMPT = [
|
|
|
25
25
|
"Respond to every task with STRICT JSON only - no prose:",
|
|
26
26
|
'{"tool":"grep","arguments":{"pattern":"<pattern>","path":"<path>"}}',
|
|
27
27
|
].join("\n");
|
|
28
|
+
export const CAPACITY_PROBE_SYSTEM_PROMPT = [
|
|
29
|
+
"You are a context-window capacity probe for a local model server.",
|
|
30
|
+
"Find the unique NEEDLE token in the user's text and echo that token only.",
|
|
31
|
+
"Do not summarize, explain, or add punctuation.",
|
|
32
|
+
].join("\n");
|
|
28
33
|
const SEARCH_PROBE_TASKS = [
|
|
29
34
|
"Where is the retry/backoff logic for HTTP requests implemented?",
|
|
30
35
|
"Which files define the settings for background research?",
|
|
@@ -138,6 +143,57 @@ function fitnessEnvelope() {
|
|
|
138
143
|
createdAt: new Date().toISOString(),
|
|
139
144
|
};
|
|
140
145
|
}
|
|
146
|
+
function buildCapacityProbePrompt(targetTokens, startNeedle, endNeedle) {
|
|
147
|
+
const targetChars = Math.max(startNeedle.length + endNeedle.length + 64, targetTokens * 4 - CAPACITY_PROBE_SYSTEM_PROMPT.length);
|
|
148
|
+
const prefix = `CAPACITY PROBE START\n${startNeedle}\n`;
|
|
149
|
+
const suffix = `\n${endNeedle}\nCAPACITY PROBE END`;
|
|
150
|
+
const fillerLength = Math.max(0, targetChars - prefix.length - suffix.length);
|
|
151
|
+
return `${prefix}${"x ".repeat(Math.ceil(fillerLength / 2)).slice(0, fillerLength)}${suffix}`;
|
|
152
|
+
}
|
|
153
|
+
async function measureServedContextWindow(args) {
|
|
154
|
+
const registered = Math.max(1, Math.floor(args.probe.registeredContextWindow));
|
|
155
|
+
const minWindow = Math.max(1, Math.floor(args.probe.minContextWindow ?? 1024));
|
|
156
|
+
const nonce = `${registered.toString(16).toUpperCase()}_${minWindow.toString(16).toUpperCase()}`;
|
|
157
|
+
const startNeedle = `NEEDLE_START_${nonce}`;
|
|
158
|
+
const endNeedle = `NEEDLE_END_${nonce}`;
|
|
159
|
+
const score = {
|
|
160
|
+
registeredContextWindow: registered,
|
|
161
|
+
servedContextWindow: 0,
|
|
162
|
+
outcomes: [],
|
|
163
|
+
meanMs: 0,
|
|
164
|
+
};
|
|
165
|
+
let candidate = registered;
|
|
166
|
+
let costUsd = 0;
|
|
167
|
+
let calls = 0;
|
|
168
|
+
while (candidate >= minWindow) {
|
|
169
|
+
const started = args.now();
|
|
170
|
+
calls++;
|
|
171
|
+
const bounded = await runBoundedCompletion({
|
|
172
|
+
maxWallClockMs: args.maxWallClockMs,
|
|
173
|
+
signal: args.signal,
|
|
174
|
+
execute: (signal) => args.complete({
|
|
175
|
+
systemPrompt: CAPACITY_PROBE_SYSTEM_PROMPT,
|
|
176
|
+
userPrompt: buildCapacityProbePrompt(candidate, startNeedle, endNeedle),
|
|
177
|
+
signal,
|
|
178
|
+
}),
|
|
179
|
+
});
|
|
180
|
+
score.meanMs += args.now() - started;
|
|
181
|
+
if (bounded.completion)
|
|
182
|
+
costUsd += bounded.completion.costUsd;
|
|
183
|
+
const output = bounded.completion?.text.trim() ?? "";
|
|
184
|
+
const recalled = !bounded.failure && output.includes(startNeedle) && output.includes(endNeedle);
|
|
185
|
+
score.outcomes.push(`${candidate}:${recalled ? "ok" : (bounded.failure?.status ?? "miss")}`);
|
|
186
|
+
if (recalled) {
|
|
187
|
+
score.servedContextWindow = candidate;
|
|
188
|
+
break;
|
|
189
|
+
}
|
|
190
|
+
candidate = Math.floor(candidate / 2);
|
|
191
|
+
}
|
|
192
|
+
if (score.servedContextWindow === 0)
|
|
193
|
+
score.servedContextWindow = minWindow;
|
|
194
|
+
score.meanMs = calls > 0 ? Math.round(score.meanMs / calls) : 0;
|
|
195
|
+
return { score, costUsd };
|
|
196
|
+
}
|
|
141
197
|
export async function runModelFitnessProbe(options) {
|
|
142
198
|
const trials = Math.max(1, Math.min(options.trials ?? 3, 20));
|
|
143
199
|
const maxWallClockMs = options.maxWallClockMs ?? 120_000;
|
|
@@ -284,8 +340,20 @@ export async function runModelFitnessProbe(options) {
|
|
|
284
340
|
toolCall.tokensPerSecond = takeSurfaceSpeed();
|
|
285
341
|
const digest = await probeSurface(CURATION_DIGEST_SYSTEM_PROMPT, DIGEST_PROBE_TASKS.map((task) => task.chunk), (text, taskIndex) => parseDigest(text, DIGEST_PROBE_TASKS[taskIndex].nonce));
|
|
286
342
|
digest.tokensPerSecond = takeSurfaceSpeed();
|
|
343
|
+
let capacity;
|
|
344
|
+
if (options.capacityProbe) {
|
|
345
|
+
const result = await measureServedContextWindow({
|
|
346
|
+
probe: options.capacityProbe,
|
|
347
|
+
complete,
|
|
348
|
+
maxWallClockMs,
|
|
349
|
+
signal: options.signal,
|
|
350
|
+
now,
|
|
351
|
+
});
|
|
352
|
+
capacity = result.score;
|
|
353
|
+
totalCostUsd += result.costUsd;
|
|
354
|
+
}
|
|
287
355
|
const tokensPerSecond = overallSpeed.evalMs > 0 ? Math.round((overallSpeed.tokens / overallSpeed.evalMs) * 1000) : undefined;
|
|
288
|
-
return { trials, tokensPerSecond, research, worker, judge, search, toolCall, digest, totalCostUsd };
|
|
356
|
+
return { trials, tokensPerSecond, research, worker, judge, search, toolCall, digest, capacity, totalCostUsd };
|
|
289
357
|
}
|
|
290
358
|
/**
|
|
291
359
|
* Pure verdict: true when the probe found ZERO successes on every LANE surface it actually graded
|
|
@@ -318,6 +386,11 @@ export function formatModelFitnessReport(model, report) {
|
|
|
318
386
|
`- search plans: ${report.search.succeeded}/${report.search.total} well-formed, mean ${report.search.meanMs}ms${speed(report.search.tokensPerSecond)}`,
|
|
319
387
|
`- tool calls: ${report.toolCall.succeeded}/${report.toolCall.total} well-formed, mean ${report.toolCall.meanMs}ms${speed(report.toolCall.tokensPerSecond)}`,
|
|
320
388
|
`- digests: ${report.digest.succeeded}/${report.digest.total} faithful, mean ${report.digest.meanMs}ms${speed(report.digest.tokensPerSecond)}`,
|
|
389
|
+
...(report.capacity
|
|
390
|
+
? [
|
|
391
|
+
`- capacity: served window ${report.capacity.servedContextWindow}/${report.capacity.registeredContextWindow} tokens, mean ${report.capacity.meanMs}ms`,
|
|
392
|
+
]
|
|
393
|
+
: []),
|
|
321
394
|
`- route judge: parsed ${report.judge.parsed}/${report.judge.total}, planning-elevated ${report.judge.planningElevated}/${report.judge.planningTotal}, trivial-cheap ${report.judge.trivialCheap}/${report.judge.trivialTotal}, mean ${report.judge.meanMs}ms${speed(report.judge.tokensPerSecond)}`,
|
|
322
395
|
...report.judge.outcomes.map((outcome) => ` ${outcome}`),
|
|
323
396
|
];
|