@kolisachint/hoocode-agent 0.4.70 → 0.4.72
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +19 -0
- package/dist/cli/args.d.ts +2 -0
- package/dist/cli/args.d.ts.map +1 -1
- package/dist/cli/args.js +11 -0
- package/dist/cli/args.js.map +1 -1
- package/dist/core/agent-session-services.d.ts +1 -0
- package/dist/core/agent-session-services.d.ts.map +1 -1
- package/dist/core/agent-session-services.js +1 -0
- package/dist/core/agent-session-services.js.map +1 -1
- package/dist/core/agent-session.d.ts +17 -1
- package/dist/core/agent-session.d.ts.map +1 -1
- package/dist/core/agent-session.js +217 -21
- package/dist/core/agent-session.js.map +1 -1
- package/dist/core/model-registry.d.ts +6 -0
- package/dist/core/model-registry.d.ts.map +1 -1
- package/dist/core/model-registry.js +34 -0
- package/dist/core/model-registry.js.map +1 -1
- package/dist/core/routing/local-inference.d.ts +129 -0
- package/dist/core/routing/local-inference.d.ts.map +1 -0
- package/dist/core/routing/local-inference.js +154 -0
- package/dist/core/routing/local-inference.js.map +1 -0
- package/dist/core/routing/metrics.d.ts +27 -0
- package/dist/core/routing/metrics.d.ts.map +1 -0
- package/dist/core/routing/metrics.js +35 -0
- package/dist/core/routing/metrics.js.map +1 -0
- package/dist/core/routing/mlx-server.d.ts +43 -0
- package/dist/core/routing/mlx-server.d.ts.map +1 -0
- package/dist/core/routing/mlx-server.js +115 -0
- package/dist/core/routing/mlx-server.js.map +1 -0
- package/dist/core/routing/tool-result-prompts.d.ts +26 -0
- package/dist/core/routing/tool-result-prompts.d.ts.map +1 -0
- package/dist/core/routing/tool-result-prompts.js +45 -0
- package/dist/core/routing/tool-result-prompts.js.map +1 -0
- package/dist/core/sdk.d.ts +6 -0
- package/dist/core/sdk.d.ts.map +1 -1
- package/dist/core/sdk.js +1 -0
- package/dist/core/sdk.js.map +1 -1
- package/dist/main.d.ts.map +1 -1
- package/dist/main.js +1 -0
- package/dist/main.js.map +1 -1
- package/examples/extensions/custom-provider-anthropic/package.json +1 -1
- package/examples/extensions/custom-provider-gitlab-duo/package.json +1 -1
- package/examples/extensions/sandbox/package.json +1 -1
- package/examples/extensions/with-deps/package.json +1 -1
- package/package.json +4 -4
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"local-inference.js","sourceRoot":"","sources":["../../../src/core/routing/local-inference.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;GAgBG;AAcH,MAAM,CAAC,MAAM,aAAa,GAA2B;IACpD,cAAc;IACd,4BAA4B;IAC5B,2BAA2B;IAC3B,iBAAiB;CACR,CAAC;AAEX;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,kBAAkB,GAAG,IAAI,GAAG,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC;AAEpD;;;;;;GAMG;AACH,MAAM,CAAC,MAAM,iBAAiB,GAAG,IAAI,CAAC;AACtC,MAAM,CAAC,MAAM,iBAAiB,GAAG,IAAI,CAAC;AAkCtC,SAAS,aAAa,CAAC,KAAc,EAAwB;IAC5D,OAAO,OAAO,KAAK,KAAK,QAAQ,IAAK,aAAmC,CAAC,QAAQ,CAAC,KAAK,CAAC,CAAC;AAAA,CACzF;AAED;;;;;;;GAOG;AACH,MAAM,UAAU,kBAAkB,CAAC,IAIlC,EAAe;IACf,MAAM,OAAO,GAAG,IAAI,CAAC,OAAO,EAAE,IAAI,EAAE,CAAC;IACrC,MAAM,YAAY,GAAG,OAAO,KAAK,SAAS,IAAI,OAAO,KAAK,EAAE,IAAI,OAAO,KAAK,cAAc,CAAC;IAC3F,MAAM,SAAS,GAAG,IAAI,CAAC,UAAU,KAAK,IAAI,IAAI,YAAY,CAAC;IAC3D,IAAI,CAAC,SAAS;QAAE,OAAO,cAAc,CAAC;IAEtC,IAAI,OAAO,IAAI,aAAa,CAAC,OAAO,CAAC;QAAE,OAAO,OAAO,CAAC;IACtD,IAAI,IAAI,CAAC,UAAU,IAAI,aAAa,CAAC,IAAI,CAAC,UAAU,CAAC;QAAE,OAAO,IAAI,CAAC,UAAU,CAAC;IAC9E,OAAO,4BAA4B,CAAC;AAAA,CACpC;AAED;;;;GAIG;AACH,MAAM,OAAO,oBAAoB;IACf,IAAI,CAAc;IAClB,cAAc,CAA6B;IAC3C,QAAQ,CAAyB;IACjC,QAAQ,CAAS;IACjB,QAAQ,CAAS;IAElC,YACC,IAAiB,EACjB,cAA0C,EAC1C,QAAgC,EAC/B;QACD,IAAI,CAAC,IAAI,GAAG,IAAI,CAAC;QACjB,IAAI,CAAC,cAAc,GAAG,cAAc,CAAC;QACrC,IAAI,CAAC,QAAQ,GAAG,QAAQ,CAAC;QACzB,IAAI,CAAC,QAAQ,GAAG,cAAc,EAAE,QAAQ,IAAI,iBAAiB,CAAC;QAC9D,IAAI,CAAC,QAAQ,GAAG,cAAc,EAAE,QAAQ,IAAI,iBAAiB,CAAC;IAAA,CAC9D;IAED,MAAM,CAAC,MAAM,CAAC,IAIb,EAAwB;QACxB,MAAM,cAAc,GAAG,IAAI,CAAC,MAAM,EAAE,QAAQ,CAAC;QAC7C,IAAI,QAAgC,CAAC;QACrC,IAAI,IAAI,CAAC,IAAI,KAAK,cAAc,IAAI,cAAc,EAAE,CAAC;YACpD,QAAQ,GAAG,IAAI,CAAC,QAAQ,CAAC,IAAI,CAAC,cAAc,CAAC,QAAQ,EAAE,cAAc,CAAC,KAAK,CAAC,CAAC;QAC9E,CAAC;QACD,OAAO,IAAI,oBAAoB,CAAC,IAAI,CAAC,IAAI,EAAE,cAAc,EAAE,QAAQ,CAAC,CAAC;IAAA,CACrE;IAED,OAAO,GAAgB;QACtB,OAAO,IAAI,CAAC,IAAI,CAAC;IAAA,CACjB;IAED,gFAAgF;IAChF,mBAAmB,GAAY;QAC9B,OAAO,IAAI,CAAC,IAAI,KAAK,cAAc,IAAI,IAAI,CAAC,QAAQ,KAAK,SAAS,CAAC;IAAA,CACnE;IAED,iBAAiB,GAA+B;QAC/C,OAAO,IAAI,CAAC,cAAc,CAAC;IAAA,CAC3B;IAED;;;;;OAKG;IACH,WAAW,CAAC,QAAkB,EAAE,OAAmB,EAAc;QAChE,IAAI,CAAC,IAAI,CAAC,mBAAmB,EAAE,IAAI,CAAC,IAAI,CAAC,QAAQ;YAAE,OAAO,OAAO,CAAC;QAClE,QAAQ,IAAI,CAAC,IAAI,EAAE,CAAC;YACnB,KAAK,4BAA4B;gBAChC,OAAO,QAAQ,KAAK,eAAe,CAAC,CAAC,CAAC,IAAI,CAAC,QAAQ,CAAC,CAAC,CAAC,OAAO,CAAC;YAC/D,KAAK,2BAA2B;gBAC/B,OAAO,QAAQ,KAAK,aAAa,CAAC,CAAC,CAAC,IAAI,CAAC,QAAQ,CAAC,CAAC,CAAC,OAAO,CAAC;YAC7D,KAAK,iBAAiB,CAAC;YACvB,KAAK,cAAc;gBAClB,OAAO,OAAO,CAAC;QACjB,CAAC;IAAA,CACD;IAED,gFAAgF;IAChF,cAAc,CAAC,KAAa,EAAW;QACtC,OAAO,KAAK,IAAI,IAAI,CAAC,QAAQ,IAAI,KAAK,IAAI,IAAI,CAAC,QAAQ,CAAC;IAAA,CACxD;IAED,yDAAyD;IACzD,WAAW,GAA2C;QACrD,OAAO,EAAE,QAAQ,EAAE,IAAI,CAAC,QAAQ,EAAE,QAAQ,EAAE,IAAI,CAAC,QAAQ,EAAE,CAAC;IAAA,CAC5D;IAED,2EAA2E;IAC3E,wBAAwB,CAAC,QAAgB,EAAE,YAAoB,EAAW;QACzE,IAAI,IAAI,CAAC,IAAI,KAAK,2BAA2B;YAAE,OAAO,KAAK,CAAC;QAC5D,IAAI,CAAC,IAAI,CAAC,mBAAmB,EAAE;YAAE,OAAO,KAAK,CAAC;QAC9C,IAAI,CAAC,kBAAkB,CAAC,GAAG,CAAC,QAAQ,CAAC;YAAE,OAAO,KAAK,CAAC;QACpD,OAAO,IAAI,CAAC,cAAc,CAAC,YAAY,CAAC,CAAC;IAAA,CACzC;IAED;;;;;OAKG;IACH,wBAAwB,CAAC,KAAa,EAAW;QAChD,IAAI,IAAI,CAAC,IAAI,KAAK,4BAA4B;YAAE,OAAO,KAAK,CAAC;QAC7D,IAAI,CAAC,IAAI,CAAC,mBAAmB,EAAE;YAAE,OAAO,KAAK,CAAC;QAC9C,OAAO,IAAI,CAAC,cAAc,CAAC,KAAK,CAAC,CAAC;IAAA,CAClC;IAED,gBAAgB,GAA2B;QAC1C,OAAO,IAAI,CAAC,QAAQ,CAAC;IAAA,CACrB;CACD","sourcesContent":["/**\n * Local-inference routing.\n *\n * Optional, opt-in routing of certain non-critical work (conversation\n * compaction, and large bash tool-result compression) to a local\n * \"executor\" model running on an OpenAI-compatible endpoint (for example an\n * MLX server), while the primary model handles all planning, reasoning, edits,\n * and tool-call synthesis.\n *\n * Everything here is INERT unless explicitly enabled via the\n * `--enable-local-inference` flag or the `HOOCODE_ROUTING_MODE` env var. On any\n * executor resolution/availability problem the caller falls back to the primary\n * model (compaction) or the raw tool result (tool-result compression). The\n * router never throws into the agent loop.\n *\n * Design and validation: see docs/local-executor-routing.md.\n */\n\nimport type { Api, Model } from \"@kolisachint/hoocode-ai\";\nimport type { ModelRegistry } from \"../model-registry.js\";\n\n/** Work that may be routed to the executor instead of the primary model. */\nexport type TurnKind = \"primary\" | \"summarization\" | \"tool-result\";\n\nexport type RoutingMode =\n\t| \"primary-only\"\n\t| \"executor-for-summarization\"\n\t| \"executor-for-tool-results\"\n\t| \"shadow-executor\";\n\nexport const ROUTING_MODES: readonly RoutingMode[] = [\n\t\"primary-only\",\n\t\"executor-for-summarization\",\n\t\"executor-for-tool-results\",\n\t\"shadow-executor\",\n] as const;\n\n/**\n * Tool names whose output is worth compressing (validated). Others pass\n * through. Only `bash` qualifies: its verbose output is mostly low-value noise\n * around a few load-bearing facts. `read` was measured to compress ~0% on real\n * source code (every line is a keep-line) and was removed. Fact-list tools\n * (grep/find/ls) were never compressible (every line is a distinct fact).\n */\nexport const COMPRESSIBLE_TOOLS = new Set([\"bash\"]);\n\n/**\n * Global size band (bytes) for local-inference routing. Applies to BOTH\n * tool-result compression and compaction summarization. Inputs below the\n * minimum are not worth offloading; inputs above the maximum are slow and risk\n * GPU OOM on small machines, so they fall back to the primary model. Tunable\n * per-machine via `minBytes`/`maxBytes` in the executor config block.\n */\nexport const DEFAULT_MIN_BYTES = 2048;\nexport const DEFAULT_MAX_BYTES = 8192;\n\n/** Optional local server the harness manages for the executor. */\nexport interface ExecutorServerConfig {\n\t/** Command to launch (default: \"mlx_lm.server\"). */\n\tcommand?: string;\n\t/** Extra args appended to the launch command. */\n\targs?: string[];\n\t/** Host to health-check and bind (default: derived from executor baseUrl or 127.0.0.1). */\n\thost?: string;\n\t/** Port to health-check and bind (default: derived from executor baseUrl or 8080). */\n\tport?: number;\n\t/** Max milliseconds to wait for the server to become healthy (default: 30000). */\n\tstartupTimeoutMs?: number;\n}\n\n/** Executor model reference as configured in models.json. */\nexport interface ExecutorConfig {\n\tprovider: string;\n\tmodel: string;\n\t/** Minimum input size (bytes) before local inference is attempted. */\n\tminBytes?: number;\n\t/** Maximum input size (bytes); larger inputs fall back to the primary model. */\n\tmaxBytes?: number;\n\t/** When set, the harness spawns/health-checks/stops this local server. */\n\tserver?: ExecutorServerConfig;\n}\n\n/** `routing` block in models.json. */\nexport interface RoutingConfig {\n\tmode?: RoutingMode;\n\texecutor?: ExecutorConfig;\n}\n\nfunction isRoutingMode(value: unknown): value is RoutingMode {\n\treturn typeof value === \"string\" && (ROUTING_MODES as readonly string[]).includes(value);\n}\n\n/**\n * Resolve the effective routing mode from CLI flag, env var, and config.\n *\n * Activation requires either the flag or the env var; config alone never\n * activates routing (decision: explicit opt-in only). When activated without an\n * explicit mode, defaults to `executor-for-summarization` (the lowest-risk\n * mode). When not activated, always `primary-only`.\n */\nexport function resolveRoutingMode(opts: {\n\tenableFlag?: boolean;\n\tenvMode?: string;\n\tconfigMode?: RoutingMode;\n}): RoutingMode {\n\tconst envMode = opts.envMode?.trim();\n\tconst envActivates = envMode !== undefined && envMode !== \"\" && envMode !== \"primary-only\";\n\tconst activated = opts.enableFlag === true || envActivates;\n\tif (!activated) return \"primary-only\";\n\n\tif (envMode && isRoutingMode(envMode)) return envMode;\n\tif (opts.configMode && isRoutingMode(opts.configMode)) return opts.configMode;\n\treturn \"executor-for-summarization\";\n}\n\n/**\n * Router that decides, per turn kind, whether to use the executor model and\n * resolves it from the registry. Holds no mutable state beyond the resolved\n * executor model.\n */\nexport class LocalInferenceRouter {\n\tprivate readonly mode: RoutingMode;\n\tprivate readonly executorConfig: ExecutorConfig | undefined;\n\tprivate readonly executor: Model<Api> | undefined;\n\tprivate readonly minBytes: number;\n\tprivate readonly maxBytes: number;\n\n\tprivate constructor(\n\t\tmode: RoutingMode,\n\t\texecutorConfig: ExecutorConfig | undefined,\n\t\texecutor: Model<Api> | undefined,\n\t) {\n\t\tthis.mode = mode;\n\t\tthis.executorConfig = executorConfig;\n\t\tthis.executor = executor;\n\t\tthis.minBytes = executorConfig?.minBytes ?? DEFAULT_MIN_BYTES;\n\t\tthis.maxBytes = executorConfig?.maxBytes ?? DEFAULT_MAX_BYTES;\n\t}\n\n\tstatic create(opts: {\n\t\tmode: RoutingMode;\n\t\tconfig: RoutingConfig | undefined;\n\t\tregistry: ModelRegistry;\n\t}): LocalInferenceRouter {\n\t\tconst executorConfig = opts.config?.executor;\n\t\tlet executor: Model<Api> | undefined;\n\t\tif (opts.mode !== \"primary-only\" && executorConfig) {\n\t\t\texecutor = opts.registry.find(executorConfig.provider, executorConfig.model);\n\t\t}\n\t\treturn new LocalInferenceRouter(opts.mode, executorConfig, executor);\n\t}\n\n\tgetMode(): RoutingMode {\n\t\treturn this.mode;\n\t}\n\n\t/** True when routing is active and an executor model is resolved and usable. */\n\tisExecutorAvailable(): boolean {\n\t\treturn this.mode !== \"primary-only\" && this.executor !== undefined;\n\t}\n\n\tgetExecutorConfig(): ExecutorConfig | undefined {\n\t\treturn this.executorConfig;\n\t}\n\n\t/**\n\t * Pick the model to use for a turn. Returns the executor when the mode routes\n\t * that turn kind and the executor is available; otherwise returns the primary\n\t * model. `shadow-executor` always returns the primary for the live path (the\n\t * executor is exercised separately for measurement).\n\t */\n\tselectModel(turnKind: TurnKind, primary: Model<Api>): Model<Api> {\n\t\tif (!this.isExecutorAvailable() || !this.executor) return primary;\n\t\tswitch (this.mode) {\n\t\t\tcase \"executor-for-summarization\":\n\t\t\t\treturn turnKind === \"summarization\" ? this.executor : primary;\n\t\t\tcase \"executor-for-tool-results\":\n\t\t\t\treturn turnKind === \"tool-result\" ? this.executor : primary;\n\t\t\tcase \"shadow-executor\":\n\t\t\tcase \"primary-only\":\n\t\t\t\treturn primary;\n\t\t}\n\t}\n\n\t/** True when an input size falls within the configured local-inference band. */\n\twithinSizeBand(bytes: number): boolean {\n\t\treturn bytes >= this.minBytes && bytes <= this.maxBytes;\n\t}\n\n\t/** The configured size band, for logging/diagnostics. */\n\tgetSizeBand(): { minBytes: number; maxBytes: number } {\n\t\treturn { minBytes: this.minBytes, maxBytes: this.maxBytes };\n\t}\n\n\t/** Whether a given tool's result should be compressed via the executor. */\n\tshouldCompressToolResult(toolName: string, contentBytes: number): boolean {\n\t\tif (this.mode !== \"executor-for-tool-results\") return false;\n\t\tif (!this.isExecutorAvailable()) return false;\n\t\tif (!COMPRESSIBLE_TOOLS.has(toolName)) return false;\n\t\treturn this.withinSizeBand(contentBytes);\n\t}\n\n\t/**\n\t * Whether to route summarization to the executor for a conversation of the\n\t * given serialized size. Requires summarization routing active, the executor\n\t * available, and the size within the band (oversized conversations fall back\n\t * to the primary to avoid slow local runs and GPU OOM).\n\t */\n\tshouldRouteSummarization(bytes: number): boolean {\n\t\tif (this.mode !== \"executor-for-summarization\") return false;\n\t\tif (!this.isExecutorAvailable()) return false;\n\t\treturn this.withinSizeBand(bytes);\n\t}\n\n\tgetExecutorModel(): Model<Api> | undefined {\n\t\treturn this.executor;\n\t}\n}\n"]}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Local-inference routing metrics and diagnostic logging.
|
|
3
|
+
*
|
|
4
|
+
* Routing is an optimization, so failures must never surface as errors in the
|
|
5
|
+
* agent UI. This module provides a single, quiet sink for routing diagnostics
|
|
6
|
+
* and per-turn metrics. Output is gated behind HOOCODE_ROUTING_DEBUG so normal
|
|
7
|
+
* sessions stay silent (decision: fall back, log silently, never hard-fail).
|
|
8
|
+
*/
|
|
9
|
+
import type { TurnKind } from "./local-inference.js";
|
|
10
|
+
/** Record that an executor turn failed and the caller fell back to primary/raw. */
|
|
11
|
+
export declare function logLocalInferenceFallback(turnKind: TurnKind, error: unknown): void;
|
|
12
|
+
/** Per-turn routing metrics for cost/perf comparison. */
|
|
13
|
+
export interface RoutingMetrics {
|
|
14
|
+
turnKind: TurnKind;
|
|
15
|
+
provider: string;
|
|
16
|
+
model: string;
|
|
17
|
+
inputTokens?: number;
|
|
18
|
+
outputTokens?: number;
|
|
19
|
+
latencyMs: number;
|
|
20
|
+
/** Bytes before/after for tool-result compression. */
|
|
21
|
+
bytesBefore?: number;
|
|
22
|
+
bytesAfter?: number;
|
|
23
|
+
fallback: boolean;
|
|
24
|
+
}
|
|
25
|
+
/** Record per-turn routing metrics (quiet unless HOOCODE_ROUTING_DEBUG). */
|
|
26
|
+
export declare function logRoutingMetrics(m: RoutingMetrics): void;
|
|
27
|
+
//# sourceMappingURL=metrics.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"metrics.d.ts","sourceRoot":"","sources":["../../../src/core/routing/metrics.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAEH,OAAO,KAAK,EAAE,QAAQ,EAAE,MAAM,sBAAsB,CAAC;AAOrD,mFAAmF;AACnF,wBAAgB,yBAAyB,CAAC,QAAQ,EAAE,QAAQ,EAAE,KAAK,EAAE,OAAO,GAAG,IAAI,CAIlF;AAED,yDAAyD;AACzD,MAAM,WAAW,cAAc;IAC9B,QAAQ,EAAE,QAAQ,CAAC;IACnB,QAAQ,EAAE,MAAM,CAAC;IACjB,KAAK,EAAE,MAAM,CAAC;IACd,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,YAAY,CAAC,EAAE,MAAM,CAAC;IACtB,SAAS,EAAE,MAAM,CAAC;IAClB,sDAAsD;IACtD,WAAW,CAAC,EAAE,MAAM,CAAC;IACrB,UAAU,CAAC,EAAE,MAAM,CAAC;IACpB,QAAQ,EAAE,OAAO,CAAC;CAClB;AAED,4EAA4E;AAC5E,wBAAgB,iBAAiB,CAAC,CAAC,EAAE,cAAc,GAAG,IAAI,CAYzD","sourcesContent":["/**\n * Local-inference routing metrics and diagnostic logging.\n *\n * Routing is an optimization, so failures must never surface as errors in the\n * agent UI. This module provides a single, quiet sink for routing diagnostics\n * and per-turn metrics. Output is gated behind HOOCODE_ROUTING_DEBUG so normal\n * sessions stay silent (decision: fall back, log silently, never hard-fail).\n */\n\nimport type { TurnKind } from \"./local-inference.js\";\n\nfunction debugEnabled(): boolean {\n\tconst v = process.env.HOOCODE_ROUTING_DEBUG;\n\treturn v === \"1\" || v === \"true\" || v === \"yes\";\n}\n\n/** Record that an executor turn failed and the caller fell back to primary/raw. */\nexport function logLocalInferenceFallback(turnKind: TurnKind, error: unknown): void {\n\tif (!debugEnabled()) return;\n\tconst message = error instanceof Error ? error.message : String(error);\n\tconsole.warn(`[local-inference] ${turnKind} fell back to primary: ${message}`);\n}\n\n/** Per-turn routing metrics for cost/perf comparison. */\nexport interface RoutingMetrics {\n\tturnKind: TurnKind;\n\tprovider: string;\n\tmodel: string;\n\tinputTokens?: number;\n\toutputTokens?: number;\n\tlatencyMs: number;\n\t/** Bytes before/after for tool-result compression. */\n\tbytesBefore?: number;\n\tbytesAfter?: number;\n\tfallback: boolean;\n}\n\n/** Record per-turn routing metrics (quiet unless HOOCODE_ROUTING_DEBUG). */\nexport function logRoutingMetrics(m: RoutingMetrics): void {\n\tif (!debugEnabled()) return;\n\tconst parts = [\n\t\t`turn=${m.turnKind}`,\n\t\t`model=${m.provider}/${m.model}`,\n\t\t`latency=${m.latencyMs}ms`,\n\t\tm.inputTokens !== undefined ? `in=${m.inputTokens}` : undefined,\n\t\tm.outputTokens !== undefined ? `out=${m.outputTokens}` : undefined,\n\t\tm.bytesBefore !== undefined && m.bytesAfter !== undefined ? `bytes=${m.bytesBefore}->${m.bytesAfter}` : undefined,\n\t\tm.fallback ? \"fallback=true\" : undefined,\n\t].filter((p): p is string => p !== undefined);\n\tconsole.error(`[local-inference] ${parts.join(\" \")}`);\n}\n"]}
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Local-inference routing metrics and diagnostic logging.
|
|
3
|
+
*
|
|
4
|
+
* Routing is an optimization, so failures must never surface as errors in the
|
|
5
|
+
* agent UI. This module provides a single, quiet sink for routing diagnostics
|
|
6
|
+
* and per-turn metrics. Output is gated behind HOOCODE_ROUTING_DEBUG so normal
|
|
7
|
+
* sessions stay silent (decision: fall back, log silently, never hard-fail).
|
|
8
|
+
*/
|
|
9
|
+
function debugEnabled() {
|
|
10
|
+
const v = process.env.HOOCODE_ROUTING_DEBUG;
|
|
11
|
+
return v === "1" || v === "true" || v === "yes";
|
|
12
|
+
}
|
|
13
|
+
/** Record that an executor turn failed and the caller fell back to primary/raw. */
|
|
14
|
+
export function logLocalInferenceFallback(turnKind, error) {
|
|
15
|
+
if (!debugEnabled())
|
|
16
|
+
return;
|
|
17
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
18
|
+
console.warn(`[local-inference] ${turnKind} fell back to primary: ${message}`);
|
|
19
|
+
}
|
|
20
|
+
/** Record per-turn routing metrics (quiet unless HOOCODE_ROUTING_DEBUG). */
|
|
21
|
+
export function logRoutingMetrics(m) {
|
|
22
|
+
if (!debugEnabled())
|
|
23
|
+
return;
|
|
24
|
+
const parts = [
|
|
25
|
+
`turn=${m.turnKind}`,
|
|
26
|
+
`model=${m.provider}/${m.model}`,
|
|
27
|
+
`latency=${m.latencyMs}ms`,
|
|
28
|
+
m.inputTokens !== undefined ? `in=${m.inputTokens}` : undefined,
|
|
29
|
+
m.outputTokens !== undefined ? `out=${m.outputTokens}` : undefined,
|
|
30
|
+
m.bytesBefore !== undefined && m.bytesAfter !== undefined ? `bytes=${m.bytesBefore}->${m.bytesAfter}` : undefined,
|
|
31
|
+
m.fallback ? "fallback=true" : undefined,
|
|
32
|
+
].filter((p) => p !== undefined);
|
|
33
|
+
console.error(`[local-inference] ${parts.join(" ")}`);
|
|
34
|
+
}
|
|
35
|
+
//# sourceMappingURL=metrics.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"metrics.js","sourceRoot":"","sources":["../../../src/core/routing/metrics.ts"],"names":[],"mappings":"AAAA;;;;;;;GAOG;AAIH,SAAS,YAAY,GAAY;IAChC,MAAM,CAAC,GAAG,OAAO,CAAC,GAAG,CAAC,qBAAqB,CAAC;IAC5C,OAAO,CAAC,KAAK,GAAG,IAAI,CAAC,KAAK,MAAM,IAAI,CAAC,KAAK,KAAK,CAAC;AAAA,CAChD;AAED,mFAAmF;AACnF,MAAM,UAAU,yBAAyB,CAAC,QAAkB,EAAE,KAAc,EAAQ;IACnF,IAAI,CAAC,YAAY,EAAE;QAAE,OAAO;IAC5B,MAAM,OAAO,GAAG,KAAK,YAAY,KAAK,CAAC,CAAC,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,MAAM,CAAC,KAAK,CAAC,CAAC;IACvE,OAAO,CAAC,IAAI,CAAC,qBAAqB,QAAQ,0BAA0B,OAAO,EAAE,CAAC,CAAC;AAAA,CAC/E;AAgBD,4EAA4E;AAC5E,MAAM,UAAU,iBAAiB,CAAC,CAAiB,EAAQ;IAC1D,IAAI,CAAC,YAAY,EAAE;QAAE,OAAO;IAC5B,MAAM,KAAK,GAAG;QACb,QAAQ,CAAC,CAAC,QAAQ,EAAE;QACpB,SAAS,CAAC,CAAC,QAAQ,IAAI,CAAC,CAAC,KAAK,EAAE;QAChC,WAAW,CAAC,CAAC,SAAS,IAAI;QAC1B,CAAC,CAAC,WAAW,KAAK,SAAS,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,WAAW,EAAE,CAAC,CAAC,CAAC,SAAS;QAC/D,CAAC,CAAC,YAAY,KAAK,SAAS,CAAC,CAAC,CAAC,OAAO,CAAC,CAAC,YAAY,EAAE,CAAC,CAAC,CAAC,SAAS;QAClE,CAAC,CAAC,WAAW,KAAK,SAAS,IAAI,CAAC,CAAC,UAAU,KAAK,SAAS,CAAC,CAAC,CAAC,SAAS,CAAC,CAAC,WAAW,KAAK,CAAC,CAAC,UAAU,EAAE,CAAC,CAAC,CAAC,SAAS;QACjH,CAAC,CAAC,QAAQ,CAAC,CAAC,CAAC,eAAe,CAAC,CAAC,CAAC,SAAS;KACxC,CAAC,MAAM,CAAC,CAAC,CAAC,EAAe,EAAE,CAAC,CAAC,KAAK,SAAS,CAAC,CAAC;IAC9C,OAAO,CAAC,KAAK,CAAC,qBAAqB,KAAK,CAAC,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,CAAC;AAAA,CACtD","sourcesContent":["/**\n * Local-inference routing metrics and diagnostic logging.\n *\n * Routing is an optimization, so failures must never surface as errors in the\n * agent UI. This module provides a single, quiet sink for routing diagnostics\n * and per-turn metrics. Output is gated behind HOOCODE_ROUTING_DEBUG so normal\n * sessions stay silent (decision: fall back, log silently, never hard-fail).\n */\n\nimport type { TurnKind } from \"./local-inference.js\";\n\nfunction debugEnabled(): boolean {\n\tconst v = process.env.HOOCODE_ROUTING_DEBUG;\n\treturn v === \"1\" || v === \"true\" || v === \"yes\";\n}\n\n/** Record that an executor turn failed and the caller fell back to primary/raw. */\nexport function logLocalInferenceFallback(turnKind: TurnKind, error: unknown): void {\n\tif (!debugEnabled()) return;\n\tconst message = error instanceof Error ? error.message : String(error);\n\tconsole.warn(`[local-inference] ${turnKind} fell back to primary: ${message}`);\n}\n\n/** Per-turn routing metrics for cost/perf comparison. */\nexport interface RoutingMetrics {\n\tturnKind: TurnKind;\n\tprovider: string;\n\tmodel: string;\n\tinputTokens?: number;\n\toutputTokens?: number;\n\tlatencyMs: number;\n\t/** Bytes before/after for tool-result compression. */\n\tbytesBefore?: number;\n\tbytesAfter?: number;\n\tfallback: boolean;\n}\n\n/** Record per-turn routing metrics (quiet unless HOOCODE_ROUTING_DEBUG). */\nexport function logRoutingMetrics(m: RoutingMetrics): void {\n\tif (!debugEnabled()) return;\n\tconst parts = [\n\t\t`turn=${m.turnKind}`,\n\t\t`model=${m.provider}/${m.model}`,\n\t\t`latency=${m.latencyMs}ms`,\n\t\tm.inputTokens !== undefined ? `in=${m.inputTokens}` : undefined,\n\t\tm.outputTokens !== undefined ? `out=${m.outputTokens}` : undefined,\n\t\tm.bytesBefore !== undefined && m.bytesAfter !== undefined ? `bytes=${m.bytesBefore}->${m.bytesAfter}` : undefined,\n\t\tm.fallback ? \"fallback=true\" : undefined,\n\t].filter((p): p is string => p !== undefined);\n\tconsole.error(`[local-inference] ${parts.join(\" \")}`);\n}\n"]}
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Local executor server lifecycle manager.
|
|
3
|
+
*
|
|
4
|
+
* When `routing.executor.server` is configured AND local-inference routing is
|
|
5
|
+
* active, the harness can spawn a local OpenAI-compatible server (for example
|
|
6
|
+
* `mlx_lm.server`), wait for it to become healthy, and stop it on shutdown.
|
|
7
|
+
*
|
|
8
|
+
* This is best-effort: if the server fails to start or never becomes healthy,
|
|
9
|
+
* `ensureStarted()` resolves to false and the caller degrades to the primary
|
|
10
|
+
* model (routing falls back). It never throws into the agent loop.
|
|
11
|
+
*/
|
|
12
|
+
import type { ExecutorServerConfig } from "./local-inference.js";
|
|
13
|
+
export interface MlxServerOptions {
|
|
14
|
+
config: ExecutorServerConfig;
|
|
15
|
+
/** Executor model id, passed as --model. */
|
|
16
|
+
modelId: string;
|
|
17
|
+
/** Executor baseUrl, used to derive host/port and health endpoint. */
|
|
18
|
+
baseUrl?: string;
|
|
19
|
+
}
|
|
20
|
+
export declare class MlxServerManager {
|
|
21
|
+
private readonly command;
|
|
22
|
+
private readonly args;
|
|
23
|
+
private readonly host;
|
|
24
|
+
private readonly port;
|
|
25
|
+
private readonly startupTimeoutMs;
|
|
26
|
+
private readonly modelId;
|
|
27
|
+
private child;
|
|
28
|
+
private startPromise;
|
|
29
|
+
constructor(opts: MlxServerOptions);
|
|
30
|
+
private healthUrl;
|
|
31
|
+
private isHealthy;
|
|
32
|
+
/**
|
|
33
|
+
* Ensure the server is running and healthy. Idempotent: concurrent callers
|
|
34
|
+
* share one start attempt. Returns false on any failure (caller degrades to
|
|
35
|
+
* primary). If a server is already healthy (user-started), reuses it without
|
|
36
|
+
* spawning.
|
|
37
|
+
*/
|
|
38
|
+
ensureStarted(signal?: AbortSignal): Promise<boolean>;
|
|
39
|
+
private start;
|
|
40
|
+
/** Stop the spawned server (no-op if we reused an external one). */
|
|
41
|
+
stop(): void;
|
|
42
|
+
}
|
|
43
|
+
//# sourceMappingURL=mlx-server.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"mlx-server.d.ts","sourceRoot":"","sources":["../../../src/core/routing/mlx-server.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAGH,OAAO,KAAK,EAAE,oBAAoB,EAAE,MAAM,sBAAsB,CAAC;AAoBjE,MAAM,WAAW,gBAAgB;IAChC,MAAM,EAAE,oBAAoB,CAAC;IAC7B,4CAA4C;IAC5C,OAAO,EAAE,MAAM,CAAC;IAChB,sEAAsE;IACtE,OAAO,CAAC,EAAE,MAAM,CAAC;CACjB;AAED,qBAAa,gBAAgB;IAC5B,OAAO,CAAC,QAAQ,CAAC,OAAO,CAAS;IACjC,OAAO,CAAC,QAAQ,CAAC,IAAI,CAAW;IAChC,OAAO,CAAC,QAAQ,CAAC,IAAI,CAAS;IAC9B,OAAO,CAAC,QAAQ,CAAC,IAAI,CAAS;IAC9B,OAAO,CAAC,QAAQ,CAAC,gBAAgB,CAAS;IAC1C,OAAO,CAAC,QAAQ,CAAC,OAAO,CAAS;IACjC,OAAO,CAAC,KAAK,CAA2B;IACxC,OAAO,CAAC,YAAY,CAA+B;IAEnD,YAAY,IAAI,EAAE,gBAAgB,EAQjC;IAED,OAAO,CAAC,SAAS;YAIH,SAAS;IASvB;;;;;OAKG;IACG,aAAa,CAAC,MAAM,CAAC,EAAE,WAAW,GAAG,OAAO,CAAC,OAAO,CAAC,CAI1D;YAEa,KAAK;IAmCnB,oEAAoE;IACpE,IAAI,IAAI,IAAI,CAMX;CACD","sourcesContent":["/**\n * Local executor server lifecycle manager.\n *\n * When `routing.executor.server` is configured AND local-inference routing is\n * active, the harness can spawn a local OpenAI-compatible server (for example\n * `mlx_lm.server`), wait for it to become healthy, and stop it on shutdown.\n *\n * This is best-effort: if the server fails to start or never becomes healthy,\n * `ensureStarted()` resolves to false and the caller degrades to the primary\n * model (routing falls back). It never throws into the agent loop.\n */\n\nimport { type ChildProcess, spawn } from \"node:child_process\";\nimport type { ExecutorServerConfig } from \"./local-inference.js\";\nimport { logLocalInferenceFallback } from \"./metrics.js\";\n\nconst DEFAULT_COMMAND = \"mlx_lm.server\";\nconst DEFAULT_HOST = \"127.0.0.1\";\nconst DEFAULT_PORT = 8080;\nconst DEFAULT_STARTUP_TIMEOUT_MS = 30_000;\nconst HEALTH_POLL_INTERVAL_MS = 500;\n\n/** Derive host/port from a baseUrl like \"http://127.0.0.1:8080/v1\". */\nfunction parseBaseUrl(baseUrl: string | undefined): { host?: string; port?: number } {\n\tif (!baseUrl) return {};\n\ttry {\n\t\tconst u = new URL(baseUrl);\n\t\treturn { host: u.hostname, port: u.port ? Number(u.port) : undefined };\n\t} catch {\n\t\treturn {};\n\t}\n}\n\nexport interface MlxServerOptions {\n\tconfig: ExecutorServerConfig;\n\t/** Executor model id, passed as --model. */\n\tmodelId: string;\n\t/** Executor baseUrl, used to derive host/port and health endpoint. */\n\tbaseUrl?: string;\n}\n\nexport class MlxServerManager {\n\tprivate readonly command: string;\n\tprivate readonly args: string[];\n\tprivate readonly host: string;\n\tprivate readonly port: number;\n\tprivate readonly startupTimeoutMs: number;\n\tprivate readonly modelId: string;\n\tprivate child: ChildProcess | undefined;\n\tprivate startPromise: Promise<boolean> | undefined;\n\n\tconstructor(opts: MlxServerOptions) {\n\t\tconst fromUrl = parseBaseUrl(opts.baseUrl);\n\t\tthis.command = opts.config.command ?? DEFAULT_COMMAND;\n\t\tthis.args = opts.config.args ?? [];\n\t\tthis.host = opts.config.host ?? fromUrl.host ?? DEFAULT_HOST;\n\t\tthis.port = opts.config.port ?? fromUrl.port ?? DEFAULT_PORT;\n\t\tthis.startupTimeoutMs = opts.config.startupTimeoutMs ?? DEFAULT_STARTUP_TIMEOUT_MS;\n\t\tthis.modelId = opts.modelId;\n\t}\n\n\tprivate healthUrl(): string {\n\t\treturn `http://${this.host}:${this.port}/v1/models`;\n\t}\n\n\tprivate async isHealthy(signal?: AbortSignal): Promise<boolean> {\n\t\ttry {\n\t\t\tconst res = await fetch(this.healthUrl(), { method: \"GET\", signal });\n\t\t\treturn res.ok;\n\t\t} catch {\n\t\t\treturn false;\n\t\t}\n\t}\n\n\t/**\n\t * Ensure the server is running and healthy. Idempotent: concurrent callers\n\t * share one start attempt. Returns false on any failure (caller degrades to\n\t * primary). If a server is already healthy (user-started), reuses it without\n\t * spawning.\n\t */\n\tasync ensureStarted(signal?: AbortSignal): Promise<boolean> {\n\t\tif (this.startPromise) return this.startPromise;\n\t\tthis.startPromise = this.start(signal);\n\t\treturn this.startPromise;\n\t}\n\n\tprivate async start(signal?: AbortSignal): Promise<boolean> {\n\t\t// Reuse an already-running server (e.g. user-started) without spawning.\n\t\tif (await this.isHealthy(signal)) return true;\n\n\t\ttry {\n\t\t\tthis.child = spawn(\n\t\t\t\tthis.command,\n\t\t\t\t[...this.args, \"--model\", this.modelId, \"--host\", this.host, \"--port\", String(this.port)],\n\t\t\t\t{\n\t\t\t\t\tstdio: \"ignore\",\n\t\t\t\t\tdetached: false,\n\t\t\t\t},\n\t\t\t);\n\t\t\tthis.child.on(\"error\", (error) => {\n\t\t\t\tlogLocalInferenceFallback(\"primary\", error);\n\t\t\t\tthis.child = undefined;\n\t\t\t});\n\t\t} catch (error) {\n\t\t\tlogLocalInferenceFallback(\"primary\", error);\n\t\t\tthis.child = undefined;\n\t\t\treturn false;\n\t\t}\n\n\t\tconst deadline = Date.now() + this.startupTimeoutMs;\n\t\twhile (Date.now() < deadline) {\n\t\t\tif (signal?.aborted) return false;\n\t\t\tif (!this.child) return false; // spawn errored\n\t\t\tif (await this.isHealthy(signal)) return true;\n\t\t\tawait new Promise((r) => setTimeout(r, HEALTH_POLL_INTERVAL_MS));\n\t\t}\n\t\t// Timed out: stop whatever we spawned and degrade.\n\t\tthis.stop();\n\t\treturn false;\n\t}\n\n\t/** Stop the spawned server (no-op if we reused an external one). */\n\tstop(): void {\n\t\tif (this.child && !this.child.killed) {\n\t\t\tthis.child.kill(\"SIGTERM\");\n\t\t}\n\t\tthis.child = undefined;\n\t\tthis.startPromise = undefined;\n\t}\n}\n"]}
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Local executor server lifecycle manager.
|
|
3
|
+
*
|
|
4
|
+
* When `routing.executor.server` is configured AND local-inference routing is
|
|
5
|
+
* active, the harness can spawn a local OpenAI-compatible server (for example
|
|
6
|
+
* `mlx_lm.server`), wait for it to become healthy, and stop it on shutdown.
|
|
7
|
+
*
|
|
8
|
+
* This is best-effort: if the server fails to start or never becomes healthy,
|
|
9
|
+
* `ensureStarted()` resolves to false and the caller degrades to the primary
|
|
10
|
+
* model (routing falls back). It never throws into the agent loop.
|
|
11
|
+
*/
|
|
12
|
+
import { spawn } from "node:child_process";
|
|
13
|
+
import { logLocalInferenceFallback } from "./metrics.js";
|
|
14
|
+
const DEFAULT_COMMAND = "mlx_lm.server";
|
|
15
|
+
const DEFAULT_HOST = "127.0.0.1";
|
|
16
|
+
const DEFAULT_PORT = 8080;
|
|
17
|
+
const DEFAULT_STARTUP_TIMEOUT_MS = 30_000;
|
|
18
|
+
const HEALTH_POLL_INTERVAL_MS = 500;
|
|
19
|
+
/** Derive host/port from a baseUrl like "http://127.0.0.1:8080/v1". */
|
|
20
|
+
function parseBaseUrl(baseUrl) {
|
|
21
|
+
if (!baseUrl)
|
|
22
|
+
return {};
|
|
23
|
+
try {
|
|
24
|
+
const u = new URL(baseUrl);
|
|
25
|
+
return { host: u.hostname, port: u.port ? Number(u.port) : undefined };
|
|
26
|
+
}
|
|
27
|
+
catch {
|
|
28
|
+
return {};
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
export class MlxServerManager {
|
|
32
|
+
command;
|
|
33
|
+
args;
|
|
34
|
+
host;
|
|
35
|
+
port;
|
|
36
|
+
startupTimeoutMs;
|
|
37
|
+
modelId;
|
|
38
|
+
child;
|
|
39
|
+
startPromise;
|
|
40
|
+
constructor(opts) {
|
|
41
|
+
const fromUrl = parseBaseUrl(opts.baseUrl);
|
|
42
|
+
this.command = opts.config.command ?? DEFAULT_COMMAND;
|
|
43
|
+
this.args = opts.config.args ?? [];
|
|
44
|
+
this.host = opts.config.host ?? fromUrl.host ?? DEFAULT_HOST;
|
|
45
|
+
this.port = opts.config.port ?? fromUrl.port ?? DEFAULT_PORT;
|
|
46
|
+
this.startupTimeoutMs = opts.config.startupTimeoutMs ?? DEFAULT_STARTUP_TIMEOUT_MS;
|
|
47
|
+
this.modelId = opts.modelId;
|
|
48
|
+
}
|
|
49
|
+
healthUrl() {
|
|
50
|
+
return `http://${this.host}:${this.port}/v1/models`;
|
|
51
|
+
}
|
|
52
|
+
async isHealthy(signal) {
|
|
53
|
+
try {
|
|
54
|
+
const res = await fetch(this.healthUrl(), { method: "GET", signal });
|
|
55
|
+
return res.ok;
|
|
56
|
+
}
|
|
57
|
+
catch {
|
|
58
|
+
return false;
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
/**
|
|
62
|
+
* Ensure the server is running and healthy. Idempotent: concurrent callers
|
|
63
|
+
* share one start attempt. Returns false on any failure (caller degrades to
|
|
64
|
+
* primary). If a server is already healthy (user-started), reuses it without
|
|
65
|
+
* spawning.
|
|
66
|
+
*/
|
|
67
|
+
async ensureStarted(signal) {
|
|
68
|
+
if (this.startPromise)
|
|
69
|
+
return this.startPromise;
|
|
70
|
+
this.startPromise = this.start(signal);
|
|
71
|
+
return this.startPromise;
|
|
72
|
+
}
|
|
73
|
+
async start(signal) {
|
|
74
|
+
// Reuse an already-running server (e.g. user-started) without spawning.
|
|
75
|
+
if (await this.isHealthy(signal))
|
|
76
|
+
return true;
|
|
77
|
+
try {
|
|
78
|
+
this.child = spawn(this.command, [...this.args, "--model", this.modelId, "--host", this.host, "--port", String(this.port)], {
|
|
79
|
+
stdio: "ignore",
|
|
80
|
+
detached: false,
|
|
81
|
+
});
|
|
82
|
+
this.child.on("error", (error) => {
|
|
83
|
+
logLocalInferenceFallback("primary", error);
|
|
84
|
+
this.child = undefined;
|
|
85
|
+
});
|
|
86
|
+
}
|
|
87
|
+
catch (error) {
|
|
88
|
+
logLocalInferenceFallback("primary", error);
|
|
89
|
+
this.child = undefined;
|
|
90
|
+
return false;
|
|
91
|
+
}
|
|
92
|
+
const deadline = Date.now() + this.startupTimeoutMs;
|
|
93
|
+
while (Date.now() < deadline) {
|
|
94
|
+
if (signal?.aborted)
|
|
95
|
+
return false;
|
|
96
|
+
if (!this.child)
|
|
97
|
+
return false; // spawn errored
|
|
98
|
+
if (await this.isHealthy(signal))
|
|
99
|
+
return true;
|
|
100
|
+
await new Promise((r) => setTimeout(r, HEALTH_POLL_INTERVAL_MS));
|
|
101
|
+
}
|
|
102
|
+
// Timed out: stop whatever we spawned and degrade.
|
|
103
|
+
this.stop();
|
|
104
|
+
return false;
|
|
105
|
+
}
|
|
106
|
+
/** Stop the spawned server (no-op if we reused an external one). */
|
|
107
|
+
stop() {
|
|
108
|
+
if (this.child && !this.child.killed) {
|
|
109
|
+
this.child.kill("SIGTERM");
|
|
110
|
+
}
|
|
111
|
+
this.child = undefined;
|
|
112
|
+
this.startPromise = undefined;
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
//# sourceMappingURL=mlx-server.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"mlx-server.js","sourceRoot":"","sources":["../../../src/core/routing/mlx-server.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;GAUG;AAEH,OAAO,EAAqB,KAAK,EAAE,MAAM,oBAAoB,CAAC;AAE9D,OAAO,EAAE,yBAAyB,EAAE,MAAM,cAAc,CAAC;AAEzD,MAAM,eAAe,GAAG,eAAe,CAAC;AACxC,MAAM,YAAY,GAAG,WAAW,CAAC;AACjC,MAAM,YAAY,GAAG,IAAI,CAAC;AAC1B,MAAM,0BAA0B,GAAG,MAAM,CAAC;AAC1C,MAAM,uBAAuB,GAAG,GAAG,CAAC;AAEpC,uEAAuE;AACvE,SAAS,YAAY,CAAC,OAA2B,EAAoC;IACpF,IAAI,CAAC,OAAO;QAAE,OAAO,EAAE,CAAC;IACxB,IAAI,CAAC;QACJ,MAAM,CAAC,GAAG,IAAI,GAAG,CAAC,OAAO,CAAC,CAAC;QAC3B,OAAO,EAAE,IAAI,EAAE,CAAC,CAAC,QAAQ,EAAE,IAAI,EAAE,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,SAAS,EAAE,CAAC;IACxE,CAAC;IAAC,MAAM,CAAC;QACR,OAAO,EAAE,CAAC;IACX,CAAC;AAAA,CACD;AAUD,MAAM,OAAO,gBAAgB;IACX,OAAO,CAAS;IAChB,IAAI,CAAW;IACf,IAAI,CAAS;IACb,IAAI,CAAS;IACb,gBAAgB,CAAS;IACzB,OAAO,CAAS;IACzB,KAAK,CAA2B;IAChC,YAAY,CAA+B;IAEnD,YAAY,IAAsB,EAAE;QACnC,MAAM,OAAO,GAAG,YAAY,CAAC,IAAI,CAAC,OAAO,CAAC,CAAC;QAC3C,IAAI,CAAC,OAAO,GAAG,IAAI,CAAC,MAAM,CAAC,OAAO,IAAI,eAAe,CAAC;QACtD,IAAI,CAAC,IAAI,GAAG,IAAI,CAAC,MAAM,CAAC,IAAI,IAAI,EAAE,CAAC;QACnC,IAAI,CAAC,IAAI,GAAG,IAAI,CAAC,MAAM,CAAC,IAAI,IAAI,OAAO,CAAC,IAAI,IAAI,YAAY,CAAC;QAC7D,IAAI,CAAC,IAAI,GAAG,IAAI,CAAC,MAAM,CAAC,IAAI,IAAI,OAAO,CAAC,IAAI,IAAI,YAAY,CAAC;QAC7D,IAAI,CAAC,gBAAgB,GAAG,IAAI,CAAC,MAAM,CAAC,gBAAgB,IAAI,0BAA0B,CAAC;QACnF,IAAI,CAAC,OAAO,GAAG,IAAI,CAAC,OAAO,CAAC;IAAA,CAC5B;IAEO,SAAS,GAAW;QAC3B,OAAO,UAAU,IAAI,CAAC,IAAI,IAAI,IAAI,CAAC,IAAI,YAAY,CAAC;IAAA,CACpD;IAEO,KAAK,CAAC,SAAS,CAAC,MAAoB,EAAoB;QAC/D,IAAI,CAAC;YACJ,MAAM,GAAG,GAAG,MAAM,KAAK,CAAC,IAAI,CAAC,SAAS,EAAE,EAAE,EAAE,MAAM,EAAE,KAAK,EAAE,MAAM,EAAE,CAAC,CAAC;YACrE,OAAO,GAAG,CAAC,EAAE,CAAC;QACf,CAAC;QAAC,MAAM,CAAC;YACR,OAAO,KAAK,CAAC;QACd,CAAC;IAAA,CACD;IAED;;;;;OAKG;IACH,KAAK,CAAC,aAAa,CAAC,MAAoB,EAAoB;QAC3D,IAAI,IAAI,CAAC,YAAY;YAAE,OAAO,IAAI,CAAC,YAAY,CAAC;QAChD,IAAI,CAAC,YAAY,GAAG,IAAI,CAAC,KAAK,CAAC,MAAM,CAAC,CAAC;QACvC,OAAO,IAAI,CAAC,YAAY,CAAC;IAAA,CACzB;IAEO,KAAK,CAAC,KAAK,CAAC,MAAoB,EAAoB;QAC3D,wEAAwE;QACxE,IAAI,MAAM,IAAI,CAAC,SAAS,CAAC,MAAM,CAAC;YAAE,OAAO,IAAI,CAAC;QAE9C,IAAI,CAAC;YACJ,IAAI,CAAC,KAAK,GAAG,KAAK,CACjB,IAAI,CAAC,OAAO,EACZ,CAAC,GAAG,IAAI,CAAC,IAAI,EAAE,SAAS,EAAE,IAAI,CAAC,OAAO,EAAE,QAAQ,EAAE,IAAI,CAAC,IAAI,EAAE,QAAQ,EAAE,MAAM,CAAC,IAAI,CAAC,IAAI,CAAC,CAAC,EACzF;gBACC,KAAK,EAAE,QAAQ;gBACf,QAAQ,EAAE,KAAK;aACf,CACD,CAAC;YACF,IAAI,CAAC,KAAK,CAAC,EAAE,CAAC,OAAO,EAAE,CAAC,KAAK,EAAE,EAAE,CAAC;gBACjC,yBAAyB,CAAC,SAAS,EAAE,KAAK,CAAC,CAAC;gBAC5C,IAAI,CAAC,KAAK,GAAG,SAAS,CAAC;YAAA,CACvB,CAAC,CAAC;QACJ,CAAC;QAAC,OAAO,KAAK,EAAE,CAAC;YAChB,yBAAyB,CAAC,SAAS,EAAE,KAAK,CAAC,CAAC;YAC5C,IAAI,CAAC,KAAK,GAAG,SAAS,CAAC;YACvB,OAAO,KAAK,CAAC;QACd,CAAC;QAED,MAAM,QAAQ,GAAG,IAAI,CAAC,GAAG,EAAE,GAAG,IAAI,CAAC,gBAAgB,CAAC;QACpD,OAAO,IAAI,CAAC,GAAG,EAAE,GAAG,QAAQ,EAAE,CAAC;YAC9B,IAAI,MAAM,EAAE,OAAO;gBAAE,OAAO,KAAK,CAAC;YAClC,IAAI,CAAC,IAAI,CAAC,KAAK;gBAAE,OAAO,KAAK,CAAC,CAAC,gBAAgB;YAC/C,IAAI,MAAM,IAAI,CAAC,SAAS,CAAC,MAAM,CAAC;gBAAE,OAAO,IAAI,CAAC;YAC9C,MAAM,IAAI,OAAO,CAAC,CAAC,CAAC,EAAE,EAAE,CAAC,UAAU,CAAC,CAAC,EAAE,uBAAuB,CAAC,CAAC,CAAC;QAClE,CAAC;QACD,mDAAmD;QACnD,IAAI,CAAC,IAAI,EAAE,CAAC;QACZ,OAAO,KAAK,CAAC;IAAA,CACb;IAED,oEAAoE;IACpE,IAAI,GAAS;QACZ,IAAI,IAAI,CAAC,KAAK,IAAI,CAAC,IAAI,CAAC,KAAK,CAAC,MAAM,EAAE,CAAC;YACtC,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,SAAS,CAAC,CAAC;QAC5B,CAAC;QACD,IAAI,CAAC,KAAK,GAAG,SAAS,CAAC;QACvB,IAAI,CAAC,YAAY,GAAG,SAAS,CAAC;IAAA,CAC9B;CACD","sourcesContent":["/**\n * Local executor server lifecycle manager.\n *\n * When `routing.executor.server` is configured AND local-inference routing is\n * active, the harness can spawn a local OpenAI-compatible server (for example\n * `mlx_lm.server`), wait for it to become healthy, and stop it on shutdown.\n *\n * This is best-effort: if the server fails to start or never becomes healthy,\n * `ensureStarted()` resolves to false and the caller degrades to the primary\n * model (routing falls back). It never throws into the agent loop.\n */\n\nimport { type ChildProcess, spawn } from \"node:child_process\";\nimport type { ExecutorServerConfig } from \"./local-inference.js\";\nimport { logLocalInferenceFallback } from \"./metrics.js\";\n\nconst DEFAULT_COMMAND = \"mlx_lm.server\";\nconst DEFAULT_HOST = \"127.0.0.1\";\nconst DEFAULT_PORT = 8080;\nconst DEFAULT_STARTUP_TIMEOUT_MS = 30_000;\nconst HEALTH_POLL_INTERVAL_MS = 500;\n\n/** Derive host/port from a baseUrl like \"http://127.0.0.1:8080/v1\". */\nfunction parseBaseUrl(baseUrl: string | undefined): { host?: string; port?: number } {\n\tif (!baseUrl) return {};\n\ttry {\n\t\tconst u = new URL(baseUrl);\n\t\treturn { host: u.hostname, port: u.port ? Number(u.port) : undefined };\n\t} catch {\n\t\treturn {};\n\t}\n}\n\nexport interface MlxServerOptions {\n\tconfig: ExecutorServerConfig;\n\t/** Executor model id, passed as --model. */\n\tmodelId: string;\n\t/** Executor baseUrl, used to derive host/port and health endpoint. */\n\tbaseUrl?: string;\n}\n\nexport class MlxServerManager {\n\tprivate readonly command: string;\n\tprivate readonly args: string[];\n\tprivate readonly host: string;\n\tprivate readonly port: number;\n\tprivate readonly startupTimeoutMs: number;\n\tprivate readonly modelId: string;\n\tprivate child: ChildProcess | undefined;\n\tprivate startPromise: Promise<boolean> | undefined;\n\n\tconstructor(opts: MlxServerOptions) {\n\t\tconst fromUrl = parseBaseUrl(opts.baseUrl);\n\t\tthis.command = opts.config.command ?? DEFAULT_COMMAND;\n\t\tthis.args = opts.config.args ?? [];\n\t\tthis.host = opts.config.host ?? fromUrl.host ?? DEFAULT_HOST;\n\t\tthis.port = opts.config.port ?? fromUrl.port ?? DEFAULT_PORT;\n\t\tthis.startupTimeoutMs = opts.config.startupTimeoutMs ?? DEFAULT_STARTUP_TIMEOUT_MS;\n\t\tthis.modelId = opts.modelId;\n\t}\n\n\tprivate healthUrl(): string {\n\t\treturn `http://${this.host}:${this.port}/v1/models`;\n\t}\n\n\tprivate async isHealthy(signal?: AbortSignal): Promise<boolean> {\n\t\ttry {\n\t\t\tconst res = await fetch(this.healthUrl(), { method: \"GET\", signal });\n\t\t\treturn res.ok;\n\t\t} catch {\n\t\t\treturn false;\n\t\t}\n\t}\n\n\t/**\n\t * Ensure the server is running and healthy. Idempotent: concurrent callers\n\t * share one start attempt. Returns false on any failure (caller degrades to\n\t * primary). If a server is already healthy (user-started), reuses it without\n\t * spawning.\n\t */\n\tasync ensureStarted(signal?: AbortSignal): Promise<boolean> {\n\t\tif (this.startPromise) return this.startPromise;\n\t\tthis.startPromise = this.start(signal);\n\t\treturn this.startPromise;\n\t}\n\n\tprivate async start(signal?: AbortSignal): Promise<boolean> {\n\t\t// Reuse an already-running server (e.g. user-started) without spawning.\n\t\tif (await this.isHealthy(signal)) return true;\n\n\t\ttry {\n\t\t\tthis.child = spawn(\n\t\t\t\tthis.command,\n\t\t\t\t[...this.args, \"--model\", this.modelId, \"--host\", this.host, \"--port\", String(this.port)],\n\t\t\t\t{\n\t\t\t\t\tstdio: \"ignore\",\n\t\t\t\t\tdetached: false,\n\t\t\t\t},\n\t\t\t);\n\t\t\tthis.child.on(\"error\", (error) => {\n\t\t\t\tlogLocalInferenceFallback(\"primary\", error);\n\t\t\t\tthis.child = undefined;\n\t\t\t});\n\t\t} catch (error) {\n\t\t\tlogLocalInferenceFallback(\"primary\", error);\n\t\t\tthis.child = undefined;\n\t\t\treturn false;\n\t\t}\n\n\t\tconst deadline = Date.now() + this.startupTimeoutMs;\n\t\twhile (Date.now() < deadline) {\n\t\t\tif (signal?.aborted) return false;\n\t\t\tif (!this.child) return false; // spawn errored\n\t\t\tif (await this.isHealthy(signal)) return true;\n\t\t\tawait new Promise((r) => setTimeout(r, HEALTH_POLL_INTERVAL_MS));\n\t\t}\n\t\t// Timed out: stop whatever we spawned and degrade.\n\t\tthis.stop();\n\t\treturn false;\n\t}\n\n\t/** Stop the spawned server (no-op if we reused an external one). */\n\tstop(): void {\n\t\tif (this.child && !this.child.killed) {\n\t\t\tthis.child.kill(\"SIGTERM\");\n\t\t}\n\t\tthis.child = undefined;\n\t\tthis.startPromise = undefined;\n\t}\n}\n"]}
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Validated per-tool extractive compression prompts.
|
|
3
|
+
*
|
|
4
|
+
* These prompts were validated against Qwen3-4B (see
|
|
5
|
+
* docs/local-executor-routing.md). Only `bash` output compresses safely: its
|
|
6
|
+
* verbose output is mostly low-value noise (progress/passing lines) around a
|
|
7
|
+
* few load-bearing facts (errors, counts, exit codes). `read` was removed after
|
|
8
|
+
* measurement showed ~0% reduction on real source code (every line is a
|
|
9
|
+
* keep-line). Fact-list outputs (grep/find/ls) are intentionally excluded:
|
|
10
|
+
* every line is a distinct fact, so compression drops matches.
|
|
11
|
+
*
|
|
12
|
+
* The prompt is extractive (keep identifiers, drop only redundant filler),
|
|
13
|
+
* not abstractive (rewrite in prose), which is what made retention reliable.
|
|
14
|
+
*/
|
|
15
|
+
export declare const TOOL_RESULT_SYSTEM_PROMPT: string;
|
|
16
|
+
/** Get the extractive compression prompt for a tool, or undefined if not compressible. */
|
|
17
|
+
export declare function getToolResultPrompt(toolName: string): string | undefined;
|
|
18
|
+
/** Build the full prompt text for compressing a tool result. */
|
|
19
|
+
export declare function buildToolResultPrompt(toolName: string, output: string): string | undefined;
|
|
20
|
+
/**
|
|
21
|
+
* Remove reasoning-model `<think>...</think>` blocks (including empty ones that
|
|
22
|
+
* Qwen3 emits even under `/no_think`) and trim the result. Reasoning models
|
|
23
|
+
* leave these tags in the text stream; they must not leak into context.
|
|
24
|
+
*/
|
|
25
|
+
export declare function stripThinkTags(text: string): string;
|
|
26
|
+
//# sourceMappingURL=tool-result-prompts.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"tool-result-prompts.d.ts","sourceRoot":"","sources":["../../../src/core/routing/tool-result-prompts.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;GAaG;AAEH,eAAO,MAAM,yBAAyB,QAEkD,CAAC;AAWzF,0FAA0F;AAC1F,wBAAgB,mBAAmB,CAAC,QAAQ,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAExE;AAED,gEAAgE;AAChE,wBAAgB,qBAAqB,CAAC,QAAQ,EAAE,MAAM,EAAE,MAAM,EAAE,MAAM,GAAG,MAAM,GAAG,SAAS,CAI1F;AAED;;;;GAIG;AACH,wBAAgB,cAAc,CAAC,IAAI,EAAE,MAAM,GAAG,MAAM,CAKnD","sourcesContent":["/**\n * Validated per-tool extractive compression prompts.\n *\n * These prompts were validated against Qwen3-4B (see\n * docs/local-executor-routing.md). Only `bash` output compresses safely: its\n * verbose output is mostly low-value noise (progress/passing lines) around a\n * few load-bearing facts (errors, counts, exit codes). `read` was removed after\n * measurement showed ~0% reduction on real source code (every line is a\n * keep-line). Fact-list outputs (grep/find/ls) are intentionally excluded:\n * every line is a distinct fact, so compression drops matches.\n *\n * The prompt is extractive (keep identifiers, drop only redundant filler),\n * not abstractive (rewrite in prose), which is what made retention reliable.\n */\n\nexport const TOOL_RESULT_SYSTEM_PROMPT =\n\t\"You compress tool output for another AI to consume. Be extractive: keep exact identifiers, \" +\n\t\"never paraphrase facts, and never invent anything. Output only the compressed result.\";\n\nconst BASH_PROMPT =\n\t\"Compress this command output. Keep ONLY: the command, every error/warning with its file:line:col \" +\n\t\"and code, and any final counts/timings/exit code. Drop progress bars, info lines, and passing/OK \" +\n\t\"lines. Keep all numbers and paths exactly. No prose.\";\n\nconst TOOL_RESULT_PROMPTS: Record<string, string> = {\n\tbash: BASH_PROMPT,\n};\n\n/** Get the extractive compression prompt for a tool, or undefined if not compressible. */\nexport function getToolResultPrompt(toolName: string): string | undefined {\n\treturn TOOL_RESULT_PROMPTS[toolName];\n}\n\n/** Build the full prompt text for compressing a tool result. */\nexport function buildToolResultPrompt(toolName: string, output: string): string | undefined {\n\tconst instruction = getToolResultPrompt(toolName);\n\tif (!instruction) return undefined;\n\treturn `${instruction}\\n\\n${output}`;\n}\n\n/**\n * Remove reasoning-model `<think>...</think>` blocks (including empty ones that\n * Qwen3 emits even under `/no_think`) and trim the result. Reasoning models\n * leave these tags in the text stream; they must not leak into context.\n */\nexport function stripThinkTags(text: string): string {\n\treturn text\n\t\t.replace(/<think>[\\s\\S]*?<\\/think>/g, \"\")\n\t\t.replace(/<\\/?think>/g, \"\")\n\t\t.trim();\n}\n"]}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Validated per-tool extractive compression prompts.
|
|
3
|
+
*
|
|
4
|
+
* These prompts were validated against Qwen3-4B (see
|
|
5
|
+
* docs/local-executor-routing.md). Only `bash` output compresses safely: its
|
|
6
|
+
* verbose output is mostly low-value noise (progress/passing lines) around a
|
|
7
|
+
* few load-bearing facts (errors, counts, exit codes). `read` was removed after
|
|
8
|
+
* measurement showed ~0% reduction on real source code (every line is a
|
|
9
|
+
* keep-line). Fact-list outputs (grep/find/ls) are intentionally excluded:
|
|
10
|
+
* every line is a distinct fact, so compression drops matches.
|
|
11
|
+
*
|
|
12
|
+
* The prompt is extractive (keep identifiers, drop only redundant filler),
|
|
13
|
+
* not abstractive (rewrite in prose), which is what made retention reliable.
|
|
14
|
+
*/
|
|
15
|
+
export const TOOL_RESULT_SYSTEM_PROMPT = "You compress tool output for another AI to consume. Be extractive: keep exact identifiers, " +
|
|
16
|
+
"never paraphrase facts, and never invent anything. Output only the compressed result.";
|
|
17
|
+
const BASH_PROMPT = "Compress this command output. Keep ONLY: the command, every error/warning with its file:line:col " +
|
|
18
|
+
"and code, and any final counts/timings/exit code. Drop progress bars, info lines, and passing/OK " +
|
|
19
|
+
"lines. Keep all numbers and paths exactly. No prose.";
|
|
20
|
+
const TOOL_RESULT_PROMPTS = {
|
|
21
|
+
bash: BASH_PROMPT,
|
|
22
|
+
};
|
|
23
|
+
/** Get the extractive compression prompt for a tool, or undefined if not compressible. */
|
|
24
|
+
export function getToolResultPrompt(toolName) {
|
|
25
|
+
return TOOL_RESULT_PROMPTS[toolName];
|
|
26
|
+
}
|
|
27
|
+
/** Build the full prompt text for compressing a tool result. */
|
|
28
|
+
export function buildToolResultPrompt(toolName, output) {
|
|
29
|
+
const instruction = getToolResultPrompt(toolName);
|
|
30
|
+
if (!instruction)
|
|
31
|
+
return undefined;
|
|
32
|
+
return `${instruction}\n\n${output}`;
|
|
33
|
+
}
|
|
34
|
+
/**
|
|
35
|
+
* Remove reasoning-model `<think>...</think>` blocks (including empty ones that
|
|
36
|
+
* Qwen3 emits even under `/no_think`) and trim the result. Reasoning models
|
|
37
|
+
* leave these tags in the text stream; they must not leak into context.
|
|
38
|
+
*/
|
|
39
|
+
export function stripThinkTags(text) {
|
|
40
|
+
return text
|
|
41
|
+
.replace(/<think>[\s\S]*?<\/think>/g, "")
|
|
42
|
+
.replace(/<\/?think>/g, "")
|
|
43
|
+
.trim();
|
|
44
|
+
}
|
|
45
|
+
//# sourceMappingURL=tool-result-prompts.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"tool-result-prompts.js","sourceRoot":"","sources":["../../../src/core/routing/tool-result-prompts.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;GAaG;AAEH,MAAM,CAAC,MAAM,yBAAyB,GACrC,6FAA6F;IAC7F,uFAAuF,CAAC;AAEzF,MAAM,WAAW,GAChB,mGAAmG;IACnG,mGAAmG;IACnG,sDAAsD,CAAC;AAExD,MAAM,mBAAmB,GAA2B;IACnD,IAAI,EAAE,WAAW;CACjB,CAAC;AAEF,0FAA0F;AAC1F,MAAM,UAAU,mBAAmB,CAAC,QAAgB,EAAsB;IACzE,OAAO,mBAAmB,CAAC,QAAQ,CAAC,CAAC;AAAA,CACrC;AAED,gEAAgE;AAChE,MAAM,UAAU,qBAAqB,CAAC,QAAgB,EAAE,MAAc,EAAsB;IAC3F,MAAM,WAAW,GAAG,mBAAmB,CAAC,QAAQ,CAAC,CAAC;IAClD,IAAI,CAAC,WAAW;QAAE,OAAO,SAAS,CAAC;IACnC,OAAO,GAAG,WAAW,OAAO,MAAM,EAAE,CAAC;AAAA,CACrC;AAED;;;;GAIG;AACH,MAAM,UAAU,cAAc,CAAC,IAAY,EAAU;IACpD,OAAO,IAAI;SACT,OAAO,CAAC,2BAA2B,EAAE,EAAE,CAAC;SACxC,OAAO,CAAC,aAAa,EAAE,EAAE,CAAC;SAC1B,IAAI,EAAE,CAAC;AAAA,CACT","sourcesContent":["/**\n * Validated per-tool extractive compression prompts.\n *\n * These prompts were validated against Qwen3-4B (see\n * docs/local-executor-routing.md). Only `bash` output compresses safely: its\n * verbose output is mostly low-value noise (progress/passing lines) around a\n * few load-bearing facts (errors, counts, exit codes). `read` was removed after\n * measurement showed ~0% reduction on real source code (every line is a\n * keep-line). Fact-list outputs (grep/find/ls) are intentionally excluded:\n * every line is a distinct fact, so compression drops matches.\n *\n * The prompt is extractive (keep identifiers, drop only redundant filler),\n * not abstractive (rewrite in prose), which is what made retention reliable.\n */\n\nexport const TOOL_RESULT_SYSTEM_PROMPT =\n\t\"You compress tool output for another AI to consume. Be extractive: keep exact identifiers, \" +\n\t\"never paraphrase facts, and never invent anything. Output only the compressed result.\";\n\nconst BASH_PROMPT =\n\t\"Compress this command output. Keep ONLY: the command, every error/warning with its file:line:col \" +\n\t\"and code, and any final counts/timings/exit code. Drop progress bars, info lines, and passing/OK \" +\n\t\"lines. Keep all numbers and paths exactly. No prose.\";\n\nconst TOOL_RESULT_PROMPTS: Record<string, string> = {\n\tbash: BASH_PROMPT,\n};\n\n/** Get the extractive compression prompt for a tool, or undefined if not compressible. */\nexport function getToolResultPrompt(toolName: string): string | undefined {\n\treturn TOOL_RESULT_PROMPTS[toolName];\n}\n\n/** Build the full prompt text for compressing a tool result. */\nexport function buildToolResultPrompt(toolName: string, output: string): string | undefined {\n\tconst instruction = getToolResultPrompt(toolName);\n\tif (!instruction) return undefined;\n\treturn `${instruction}\\n\\n${output}`;\n}\n\n/**\n * Remove reasoning-model `<think>...</think>` blocks (including empty ones that\n * Qwen3 emits even under `/no_think`) and trim the result. Reasoning models\n * leave these tags in the text stream; they must not leak into context.\n */\nexport function stripThinkTags(text: string): string {\n\treturn text\n\t\t.replace(/<think>[\\s\\S]*?<\\/think>/g, \"\")\n\t\t.replace(/<\\/?think>/g, \"\")\n\t\t.trim();\n}\n"]}
|
package/dist/core/sdk.d.ts
CHANGED
|
@@ -57,6 +57,12 @@ export interface CreateAgentSessionOptions {
|
|
|
57
57
|
settingsManager?: SettingsManager;
|
|
58
58
|
/** Session start event metadata for extension runtime startup. */
|
|
59
59
|
sessionStartEvent?: SessionStartEvent;
|
|
60
|
+
/**
|
|
61
|
+
* Master gate for local-inference routing (compaction + tool-result
|
|
62
|
+
* compression via a local executor model). Off by default. Routing also
|
|
63
|
+
* requires a `routing` block in models.json. See docs/local-executor-routing.md.
|
|
64
|
+
*/
|
|
65
|
+
enableLocalInference?: boolean;
|
|
60
66
|
}
|
|
61
67
|
/** Result from createAgentSession */
|
|
62
68
|
export interface CreateAgentSessionResult {
|
package/dist/core/sdk.d.ts.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"sdk.d.ts","sourceRoot":"","sources":["../../src/core/sdk.ts"],"names":[],"mappings":"AACA,OAAO,EAA4B,KAAK,aAAa,EAAE,MAAM,iCAAiC,CAAC;AAC/F,OAAO,EAAoC,KAAK,KAAK,EAAgB,MAAM,yBAAyB,CAAC;AAErG,OAAO,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAElD,OAAO,EAAE,WAAW,EAAE,MAAM,mBAAmB,CAAC;AAEhD,OAAO,KAAK,EAAmB,oBAAoB,EAAE,iBAAiB,EAAE,cAAc,EAAE,MAAM,uBAAuB,CAAC;AAEtH,OAAO,EAAE,aAAa,EAAE,MAAM,qBAAqB,CAAC;AAEpD,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,sBAAsB,CAAC;AAE3D,OAAO,EAAwB,cAAc,EAAE,MAAM,sBAAsB,CAAC;AAC5E,OAAO,EAAE,eAAe,EAAE,MAAM,uBAAuB,CAAC;AAIxD,OAAO,EACN,cAAc,EACd,iBAAiB,EACjB,cAAc,EACd,cAAc,EACd,cAAc,EACd,YAAY,EACZ,mBAAmB,EACnB,cAAc,EACd,eAAe,EAEf,qBAAqB,EACrB,MAAM,kBAAkB,CAAC;AAE1B,MAAM,WAAW,yBAAyB;IACzC,4EAA4E;IAC5E,GAAG,CAAC,EAAE,MAAM,CAAC;IACb,yDAAyD;IACzD,QAAQ,CAAC,EAAE,MAAM,CAAC;IAElB,oFAAoF;IACpF,WAAW,CAAC,EAAE,WAAW,CAAC;IAC1B,uFAAuF;IACvF,aAAa,CAAC,EAAE,aAAa,CAAC;IAE9B,iEAAiE;IACjE,KAAK,CAAC,EAAE,KAAK,CAAC,GAAG,CAAC,CAAC;IACnB,4FAA4F;IAC5F,aAAa,CAAC,EAAE,aAAa,CAAC;IAC9B,gEAAgE;IAChE,YAAY,CAAC,EAAE,KAAK,CAAC;QAAE,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,CAAC;QAAC,aAAa,CAAC,EAAE,aAAa,CAAA;KAAE,CAAC,CAAC;IAE3E;;;;;;OAMG;IACH,OAAO,CAAC,EAAE,KAAK,GAAG,SAAS,CAAC;IAC5B;;;;;;OAMG;IACH,KAAK,CAAC,EAAE,MAAM,EAAE,CAAC;IACjB;;;OAGG;IACH,eAAe,CAAC,EAAE,MAAM,EAAE,CAAC;IAC3B,gEAAgE;IAChE,WAAW,CAAC,EAAE,cAAc,EAAE,CAAC;IAE/B,oEAAoE;IACpE,cAAc,CAAC,EAAE,cAAc,CAAC;IAEhC,2DAA2D;IAC3D,cAAc,CAAC,EAAE,cAAc,CAAC;IAEhC,uEAAuE;IACvE,eAAe,CAAC,EAAE,eAAe,CAAC;IAClC,kEAAkE;IAClE,iBAAiB,CAAC,EAAE,iBAAiB,CAAC;CACtC;AAED,qCAAqC;AACrC,MAAM,WAAW,wBAAwB;IACxC,0BAA0B;IAC1B,OAAO,EAAE,YAAY,CAAC;IACtB,mEAAmE;IACnE,gBAAgB,EAAE,oBAAoB,CAAC;IACvC,wEAAwE;IACxE,oBAAoB,CAAC,EAAE,MAAM,CAAC;CAC9B;AAID,YAAY,EAAE,eAAe,EAAE,WAAW,EAAE,MAAM,wBAAwB,CAAC;AAC3E,OAAO,EAAE,aAAa,EAAE,qBAAqB,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AAC9F,cAAc,4BAA4B,CAAC;AAC3C,YAAY,EACX,YAAY,EACZ,uBAAuB,EACvB,gBAAgB,EAChB,gBAAgB,EAChB,gBAAgB,EAChB,kBAAkB,EAClB,cAAc,GACd,MAAM,uBAAuB,CAAC;AAC/B,YAAY,EAAE,cAAc,EAAE,MAAM,uBAAuB,CAAC;AAC5D,YAAY,EAAE,KAAK,EAAE,MAAM,aAAa,CAAC;AACzC,YAAY,EAAE,IAAI,EAAE,MAAM,kBAAkB,CAAC;AAE7C,OAAO,EACN,qBAAqB,EAErB,iBAAiB,EACjB,mBAAmB,EACnB,cAAc,EACd,cAAc,EACd,cAAc,EACd,eAAe,EACf,cAAc,EACd,cAAc,EACd,YAAY,GACZ,CAAC;AAsCF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkCG;AACH,wBAAsB,kBAAkB,CAAC,OAAO,GAAE,yBAA8B,GAAG,OAAO,CAAC,wBAAwB,CAAC,CAqOnH","sourcesContent":["import { join } from \"node:path\";\nimport { Agent, type AgentMessage, type ThinkingLevel } from \"@kolisachint/hoocode-agent-core\";\nimport { clampThinkingLevel, type Message, type Model, streamSimple } from \"@kolisachint/hoocode-ai\";\nimport { getAgentDir } from \"../config.js\";\nimport { AgentSession } from \"./agent-session.js\";\nimport { formatNoModelsAvailableMessage } from \"./auth-guidance.js\";\nimport { AuthStorage } from \"./auth-storage.js\";\nimport { DEFAULT_THINKING_LEVEL } from \"./defaults.js\";\nimport type { ExtensionRunner, LoadExtensionsResult, SessionStartEvent, ToolDefinition } from \"./extensions/index.js\";\nimport { convertToLlm, createBackgroundPlaceholderText, createBackgroundTaskMessage } from \"./messages.js\";\nimport { ModelRegistry } from \"./model-registry.js\";\nimport { findInitialModel } from \"./model-resolver.js\";\nimport type { ResourceLoader } from \"./resource-loader.js\";\nimport { DefaultResourceLoader } from \"./resource-loader.js\";\nimport { getDefaultSessionDir, SessionManager } from \"./session-manager.js\";\nimport { SettingsManager } from \"./settings-manager.js\";\nimport { peekSubagentPool } from \"./subagent-pool-instance.js\";\nimport { isInstallTelemetryEnabled } from \"./telemetry.js\";\nimport { time } from \"./timings.js\";\nimport {\n\tcreateBashTool,\n\tcreateCodingTools,\n\tcreateEditTool,\n\tcreateFindTool,\n\tcreateGrepTool,\n\tcreateLsTool,\n\tcreateReadOnlyTools,\n\tcreateReadTool,\n\tcreateWriteTool,\n\ttype ToolName,\n\twithFileMutationQueue,\n} from \"./tools/index.js\";\n\nexport interface CreateAgentSessionOptions {\n\t/** Working directory for project-local discovery. Default: process.cwd() */\n\tcwd?: string;\n\t/** Global config directory. Default: ~/.hoocode/agent */\n\tagentDir?: string;\n\n\t/** Auth storage for credentials. Default: AuthStorage.create(agentDir/auth.json) */\n\tauthStorage?: AuthStorage;\n\t/** Model registry. Default: ModelRegistry.create(authStorage, agentDir/models.json) */\n\tmodelRegistry?: ModelRegistry;\n\n\t/** Model to use. Default: from settings, else first available */\n\tmodel?: Model<any>;\n\t/** Thinking level. Default: from settings, else 'medium' (clamped to model capabilities) */\n\tthinkingLevel?: ThinkingLevel;\n\t/** Models available for cycling (Ctrl+P in interactive mode) */\n\tscopedModels?: Array<{ model: Model<any>; thinkingLevel?: ThinkingLevel }>;\n\n\t/**\n\t * Optional default tool suppression mode when no explicit allowlist is provided.\n\t *\n\t * - \"all\": start with no tools enabled\n\t * - \"builtin\": disable the default built-in tools (read, bash, edit, write)\n\t * but keep extension/custom tools enabled\n\t */\n\tnoTools?: \"all\" | \"builtin\";\n\t/**\n\t * Optional allowlist of tool names.\n\t *\n\t * When omitted, hoocode enables the default built-in tools (read, bash, edit, write)\n\t * and leaves extension/custom tools enabled unless `noTools` changes that default.\n\t * When provided, only the listed tool names are enabled.\n\t */\n\ttools?: string[];\n\t/**\n\t * Optional denylist of tool names, subtracted from whatever set is otherwise\n\t * enabled (allowlist or default). Applied to built-in, extension, and custom tools.\n\t */\n\tdisallowedTools?: string[];\n\t/** Custom tools to register (in addition to built-in tools). */\n\tcustomTools?: ToolDefinition[];\n\n\t/** Resource loader. When omitted, DefaultResourceLoader is used. */\n\tresourceLoader?: ResourceLoader;\n\n\t/** Session manager. Default: SessionManager.create(cwd) */\n\tsessionManager?: SessionManager;\n\n\t/** Settings manager. Default: SettingsManager.create(cwd, agentDir) */\n\tsettingsManager?: SettingsManager;\n\t/** Session start event metadata for extension runtime startup. */\n\tsessionStartEvent?: SessionStartEvent;\n}\n\n/** Result from createAgentSession */\nexport interface CreateAgentSessionResult {\n\t/** The created session */\n\tsession: AgentSession;\n\t/** Extensions result (for UI context setup in interactive mode) */\n\textensionsResult: LoadExtensionsResult;\n\t/** Warning if session was restored with a different model than saved */\n\tmodelFallbackMessage?: string;\n}\n\n// Re-exports\n\nexport type { AgentDefinition, AgentSource } from \"./agent-frontmatter.js\";\nexport { AgentRegistry, formatAgentsForPrompt, loadAgentRegistry } from \"./agent-registry.js\";\nexport * from \"./agent-session-runtime.js\";\nexport type {\n\tExtensionAPI,\n\tExtensionCommandContext,\n\tExtensionContext,\n\tExtensionFactory,\n\tSlashCommandInfo,\n\tSlashCommandSource,\n\tToolDefinition,\n} from \"./extensions/index.js\";\nexport type { PromptTemplate } from \"./prompt-templates.js\";\nexport type { Skill } from \"./skills.js\";\nexport type { Tool } from \"./tools/index.js\";\n\nexport {\n\twithFileMutationQueue,\n\t// Tool factories (for custom cwd)\n\tcreateCodingTools,\n\tcreateReadOnlyTools,\n\tcreateReadTool,\n\tcreateBashTool,\n\tcreateEditTool,\n\tcreateWriteTool,\n\tcreateGrepTool,\n\tcreateFindTool,\n\tcreateLsTool,\n};\n\n// Helper Functions\n\nfunction getDefaultAgentDir(): string {\n\treturn getAgentDir();\n}\n\nfunction getAttributionHeaders(\n\tmodel: Model<any>,\n\tsettingsManager: SettingsManager,\n): Record<string, string> | undefined {\n\tif (!isInstallTelemetryEnabled(settingsManager)) {\n\t\treturn undefined;\n\t}\n\n\tif (model.provider === \"openrouter\" || model.baseUrl.includes(\"openrouter.ai\")) {\n\t\treturn {\n\t\t\t\"HTTP-Referer\": \"https://github.com/kolisachint/hoocode\",\n\t\t\t\"X-OpenRouter-Title\": \"hoocode\",\n\t\t\t\"X-OpenRouter-Categories\": \"cli-agent\",\n\t\t};\n\t}\n\n\tif (\n\t\tmodel.provider === \"cloudflare-workers-ai\" ||\n\t\tmodel.provider === \"cloudflare-ai-gateway\" ||\n\t\tmodel.baseUrl.includes(\"api.cloudflare.com\") ||\n\t\tmodel.baseUrl.includes(\"gateway.ai.cloudflare.com\")\n\t) {\n\t\treturn {\n\t\t\t\"User-Agent\": \"hoocode\",\n\t\t};\n\t}\n\n\treturn undefined;\n}\n\n/**\n * Create an AgentSession with the specified options.\n *\n * @example\n * ```typescript\n * // Minimal - uses defaults\n * const { session } = await createAgentSession();\n *\n * // With explicit model\n * import { getModel } from '@kolisachint/hoocode-ai';\n * const { session } = await createAgentSession({\n * model: getModel('anthropic', 'claude-opus-4-5'),\n * thinkingLevel: 'high',\n * });\n *\n * // Continue previous session\n * const { session, modelFallbackMessage } = await createAgentSession({\n * continueSession: true,\n * });\n *\n * // Full control\n * const loader = new DefaultResourceLoader({\n * cwd: process.cwd(),\n * agentDir: getAgentDir(),\n * settingsManager: SettingsManager.create(),\n * });\n * await loader.reload();\n * const { session } = await createAgentSession({\n * model: myModel,\n * tools: [readTool, bashTool],\n * resourceLoader: loader,\n * sessionManager: SessionManager.inMemory(),\n * });\n * ```\n */\nexport async function createAgentSession(options: CreateAgentSessionOptions = {}): Promise<CreateAgentSessionResult> {\n\tconst cwd = options.cwd ?? options.sessionManager?.getCwd() ?? process.cwd();\n\tconst agentDir = options.agentDir ?? getDefaultAgentDir();\n\tlet resourceLoader = options.resourceLoader;\n\n\t// Use provided or create AuthStorage and ModelRegistry\n\tconst authPath = options.agentDir ? join(agentDir, \"auth.json\") : undefined;\n\tconst modelsPath = options.agentDir ? join(agentDir, \"models.json\") : undefined;\n\tconst authStorage = options.authStorage ?? AuthStorage.create(authPath);\n\tconst modelRegistry = options.modelRegistry ?? ModelRegistry.create(authStorage, modelsPath);\n\n\tconst settingsManager = options.settingsManager ?? SettingsManager.create(cwd, agentDir);\n\tconst sessionManager = options.sessionManager ?? SessionManager.create(cwd, getDefaultSessionDir(cwd, agentDir));\n\n\tif (!resourceLoader) {\n\t\tresourceLoader = new DefaultResourceLoader({ cwd, agentDir, settingsManager });\n\t\tawait resourceLoader.reload();\n\t\ttime(\"resourceLoader.reload\");\n\t}\n\n\t// Check if session has existing data to restore\n\tconst existingSession = sessionManager.buildSessionContext();\n\tconst hasExistingSession = existingSession.messages.length > 0;\n\tconst hasThinkingEntry = sessionManager.getBranch().some((entry) => entry.type === \"thinking_level_change\");\n\n\tlet model = options.model;\n\tlet modelFallbackMessage: string | undefined;\n\n\t// If session has data, try to restore model from it\n\tif (!model && hasExistingSession && existingSession.model) {\n\t\tconst restoredModel = modelRegistry.find(existingSession.model.provider, existingSession.model.modelId);\n\t\tif (restoredModel && modelRegistry.hasConfiguredAuth(restoredModel)) {\n\t\t\tmodel = restoredModel;\n\t\t}\n\t\tif (!model) {\n\t\t\tmodelFallbackMessage = `Could not restore model ${existingSession.model.provider}/${existingSession.model.modelId}`;\n\t\t}\n\t}\n\n\t// If still no model, use findInitialModel (checks settings default, then provider defaults)\n\tif (!model) {\n\t\tconst result = await findInitialModel({\n\t\t\tscopedModels: [],\n\t\t\tisContinuing: hasExistingSession,\n\t\t\tdefaultProvider: settingsManager.getDefaultProvider(),\n\t\t\tdefaultModelId: settingsManager.getDefaultModel(),\n\t\t\tdefaultThinkingLevel: settingsManager.getDefaultThinkingLevel(),\n\t\t\tmodelRegistry,\n\t\t});\n\t\tmodel = result.model;\n\t\tif (!model) {\n\t\t\tmodelFallbackMessage = formatNoModelsAvailableMessage();\n\t\t} else if (modelFallbackMessage) {\n\t\t\tmodelFallbackMessage += `. Using ${model.provider}/${model.id}`;\n\t\t}\n\t}\n\n\tlet thinkingLevel = options.thinkingLevel;\n\n\t// If session has data, restore thinking level from it\n\tif (thinkingLevel === undefined && hasExistingSession) {\n\t\tthinkingLevel = hasThinkingEntry\n\t\t\t? (existingSession.thinkingLevel as ThinkingLevel)\n\t\t\t: (settingsManager.getDefaultThinkingLevel() ?? DEFAULT_THINKING_LEVEL);\n\t}\n\n\t// Fall back to settings default\n\tif (thinkingLevel === undefined) {\n\t\tthinkingLevel = settingsManager.getDefaultThinkingLevel() ?? DEFAULT_THINKING_LEVEL;\n\t}\n\n\t// Clamp to model capabilities\n\tif (!model) {\n\t\tthinkingLevel = \"off\";\n\t} else {\n\t\tthinkingLevel = clampThinkingLevel(model, thinkingLevel) as ThinkingLevel;\n\t}\n\n\tconst defaultActiveToolNames: ToolName[] = [\"read\", \"bash\", \"edit\", \"write\", \"grep\", \"find\", \"ls\"];\n\tconst allowedToolNames = options.tools ?? (options.noTools === \"all\" ? [] : undefined);\n\tconst initialActiveToolNames: string[] = options.tools\n\t\t? [...options.tools]\n\t\t: options.noTools\n\t\t\t? []\n\t\t\t: defaultActiveToolNames;\n\n\tlet agent: Agent;\n\n\t// Create convertToLlm wrapper that filters images if blockImages is enabled (defense-in-depth)\n\tconst convertToLlmWithBlockImages = (messages: AgentMessage[]): Message[] => {\n\t\tconst converted = convertToLlm(messages);\n\t\t// Check setting dynamically so mid-session changes take effect\n\t\tif (!settingsManager.getBlockImages()) {\n\t\t\treturn converted;\n\t\t}\n\t\t// Filter out ImageContent from all messages, replacing with text placeholder\n\t\treturn converted.map((msg) => {\n\t\t\tif (msg.role === \"user\" || msg.role === \"toolResult\") {\n\t\t\t\tconst content = msg.content;\n\t\t\t\tif (Array.isArray(content)) {\n\t\t\t\t\tconst hasImages = content.some((c) => c.type === \"image\");\n\t\t\t\t\tif (hasImages) {\n\t\t\t\t\t\tconst filteredContent = content\n\t\t\t\t\t\t\t.map((c) =>\n\t\t\t\t\t\t\t\tc.type === \"image\" ? { type: \"text\" as const, text: \"Image reading is disabled.\" } : c,\n\t\t\t\t\t\t\t)\n\t\t\t\t\t\t\t.filter(\n\t\t\t\t\t\t\t\t(c, i, arr) =>\n\t\t\t\t\t\t\t\t\t// Dedupe consecutive \"Image reading is disabled.\" texts\n\t\t\t\t\t\t\t\t\t!(\n\t\t\t\t\t\t\t\t\t\tc.type === \"text\" &&\n\t\t\t\t\t\t\t\t\t\tc.text === \"Image reading is disabled.\" &&\n\t\t\t\t\t\t\t\t\t\ti > 0 &&\n\t\t\t\t\t\t\t\t\t\tarr[i - 1].type === \"text\" &&\n\t\t\t\t\t\t\t\t\t\t(arr[i - 1] as { type: \"text\"; text: string }).text === \"Image reading is disabled.\"\n\t\t\t\t\t\t\t\t\t),\n\t\t\t\t\t\t\t);\n\t\t\t\t\t\treturn { ...msg, content: filteredContent };\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}\n\t\t\treturn msg;\n\t\t});\n\t};\n\n\tconst extensionRunnerRef: { current?: ExtensionRunner } = {};\n\n\tagent = new Agent({\n\t\tinitialState: {\n\t\t\tsystemPrompt: \"\",\n\t\t\tmodel,\n\t\t\tthinkingLevel,\n\t\t\ttools: [],\n\t\t},\n\t\tconvertToLlm: convertToLlmWithBlockImages,\n\t\tcreateBackgroundResultMessage: createBackgroundTaskMessage,\n\t\tcreateBackgroundPlaceholder: (toolCall) => createBackgroundPlaceholderText(toolCall),\n\t\t// Report in-process background tool load (e.g. background MCP tools) to the\n\t\t// subagent lifeguard so it widens its heartbeat/timeout tolerance for\n\t\t// concurrently-monitored subagents. Peek (don't create) the pool: background\n\t\t// tools can run before any subagent is ever dispatched.\n\t\tonBackgroundTaskCountChange: (count) => peekSubagentPool()?.setExternalLoad(count),\n\t\tstreamFn: async (model, context, options) => {\n\t\t\tconst auth = await modelRegistry.getApiKeyAndHeaders(model);\n\t\t\tif (!auth.ok) {\n\t\t\t\tthrow new Error(auth.error);\n\t\t\t}\n\t\t\tconst providerRetrySettings = settingsManager.getProviderRetrySettings();\n\t\t\tconst attributionHeaders = getAttributionHeaders(model, settingsManager);\n\t\t\treturn streamSimple(model, context, {\n\t\t\t\t...options,\n\t\t\t\tapiKey: auth.apiKey,\n\t\t\t\ttimeoutMs: options?.timeoutMs ?? providerRetrySettings.timeoutMs,\n\t\t\t\tmaxRetries: options?.maxRetries ?? providerRetrySettings.maxRetries,\n\t\t\t\tmaxRetryDelayMs: options?.maxRetryDelayMs ?? providerRetrySettings.maxRetryDelayMs,\n\t\t\t\theaders:\n\t\t\t\t\tattributionHeaders || auth.headers || options?.headers\n\t\t\t\t\t\t? { ...attributionHeaders, ...auth.headers, ...options?.headers }\n\t\t\t\t\t\t: undefined,\n\t\t\t});\n\t\t},\n\t\tonPayload: async (payload, _model) => {\n\t\t\tconst runner = extensionRunnerRef.current;\n\t\t\tif (!runner?.hasHandlers(\"before_provider_request\")) {\n\t\t\t\treturn payload;\n\t\t\t}\n\t\t\treturn runner.emitBeforeProviderRequest(payload);\n\t\t},\n\t\tonResponse: async (response, _model) => {\n\t\t\tconst runner = extensionRunnerRef.current;\n\t\t\tif (!runner?.hasHandlers(\"after_provider_response\")) {\n\t\t\t\treturn;\n\t\t\t}\n\t\t\tawait runner.emit({\n\t\t\t\ttype: \"after_provider_response\",\n\t\t\t\tstatus: response.status,\n\t\t\t\theaders: response.headers,\n\t\t\t});\n\t\t},\n\t\tsessionId: sessionManager.getSessionId(),\n\t\ttransformContext: async (messages) => {\n\t\t\tconst runner = extensionRunnerRef.current;\n\t\t\tif (!runner) return messages;\n\t\t\treturn runner.emitContext(messages);\n\t\t},\n\t\tsteeringMode: settingsManager.getSteeringMode(),\n\t\tfollowUpMode: settingsManager.getFollowUpMode(),\n\t\ttransport: settingsManager.getTransport(),\n\t\tthinkingBudgets: settingsManager.getThinkingBudgets(),\n\t\tthinkingDisplay: settingsManager.getThinkingDisplay(),\n\t\tmaxRetryDelayMs: settingsManager.getProviderRetrySettings().maxRetryDelayMs,\n\t});\n\n\t// Restore messages if session has existing data\n\tif (hasExistingSession) {\n\t\tagent.state.messages = existingSession.messages;\n\t\tif (!hasThinkingEntry) {\n\t\t\tsessionManager.appendThinkingLevelChange(thinkingLevel);\n\t\t}\n\t} else {\n\t\t// Save initial model and thinking level for new sessions so they can be restored on resume\n\t\tif (model) {\n\t\t\tsessionManager.appendModelChange(model.provider, model.id);\n\t\t}\n\t\tsessionManager.appendThinkingLevelChange(thinkingLevel);\n\t}\n\n\tconst session = new AgentSession({\n\t\tagent,\n\t\tsessionManager,\n\t\tsettingsManager,\n\t\tcwd,\n\t\tscopedModels: options.scopedModels,\n\t\tresourceLoader,\n\t\tcustomTools: options.customTools,\n\t\tmodelRegistry,\n\t\tinitialActiveToolNames,\n\t\tallowedToolNames,\n\t\tdisallowedToolNames: options.disallowedTools,\n\t\textensionRunnerRef,\n\t\tsessionStartEvent: options.sessionStartEvent,\n\t});\n\tconst extensionsResult = resourceLoader.getExtensions();\n\n\treturn {\n\t\tsession,\n\t\textensionsResult,\n\t\tmodelFallbackMessage,\n\t};\n}\n"]}
|
|
1
|
+
{"version":3,"file":"sdk.d.ts","sourceRoot":"","sources":["../../src/core/sdk.ts"],"names":[],"mappings":"AACA,OAAO,EAA4B,KAAK,aAAa,EAAE,MAAM,iCAAiC,CAAC;AAC/F,OAAO,EAAoC,KAAK,KAAK,EAAgB,MAAM,yBAAyB,CAAC;AAErG,OAAO,EAAE,YAAY,EAAE,MAAM,oBAAoB,CAAC;AAElD,OAAO,EAAE,WAAW,EAAE,MAAM,mBAAmB,CAAC;AAEhD,OAAO,KAAK,EAAmB,oBAAoB,EAAE,iBAAiB,EAAE,cAAc,EAAE,MAAM,uBAAuB,CAAC;AAEtH,OAAO,EAAE,aAAa,EAAE,MAAM,qBAAqB,CAAC;AAEpD,OAAO,KAAK,EAAE,cAAc,EAAE,MAAM,sBAAsB,CAAC;AAE3D,OAAO,EAAwB,cAAc,EAAE,MAAM,sBAAsB,CAAC;AAC5E,OAAO,EAAE,eAAe,EAAE,MAAM,uBAAuB,CAAC;AAIxD,OAAO,EACN,cAAc,EACd,iBAAiB,EACjB,cAAc,EACd,cAAc,EACd,cAAc,EACd,YAAY,EACZ,mBAAmB,EACnB,cAAc,EACd,eAAe,EAEf,qBAAqB,EACrB,MAAM,kBAAkB,CAAC;AAE1B,MAAM,WAAW,yBAAyB;IACzC,4EAA4E;IAC5E,GAAG,CAAC,EAAE,MAAM,CAAC;IACb,yDAAyD;IACzD,QAAQ,CAAC,EAAE,MAAM,CAAC;IAElB,oFAAoF;IACpF,WAAW,CAAC,EAAE,WAAW,CAAC;IAC1B,uFAAuF;IACvF,aAAa,CAAC,EAAE,aAAa,CAAC;IAE9B,iEAAiE;IACjE,KAAK,CAAC,EAAE,KAAK,CAAC,GAAG,CAAC,CAAC;IACnB,4FAA4F;IAC5F,aAAa,CAAC,EAAE,aAAa,CAAC;IAC9B,gEAAgE;IAChE,YAAY,CAAC,EAAE,KAAK,CAAC;QAAE,KAAK,EAAE,KAAK,CAAC,GAAG,CAAC,CAAC;QAAC,aAAa,CAAC,EAAE,aAAa,CAAA;KAAE,CAAC,CAAC;IAE3E;;;;;;OAMG;IACH,OAAO,CAAC,EAAE,KAAK,GAAG,SAAS,CAAC;IAC5B;;;;;;OAMG;IACH,KAAK,CAAC,EAAE,MAAM,EAAE,CAAC;IACjB;;;OAGG;IACH,eAAe,CAAC,EAAE,MAAM,EAAE,CAAC;IAC3B,gEAAgE;IAChE,WAAW,CAAC,EAAE,cAAc,EAAE,CAAC;IAE/B,oEAAoE;IACpE,cAAc,CAAC,EAAE,cAAc,CAAC;IAEhC,2DAA2D;IAC3D,cAAc,CAAC,EAAE,cAAc,CAAC;IAEhC,uEAAuE;IACvE,eAAe,CAAC,EAAE,eAAe,CAAC;IAClC,kEAAkE;IAClE,iBAAiB,CAAC,EAAE,iBAAiB,CAAC;IACtC;;;;OAIG;IACH,oBAAoB,CAAC,EAAE,OAAO,CAAC;CAC/B;AAED,qCAAqC;AACrC,MAAM,WAAW,wBAAwB;IACxC,0BAA0B;IAC1B,OAAO,EAAE,YAAY,CAAC;IACtB,mEAAmE;IACnE,gBAAgB,EAAE,oBAAoB,CAAC;IACvC,wEAAwE;IACxE,oBAAoB,CAAC,EAAE,MAAM,CAAC;CAC9B;AAID,YAAY,EAAE,eAAe,EAAE,WAAW,EAAE,MAAM,wBAAwB,CAAC;AAC3E,OAAO,EAAE,aAAa,EAAE,qBAAqB,EAAE,iBAAiB,EAAE,MAAM,qBAAqB,CAAC;AAC9F,cAAc,4BAA4B,CAAC;AAC3C,YAAY,EACX,YAAY,EACZ,uBAAuB,EACvB,gBAAgB,EAChB,gBAAgB,EAChB,gBAAgB,EAChB,kBAAkB,EAClB,cAAc,GACd,MAAM,uBAAuB,CAAC;AAC/B,YAAY,EAAE,cAAc,EAAE,MAAM,uBAAuB,CAAC;AAC5D,YAAY,EAAE,KAAK,EAAE,MAAM,aAAa,CAAC;AACzC,YAAY,EAAE,IAAI,EAAE,MAAM,kBAAkB,CAAC;AAE7C,OAAO,EACN,qBAAqB,EAErB,iBAAiB,EACjB,mBAAmB,EACnB,cAAc,EACd,cAAc,EACd,cAAc,EACd,eAAe,EACf,cAAc,EACd,cAAc,EACd,YAAY,GACZ,CAAC;AAsCF;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;GAkCG;AACH,wBAAsB,kBAAkB,CAAC,OAAO,GAAE,yBAA8B,GAAG,OAAO,CAAC,wBAAwB,CAAC,CAsOnH","sourcesContent":["import { join } from \"node:path\";\nimport { Agent, type AgentMessage, type ThinkingLevel } from \"@kolisachint/hoocode-agent-core\";\nimport { clampThinkingLevel, type Message, type Model, streamSimple } from \"@kolisachint/hoocode-ai\";\nimport { getAgentDir } from \"../config.js\";\nimport { AgentSession } from \"./agent-session.js\";\nimport { formatNoModelsAvailableMessage } from \"./auth-guidance.js\";\nimport { AuthStorage } from \"./auth-storage.js\";\nimport { DEFAULT_THINKING_LEVEL } from \"./defaults.js\";\nimport type { ExtensionRunner, LoadExtensionsResult, SessionStartEvent, ToolDefinition } from \"./extensions/index.js\";\nimport { convertToLlm, createBackgroundPlaceholderText, createBackgroundTaskMessage } from \"./messages.js\";\nimport { ModelRegistry } from \"./model-registry.js\";\nimport { findInitialModel } from \"./model-resolver.js\";\nimport type { ResourceLoader } from \"./resource-loader.js\";\nimport { DefaultResourceLoader } from \"./resource-loader.js\";\nimport { getDefaultSessionDir, SessionManager } from \"./session-manager.js\";\nimport { SettingsManager } from \"./settings-manager.js\";\nimport { peekSubagentPool } from \"./subagent-pool-instance.js\";\nimport { isInstallTelemetryEnabled } from \"./telemetry.js\";\nimport { time } from \"./timings.js\";\nimport {\n\tcreateBashTool,\n\tcreateCodingTools,\n\tcreateEditTool,\n\tcreateFindTool,\n\tcreateGrepTool,\n\tcreateLsTool,\n\tcreateReadOnlyTools,\n\tcreateReadTool,\n\tcreateWriteTool,\n\ttype ToolName,\n\twithFileMutationQueue,\n} from \"./tools/index.js\";\n\nexport interface CreateAgentSessionOptions {\n\t/** Working directory for project-local discovery. Default: process.cwd() */\n\tcwd?: string;\n\t/** Global config directory. Default: ~/.hoocode/agent */\n\tagentDir?: string;\n\n\t/** Auth storage for credentials. Default: AuthStorage.create(agentDir/auth.json) */\n\tauthStorage?: AuthStorage;\n\t/** Model registry. Default: ModelRegistry.create(authStorage, agentDir/models.json) */\n\tmodelRegistry?: ModelRegistry;\n\n\t/** Model to use. Default: from settings, else first available */\n\tmodel?: Model<any>;\n\t/** Thinking level. Default: from settings, else 'medium' (clamped to model capabilities) */\n\tthinkingLevel?: ThinkingLevel;\n\t/** Models available for cycling (Ctrl+P in interactive mode) */\n\tscopedModels?: Array<{ model: Model<any>; thinkingLevel?: ThinkingLevel }>;\n\n\t/**\n\t * Optional default tool suppression mode when no explicit allowlist is provided.\n\t *\n\t * - \"all\": start with no tools enabled\n\t * - \"builtin\": disable the default built-in tools (read, bash, edit, write)\n\t * but keep extension/custom tools enabled\n\t */\n\tnoTools?: \"all\" | \"builtin\";\n\t/**\n\t * Optional allowlist of tool names.\n\t *\n\t * When omitted, hoocode enables the default built-in tools (read, bash, edit, write)\n\t * and leaves extension/custom tools enabled unless `noTools` changes that default.\n\t * When provided, only the listed tool names are enabled.\n\t */\n\ttools?: string[];\n\t/**\n\t * Optional denylist of tool names, subtracted from whatever set is otherwise\n\t * enabled (allowlist or default). Applied to built-in, extension, and custom tools.\n\t */\n\tdisallowedTools?: string[];\n\t/** Custom tools to register (in addition to built-in tools). */\n\tcustomTools?: ToolDefinition[];\n\n\t/** Resource loader. When omitted, DefaultResourceLoader is used. */\n\tresourceLoader?: ResourceLoader;\n\n\t/** Session manager. Default: SessionManager.create(cwd) */\n\tsessionManager?: SessionManager;\n\n\t/** Settings manager. Default: SettingsManager.create(cwd, agentDir) */\n\tsettingsManager?: SettingsManager;\n\t/** Session start event metadata for extension runtime startup. */\n\tsessionStartEvent?: SessionStartEvent;\n\t/**\n\t * Master gate for local-inference routing (compaction + tool-result\n\t * compression via a local executor model). Off by default. Routing also\n\t * requires a `routing` block in models.json. See docs/local-executor-routing.md.\n\t */\n\tenableLocalInference?: boolean;\n}\n\n/** Result from createAgentSession */\nexport interface CreateAgentSessionResult {\n\t/** The created session */\n\tsession: AgentSession;\n\t/** Extensions result (for UI context setup in interactive mode) */\n\textensionsResult: LoadExtensionsResult;\n\t/** Warning if session was restored with a different model than saved */\n\tmodelFallbackMessage?: string;\n}\n\n// Re-exports\n\nexport type { AgentDefinition, AgentSource } from \"./agent-frontmatter.js\";\nexport { AgentRegistry, formatAgentsForPrompt, loadAgentRegistry } from \"./agent-registry.js\";\nexport * from \"./agent-session-runtime.js\";\nexport type {\n\tExtensionAPI,\n\tExtensionCommandContext,\n\tExtensionContext,\n\tExtensionFactory,\n\tSlashCommandInfo,\n\tSlashCommandSource,\n\tToolDefinition,\n} from \"./extensions/index.js\";\nexport type { PromptTemplate } from \"./prompt-templates.js\";\nexport type { Skill } from \"./skills.js\";\nexport type { Tool } from \"./tools/index.js\";\n\nexport {\n\twithFileMutationQueue,\n\t// Tool factories (for custom cwd)\n\tcreateCodingTools,\n\tcreateReadOnlyTools,\n\tcreateReadTool,\n\tcreateBashTool,\n\tcreateEditTool,\n\tcreateWriteTool,\n\tcreateGrepTool,\n\tcreateFindTool,\n\tcreateLsTool,\n};\n\n// Helper Functions\n\nfunction getDefaultAgentDir(): string {\n\treturn getAgentDir();\n}\n\nfunction getAttributionHeaders(\n\tmodel: Model<any>,\n\tsettingsManager: SettingsManager,\n): Record<string, string> | undefined {\n\tif (!isInstallTelemetryEnabled(settingsManager)) {\n\t\treturn undefined;\n\t}\n\n\tif (model.provider === \"openrouter\" || model.baseUrl.includes(\"openrouter.ai\")) {\n\t\treturn {\n\t\t\t\"HTTP-Referer\": \"https://github.com/kolisachint/hoocode\",\n\t\t\t\"X-OpenRouter-Title\": \"hoocode\",\n\t\t\t\"X-OpenRouter-Categories\": \"cli-agent\",\n\t\t};\n\t}\n\n\tif (\n\t\tmodel.provider === \"cloudflare-workers-ai\" ||\n\t\tmodel.provider === \"cloudflare-ai-gateway\" ||\n\t\tmodel.baseUrl.includes(\"api.cloudflare.com\") ||\n\t\tmodel.baseUrl.includes(\"gateway.ai.cloudflare.com\")\n\t) {\n\t\treturn {\n\t\t\t\"User-Agent\": \"hoocode\",\n\t\t};\n\t}\n\n\treturn undefined;\n}\n\n/**\n * Create an AgentSession with the specified options.\n *\n * @example\n * ```typescript\n * // Minimal - uses defaults\n * const { session } = await createAgentSession();\n *\n * // With explicit model\n * import { getModel } from '@kolisachint/hoocode-ai';\n * const { session } = await createAgentSession({\n * model: getModel('anthropic', 'claude-opus-4-5'),\n * thinkingLevel: 'high',\n * });\n *\n * // Continue previous session\n * const { session, modelFallbackMessage } = await createAgentSession({\n * continueSession: true,\n * });\n *\n * // Full control\n * const loader = new DefaultResourceLoader({\n * cwd: process.cwd(),\n * agentDir: getAgentDir(),\n * settingsManager: SettingsManager.create(),\n * });\n * await loader.reload();\n * const { session } = await createAgentSession({\n * model: myModel,\n * tools: [readTool, bashTool],\n * resourceLoader: loader,\n * sessionManager: SessionManager.inMemory(),\n * });\n * ```\n */\nexport async function createAgentSession(options: CreateAgentSessionOptions = {}): Promise<CreateAgentSessionResult> {\n\tconst cwd = options.cwd ?? options.sessionManager?.getCwd() ?? process.cwd();\n\tconst agentDir = options.agentDir ?? getDefaultAgentDir();\n\tlet resourceLoader = options.resourceLoader;\n\n\t// Use provided or create AuthStorage and ModelRegistry\n\tconst authPath = options.agentDir ? join(agentDir, \"auth.json\") : undefined;\n\tconst modelsPath = options.agentDir ? join(agentDir, \"models.json\") : undefined;\n\tconst authStorage = options.authStorage ?? AuthStorage.create(authPath);\n\tconst modelRegistry = options.modelRegistry ?? ModelRegistry.create(authStorage, modelsPath);\n\n\tconst settingsManager = options.settingsManager ?? SettingsManager.create(cwd, agentDir);\n\tconst sessionManager = options.sessionManager ?? SessionManager.create(cwd, getDefaultSessionDir(cwd, agentDir));\n\n\tif (!resourceLoader) {\n\t\tresourceLoader = new DefaultResourceLoader({ cwd, agentDir, settingsManager });\n\t\tawait resourceLoader.reload();\n\t\ttime(\"resourceLoader.reload\");\n\t}\n\n\t// Check if session has existing data to restore\n\tconst existingSession = sessionManager.buildSessionContext();\n\tconst hasExistingSession = existingSession.messages.length > 0;\n\tconst hasThinkingEntry = sessionManager.getBranch().some((entry) => entry.type === \"thinking_level_change\");\n\n\tlet model = options.model;\n\tlet modelFallbackMessage: string | undefined;\n\n\t// If session has data, try to restore model from it\n\tif (!model && hasExistingSession && existingSession.model) {\n\t\tconst restoredModel = modelRegistry.find(existingSession.model.provider, existingSession.model.modelId);\n\t\tif (restoredModel && modelRegistry.hasConfiguredAuth(restoredModel)) {\n\t\t\tmodel = restoredModel;\n\t\t}\n\t\tif (!model) {\n\t\t\tmodelFallbackMessage = `Could not restore model ${existingSession.model.provider}/${existingSession.model.modelId}`;\n\t\t}\n\t}\n\n\t// If still no model, use findInitialModel (checks settings default, then provider defaults)\n\tif (!model) {\n\t\tconst result = await findInitialModel({\n\t\t\tscopedModels: [],\n\t\t\tisContinuing: hasExistingSession,\n\t\t\tdefaultProvider: settingsManager.getDefaultProvider(),\n\t\t\tdefaultModelId: settingsManager.getDefaultModel(),\n\t\t\tdefaultThinkingLevel: settingsManager.getDefaultThinkingLevel(),\n\t\t\tmodelRegistry,\n\t\t});\n\t\tmodel = result.model;\n\t\tif (!model) {\n\t\t\tmodelFallbackMessage = formatNoModelsAvailableMessage();\n\t\t} else if (modelFallbackMessage) {\n\t\t\tmodelFallbackMessage += `. Using ${model.provider}/${model.id}`;\n\t\t}\n\t}\n\n\tlet thinkingLevel = options.thinkingLevel;\n\n\t// If session has data, restore thinking level from it\n\tif (thinkingLevel === undefined && hasExistingSession) {\n\t\tthinkingLevel = hasThinkingEntry\n\t\t\t? (existingSession.thinkingLevel as ThinkingLevel)\n\t\t\t: (settingsManager.getDefaultThinkingLevel() ?? DEFAULT_THINKING_LEVEL);\n\t}\n\n\t// Fall back to settings default\n\tif (thinkingLevel === undefined) {\n\t\tthinkingLevel = settingsManager.getDefaultThinkingLevel() ?? DEFAULT_THINKING_LEVEL;\n\t}\n\n\t// Clamp to model capabilities\n\tif (!model) {\n\t\tthinkingLevel = \"off\";\n\t} else {\n\t\tthinkingLevel = clampThinkingLevel(model, thinkingLevel) as ThinkingLevel;\n\t}\n\n\tconst defaultActiveToolNames: ToolName[] = [\"read\", \"bash\", \"edit\", \"write\", \"grep\", \"find\", \"ls\"];\n\tconst allowedToolNames = options.tools ?? (options.noTools === \"all\" ? [] : undefined);\n\tconst initialActiveToolNames: string[] = options.tools\n\t\t? [...options.tools]\n\t\t: options.noTools\n\t\t\t? []\n\t\t\t: defaultActiveToolNames;\n\n\tlet agent: Agent;\n\n\t// Create convertToLlm wrapper that filters images if blockImages is enabled (defense-in-depth)\n\tconst convertToLlmWithBlockImages = (messages: AgentMessage[]): Message[] => {\n\t\tconst converted = convertToLlm(messages);\n\t\t// Check setting dynamically so mid-session changes take effect\n\t\tif (!settingsManager.getBlockImages()) {\n\t\t\treturn converted;\n\t\t}\n\t\t// Filter out ImageContent from all messages, replacing with text placeholder\n\t\treturn converted.map((msg) => {\n\t\t\tif (msg.role === \"user\" || msg.role === \"toolResult\") {\n\t\t\t\tconst content = msg.content;\n\t\t\t\tif (Array.isArray(content)) {\n\t\t\t\t\tconst hasImages = content.some((c) => c.type === \"image\");\n\t\t\t\t\tif (hasImages) {\n\t\t\t\t\t\tconst filteredContent = content\n\t\t\t\t\t\t\t.map((c) =>\n\t\t\t\t\t\t\t\tc.type === \"image\" ? { type: \"text\" as const, text: \"Image reading is disabled.\" } : c,\n\t\t\t\t\t\t\t)\n\t\t\t\t\t\t\t.filter(\n\t\t\t\t\t\t\t\t(c, i, arr) =>\n\t\t\t\t\t\t\t\t\t// Dedupe consecutive \"Image reading is disabled.\" texts\n\t\t\t\t\t\t\t\t\t!(\n\t\t\t\t\t\t\t\t\t\tc.type === \"text\" &&\n\t\t\t\t\t\t\t\t\t\tc.text === \"Image reading is disabled.\" &&\n\t\t\t\t\t\t\t\t\t\ti > 0 &&\n\t\t\t\t\t\t\t\t\t\tarr[i - 1].type === \"text\" &&\n\t\t\t\t\t\t\t\t\t\t(arr[i - 1] as { type: \"text\"; text: string }).text === \"Image reading is disabled.\"\n\t\t\t\t\t\t\t\t\t),\n\t\t\t\t\t\t\t);\n\t\t\t\t\t\treturn { ...msg, content: filteredContent };\n\t\t\t\t\t}\n\t\t\t\t}\n\t\t\t}\n\t\t\treturn msg;\n\t\t});\n\t};\n\n\tconst extensionRunnerRef: { current?: ExtensionRunner } = {};\n\n\tagent = new Agent({\n\t\tinitialState: {\n\t\t\tsystemPrompt: \"\",\n\t\t\tmodel,\n\t\t\tthinkingLevel,\n\t\t\ttools: [],\n\t\t},\n\t\tconvertToLlm: convertToLlmWithBlockImages,\n\t\tcreateBackgroundResultMessage: createBackgroundTaskMessage,\n\t\tcreateBackgroundPlaceholder: (toolCall) => createBackgroundPlaceholderText(toolCall),\n\t\t// Report in-process background tool load (e.g. background MCP tools) to the\n\t\t// subagent lifeguard so it widens its heartbeat/timeout tolerance for\n\t\t// concurrently-monitored subagents. Peek (don't create) the pool: background\n\t\t// tools can run before any subagent is ever dispatched.\n\t\tonBackgroundTaskCountChange: (count) => peekSubagentPool()?.setExternalLoad(count),\n\t\tstreamFn: async (model, context, options) => {\n\t\t\tconst auth = await modelRegistry.getApiKeyAndHeaders(model);\n\t\t\tif (!auth.ok) {\n\t\t\t\tthrow new Error(auth.error);\n\t\t\t}\n\t\t\tconst providerRetrySettings = settingsManager.getProviderRetrySettings();\n\t\t\tconst attributionHeaders = getAttributionHeaders(model, settingsManager);\n\t\t\treturn streamSimple(model, context, {\n\t\t\t\t...options,\n\t\t\t\tapiKey: auth.apiKey,\n\t\t\t\ttimeoutMs: options?.timeoutMs ?? providerRetrySettings.timeoutMs,\n\t\t\t\tmaxRetries: options?.maxRetries ?? providerRetrySettings.maxRetries,\n\t\t\t\tmaxRetryDelayMs: options?.maxRetryDelayMs ?? providerRetrySettings.maxRetryDelayMs,\n\t\t\t\theaders:\n\t\t\t\t\tattributionHeaders || auth.headers || options?.headers\n\t\t\t\t\t\t? { ...attributionHeaders, ...auth.headers, ...options?.headers }\n\t\t\t\t\t\t: undefined,\n\t\t\t});\n\t\t},\n\t\tonPayload: async (payload, _model) => {\n\t\t\tconst runner = extensionRunnerRef.current;\n\t\t\tif (!runner?.hasHandlers(\"before_provider_request\")) {\n\t\t\t\treturn payload;\n\t\t\t}\n\t\t\treturn runner.emitBeforeProviderRequest(payload);\n\t\t},\n\t\tonResponse: async (response, _model) => {\n\t\t\tconst runner = extensionRunnerRef.current;\n\t\t\tif (!runner?.hasHandlers(\"after_provider_response\")) {\n\t\t\t\treturn;\n\t\t\t}\n\t\t\tawait runner.emit({\n\t\t\t\ttype: \"after_provider_response\",\n\t\t\t\tstatus: response.status,\n\t\t\t\theaders: response.headers,\n\t\t\t});\n\t\t},\n\t\tsessionId: sessionManager.getSessionId(),\n\t\ttransformContext: async (messages) => {\n\t\t\tconst runner = extensionRunnerRef.current;\n\t\t\tif (!runner) return messages;\n\t\t\treturn runner.emitContext(messages);\n\t\t},\n\t\tsteeringMode: settingsManager.getSteeringMode(),\n\t\tfollowUpMode: settingsManager.getFollowUpMode(),\n\t\ttransport: settingsManager.getTransport(),\n\t\tthinkingBudgets: settingsManager.getThinkingBudgets(),\n\t\tthinkingDisplay: settingsManager.getThinkingDisplay(),\n\t\tmaxRetryDelayMs: settingsManager.getProviderRetrySettings().maxRetryDelayMs,\n\t});\n\n\t// Restore messages if session has existing data\n\tif (hasExistingSession) {\n\t\tagent.state.messages = existingSession.messages;\n\t\tif (!hasThinkingEntry) {\n\t\t\tsessionManager.appendThinkingLevelChange(thinkingLevel);\n\t\t}\n\t} else {\n\t\t// Save initial model and thinking level for new sessions so they can be restored on resume\n\t\tif (model) {\n\t\t\tsessionManager.appendModelChange(model.provider, model.id);\n\t\t}\n\t\tsessionManager.appendThinkingLevelChange(thinkingLevel);\n\t}\n\n\tconst session = new AgentSession({\n\t\tagent,\n\t\tsessionManager,\n\t\tsettingsManager,\n\t\tcwd,\n\t\tscopedModels: options.scopedModels,\n\t\tresourceLoader,\n\t\tcustomTools: options.customTools,\n\t\tmodelRegistry,\n\t\tinitialActiveToolNames,\n\t\tallowedToolNames,\n\t\tdisallowedToolNames: options.disallowedTools,\n\t\textensionRunnerRef,\n\t\tsessionStartEvent: options.sessionStartEvent,\n\t\tenableLocalInference: options.enableLocalInference,\n\t});\n\tconst extensionsResult = resourceLoader.getExtensions();\n\n\treturn {\n\t\tsession,\n\t\textensionsResult,\n\t\tmodelFallbackMessage,\n\t};\n}\n"]}
|
package/dist/core/sdk.js
CHANGED
|
@@ -281,6 +281,7 @@ export async function createAgentSession(options = {}) {
|
|
|
281
281
|
disallowedToolNames: options.disallowedTools,
|
|
282
282
|
extensionRunnerRef,
|
|
283
283
|
sessionStartEvent: options.sessionStartEvent,
|
|
284
|
+
enableLocalInference: options.enableLocalInference,
|
|
284
285
|
});
|
|
285
286
|
const extensionsResult = resourceLoader.getExtensions();
|
|
286
287
|
return {
|