headlesscode 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ATTRIBUTION.md +53 -0
- package/CODE_OF_CONDUCT.md +130 -0
- package/CONTRIBUTING.md +107 -0
- package/LICENSE +202 -0
- package/README.md +486 -0
- package/SECURITY.md +211 -0
- package/bin/headlesscode.mjs +83 -0
- package/package.json +63 -0
- package/shared/prompts/review-mode-prompt-short.md +93 -0
- package/shared/prompts/review-mode-prompt.md +281 -0
- package/shared/rules-code/rules.md +22 -0
- package/shared/stacks/cpp/rules.md +30 -0
- package/shared/stacks/fastapi/rules.md +30 -0
- package/shared/stacks/javascript/rules.md +37 -0
- package/shared/stacks/postgresql/rules.md +31 -0
- package/shared/stacks/python/rules.md +35 -0
- package/shared/stacks/react/rules.md +11 -0
- package/shared/stacks/typescript/rules.md +10 -0
- package/src/budget/budget.ts +221 -0
- package/src/budget/concurrency.ts +126 -0
- package/src/budget/cost.ts +309 -0
- package/src/budget/index.ts +8 -0
- package/src/checkpoints/cli.ts +256 -0
- package/src/checkpoints/service.ts +227 -0
- package/src/cli.ts +1535 -0
- package/src/cloud/docker-provider.ts +334 -0
- package/src/cloud/provider.ts +300 -0
- package/src/codeintel/call-graph.ts +78 -0
- package/src/codeintel/find-references.ts +123 -0
- package/src/codeintel/go-to-definition.ts +193 -0
- package/src/codeintel/handlers.ts +190 -0
- package/src/codeintel/import-graph.ts +173 -0
- package/src/codeintel/outline.ts +180 -0
- package/src/codeintel/position.ts +77 -0
- package/src/codeintel/program.ts +350 -0
- package/src/codeintel/rename-symbol.ts +213 -0
- package/src/codeintel/tools.ts +280 -0
- package/src/codemap/build.ts +135 -0
- package/src/codemap/cli.ts +190 -0
- package/src/codemap/extract.ts +339 -0
- package/src/codemap/files.ts +236 -0
- package/src/codemap/fingerprint.ts +65 -0
- package/src/codemap/flows.ts +62 -0
- package/src/codemap/html.ts +451 -0
- package/src/codemap/lock.ts +80 -0
- package/src/codemap/types.ts +101 -0
- package/src/codesearch/airunner-embedder.ts +185 -0
- package/src/codesearch/chunk.ts +339 -0
- package/src/codesearch/cli.ts +223 -0
- package/src/codesearch/embedder.ts +332 -0
- package/src/codesearch/files.ts +280 -0
- package/src/codesearch/index.ts +469 -0
- package/src/codesearch/ollama-embedder.ts +205 -0
- package/src/codesearch/search.ts +141 -0
- package/src/codesearch/types.ts +100 -0
- package/src/config/mode-models.ts +218 -0
- package/src/dashboard/aggregate.ts +364 -0
- package/src/dashboard/chat-thread.ts +141 -0
- package/src/dashboard/checkpoints.ts +124 -0
- package/src/dashboard/cli.ts +193 -0
- package/src/dashboard/codemap.ts +44 -0
- package/src/dashboard/files.ts +121 -0
- package/src/dashboard/page.ts +2803 -0
- package/src/dashboard/self-improvement-metrics.ts +282 -0
- package/src/dashboard/server.ts +1103 -0
- package/src/dashboard/session-launch.ts +310 -0
- package/src/dashboard/timeline.ts +273 -0
- package/src/dashboard/tool-exec.ts +107 -0
- package/src/dashboard/trend-cli.ts +141 -0
- package/src/dashboard/trend.ts +413 -0
- package/src/decision-proxy/cli.ts +261 -0
- package/src/decision-proxy/proxy.ts +569 -0
- package/src/deploy/gate-cli.ts +147 -0
- package/src/deploy/gate.ts +254 -0
- package/src/engine/condense.ts +512 -0
- package/src/engine/events.ts +428 -0
- package/src/engine/handoff.ts +71 -0
- package/src/engine/lazy-tools.ts +160 -0
- package/src/engine/local-explore.ts +653 -0
- package/src/engine/logger.ts +96 -0
- package/src/engine/loop.ts +5517 -0
- package/src/engine/parser.ts +347 -0
- package/src/engine/prompt.ts +860 -0
- package/src/engine/reports.ts +47 -0
- package/src/engine/stacks.ts +448 -0
- package/src/engine/types.ts +291 -0
- package/src/engine/usage.ts +186 -0
- package/src/github/app-auth.ts +161 -0
- package/src/github/cli.ts +448 -0
- package/src/github/installations.ts +133 -0
- package/src/github/pr.ts +321 -0
- package/src/github/provision.ts +118 -0
- package/src/github/push.ts +122 -0
- package/src/index-util.ts +50 -0
- package/src/index.ts +81 -0
- package/src/init/cli.ts +248 -0
- package/src/init/gitignore.ts +74 -0
- package/src/llm/ollama.ts +308 -0
- package/src/llm/openrouter.ts +868 -0
- package/src/llm/preflight.ts +367 -0
- package/src/llm/transcript-capture.ts +84 -0
- package/src/memory/embed.ts +110 -0
- package/src/memory/index.ts +22 -0
- package/src/memory/local.ts +259 -0
- package/src/memory/summarizer.ts +283 -0
- package/src/memory/types.ts +153 -0
- package/src/memory/uwuchat.ts +157 -0
- package/src/migrate/cli.ts +115 -0
- package/src/orchestrator/analyze-cli.ts +104 -0
- package/src/orchestrator/auto-split.ts +206 -0
- package/src/orchestrator/cleanup.ts +1003 -0
- package/src/orchestrator/cli.ts +3571 -0
- package/src/orchestrator/cost-estimate.ts +564 -0
- package/src/orchestrator/cost-history-cli.ts +242 -0
- package/src/orchestrator/cost-history.ts +397 -0
- package/src/orchestrator/git-sync.ts +250 -0
- package/src/orchestrator/index.ts +153 -0
- package/src/orchestrator/log-analysis.ts +0 -0
- package/src/orchestrator/merge-check.ts +108 -0
- package/src/orchestrator/pipeline.ts +411 -0
- package/src/orchestrator/resume.ts +1940 -0
- package/src/orchestrator/reviewer.ts +503 -0
- package/src/orchestrator/split.ts +296 -0
- package/src/orchestrator/state.ts +542 -0
- package/src/orchestrator/status.ts +697 -0
- package/src/orchestrator/verification-gate.ts +134 -0
- package/src/orchestrator/watch.ts +898 -0
- package/src/permissions/commands.ts +1083 -0
- package/src/permissions/config.ts +241 -0
- package/src/permissions/index.ts +12 -0
- package/src/permissions/protected-files.ts +96 -0
- package/src/permissions/store-protection.ts +272 -0
- package/src/project-store.ts +648 -0
- package/src/projects/cli.ts +382 -0
- package/src/qa/qa.ts +487 -0
- package/src/tools/browser/handler.ts +346 -0
- package/src/tools/browser/service.ts +406 -0
- package/src/tools/browser/smoke.ts +78 -0
- package/src/tools/browser/tool.ts +99 -0
- package/src/tools/executor.ts +2575 -0
- package/src/tools/language-detect.ts +183 -0
- package/src/tools/output-summarizer.ts +369 -0
- package/src/tools/run-tests.ts +302 -0
- package/src/tools/set-indentation-tool.ts +49 -0
- package/src/tools/test-selection.ts +160 -0
- package/src/vendor/tests/smoke.ts +103 -0
- package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
- package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
- package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
- package/src/vendor/zoo-code/shim/os-name.ts +18 -0
- package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
- package/src/vendor/zoo-code/shim/vscode.ts +76 -0
- package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
- package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
- package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
- package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
- package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
- package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
- package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
- package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
- package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
- package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
- package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
- package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
- package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
- package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
- package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
- package/src/vendor/zoo-code/src/shared/language.ts +43 -0
- package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
- package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
- package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
- package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
- package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
- package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
- package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
- package/src/vendor/zoo-code/src/utils/object.ts +18 -0
- package/src/vendor/zoo-code/src/utils/path.ts +94 -0
- package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
- package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
- package/src/vendor/zoo-code/types/global-settings.ts +19 -0
- package/src/vendor/zoo-code/types/index.ts +22 -0
- package/src/vendor/zoo-code/types/message.ts +375 -0
- package/src/vendor/zoo-code/types/mode.ts +241 -0
- package/src/vendor/zoo-code/types/todo.ts +19 -0
- package/src/vendor/zoo-code/types/tool-params.ts +116 -0
- package/src/vendor/zoo-code/types/tool.ts +67 -0
- package/src/vendor/zoo-code/types/vscode.ts +84 -0
- package/src/vision/describe.ts +242 -0
- package/src/vision/tool.ts +91 -0
- package/src/watcher/cli.ts +369 -0
- package/src/watcher/github.ts +304 -0
- package/src/watcher/index.ts +59 -0
- package/src/watcher/state.ts +254 -0
- package/src/watcher/watch.ts +562 -0
- package/tsconfig.json +18 -0
|
@@ -0,0 +1,367 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Preflight probe for `orchestrate` (issue #13).
|
|
3
|
+
*
|
|
4
|
+
* Before ANY worker is spawned, make one cheap 1-token completion using the
|
|
5
|
+
* EXACT model + provider pin a real worker will use (the pin is applied by
|
|
6
|
+
* `buildRequestBody` in src/llm/openrouter.ts for deepseek/* models, so a
|
|
7
|
+
* probe through `OpenRouterClient.createChatCompletion` reproduces it by
|
|
8
|
+
* construction — this is the whole point: a naive manual test request WITHOUT
|
|
9
|
+
* the pin silently routes to a different, reachable provider and gives a
|
|
10
|
+
* false "it's fine" signal).
|
|
11
|
+
*
|
|
12
|
+
* The failure this catches would otherwise only surface 30-80 iterations
|
|
13
|
+
* (10-15 minutes of wall time and real spend) into a round. The classification
|
|
14
|
+
* also separates three failure modes a generic error message conflates:
|
|
15
|
+
* - the API key is invalid/missing (HTTP 401/403);
|
|
16
|
+
* - the OpenRouter account balance is exhausted (402 / insufficient_balance
|
|
17
|
+
* in the body or a 200 error envelope);
|
|
18
|
+
* - the PINNED provider itself is unreachable/depleted (5xx/429 — distinct
|
|
19
|
+
* from account balance; this is the case that misled a real session).
|
|
20
|
+
*
|
|
21
|
+
* `--no-preflight` (orchestrate) skips the extra round-trip for CI /
|
|
22
|
+
* non-interactive contexts; on by default given the cost of getting this
|
|
23
|
+
* wrong.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
import type { PricingTable } from "../budget/cost.js"
|
|
27
|
+
import { estimateCost, loadPricingTable } from "../budget/cost.js"
|
|
28
|
+
import { OllamaClient, DEFAULT_OLLAMA_URL } from "./ollama.js"
|
|
29
|
+
import { OpenRouterClient, OpenRouterError, OPENROUTER_BASE_URL, DEFAULT_MODEL } from "./openrouter.js"
|
|
30
|
+
import type { LlmRequest } from "../engine/types.js"
|
|
31
|
+
|
|
32
|
+
/**
|
|
33
|
+
* Outcome of the preflight probe. `"ok"` is the only status that means
|
|
34
|
+
* "spawn the round"; everything else is a distinct, diagnosable failure.
|
|
35
|
+
*/
|
|
36
|
+
export type PreflightStatus = "ok" | "no-api-key" | "invalid-key" | "balance" | "provider" | "network" | "other"
|
|
37
|
+
|
|
38
|
+
export interface PreflightResult {
|
|
39
|
+
status: PreflightStatus
|
|
40
|
+
/** The model id the probe ran with (the exact model a worker would use). */
|
|
41
|
+
model: string
|
|
42
|
+
/** The OpenRouter base URL probed. */
|
|
43
|
+
baseUrl: string
|
|
44
|
+
/** A single clear, human-readable preflight line (for the CLI + dashboard). */
|
|
45
|
+
line: string
|
|
46
|
+
/** Probe wall time, ms. */
|
|
47
|
+
latencyMs: number
|
|
48
|
+
/** Estimated USD cost of the 1-token probe itself. */
|
|
49
|
+
probeCostUsd: number
|
|
50
|
+
/**
|
|
51
|
+
* Order-of-magnitude estimate of the round's LLM cost (nominal per-session
|
|
52
|
+
* token profile × worker session count × the model's price). undefined when
|
|
53
|
+
* the caller didn't provide a worker session count.
|
|
54
|
+
*/
|
|
55
|
+
roundCostEstimateUsd?: number
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
export interface PreflightOptions {
|
|
59
|
+
/** The model id a worker would use (defaults to the client default). */
|
|
60
|
+
model?: string
|
|
61
|
+
/** API key (default: $HEADLESSCODE_OPENROUTER_API_KEY). */
|
|
62
|
+
apiKey?: string
|
|
63
|
+
/** OpenRouter base URL (default: $OPENROUTER_BASE_URL or the built-in). */
|
|
64
|
+
baseUrl?: string
|
|
65
|
+
/**
|
|
66
|
+
* Number of worker sessions the round will spawn — used for the round cost
|
|
67
|
+
* estimate. Omit/0 to skip the estimate.
|
|
68
|
+
*/
|
|
69
|
+
workerSessions?: number
|
|
70
|
+
/**
|
|
71
|
+
* Per-session cost cap ($HEADLESSCODE_MAX_COST_USD) — when set, the
|
|
72
|
+
* preflight line also shows the round ceiling (cap × sessions).
|
|
73
|
+
*/
|
|
74
|
+
maxCostUsd?: number
|
|
75
|
+
/** Env to read key/base-url from (default: process.env). */
|
|
76
|
+
env?: NodeJS.ProcessEnv
|
|
77
|
+
signal?: AbortSignal
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/**
|
|
81
|
+
* Nominal per-session token profile for the round cost estimate. Deliberately
|
|
82
|
+
* an ORDER-OF-MAGNITUDE heuristic, clearly labeled as such in the output: the
|
|
83
|
+
* estimate is for deciding whether a round is affordable, not for billing.
|
|
84
|
+
*/
|
|
85
|
+
/** Prefix/first-request size per session (system prompt + tool catalog). */
|
|
86
|
+
export const NOMINAL_SESSION_INPUT_TOKENS = 40_000
|
|
87
|
+
/** Single-request output baseline (kept for compatibility/testing). */
|
|
88
|
+
export const NOMINAL_SESSION_OUTPUT_TOKENS = 10_000
|
|
89
|
+
/** Nominal iterations per worker session for the estimate. */
|
|
90
|
+
export const NOMINAL_SESSION_ITERATIONS = 120
|
|
91
|
+
/** History growth per iteration (tokens) — tool-call turns accumulate. */
|
|
92
|
+
export const NOMINAL_HISTORY_GROWTH_PER_ITERATION = 700
|
|
93
|
+
/** Output tokens per iteration (a tool-call turn, not a big generation). */
|
|
94
|
+
export const NOMINAL_OUTPUT_TOKENS_PER_ITERATION = 1_000
|
|
95
|
+
/**
|
|
96
|
+
* Cache-hit rate for resends of the growing history. Live rounds measured
|
|
97
|
+
* ~94% (round1-consolidation 2026-08-16, deepseek-v4-flash) — context resends
|
|
98
|
+
* are cheap cache reads; only the prefix and misses pay full input price.
|
|
99
|
+
*/
|
|
100
|
+
export const NOMINAL_CACHE_HIT_RATE = 0.9
|
|
101
|
+
|
|
102
|
+
/**
|
|
103
|
+
* Estimate the round's LLM cost by modeling the ITERATIVE loop, not a single
|
|
104
|
+
* request: history grows ~linearly each iteration, so total input ≈
|
|
105
|
+
* iterations × average-history-size, most of it cache-read; output scales with
|
|
106
|
+
* iterations. Calibrated against a live 4-worker round (2026-08-16) that
|
|
107
|
+
* measured $0.09–$0.39/session over 83–241 iterations — the old single-request
|
|
108
|
+
* model (40k in / 10k out) estimated $0.0336 for the whole round, ~27× under
|
|
109
|
+
* the ~$0.90 actual. Uses the effective pricing table (defaults +
|
|
110
|
+
* $HEADLESSCODE_PRICING_JSON overrides), same as the budget guardrail.
|
|
111
|
+
*/
|
|
112
|
+
export function estimateRoundCost(
|
|
113
|
+
model: string,
|
|
114
|
+
workerSessions: number,
|
|
115
|
+
options: { pricing?: PricingTable; env?: NodeJS.ProcessEnv; iterationsPerSession?: number } = {},
|
|
116
|
+
): number {
|
|
117
|
+
const table = options.pricing ?? loadPricingTable(options.env ?? process.env)
|
|
118
|
+
const iterations = options.iterationsPerSession ?? NOMINAL_SESSION_ITERATIONS
|
|
119
|
+
// Average history size over the session ≈ prefix + (iterations/2) × growth.
|
|
120
|
+
const avgHistoryTokens = NOMINAL_SESSION_INPUT_TOKENS + (iterations / 2) * NOMINAL_HISTORY_GROWTH_PER_ITERATION
|
|
121
|
+
const totalInputTokens = avgHistoryTokens * iterations
|
|
122
|
+
const perSession = estimateCost({
|
|
123
|
+
model,
|
|
124
|
+
inputTokens: totalInputTokens,
|
|
125
|
+
cachedTokens: Math.round(totalInputTokens * NOMINAL_CACHE_HIT_RATE),
|
|
126
|
+
outputTokens: iterations * NOMINAL_OUTPUT_TOKENS_PER_ITERATION,
|
|
127
|
+
pricing: table,
|
|
128
|
+
})
|
|
129
|
+
return perSession * Math.max(1, workerSessions)
|
|
130
|
+
}
|
|
131
|
+
|
|
132
|
+
/** Round cost ceiling when a per-session cap is configured (cap × sessions). */
|
|
133
|
+
export function roundCostCeilingUsd(workerSessions: number, maxCostUsd?: number): number | undefined {
|
|
134
|
+
if (maxCostUsd === undefined || !Number.isFinite(maxCostUsd) || maxCostUsd <= 0) {
|
|
135
|
+
return undefined
|
|
136
|
+
}
|
|
137
|
+
return maxCostUsd * Math.max(1, workerSessions)
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
/** Balance/credit keywords — the highest-signal text signal, checked FIRST. */
|
|
141
|
+
const BALANCE_RE = /insufficient_balance|insufficient balance|insufficient_quota|insufficient quota|payment required|out of credit|out of quota|billing|account balance|402/i
|
|
142
|
+
|
|
143
|
+
/** Pinned-provider-down language, used to name 5xx/429/200-envelope failures. */
|
|
144
|
+
const PROVIDER_DOWN_RE = /no available providers?|all providers?|provider.*(?:unavailable|down|error|failed)|endpoint.*(?:unavailable|down|error|failed)|upstream|capacity|overloaded|5[0-9]{2}|520|529|530/i
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Classify an OpenRouter probe failure into the issue's named failure modes.
|
|
148
|
+
* Exported for direct unit testing; the text checks run before the status
|
|
149
|
+
* checks because OpenRouter reports the same underlying problem (e.g. a dead
|
|
150
|
+
* account) as 402, 429, or a 200-with-error-envelope depending on which
|
|
151
|
+
* gateway layer answers.
|
|
152
|
+
*/
|
|
153
|
+
export function classifyError(err: unknown): PreflightStatus {
|
|
154
|
+
if (!(err instanceof OpenRouterError)) {
|
|
155
|
+
return "other"
|
|
156
|
+
}
|
|
157
|
+
const haystack = `${err.message} ${err.body ?? ""}`
|
|
158
|
+
if (BALANCE_RE.test(haystack)) {
|
|
159
|
+
return "balance"
|
|
160
|
+
}
|
|
161
|
+
if (err.status === 401 || err.status === 403) {
|
|
162
|
+
return "invalid-key"
|
|
163
|
+
}
|
|
164
|
+
if (err.status === 402) {
|
|
165
|
+
return "balance"
|
|
166
|
+
}
|
|
167
|
+
if (err.status === 429 || (err.status !== undefined && err.status >= 500)) {
|
|
168
|
+
// 429 = OpenRouter's "no available provider / capacity" signal (with
|
|
169
|
+
// allow_fallbacks:false the pinned provider's own quota/depletion
|
|
170
|
+
// surfaces here); 5xx = the upstream endpoint or the OpenRouter
|
|
171
|
+
// gateway failed. Both are "the pinned endpoint isn't serving right
|
|
172
|
+
// now", distinct from the account-balance case above.
|
|
173
|
+
return "provider"
|
|
174
|
+
}
|
|
175
|
+
if (err.status === undefined) {
|
|
176
|
+
if (/^Network error/.test(err.message)) {
|
|
177
|
+
return "network"
|
|
178
|
+
}
|
|
179
|
+
if (PROVIDER_DOWN_RE.test(haystack)) {
|
|
180
|
+
return "provider"
|
|
181
|
+
}
|
|
182
|
+
return "other"
|
|
183
|
+
}
|
|
184
|
+
return "other"
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
function resolveBaseUrl(options: PreflightOptions): string {
|
|
188
|
+
const env = options.env ?? process.env
|
|
189
|
+
return (options.baseUrl ?? env.OPENROUTER_BASE_URL ?? OPENROUTER_BASE_URL).replace(/\/+$/, "")
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
function excerpt(text: string, max = 160): string {
|
|
193
|
+
const t = text.trim().replace(/\s+/g, " ")
|
|
194
|
+
return t.length > max ? `${t.slice(0, max)}…` : t
|
|
195
|
+
}
|
|
196
|
+
|
|
197
|
+
/** Compose the single human-readable preflight line for a result. */
|
|
198
|
+
export function buildPreflightLine(
|
|
199
|
+
result: Omit<PreflightResult, "line"> & { workerSessions: number; maxCostUsd?: number },
|
|
200
|
+
): string {
|
|
201
|
+
const { status, model, latencyMs, probeCostUsd, roundCostEstimateUsd, workerSessions, maxCostUsd } = result
|
|
202
|
+
const pinNote = model.startsWith("deepseek/") ? ' (pinned to provider "deepseek")' : ""
|
|
203
|
+
const costPart =
|
|
204
|
+
roundCostEstimateUsd !== undefined
|
|
205
|
+
? `; estimated round cost ≈ $${roundCostEstimateUsd.toFixed(3)} (${workerSessions} worker session(s), order-of-magnitude)`
|
|
206
|
+
: ""
|
|
207
|
+
const ceiling = roundCostCeilingUsd(workerSessions, maxCostUsd)
|
|
208
|
+
const ceilingPart = ceiling !== undefined ? `; per-session cap $${maxCostUsd} → round ceiling $${ceiling.toFixed(3)}` : ""
|
|
209
|
+
|
|
210
|
+
switch (status) {
|
|
211
|
+
case "ok":
|
|
212
|
+
return (
|
|
213
|
+
`${model}${pinNote}: all clear — 1-token probe OK in ${latencyMs}ms (probe ≈ $${probeCostUsd.toFixed(6)})` +
|
|
214
|
+
costPart +
|
|
215
|
+
ceilingPart
|
|
216
|
+
)
|
|
217
|
+
case "no-api-key":
|
|
218
|
+
return "HEADLESSCODE_OPENROUTER_API_KEY is not set — workers cannot reach OpenRouter"
|
|
219
|
+
case "invalid-key":
|
|
220
|
+
return `${model}${pinNote}: API key invalid or unauthorized (HTTP 401/403) — check HEADLESSCODE_OPENROUTER_API_KEY`
|
|
221
|
+
case "balance":
|
|
222
|
+
return `${model}${pinNote}: OpenRouter account balance exhausted (insufficient_balance / HTTP 402) — top up the account before spawning workers`
|
|
223
|
+
case "provider":
|
|
224
|
+
return `${model}${pinNote}: pinned provider unreachable or depleted (HTTP 5xx/429) — distinct from account balance; the pinned official endpoint is not serving this model right now`
|
|
225
|
+
case "network":
|
|
226
|
+
return `network error reaching OpenRouter (${result.baseUrl}) — check connectivity`
|
|
227
|
+
default:
|
|
228
|
+
return `${model}${pinNote}: preflight probe failed with an unexpected error — inspect the message`
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
/**
|
|
233
|
+
* Run the preflight probe. Never throws: every failure mode is folded into a
|
|
234
|
+
* `PreflightResult` with a clear status + line (the CLI aborts on any
|
|
235
|
+
* non-"ok" status).
|
|
236
|
+
*/
|
|
237
|
+
export async function runPreflight(options: PreflightOptions): Promise<PreflightResult> {
|
|
238
|
+
const env = options.env ?? process.env
|
|
239
|
+
const apiKey = options.apiKey ?? env.HEADLESSCODE_OPENROUTER_API_KEY
|
|
240
|
+
const model = (options.model?.trim() || DEFAULT_MODEL).trim()
|
|
241
|
+
const baseUrl = resolveBaseUrl(options)
|
|
242
|
+
const workerSessions = Math.max(1, options.workerSessions ?? 0)
|
|
243
|
+
const roundCostEstimateUsd =
|
|
244
|
+
options.workerSessions !== undefined && options.workerSessions > 0 ? estimateRoundCost(model, workerSessions, { env }) : undefined
|
|
245
|
+
|
|
246
|
+
if (!apiKey) {
|
|
247
|
+
const result: PreflightResult = {
|
|
248
|
+
status: "no-api-key",
|
|
249
|
+
model,
|
|
250
|
+
baseUrl,
|
|
251
|
+
line: "",
|
|
252
|
+
latencyMs: 0,
|
|
253
|
+
probeCostUsd: 0,
|
|
254
|
+
roundCostEstimateUsd,
|
|
255
|
+
}
|
|
256
|
+
result.line = buildPreflightLine({ ...result, workerSessions, maxCostUsd: options.maxCostUsd })
|
|
257
|
+
return result
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
const client = new OpenRouterClient({ apiKey, baseUrl, defaultModel: model })
|
|
261
|
+
const startedAt = Date.now()
|
|
262
|
+
let status: PreflightStatus
|
|
263
|
+
let probeCostUsd = 0
|
|
264
|
+
let detail = ""
|
|
265
|
+
try {
|
|
266
|
+
const probeRequest: LlmRequest = {
|
|
267
|
+
model,
|
|
268
|
+
messages: [{ role: "user", content: "ping" }],
|
|
269
|
+
maxTokens: 1,
|
|
270
|
+
temperature: 0,
|
|
271
|
+
signal: options.signal,
|
|
272
|
+
}
|
|
273
|
+
const response = await client.createChatCompletion(probeRequest)
|
|
274
|
+
status = "ok"
|
|
275
|
+
const usage = response.usage
|
|
276
|
+
probeCostUsd = estimateCost({
|
|
277
|
+
model,
|
|
278
|
+
inputTokens: usage?.promptTokens ?? 0,
|
|
279
|
+
outputTokens: usage?.completionTokens ?? 0,
|
|
280
|
+
cachedTokens: usage?.cachedTokens ?? 0,
|
|
281
|
+
})
|
|
282
|
+
} catch (err) {
|
|
283
|
+
status = classifyError(err)
|
|
284
|
+
detail = err instanceof Error ? err.message : String(err)
|
|
285
|
+
}
|
|
286
|
+
const latencyMs = Date.now() - startedAt
|
|
287
|
+
|
|
288
|
+
const result: PreflightResult = {
|
|
289
|
+
status,
|
|
290
|
+
model,
|
|
291
|
+
baseUrl,
|
|
292
|
+
line: "",
|
|
293
|
+
latencyMs,
|
|
294
|
+
probeCostUsd,
|
|
295
|
+
roundCostEstimateUsd,
|
|
296
|
+
}
|
|
297
|
+
const line = buildPreflightLine({ ...result, workerSessions, maxCostUsd: options.maxCostUsd })
|
|
298
|
+
result.line = status === "ok" ? line : `${line} — ${excerpt(detail)}`
|
|
299
|
+
return result
|
|
300
|
+
}
|
|
301
|
+
|
|
302
|
+
/**
|
|
303
|
+
* Local-backend counterpart to `runPreflight` — 2026-08-27: `orchestrate`
|
|
304
|
+
* unconditionally ran the OpenRouter probe above even when a round's worker/
|
|
305
|
+
* reviewer/QA mode was configured for the local Ollama-compat daemon
|
|
306
|
+
* (`useLocalCodeBackend` in cli.ts), so switching a project fully to local
|
|
307
|
+
* inference made every `orchestrate` invocation fail preflight against a
|
|
308
|
+
* cloud key/model that round was never going to use (verified live: a round
|
|
309
|
+
* with only local env vars set still probed `deepseek/deepseek-v4-flash-0731`
|
|
310
|
+
* and aborted on a stale/invalid OpenRouter key that was irrelevant to the
|
|
311
|
+
* actual run). This does the same "one cheap real completion before spawning
|
|
312
|
+
* anything" check, but against the local daemon — catching the daemon being
|
|
313
|
+
* down, the model not loaded, or a wrong base URL before burning iterations
|
|
314
|
+
* on it, the same value `runPreflight` provides for the cloud path.
|
|
315
|
+
*
|
|
316
|
+
* Deliberately NOT a variant of `PreflightStatus` above (no-api-key/balance/
|
|
317
|
+
* provider-pin are cloud-specific concepts with no local equivalent) — a
|
|
318
|
+
* local daemon is either reachable-and-generating or it isn't.
|
|
319
|
+
*/
|
|
320
|
+
export type LocalPreflightStatus = "ok" | "unreachable" | "other"
|
|
321
|
+
|
|
322
|
+
export interface LocalPreflightResult {
|
|
323
|
+
status: LocalPreflightStatus
|
|
324
|
+
model: string
|
|
325
|
+
baseUrl: string
|
|
326
|
+
line: string
|
|
327
|
+
latencyMs: number
|
|
328
|
+
}
|
|
329
|
+
|
|
330
|
+
export interface LocalPreflightOptions {
|
|
331
|
+
/** The model id a worker would use (default: OllamaClient's own default). */
|
|
332
|
+
model?: string
|
|
333
|
+
/** Ollama-compat base URL (default: $HEADLESSCODE_OLLAMA_URL or the built-in). */
|
|
334
|
+
baseUrl?: string
|
|
335
|
+
env?: NodeJS.ProcessEnv
|
|
336
|
+
signal?: AbortSignal
|
|
337
|
+
}
|
|
338
|
+
|
|
339
|
+
export async function runLocalPreflight(options: LocalPreflightOptions): Promise<LocalPreflightResult> {
|
|
340
|
+
const env = options.env ?? process.env
|
|
341
|
+
const baseUrl = (options.baseUrl ?? env.HEADLESSCODE_OLLAMA_URL ?? DEFAULT_OLLAMA_URL).replace(/\/+$/, "")
|
|
342
|
+
const model = options.model?.trim() || "(daemon default)"
|
|
343
|
+
const client = new OllamaClient({ baseUrl, defaultModel: options.model })
|
|
344
|
+
const startedAt = Date.now()
|
|
345
|
+
let status: LocalPreflightStatus
|
|
346
|
+
let detail = ""
|
|
347
|
+
try {
|
|
348
|
+
await client.createChatCompletion({
|
|
349
|
+
model,
|
|
350
|
+
messages: [{ role: "user", content: "ping" }],
|
|
351
|
+
maxTokens: 1,
|
|
352
|
+
temperature: 0,
|
|
353
|
+
signal: options.signal,
|
|
354
|
+
})
|
|
355
|
+
status = "ok"
|
|
356
|
+
} catch (err) {
|
|
357
|
+
detail = err instanceof Error ? err.message : String(err)
|
|
358
|
+
const cause = err instanceof Error ? (err.cause as { code?: string } | undefined) : undefined
|
|
359
|
+
status = /ECONNREFUSED|ENOTFOUND|fetch failed|is the daemon running/i.test(detail) || cause?.code === "ECONNREFUSED" ? "unreachable" : "other"
|
|
360
|
+
}
|
|
361
|
+
const latencyMs = Date.now() - startedAt
|
|
362
|
+
const line =
|
|
363
|
+
status === "ok"
|
|
364
|
+
? `${model} @ ${baseUrl}: all clear — 1-token probe OK in ${latencyMs}ms (local daemon, $0)`
|
|
365
|
+
: `${model} @ ${baseUrl}: ${status === "unreachable" ? "daemon unreachable" : "probe failed"} — ${excerpt(detail)}`
|
|
366
|
+
return { status, model, baseUrl, line, latencyMs }
|
|
367
|
+
}
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Opt-in, full-fidelity LLM call transcript capture — the prerequisite for
|
|
3
|
+
* ever fine-tuning/distilling on this project's own sessions.
|
|
4
|
+
*
|
|
5
|
+
* `.headlesscode/events/*.jsonl` (see src/engine/loop.ts's eventFeed) is a
|
|
6
|
+
* lightweight MONITORING log, not a training-data source: verified live
|
|
7
|
+
* 2026-08-27 that a `write_to_file` tool call there logs only the target
|
|
8
|
+
* file path, never the content the model actually generated. It answers
|
|
9
|
+
* "what happened" (which is all it was ever built for) but not "what did
|
|
10
|
+
* the model actually write" — the second one is what imitation-based
|
|
11
|
+
* fine-tuning needs. This module captures the real thing: the exact
|
|
12
|
+
* `LlmRequest.messages` sent and the exact `LlmResponse` received, for
|
|
13
|
+
* every call, from both providers (OllamaClient and OpenRouterClient) —
|
|
14
|
+
* so a DeepSeek session's successful trajectory and a Qwen session's
|
|
15
|
+
* failing one on the same task become directly comparable training pairs.
|
|
16
|
+
*
|
|
17
|
+
* Off by default (matches this project's convention for every experimental
|
|
18
|
+
* feature — see e.g. HEADLESSCODE_LOCAL_EXPLORE's own doc comment). Opt in
|
|
19
|
+
* with HEADLESSCODE_CAPTURE_TRANSCRIPT_DIR set to a directory; one JSONL
|
|
20
|
+
* file per provider per UTC day, so a long-running orchestrate invocation
|
|
21
|
+
* doesn't produce thousands of tiny files. Capture failures are swallowed
|
|
22
|
+
* (never allowed to break a real LLM call over a logging side effect).
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
import * as fs from "node:fs"
|
|
26
|
+
import * as path from "node:path"
|
|
27
|
+
import type { ChatMessage, ChatTool, LlmResponse } from "../engine/types.js"
|
|
28
|
+
|
|
29
|
+
export const TRANSCRIPT_CAPTURE_DIR_ENV = "HEADLESSCODE_CAPTURE_TRANSCRIPT_DIR"
|
|
30
|
+
|
|
31
|
+
export function isTranscriptCaptureEnabled(env: NodeJS.ProcessEnv = process.env): boolean {
|
|
32
|
+
return Boolean(env[TRANSCRIPT_CAPTURE_DIR_ENV]?.trim())
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export interface TranscriptContext {
|
|
36
|
+
/** "ollama" | "openrouter" — which client made this call. */
|
|
37
|
+
provider: "ollama" | "openrouter"
|
|
38
|
+
/** Resolved model id actually used for the call. */
|
|
39
|
+
model: string
|
|
40
|
+
/** Session/task identifiers, when the caller has them, purely for later
|
|
41
|
+
* filtering/joining against harness.log — never required. */
|
|
42
|
+
sessionId?: string
|
|
43
|
+
mode?: string
|
|
44
|
+
workspaceRoot?: string
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export interface TranscriptOutcome {
|
|
48
|
+
/** Full engine-level request messages, exactly as sent (post-condensation,
|
|
49
|
+
* pre-wire-format-translation — the same shape fed back into the next
|
|
50
|
+
* turn, which is what a training example needs to reproduce). */
|
|
51
|
+
messages: ChatMessage[]
|
|
52
|
+
tools?: ChatTool[]
|
|
53
|
+
temperature?: number
|
|
54
|
+
/** Present on success. */
|
|
55
|
+
response?: LlmResponse
|
|
56
|
+
/** Present on failure — the error message, not a thrown object. */
|
|
57
|
+
error?: string
|
|
58
|
+
durationMs: number
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Append one call's full request/response to today's capture file for this
|
|
63
|
+
* provider. Fire-and-forget: never throws, never awaited by the caller —
|
|
64
|
+
* a capture failure must not affect the real LLM call it's recording.
|
|
65
|
+
*/
|
|
66
|
+
export function captureTranscript(context: TranscriptContext, outcome: TranscriptOutcome): void {
|
|
67
|
+
const dir = process.env[TRANSCRIPT_CAPTURE_DIR_ENV]?.trim()
|
|
68
|
+
if (!dir) {
|
|
69
|
+
return
|
|
70
|
+
}
|
|
71
|
+
try {
|
|
72
|
+
fs.mkdirSync(dir, { recursive: true })
|
|
73
|
+
const day = new Date().toISOString().slice(0, 10)
|
|
74
|
+
const filePath = path.join(dir, `${context.provider}-${day}.jsonl`)
|
|
75
|
+
const record = {
|
|
76
|
+
ts: new Date().toISOString(),
|
|
77
|
+
...context,
|
|
78
|
+
...outcome,
|
|
79
|
+
}
|
|
80
|
+
fs.appendFileSync(filePath, `${JSON.stringify(record)}\n`, "utf-8")
|
|
81
|
+
} catch {
|
|
82
|
+
// Never let a capture failure affect the real call.
|
|
83
|
+
}
|
|
84
|
+
}
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Local embedder — Phase 3 placeholder for a real local embedding model.
|
|
3
|
+
*
|
|
4
|
+
* `createLocalEmbedder()` returns a dependency-free, deterministic embedder:
|
|
5
|
+
* the text is tokenized into lowercase word tokens + character bigrams, each
|
|
6
|
+
* token is hashed into a fixed-dimension vector (default 256) with a stable
|
|
7
|
+
* 32-bit hash, and the vector is L2-normalized. Cosine similarity over these
|
|
8
|
+
* vectors gives a cheap lexical-overlap similarity signal.
|
|
9
|
+
*
|
|
10
|
+
* This is the ZERO-DEPENDENCY stand-in the spec calls for (a local embedding
|
|
11
|
+
* model with no cloud routing / no rate limiter / no encryption). A real local
|
|
12
|
+
* embedding model can be swapped in behind the `Embedder` interface
|
|
13
|
+
* (src/memory/types.ts) later WITHOUT touching any caller.
|
|
14
|
+
*
|
|
15
|
+
* Determinism guarantees:
|
|
16
|
+
* - Same input string → same vector (across calls AND across processes).
|
|
17
|
+
* - No Math.random, no platform-dependent iteration order.
|
|
18
|
+
* - Hashing is pure FNV-1a; dims must be ≥ 1.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import type { Embedder } from "./types.js"
|
|
22
|
+
|
|
23
|
+
/** FNV-1a 32-bit hash — pure, deterministic, platform-independent. */
|
|
24
|
+
export function fnv1a(text: string): number {
|
|
25
|
+
let hash = 0x811c9dc5
|
|
26
|
+
for (let i = 0; i < text.length; i++) {
|
|
27
|
+
hash ^= text.charCodeAt(i)
|
|
28
|
+
hash = Math.imul(hash, 0x01000193)
|
|
29
|
+
}
|
|
30
|
+
return hash >>> 0
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Tokenize into lowercase word tokens (len > 1) + character bigrams of the
|
|
35
|
+
* lowercased string. Deterministic order: words first, then bigrams.
|
|
36
|
+
*/
|
|
37
|
+
export function tokenize(text: string): string[] {
|
|
38
|
+
const lower = text.toLowerCase()
|
|
39
|
+
const words = lower.match(/[a-z0-9]+/g) ?? []
|
|
40
|
+
const tokens = words.filter((w) => w.length > 1)
|
|
41
|
+
for (let i = 0; i + 1 < lower.length; i++) {
|
|
42
|
+
const bigram = lower.slice(i, i + 2)
|
|
43
|
+
if (/[a-z0-9]/.test(bigram[0]) && /[a-z0-9]/.test(bigram[1])) {
|
|
44
|
+
tokens.push(bigram)
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
return tokens
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Create a deterministic local embedder.
|
|
52
|
+
*
|
|
53
|
+
* @param dims fixed vector dimension (default 256). Must be a positive int.
|
|
54
|
+
*/
|
|
55
|
+
export function createLocalEmbedder(dims = 256): Embedder {
|
|
56
|
+
if (!Number.isInteger(dims) || dims <= 0) {
|
|
57
|
+
throw new Error(`createLocalEmbedder: dims must be a positive integer (got ${dims})`)
|
|
58
|
+
}
|
|
59
|
+
return {
|
|
60
|
+
embed(text: string): number[] {
|
|
61
|
+
const vector = new Array<number>(dims).fill(0)
|
|
62
|
+
for (const token of tokenize(text)) {
|
|
63
|
+
vector[fnv1a(token) % dims] += 1
|
|
64
|
+
}
|
|
65
|
+
return l2Normalize(vector)
|
|
66
|
+
},
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* Cosine similarity between two vectors. Zero vectors (or zero norms) yield 0.
|
|
72
|
+
* Handles different lengths by treating missing entries as 0.
|
|
73
|
+
*/
|
|
74
|
+
export function cosine(a: number[], b: number[]): number {
|
|
75
|
+
let dot = 0
|
|
76
|
+
let normA = 0
|
|
77
|
+
let normB = 0
|
|
78
|
+
const len = Math.max(a.length, b.length)
|
|
79
|
+
for (let i = 0; i < len; i++) {
|
|
80
|
+
const av = a[i] ?? 0
|
|
81
|
+
const bv = b[i] ?? 0
|
|
82
|
+
dot += av * bv
|
|
83
|
+
normA += av * av
|
|
84
|
+
normB += bv * bv
|
|
85
|
+
}
|
|
86
|
+
if (normA === 0 || normB === 0) {
|
|
87
|
+
return 0
|
|
88
|
+
}
|
|
89
|
+
return dot / (Math.sqrt(normA) * Math.sqrt(normB))
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/** Convenience: cosine similarity of two texts through an embedder. */
|
|
93
|
+
export function similarity(embedder: Embedder, a: string, b: string): number {
|
|
94
|
+
return cosine(embedder.embed(a), embedder.embed(b))
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
function l2Normalize(vector: number[]): number[] {
|
|
98
|
+
let norm = 0
|
|
99
|
+
for (const v of vector) {
|
|
100
|
+
norm += v * v
|
|
101
|
+
}
|
|
102
|
+
if (norm === 0) {
|
|
103
|
+
return vector
|
|
104
|
+
}
|
|
105
|
+
const scale = 1 / Math.sqrt(norm)
|
|
106
|
+
for (let i = 0; i < vector.length; i++) {
|
|
107
|
+
vector[i] *= scale
|
|
108
|
+
}
|
|
109
|
+
return vector
|
|
110
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Phase 3 memory subsystem — public API boundary.
|
|
3
|
+
*
|
|
4
|
+
* Re-exports the memory contracts, the fully working local backend, the local
|
|
5
|
+
* embedder, the deterministic summarizer, and the UwUChat client stub so
|
|
6
|
+
* callers (HeadlessSession, the CLI, and future consumers) depend on this
|
|
7
|
+
* module — never on file paths into `src/memory/`.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
export * from "./types.js"
|
|
11
|
+
export { LocalMemoryStore, sanitizeProject, type LocalMemoryStoreOptions } from "./local.js"
|
|
12
|
+
export { createLocalEmbedder, cosine, similarity, fnv1a, tokenize } from "./embed.js"
|
|
13
|
+
export {
|
|
14
|
+
extractSessionSummary,
|
|
15
|
+
extractFacts,
|
|
16
|
+
buildRollingSummary,
|
|
17
|
+
classifyLine,
|
|
18
|
+
NoopSummarizer,
|
|
19
|
+
type SummarizeWithLlm,
|
|
20
|
+
type ExtractSummaryOptions,
|
|
21
|
+
} from "./summarizer.js"
|
|
22
|
+
export { UwUChatMemoryStore, UwUChatMemoryError, API_PREFIX, type UwUChatMemoryOptions } from "./uwuchat.js"
|