headlesscode 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ATTRIBUTION.md +53 -0
- package/CODE_OF_CONDUCT.md +130 -0
- package/CONTRIBUTING.md +107 -0
- package/LICENSE +202 -0
- package/README.md +486 -0
- package/SECURITY.md +211 -0
- package/bin/headlesscode.mjs +83 -0
- package/package.json +63 -0
- package/shared/prompts/review-mode-prompt-short.md +93 -0
- package/shared/prompts/review-mode-prompt.md +281 -0
- package/shared/rules-code/rules.md +22 -0
- package/shared/stacks/cpp/rules.md +30 -0
- package/shared/stacks/fastapi/rules.md +30 -0
- package/shared/stacks/javascript/rules.md +37 -0
- package/shared/stacks/postgresql/rules.md +31 -0
- package/shared/stacks/python/rules.md +35 -0
- package/shared/stacks/react/rules.md +11 -0
- package/shared/stacks/typescript/rules.md +10 -0
- package/src/budget/budget.ts +221 -0
- package/src/budget/concurrency.ts +126 -0
- package/src/budget/cost.ts +309 -0
- package/src/budget/index.ts +8 -0
- package/src/checkpoints/cli.ts +256 -0
- package/src/checkpoints/service.ts +227 -0
- package/src/cli.ts +1535 -0
- package/src/cloud/docker-provider.ts +334 -0
- package/src/cloud/provider.ts +300 -0
- package/src/codeintel/call-graph.ts +78 -0
- package/src/codeintel/find-references.ts +123 -0
- package/src/codeintel/go-to-definition.ts +193 -0
- package/src/codeintel/handlers.ts +190 -0
- package/src/codeintel/import-graph.ts +173 -0
- package/src/codeintel/outline.ts +180 -0
- package/src/codeintel/position.ts +77 -0
- package/src/codeintel/program.ts +350 -0
- package/src/codeintel/rename-symbol.ts +213 -0
- package/src/codeintel/tools.ts +280 -0
- package/src/codemap/build.ts +135 -0
- package/src/codemap/cli.ts +190 -0
- package/src/codemap/extract.ts +339 -0
- package/src/codemap/files.ts +236 -0
- package/src/codemap/fingerprint.ts +65 -0
- package/src/codemap/flows.ts +62 -0
- package/src/codemap/html.ts +451 -0
- package/src/codemap/lock.ts +80 -0
- package/src/codemap/types.ts +101 -0
- package/src/codesearch/airunner-embedder.ts +185 -0
- package/src/codesearch/chunk.ts +339 -0
- package/src/codesearch/cli.ts +223 -0
- package/src/codesearch/embedder.ts +332 -0
- package/src/codesearch/files.ts +280 -0
- package/src/codesearch/index.ts +469 -0
- package/src/codesearch/ollama-embedder.ts +205 -0
- package/src/codesearch/search.ts +141 -0
- package/src/codesearch/types.ts +100 -0
- package/src/config/mode-models.ts +218 -0
- package/src/dashboard/aggregate.ts +364 -0
- package/src/dashboard/chat-thread.ts +141 -0
- package/src/dashboard/checkpoints.ts +124 -0
- package/src/dashboard/cli.ts +193 -0
- package/src/dashboard/codemap.ts +44 -0
- package/src/dashboard/files.ts +121 -0
- package/src/dashboard/page.ts +2803 -0
- package/src/dashboard/self-improvement-metrics.ts +282 -0
- package/src/dashboard/server.ts +1103 -0
- package/src/dashboard/session-launch.ts +310 -0
- package/src/dashboard/timeline.ts +273 -0
- package/src/dashboard/tool-exec.ts +107 -0
- package/src/dashboard/trend-cli.ts +141 -0
- package/src/dashboard/trend.ts +413 -0
- package/src/decision-proxy/cli.ts +261 -0
- package/src/decision-proxy/proxy.ts +569 -0
- package/src/deploy/gate-cli.ts +147 -0
- package/src/deploy/gate.ts +254 -0
- package/src/engine/condense.ts +512 -0
- package/src/engine/events.ts +428 -0
- package/src/engine/handoff.ts +71 -0
- package/src/engine/lazy-tools.ts +160 -0
- package/src/engine/local-explore.ts +653 -0
- package/src/engine/logger.ts +96 -0
- package/src/engine/loop.ts +5517 -0
- package/src/engine/parser.ts +347 -0
- package/src/engine/prompt.ts +860 -0
- package/src/engine/reports.ts +47 -0
- package/src/engine/stacks.ts +448 -0
- package/src/engine/types.ts +291 -0
- package/src/engine/usage.ts +186 -0
- package/src/github/app-auth.ts +161 -0
- package/src/github/cli.ts +448 -0
- package/src/github/installations.ts +133 -0
- package/src/github/pr.ts +321 -0
- package/src/github/provision.ts +118 -0
- package/src/github/push.ts +122 -0
- package/src/index-util.ts +50 -0
- package/src/index.ts +81 -0
- package/src/init/cli.ts +248 -0
- package/src/init/gitignore.ts +74 -0
- package/src/llm/ollama.ts +308 -0
- package/src/llm/openrouter.ts +868 -0
- package/src/llm/preflight.ts +367 -0
- package/src/llm/transcript-capture.ts +84 -0
- package/src/memory/embed.ts +110 -0
- package/src/memory/index.ts +22 -0
- package/src/memory/local.ts +259 -0
- package/src/memory/summarizer.ts +283 -0
- package/src/memory/types.ts +153 -0
- package/src/memory/uwuchat.ts +157 -0
- package/src/migrate/cli.ts +115 -0
- package/src/orchestrator/analyze-cli.ts +104 -0
- package/src/orchestrator/auto-split.ts +206 -0
- package/src/orchestrator/cleanup.ts +1003 -0
- package/src/orchestrator/cli.ts +3571 -0
- package/src/orchestrator/cost-estimate.ts +564 -0
- package/src/orchestrator/cost-history-cli.ts +242 -0
- package/src/orchestrator/cost-history.ts +397 -0
- package/src/orchestrator/git-sync.ts +250 -0
- package/src/orchestrator/index.ts +153 -0
- package/src/orchestrator/log-analysis.ts +0 -0
- package/src/orchestrator/merge-check.ts +108 -0
- package/src/orchestrator/pipeline.ts +411 -0
- package/src/orchestrator/resume.ts +1940 -0
- package/src/orchestrator/reviewer.ts +503 -0
- package/src/orchestrator/split.ts +296 -0
- package/src/orchestrator/state.ts +542 -0
- package/src/orchestrator/status.ts +697 -0
- package/src/orchestrator/verification-gate.ts +134 -0
- package/src/orchestrator/watch.ts +898 -0
- package/src/permissions/commands.ts +1083 -0
- package/src/permissions/config.ts +241 -0
- package/src/permissions/index.ts +12 -0
- package/src/permissions/protected-files.ts +96 -0
- package/src/permissions/store-protection.ts +272 -0
- package/src/project-store.ts +648 -0
- package/src/projects/cli.ts +382 -0
- package/src/qa/qa.ts +487 -0
- package/src/tools/browser/handler.ts +346 -0
- package/src/tools/browser/service.ts +406 -0
- package/src/tools/browser/smoke.ts +78 -0
- package/src/tools/browser/tool.ts +99 -0
- package/src/tools/executor.ts +2575 -0
- package/src/tools/language-detect.ts +183 -0
- package/src/tools/output-summarizer.ts +369 -0
- package/src/tools/run-tests.ts +302 -0
- package/src/tools/set-indentation-tool.ts +49 -0
- package/src/tools/test-selection.ts +160 -0
- package/src/vendor/tests/smoke.ts +103 -0
- package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
- package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
- package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
- package/src/vendor/zoo-code/shim/os-name.ts +18 -0
- package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
- package/src/vendor/zoo-code/shim/vscode.ts +76 -0
- package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
- package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
- package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
- package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
- package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
- package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
- package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
- package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
- package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
- package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
- package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
- package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
- package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
- package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
- package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
- package/src/vendor/zoo-code/src/shared/language.ts +43 -0
- package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
- package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
- package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
- package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
- package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
- package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
- package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
- package/src/vendor/zoo-code/src/utils/object.ts +18 -0
- package/src/vendor/zoo-code/src/utils/path.ts +94 -0
- package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
- package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
- package/src/vendor/zoo-code/types/global-settings.ts +19 -0
- package/src/vendor/zoo-code/types/index.ts +22 -0
- package/src/vendor/zoo-code/types/message.ts +375 -0
- package/src/vendor/zoo-code/types/mode.ts +241 -0
- package/src/vendor/zoo-code/types/todo.ts +19 -0
- package/src/vendor/zoo-code/types/tool-params.ts +116 -0
- package/src/vendor/zoo-code/types/tool.ts +67 -0
- package/src/vendor/zoo-code/types/vscode.ts +84 -0
- package/src/vision/describe.ts +242 -0
- package/src/vision/tool.ts +91 -0
- package/src/watcher/cli.ts +369 -0
- package/src/watcher/github.ts +304 -0
- package/src/watcher/index.ts +59 -0
- package/src/watcher/state.ts +254 -0
- package/src/watcher/watch.ts +562 -0
- package/tsconfig.json +18 -0
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Phase 6 — per-session cost/time/iteration budget enforcement.
|
|
3
|
+
*
|
|
4
|
+
* `BudgetTracker` is the pure, unit-testable core of the per-session budget
|
|
5
|
+
* guardrail (spec 6.3): it sits inside a `HeadlessSession` (or any caller
|
|
6
|
+
* that makes LLM calls) and trips a `BudgetExceededError` when ANY of these
|
|
7
|
+
* limits is crossed:
|
|
8
|
+
*
|
|
9
|
+
* - `maxCostUsd` — accumulated estimated USD cost (tokens × pricing);
|
|
10
|
+
* - `maxDurationMs` — wall-clock elapsed since the tracker was created;
|
|
11
|
+
* - `maxIterations` — number of LLM calls (ticks) performed.
|
|
12
|
+
*
|
|
13
|
+
* Enforcement points (see `check` / `tick` / `record`):
|
|
14
|
+
* - `tick()` is called BEFORE each LLM call and re-checks elapsed time,
|
|
15
|
+
* iteration count AND accumulated cost (a previous call's `record()` may
|
|
16
|
+
* have pushed cost past the cap — the next `tick()` catches it before any
|
|
17
|
+
* further spend);
|
|
18
|
+
* - `record()` is called AFTER each LLM call with the provider's usage token
|
|
19
|
+
* counts and accumulates cost (checking the cost cap again).
|
|
20
|
+
*
|
|
21
|
+
* Pure by construction: no fs, no network, no global state. Time is
|
|
22
|
+
* injectable (`now`), so duration tests use a fake clock; pricing is
|
|
23
|
+
* injectable so cost tests don't depend on the env pricing file.
|
|
24
|
+
*
|
|
25
|
+
* When a budget is NOT configured the tracker is simply never created — zero
|
|
26
|
+
* behavior change (the session default is `budget: null`).
|
|
27
|
+
*/
|
|
28
|
+
|
|
29
|
+
import { estimateCost, loadPricingTable, type PricingTable } from "./cost.js"
|
|
30
|
+
|
|
31
|
+
/** Per-session budget (all limits optional — only the set ones are enforced). */
|
|
32
|
+
export interface SessionBudget {
|
|
33
|
+
/** Max estimated USD spend for the session (accumulated across LLM calls). */
|
|
34
|
+
maxCostUsd?: number
|
|
35
|
+
/** Max wall-clock duration for the session, ms (checked before each call). */
|
|
36
|
+
maxDurationMs?: number
|
|
37
|
+
/** Max LLM-call iterations (a stricter cap than the loop's maxIterations). */
|
|
38
|
+
maxIterations?: number
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export type BudgetLimitReason = "cost" | "duration" | "iterations"
|
|
42
|
+
|
|
43
|
+
/** Thrown by `tick()`/`record()` when a budget limit trips. */
|
|
44
|
+
export class BudgetExceededError extends Error {
|
|
45
|
+
readonly reason: BudgetLimitReason
|
|
46
|
+
readonly costUsd: number
|
|
47
|
+
readonly elapsedMs: number
|
|
48
|
+
readonly iterations: number
|
|
49
|
+
|
|
50
|
+
constructor(reason: BudgetLimitReason, costUsd: number, elapsedMs: number, iterations: number) {
|
|
51
|
+
super(`Budget exceeded: ${reason} (cost $${costUsd.toFixed(6)}, elapsed ${elapsedMs}ms, iterations ${iterations})`)
|
|
52
|
+
this.name = "BudgetExceededError"
|
|
53
|
+
this.reason = reason
|
|
54
|
+
this.costUsd = costUsd
|
|
55
|
+
this.elapsedMs = elapsedMs
|
|
56
|
+
this.iterations = iterations
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
export interface BudgetTrackerOptions {
|
|
61
|
+
/** Clock (default Date.now) — inject a fake for duration tests. */
|
|
62
|
+
now?: () => number
|
|
63
|
+
/** Pricing table (default: env-merged defaults) — inject for cost tests. */
|
|
64
|
+
pricing?: PricingTable
|
|
65
|
+
/**
|
|
66
|
+
* When false, record() never accumulates cost (stays 0 forever) and
|
|
67
|
+
* maxCostUsd checks never trip. Default true. Issue #144: a local-
|
|
68
|
+
* backend session has no real dollar cost — computing and logging a
|
|
69
|
+
* fabricated figure for it is noise at best, misleading at worst.
|
|
70
|
+
*/
|
|
71
|
+
trackCost?: boolean
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** Non-throwing snapshot of the budget state (surfaced on SessionResult). */
|
|
75
|
+
export interface BudgetCheck {
|
|
76
|
+
ok: boolean
|
|
77
|
+
reason?: BudgetLimitReason
|
|
78
|
+
costUsd: number
|
|
79
|
+
elapsedMs: number
|
|
80
|
+
iterations: number
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
export class BudgetTracker {
|
|
84
|
+
private readonly budget: SessionBudget
|
|
85
|
+
private readonly now: () => number
|
|
86
|
+
private readonly pricing: PricingTable
|
|
87
|
+
private readonly trackCost: boolean
|
|
88
|
+
private readonly startedAt: number
|
|
89
|
+
private iterations = 0
|
|
90
|
+
private costUsd = 0
|
|
91
|
+
/** Cumulative ms spent paused (see pauseClock/resumeClock), excluded from elapsedMs. */
|
|
92
|
+
private blockedMs = 0
|
|
93
|
+
/** Set while paused (the `now()` value pauseClock() was called at); null when running. */
|
|
94
|
+
private blockStartedAt: number | null = null
|
|
95
|
+
|
|
96
|
+
constructor(budget: SessionBudget, options: BudgetTrackerOptions = {}) {
|
|
97
|
+
this.budget = budget
|
|
98
|
+
this.now = options.now ?? (() => Date.now())
|
|
99
|
+
this.pricing = options.pricing ?? loadPricingTable()
|
|
100
|
+
this.trackCost = options.trackCost ?? true
|
|
101
|
+
this.startedAt = this.now()
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Decision escalation (workstream 2): call before blocking on an external
|
|
106
|
+
* answer (e.g. ask_followup_question waiting on `.harness.decision-answer`)
|
|
107
|
+
* so the wait doesn't count against `maxDurationMs`. Idempotent — a second
|
|
108
|
+
* call while already paused is a no-op.
|
|
109
|
+
*/
|
|
110
|
+
pauseClock(): void {
|
|
111
|
+
if (this.blockStartedAt === null) {
|
|
112
|
+
this.blockStartedAt = this.now()
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* Resume the clock after a pause, folding the paused interval into
|
|
118
|
+
* `blockedMs`. Idempotent — a call while not paused is a no-op.
|
|
119
|
+
*/
|
|
120
|
+
resumeClock(): void {
|
|
121
|
+
if (this.blockStartedAt !== null) {
|
|
122
|
+
this.blockedMs += Math.max(0, this.now() - this.blockStartedAt)
|
|
123
|
+
this.blockStartedAt = null
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
get elapsedMs(): number {
|
|
128
|
+
const raw = Math.max(0, this.now() - this.startedAt)
|
|
129
|
+
const inProgressBlock = this.blockStartedAt !== null ? Math.max(0, this.now() - this.blockStartedAt) : 0
|
|
130
|
+
return Math.max(0, raw - this.blockedMs - inProgressBlock)
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
get iterationCount(): number {
|
|
134
|
+
return this.iterations
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
get totalCostUsd(): number {
|
|
138
|
+
return this.costUsd
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* Call BEFORE each LLM call. Increments the iteration counter and re-checks
|
|
143
|
+
* duration, iterations and accumulated cost. Throws `BudgetExceededError`
|
|
144
|
+
* when a limit is crossed.
|
|
145
|
+
*/
|
|
146
|
+
tick(): void {
|
|
147
|
+
this.iterations++
|
|
148
|
+
this.throwIfExceeded()
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
/**
|
|
152
|
+
* Call AFTER each LLM call with the provider's usage token counts.
|
|
153
|
+
* Accumulates the estimated cost and re-checks the cost cap. Throws
|
|
154
|
+
* `BudgetExceededError` when the cost limit trips.
|
|
155
|
+
*/
|
|
156
|
+
record(input: { model: string; inputTokens?: number; outputTokens?: number; cachedTokens?: number }): void {
|
|
157
|
+
if (!this.trackCost) {
|
|
158
|
+
return
|
|
159
|
+
}
|
|
160
|
+
const cost = estimateCost({
|
|
161
|
+
model: input.model,
|
|
162
|
+
inputTokens: input.inputTokens ?? 0,
|
|
163
|
+
outputTokens: input.outputTokens ?? 0,
|
|
164
|
+
cachedTokens: input.cachedTokens ?? 0,
|
|
165
|
+
pricing: this.pricing,
|
|
166
|
+
})
|
|
167
|
+
this.costUsd += cost
|
|
168
|
+
this.throwIfExceeded()
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
/** Non-throwing snapshot: { ok, reason?, costUsd, elapsedMs, iterations }. */
|
|
172
|
+
check(): BudgetCheck {
|
|
173
|
+
const { reason } = this.exceededReason()
|
|
174
|
+
return {
|
|
175
|
+
ok: reason === undefined,
|
|
176
|
+
reason,
|
|
177
|
+
costUsd: this.costUsd,
|
|
178
|
+
elapsedMs: this.elapsedMs,
|
|
179
|
+
iterations: this.iterations,
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
private exceededReason(): { reason?: BudgetLimitReason } {
|
|
184
|
+
if (this.budget.maxDurationMs !== undefined && this.elapsedMs >= this.budget.maxDurationMs) {
|
|
185
|
+
return { reason: "duration" }
|
|
186
|
+
}
|
|
187
|
+
if (this.budget.maxIterations !== undefined && this.iterations > this.budget.maxIterations) {
|
|
188
|
+
return { reason: "iterations" }
|
|
189
|
+
}
|
|
190
|
+
if (this.budget.maxCostUsd !== undefined && this.costUsd >= this.budget.maxCostUsd) {
|
|
191
|
+
return { reason: "cost" }
|
|
192
|
+
}
|
|
193
|
+
return {}
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
private throwIfExceeded(): void {
|
|
197
|
+
const { reason } = this.exceededReason()
|
|
198
|
+
if (reason !== undefined) {
|
|
199
|
+
throw new BudgetExceededError(reason, this.costUsd, this.elapsedMs, this.iterations)
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Build a `SessionBudget` from the harness env vars used by the CLI +
|
|
206
|
+
* run-worker.sh/run-qa.sh. Returns undefined when neither cost nor duration is
|
|
207
|
+
* set (budget OFF). Invalid values are ignored (validation lives in the CLI).
|
|
208
|
+
*/
|
|
209
|
+
export function sessionBudgetFromEnv(env: NodeJS.ProcessEnv = process.env): SessionBudget | undefined {
|
|
210
|
+
const costRaw = env.HEADLESSCODE_MAX_COST_USD
|
|
211
|
+
const durationRaw = env.HEADLESSCODE_MAX_DURATION_MS
|
|
212
|
+
const cost = costRaw !== undefined && costRaw !== "" ? Number(costRaw) : undefined
|
|
213
|
+
const duration = durationRaw !== undefined && durationRaw !== "" ? Number(durationRaw) : undefined
|
|
214
|
+
if ((cost !== undefined && Number.isFinite(cost) && cost > 0) || (duration !== undefined && Number.isFinite(duration) && duration > 0)) {
|
|
215
|
+
return {
|
|
216
|
+
...(cost !== undefined && Number.isFinite(cost) && cost > 0 ? { maxCostUsd: cost } : {}),
|
|
217
|
+
...(duration !== undefined && Number.isFinite(duration) && duration > 0 ? { maxDurationMs: duration } : {}),
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
return undefined
|
|
221
|
+
}
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Phase 6 — global concurrency guardrail (spec 6.3): a hard cap on how many
|
|
3
|
+
* harness sessions may run at once, enforced BEFORE anything is spawned.
|
|
4
|
+
*
|
|
5
|
+
* Two layers, both in this module:
|
|
6
|
+
*
|
|
7
|
+
* 1. `ConcurrencyLimiter` — an in-process counting semaphore. We deliberately
|
|
8
|
+
* chose FAIL-FAST over queueing for headless spawners: a queued spawner
|
|
9
|
+
* has nowhere to hold "waiting" work unless it is the watcher (which
|
|
10
|
+
* already has a durable pending mechanism), and a silently queued
|
|
11
|
+
* orchestrate run is worse than a loud abort. The watcher's "defer with
|
|
12
|
+
* durable pending state" is the queue — implemented in the watcher, not
|
|
13
|
+
* the limiter.
|
|
14
|
+
*
|
|
15
|
+
* 2. `activeSessionCount(...)` — a CROSS-PROCESS view of how many sessions
|
|
16
|
+
* are currently running, derived from the durable state files:
|
|
17
|
+
* - `.orchestrator-state.json` groups with status `spawned` | `running`;
|
|
18
|
+
* - `.watcher-state.json` entries with status `spawned` (in-flight
|
|
19
|
+
* write-ahead spawns).
|
|
20
|
+
* Each process (orchestrate, watcher, a cron watcher, a manual run) reads
|
|
21
|
+
* these files before spawning, so the cap holds even when several
|
|
22
|
+
* processes run concurrently against the same repo — the in-process
|
|
23
|
+
* limiter covers the window INSIDE one process (e.g. a sweep spawning
|
|
24
|
+
* several batches back-to-back), the state files cover the window ACROSS
|
|
25
|
+
* processes.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
import * as path from "node:path"
|
|
29
|
+
|
|
30
|
+
import { loadStateSync, type OrchestratorState } from "../orchestrator/state.js"
|
|
31
|
+
import { loadWatcherStateSync, type WatcherState } from "../watcher/state.js"
|
|
32
|
+
|
|
33
|
+
/** Default global cap: HEADLESSCODE_MAX_CONCURRENT_SESSIONS or 3. */
|
|
34
|
+
export const DEFAULT_MAX_CONCURRENT_SESSIONS = 3
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* In-process counting semaphore with fail-fast acquisition. The cap counts
|
|
38
|
+
* SESSIONS (one worktree group = one harness session), not worktrees.
|
|
39
|
+
*/
|
|
40
|
+
export class ConcurrencyLimiter {
|
|
41
|
+
readonly maxSessions: number
|
|
42
|
+
private active = 0
|
|
43
|
+
|
|
44
|
+
constructor(maxSessions: number) {
|
|
45
|
+
if (!Number.isInteger(maxSessions) || maxSessions <= 0) {
|
|
46
|
+
throw new Error(`ConcurrencyLimiter: maxSessions must be a positive integer (got ${maxSessions})`)
|
|
47
|
+
}
|
|
48
|
+
this.maxSessions = maxSessions
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Try to reserve a slot. Fail-fast: returns { ok: false, reason } instead
|
|
53
|
+
* of queueing — a caller that cannot spawn right now must either defer
|
|
54
|
+
* (watcher → durable pending) or abort loudly (orchestrate), never wait
|
|
55
|
+
* silently. (Decision documented in docs/phase6-cloud.md.)
|
|
56
|
+
*/
|
|
57
|
+
acquire(): { ok: boolean; reason?: string } {
|
|
58
|
+
if (this.active >= this.maxSessions) {
|
|
59
|
+
return {
|
|
60
|
+
ok: false,
|
|
61
|
+
reason: `concurrency cap reached: ${this.maxSessions} active session(s) (HEADLESSCODE_MAX_CONCURRENT_SESSIONS)`,
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
this.active++
|
|
65
|
+
return { ok: true }
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/** Release a slot previously reserved with acquire(). Never goes below 0. */
|
|
69
|
+
release(): void {
|
|
70
|
+
this.active = Math.max(0, this.active - 1)
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
/** Current number of reserved (active) slots in this process. */
|
|
74
|
+
current(): number {
|
|
75
|
+
return this.active
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
/** Parsed env cap: HEADLESSCODE_MAX_CONCURRENT_SESSIONS (default 3). */
|
|
80
|
+
export function maxConcurrentSessionsFromEnv(env: NodeJS.ProcessEnv = process.env): number {
|
|
81
|
+
const raw = env.HEADLESSCODE_MAX_CONCURRENT_SESSIONS
|
|
82
|
+
if (raw === undefined || raw === "") {
|
|
83
|
+
return DEFAULT_MAX_CONCURRENT_SESSIONS
|
|
84
|
+
}
|
|
85
|
+
const n = Number(raw)
|
|
86
|
+
return Number.isInteger(n) && n > 0 ? n : DEFAULT_MAX_CONCURRENT_SESSIONS
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Cross-process active-session count: orchestrator groups in `spawned`/
|
|
91
|
+
* `running` + watcher entries in `spawned` (write-ahead, spawn in flight).
|
|
92
|
+
* Pass either state as null/undefined when that pipeline isn't in use.
|
|
93
|
+
*/
|
|
94
|
+
export function activeSessionCount(
|
|
95
|
+
orchState?: OrchestratorState | null,
|
|
96
|
+
watcherState?: WatcherState | null,
|
|
97
|
+
): number {
|
|
98
|
+
let count = 0
|
|
99
|
+
if (orchState) {
|
|
100
|
+
for (const group of orchState.groups) {
|
|
101
|
+
if (group.status === "spawned" || group.status === "running") {
|
|
102
|
+
count++
|
|
103
|
+
}
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
if (watcherState) {
|
|
107
|
+
for (const entry of Object.values(watcherState.processed)) {
|
|
108
|
+
if (entry.status === "spawned") {
|
|
109
|
+
count++
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
return count
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
/**
|
|
117
|
+
* Convenience: load BOTH durable state files for a repo root and return the
|
|
118
|
+
* combined active count. Missing files are tolerated (fresh default state).
|
|
119
|
+
* Paths follow the repo conventions: `<repoRoot>/.worktrees/`.
|
|
120
|
+
*/
|
|
121
|
+
export function activeSessionCountForRepo(repoRoot: string): number {
|
|
122
|
+
const worktrees = path.join(path.resolve(repoRoot), ".worktrees")
|
|
123
|
+
const orchState = loadStateSync(path.join(worktrees, ".orchestrator-state.json"))
|
|
124
|
+
const watcherState = loadWatcherStateSync(path.join(worktrees, ".watcher-state.json"))
|
|
125
|
+
return activeSessionCount(orchState, watcherState)
|
|
126
|
+
}
|
|
@@ -0,0 +1,309 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Phase 6 — model pricing + LLM cost estimation (the numeric heart of the
|
|
3
|
+
* per-session cost budget).
|
|
4
|
+
*
|
|
5
|
+
* The harness pays OpenRouter per token (input + output). To enforce a
|
|
6
|
+
* per-session cost cap without post-hoc reconciliation we estimate the USD
|
|
7
|
+
* cost of every LLM call from the `usage` object the provider returns
|
|
8
|
+
* (prompt/completion tokens) × the model's price.
|
|
9
|
+
*
|
|
10
|
+
* Pricing table: USD per 1M tokens, keyed by model id. Defaults cover the
|
|
11
|
+
* models this project actually uses (DeepSeek family via OpenRouter) plus one
|
|
12
|
+
* general model as a sanity reference. The table is overridable:
|
|
13
|
+
*
|
|
14
|
+
* - in-code: pass a `PricingTable` to `estimateCost` / `accumulateCost` /
|
|
15
|
+
* `BudgetTracker` (the constructor option);
|
|
16
|
+
* - via env: `HEADLESSCODE_PRICING_JSON` = path to a JSON file of the same
|
|
17
|
+
* shape (`{ "<model>": { "input": <per1M USD>, "output": <per1M USD> } }`),
|
|
18
|
+
* deep-merged over the defaults (per-model override; unlisted models keep
|
|
19
|
+
* the built-in price).
|
|
20
|
+
*
|
|
21
|
+
* Unknown models fall back to a CONSERVATIVE general rate (`FALLBACK_MODEL_PRICE`)
|
|
22
|
+
* rather than $0 — a runaway session on a model we don't have a price for must
|
|
23
|
+
* still be capped, not silently free.
|
|
24
|
+
*
|
|
25
|
+
* Prices below are the list rates as of the 2026-08 baseline; they are input
|
|
26
|
+
* to a GUARDRAIL (cap enforcement), so being slightly stale is safe (the cap
|
|
27
|
+
* direction — "did we cross the budget" — is what matters).
|
|
28
|
+
*/
|
|
29
|
+
|
|
30
|
+
import * as fs from "node:fs"
|
|
31
|
+
import * as path from "node:path"
|
|
32
|
+
|
|
33
|
+
/** USD per 1M tokens for one model. */
|
|
34
|
+
export interface ModelPrice {
|
|
35
|
+
/** Price per 1M INPUT (prompt) tokens, USD. */
|
|
36
|
+
input: number
|
|
37
|
+
/** Price per 1M OUTPUT (completion) tokens, USD. */
|
|
38
|
+
output: number
|
|
39
|
+
/**
|
|
40
|
+
* Price per 1M CACHED input tokens, USD (a provider-side prompt-cache hit —
|
|
41
|
+
* see LlmResponse.usage.cachedTokens). Optional and deliberately NOT
|
|
42
|
+
* defaulted to a discounted guess: when absent, cached tokens are priced
|
|
43
|
+
* at the full `input` rate, i.e. zero behavior change from before cache
|
|
44
|
+
* accounting existed. Only set this once you've confirmed the model's
|
|
45
|
+
* real cache-hit rate (e.g. from OpenRouter/provider docs) — this is a
|
|
46
|
+
* cost GUARDRAIL, so a wrong optimistic discount could let real spend
|
|
47
|
+
* exceed a configured cap.
|
|
48
|
+
*/
|
|
49
|
+
cacheRead?: number
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** Model id → price. */
|
|
53
|
+
export type PricingTable = Record<string, ModelPrice>
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* Default pricing table (OpenRouter list rates, USD per 1M tokens).
|
|
57
|
+
*
|
|
58
|
+
* - `deepseek/deepseek-v4-flash-0731` — the harness default (Phase 1/2/5
|
|
59
|
+
* workers; see `DEFAULT_MODEL` in src/llm/openrouter.ts). Priced the same
|
|
60
|
+
* as the prior default, `deepseek/deepseek-v4-flash` (kept below for
|
|
61
|
+
* override compat) — same model line, dated point-release id. The rate is
|
|
62
|
+
* the OFFICIAL DeepSeek provider's own rate specifically (verified
|
|
63
|
+
* against OpenRouter's own
|
|
64
|
+
* `/api/v1/models/deepseek/deepseek-v4-flash/endpoints` on 2026-08-01,
|
|
65
|
+
* cross-checked against OpenRouter's own pricing UI). This price is only
|
|
66
|
+
* actually correct because `src/llm/openrouter.ts` pins routing to this
|
|
67
|
+
* exact provider (`provider: { order: ["deepseek"], allow_fallbacks:
|
|
68
|
+
* false }`) for deepseek/* models — OTHER routed endpoints behind the
|
|
69
|
+
* same model id (DeepInfra, Baidu, Mancer, ...) have meaningfully
|
|
70
|
+
* different prices, especially for cache reads, so this number would be
|
|
71
|
+
* wrong/unverifiable without that pin. If the pin is ever removed, this
|
|
72
|
+
* price must be revisited, not left as a stale guess.
|
|
73
|
+
* Missing entirely from this table before was the actual root cause of a
|
|
74
|
+
* real incident: sessions silently fell through to `FALLBACK_MODEL_PRICE`
|
|
75
|
+
* (7x+ this model's real input rate), producing wildly inflated internal
|
|
76
|
+
* cost estimates and at least one false-positive budget abort. A second,
|
|
77
|
+
* smaller error (`cacheRead` off by 10x — $0.028 instead of the real
|
|
78
|
+
* $0.0028/1M) was caught and fixed the same day.
|
|
79
|
+
* - `deepseek/deepseek-v4-flash` — the previous default, kept priced so an
|
|
80
|
+
* explicit override to the un-dated id still estimates correctly.
|
|
81
|
+
* - `deepseek/deepseek-chat` — the legacy default, still priced so an
|
|
82
|
+
* explicit `OPENROUTER_MODEL=deepseek/deepseek-chat` override accounts
|
|
83
|
+
* correctly.
|
|
84
|
+
* - `deepseek/deepseek-reasoner` — the reasoning variant used for review/QA.
|
|
85
|
+
* - `anthropic/claude-3.5-sonnet` — a general-purpose reference model.
|
|
86
|
+
* - `qwen/qwen3-embedding-4b` / `qwen/qwen3-embedding-8b` — the codebase-index
|
|
87
|
+
* embedding models (src/codesearch/embedder.ts; 8b is the default since
|
|
88
|
+
* the central-store round, 4b kept priced so a legacy index/override still
|
|
89
|
+
* estimates correctly). Embedding responses carry no completion tokens
|
|
90
|
+
* (output is always 0). The rates are CONSERVATIVE input-only estimates:
|
|
91
|
+
* $0.15/1M for 4b was observed live on OpenRouter's Google embedding
|
|
92
|
+
* endpoint (2026-08-01), and 8b is estimated at $0.30/1M (2x the 4b rate,
|
|
93
|
+
* parameter-scaled) because embedding models are absent from OpenRouter's
|
|
94
|
+
* /api/v1/models listing so neither list price is API-verifiable (see
|
|
95
|
+
* src/codesearch/embedder.ts's header). Pricing a guardrail means
|
|
96
|
+
* fail-closed: estimate rather than $0, so an index build on a large repo
|
|
97
|
+
* still counts against a cost cap.
|
|
98
|
+
*/
|
|
99
|
+
export const DEFAULT_PRICING_TABLE: PricingTable = {
|
|
100
|
+
"deepseek/deepseek-chat": { input: 0.27, output: 1.1 },
|
|
101
|
+
"deepseek/deepseek-reasoner": { input: 0.55, output: 2.19 },
|
|
102
|
+
"deepseek/deepseek-v4-flash-0731": { input: 0.14, output: 0.28, cacheRead: 0.0028 },
|
|
103
|
+
"deepseek/deepseek-v4-flash": { input: 0.14, output: 0.28, cacheRead: 0.0028 },
|
|
104
|
+
"anthropic/claude-3.5-sonnet": { input: 3.0, output: 15.0 },
|
|
105
|
+
"qwen/qwen3-embedding-4b": { input: 0.15, output: 0.15 },
|
|
106
|
+
"qwen/qwen3-embedding-8b": { input: 0.3, output: 0.3 },
|
|
107
|
+
// Cloud vision captioning (src/vision/describe.ts). Live OpenRouter list
|
|
108
|
+
// rates pulled 2026-08-02 for image-input models; the 12b is the default
|
|
109
|
+
// (cheapest candidate that wasn't materially worse than the quality anchor
|
|
110
|
+
// in the image-support evaluation — see plans/image-support.md), and the
|
|
111
|
+
// 27b is priced too so an override to the quality anchor still estimates
|
|
112
|
+
// accurately instead of falling through to FALLBACK_MODEL_PRICE. Both were
|
|
113
|
+
// verified live against the API during the evaluation.
|
|
114
|
+
"google/gemma-3-12b-it": { input: 0.05, output: 0.15 },
|
|
115
|
+
"google/gemma-3-27b-it": { input: 0.08, output: 0.45 },
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Conservative fallback for models missing from the table (USD per 1M).
|
|
120
|
+
* Deliberately a mid-range general rate, NOT $0 — unknown models must still
|
|
121
|
+
* count against the budget (fail-closed: over-estimate slightly rather than
|
|
122
|
+
* underestimate, so a cost cap is never bypassed by an unlisted model id).
|
|
123
|
+
*/
|
|
124
|
+
export const FALLBACK_MODEL_PRICE: ModelPrice = { input: 2.0, output: 8.0 }
|
|
125
|
+
|
|
126
|
+
/** One endpoint entry of OpenRouter's `/api/v1/models/<id>/endpoints` response. */
|
|
127
|
+
export interface EndpointPricingEntry {
|
|
128
|
+
/** Provider slug as OpenRouter reports it (e.g. "DeepSeek", "DeepInfra"). */
|
|
129
|
+
provider_name?: unknown
|
|
130
|
+
/** Per-token prices as STRINGS, USD per token (verified live 2026-08-04). */
|
|
131
|
+
pricing?: {
|
|
132
|
+
prompt?: unknown
|
|
133
|
+
completion?: unknown
|
|
134
|
+
input_cache_read?: unknown
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/** Parse a per-token USD string like "0.00000014" into per-1M-token USD. */
|
|
139
|
+
function perTokenToPerMillion(value: unknown): number | undefined {
|
|
140
|
+
if (typeof value !== "string" && typeof value !== "number") {
|
|
141
|
+
return undefined
|
|
142
|
+
}
|
|
143
|
+
const n = typeof value === "string" ? Number(value) : value
|
|
144
|
+
if (!Number.isFinite(n) || n < 0) {
|
|
145
|
+
return undefined
|
|
146
|
+
}
|
|
147
|
+
return n * 1_000_000
|
|
148
|
+
}
|
|
149
|
+
|
|
150
|
+
/**
|
|
151
|
+
* Parse live endpoint pricing into a `ModelPrice` (USD per 1M tokens).
|
|
152
|
+
*
|
|
153
|
+
* Selection rules, mirroring how the request actually routes:
|
|
154
|
+
* - When `providerPreference` is given (e.g. "DeepSeek" for the deepseek/*
|
|
155
|
+
* chat pin), the FIRST endpoint whose provider_name matches (case-
|
|
156
|
+
* insensitive) is used verbatim — that is the price actually charged.
|
|
157
|
+
* - Otherwise the MAX of each field across all endpoints that expose it is
|
|
158
|
+
* used. This is a deliberate fail-closed choice: for auto-routed models the
|
|
159
|
+
* harness cannot know which endpoint served a call, and this table is a
|
|
160
|
+
* GUARDRAIL (the codebase's FALLBACK_MODEL_PRICE rationale) — over-
|
|
161
|
+
* estimating is safe, under-estimating lets a cap be bypassed.
|
|
162
|
+
*
|
|
163
|
+
* Returns undefined when no endpoint exposes usable pricing.
|
|
164
|
+
*/
|
|
165
|
+
export function parseEndpointPricing(
|
|
166
|
+
endpoints: EndpointPricingEntry[],
|
|
167
|
+
providerPreference?: string,
|
|
168
|
+
): ModelPrice | undefined {
|
|
169
|
+
let price: ModelPrice | undefined
|
|
170
|
+
for (const ep of endpoints) {
|
|
171
|
+
const pricing = ep?.pricing
|
|
172
|
+
if (!pricing || typeof pricing !== "object") {
|
|
173
|
+
continue
|
|
174
|
+
}
|
|
175
|
+
const parsed: ModelPrice = {
|
|
176
|
+
input: perTokenToPerMillion(pricing.prompt) ?? 0,
|
|
177
|
+
output: perTokenToPerMillion(pricing.completion) ?? 0,
|
|
178
|
+
}
|
|
179
|
+
if (parsed.input === 0 && parsed.output === 0) {
|
|
180
|
+
continue
|
|
181
|
+
}
|
|
182
|
+
const cacheRead = perTokenToPerMillion(pricing.input_cache_read)
|
|
183
|
+
if (cacheRead !== undefined) {
|
|
184
|
+
parsed.cacheRead = cacheRead
|
|
185
|
+
}
|
|
186
|
+
// Provider-preference match wins outright (first match — a match
|
|
187
|
+
// returns immediately, so reaching the end without returning means no
|
|
188
|
+
// usable match existed).
|
|
189
|
+
if (providerPreference) {
|
|
190
|
+
const name = typeof ep?.provider_name === "string" ? ep.provider_name : ""
|
|
191
|
+
if (name.toLowerCase() === providerPreference.toLowerCase()) {
|
|
192
|
+
return parsed
|
|
193
|
+
}
|
|
194
|
+
}
|
|
195
|
+
// Otherwise keep the conservative per-field max.
|
|
196
|
+
price = {
|
|
197
|
+
input: Math.max(price?.input ?? 0, parsed.input),
|
|
198
|
+
output: Math.max(price?.output ?? 0, parsed.output),
|
|
199
|
+
...(cacheRead !== undefined
|
|
200
|
+
? { cacheRead: Math.max(price?.cacheRead ?? 0, cacheRead) }
|
|
201
|
+
: price?.cacheRead !== undefined
|
|
202
|
+
? { cacheRead: price.cacheRead }
|
|
203
|
+
: {}),
|
|
204
|
+
}
|
|
205
|
+
}
|
|
206
|
+
// A preference was requested but no endpoint matched it — return nothing
|
|
207
|
+
// so the caller falls back to the hardcoded table (for deepseek/* that
|
|
208
|
+
// table IS the official endpoint's rate; using some other host's numbers
|
|
209
|
+
// for a pinned model would be wrong).
|
|
210
|
+
if (providerPreference) {
|
|
211
|
+
return undefined
|
|
212
|
+
}
|
|
213
|
+
return price
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
/** Merge a live-resolved price into a pricing table under a model id. */
|
|
217
|
+
export function mergeLivePrice(table: PricingTable, model: string, price: ModelPrice): PricingTable {
|
|
218
|
+
return { ...table, [model]: price }
|
|
219
|
+
}
|
|
220
|
+
|
|
221
|
+
/**
|
|
222
|
+
* Load the effective pricing table: defaults deep-merged with overrides from
|
|
223
|
+
* `HEADLESSCODE_PRICING_JSON` (when set). A missing/unreadable/malformed file
|
|
224
|
+
* throws — a broken pricing override must fail loudly, never silently weaken
|
|
225
|
+
* a cost cap.
|
|
226
|
+
*/
|
|
227
|
+
export function loadPricingTable(
|
|
228
|
+
env: NodeJS.ProcessEnv = process.env,
|
|
229
|
+
base: PricingTable = DEFAULT_PRICING_TABLE,
|
|
230
|
+
): PricingTable {
|
|
231
|
+
const pricingPath = env.HEADLESSCODE_PRICING_JSON
|
|
232
|
+
if (!pricingPath) {
|
|
233
|
+
return base
|
|
234
|
+
}
|
|
235
|
+
const file = path.resolve(pricingPath)
|
|
236
|
+
let raw: string
|
|
237
|
+
try {
|
|
238
|
+
raw = fs.readFileSync(file, "utf-8")
|
|
239
|
+
} catch (err) {
|
|
240
|
+
throw new Error(
|
|
241
|
+
`HEADLESSCODE_PRICING_JSON: cannot read pricing file '${file}': ${err instanceof Error ? err.message : String(err)}`,
|
|
242
|
+
)
|
|
243
|
+
}
|
|
244
|
+
let parsed: unknown
|
|
245
|
+
try {
|
|
246
|
+
parsed = JSON.parse(raw)
|
|
247
|
+
} catch (err) {
|
|
248
|
+
throw new Error(
|
|
249
|
+
`HEADLESSCODE_PRICING_JSON: invalid JSON in '${file}': ${err instanceof Error ? err.message : String(err)}`,
|
|
250
|
+
)
|
|
251
|
+
}
|
|
252
|
+
if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) {
|
|
253
|
+
throw new Error(`HEADLESSCODE_PRICING_JSON: '${file}' must be a JSON object of {model: {input, output}}`)
|
|
254
|
+
}
|
|
255
|
+
const overrides: PricingTable = {}
|
|
256
|
+
for (const [model, value] of Object.entries(parsed)) {
|
|
257
|
+
const v = value as { input?: unknown; output?: unknown }
|
|
258
|
+
if (typeof v.input === "number" && Number.isFinite(v.input) && v.input >= 0 && typeof v.output === "number" && Number.isFinite(v.output) && v.output >= 0) {
|
|
259
|
+
overrides[model] = { input: v.input, output: v.output }
|
|
260
|
+
}
|
|
261
|
+
}
|
|
262
|
+
return { ...base, ...overrides }
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
/** Resolve the price for a model (fallback when unlisted). */
|
|
266
|
+
export function priceFor(model: string, pricing: PricingTable = DEFAULT_PRICING_TABLE): ModelPrice {
|
|
267
|
+
return pricing[model] ?? FALLBACK_MODEL_PRICE
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
export interface CostInput {
|
|
271
|
+
model: string
|
|
272
|
+
inputTokens?: number
|
|
273
|
+
outputTokens?: number
|
|
274
|
+
/**
|
|
275
|
+
* Prompt tokens served from the provider's cache (a SUBSET of
|
|
276
|
+
* inputTokens, not additional — see LlmResponse.usage.cachedTokens).
|
|
277
|
+
* Priced at price.cacheRead when set, else at the full input rate.
|
|
278
|
+
*/
|
|
279
|
+
cachedTokens?: number
|
|
280
|
+
/** Explicit pricing table (default: env-merged defaults). */
|
|
281
|
+
pricing?: PricingTable
|
|
282
|
+
}
|
|
283
|
+
|
|
284
|
+
/**
|
|
285
|
+
* Estimate the USD cost of one LLM call from its usage token counts.
|
|
286
|
+
* cost = (inputTokens - cachedTokens) × price.input / 1e6
|
|
287
|
+
* + cachedTokens × (price.cacheRead ?? price.input) / 1e6
|
|
288
|
+
* + outputTokens × price.output / 1e6
|
|
289
|
+
* `cachedTokens` is clamped to `inputTokens` (a provider reporting more
|
|
290
|
+
* cached than total prompt tokens would otherwise produce a negative
|
|
291
|
+
* "uncached" count).
|
|
292
|
+
*/
|
|
293
|
+
export function estimateCost(input: CostInput): number {
|
|
294
|
+
const { model, inputTokens = 0, outputTokens = 0 } = input
|
|
295
|
+
const table = input.pricing ?? loadPricingTable()
|
|
296
|
+
const price = priceFor(model, table)
|
|
297
|
+
const cachedTokens = Math.min(Math.max(input.cachedTokens ?? 0, 0), inputTokens)
|
|
298
|
+
const uncachedTokens = inputTokens - cachedTokens
|
|
299
|
+
const cacheRate = price.cacheRead ?? price.input
|
|
300
|
+
return (uncachedTokens * price.input + cachedTokens * cacheRate + outputTokens * price.output) / 1_000_000
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
/** Sum the estimated cost across multiple calls (runs) — the session total. */
|
|
304
|
+
export function accumulateCost(
|
|
305
|
+
runs: Array<{ model: string; inputTokens?: number; outputTokens?: number; cachedTokens?: number }>,
|
|
306
|
+
pricing?: PricingTable,
|
|
307
|
+
): number {
|
|
308
|
+
return runs.reduce((sum, run) => sum + estimateCost({ ...run, pricing }), 0)
|
|
309
|
+
}
|