@bitkyc08/opencodex 2.55.0 → 2.57.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/ocx.mjs +10 -0
- package/gui/dist/assets/{index-BBOZWGB6.css → index-C5-RdDmD.css} +1 -1
- package/gui/dist/assets/{index-VuoiWj9J.js → index-Cz7CLdif.js} +21 -21
- package/gui/dist/index.html +2 -2
- package/package.json +4 -3
- package/src/adapters/base.ts +21 -0
- package/src/adapters/codebuddy/adapter.ts +2 -1
- package/src/adapters/codebuddy/scaffold-guard.ts +248 -0
- package/src/adapters/command-code.ts +1 -1
- package/src/adapters/cursor/envelope-echo.ts +8 -2
- package/src/adapters/cursor/transport-retry.ts +46 -1
- package/src/adapters/cursor.ts +4 -0
- package/src/adapters/google.ts +7 -7
- package/src/adapters/kiro/adapter.ts +42 -1
- package/src/adapters/kiro/payload.ts +17 -3
- package/src/adapters/kiro/reasoning.ts +70 -7
- package/src/adapters/kiro/stream.ts +8 -2
- package/src/adapters/kiro/wire.ts +2 -1
- package/src/adapters/kiro-events.ts +21 -13
- package/src/adapters/kiro-retry.ts +23 -4
- package/src/adapters/openai-chat/errors.ts +116 -0
- package/src/adapters/openai-chat/messages.ts +346 -0
- package/src/adapters/openai-chat/passthrough.ts +146 -0
- package/src/adapters/openai-chat/response-events.ts +117 -0
- package/src/adapters/openai-chat/tool-call-validation.ts +200 -0
- package/src/adapters/openai-chat/tool-name-registry.ts +166 -0
- package/src/adapters/openai-chat/tool-schema.ts +495 -0
- package/src/adapters/openai-chat/wire.ts +50 -0
- package/src/adapters/openai-chat.ts +40 -1452
- package/src/adapters/openai-responses/canonical-forward.ts +202 -0
- package/src/adapters/openai-responses/image-gen.ts +406 -0
- package/src/adapters/openai-responses/internal.ts +3 -0
- package/src/adapters/openai-responses/passthrough.ts +642 -0
- package/src/adapters/openai-responses/prompt-cache.ts +83 -0
- package/src/adapters/openai-responses/reasoning.ts +220 -0
- package/src/adapters/openai-responses/request-strips.ts +185 -0
- package/src/adapters/openai-responses/tool-output-recovery.ts +509 -0
- package/src/adapters/openai-responses/tool-schema.ts +293 -0
- package/src/adapters/openai-responses/web-search.ts +156 -0
- package/src/adapters/openai-responses.ts +4 -2625
- package/src/bridge/errors.ts +58 -0
- package/src/bridge/internal.ts +174 -0
- package/src/bridge/response-json.ts +630 -0
- package/src/bridge/sse.ts +1462 -0
- package/src/bridge.ts +5 -2204
- package/src/chat/inbound.ts +12 -1
- package/src/claude/desktop-profile.ts +66 -9
- package/src/claude/outbound.ts +18 -0
- package/src/cli/account-main.ts +1 -1
- package/src/cli/capabilities.ts +2 -2
- package/src/cli/combo.ts +10 -1
- package/src/cli/index.ts +48 -5
- package/src/cli/registry.ts +2 -1
- package/src/cli/system-command.ts +4 -4
- package/src/clients/config-export.ts +7 -3
- package/src/codex/account-label.ts +14 -3
- package/src/codex/account-lifecycle.ts +3 -0
- package/src/codex/account-store.ts +184 -35
- package/src/codex/account-usability.ts +21 -0
- package/src/codex/auth-api/account-list.ts +507 -0
- package/src/codex/auth-api/http.ts +32 -0
- package/src/codex/auth-api/login-flow.ts +566 -0
- package/src/codex/auth-api/login-state.ts +64 -0
- package/src/codex/auth-api/main-account-probe.ts +331 -0
- package/src/codex/auth-api/pool-mode-gate.ts +274 -0
- package/src/codex/auth-api/pool-quota-probe.ts +512 -0
- package/src/codex/auth-api/reset-credit-service.ts +431 -0
- package/src/codex/auth-api/routes.ts +425 -0
- package/src/codex/auth-api/runtime-config.ts +48 -0
- package/src/codex/auth-api.ts +27 -3118
- package/src/codex/auth-context.ts +252 -35
- package/src/codex/catalog/aggregation.ts +80 -1
- package/src/codex/catalog/auto-review.ts +507 -0
- package/src/codex/catalog/build-entries.ts +981 -0
- package/src/codex/catalog/combo-member.ts +375 -0
- package/src/codex/catalog/derive-entry.ts +229 -0
- package/src/codex/catalog/effort.ts +0 -1
- package/src/codex/catalog/gated-native-warn.ts +63 -0
- package/src/codex/catalog/gather-capture.ts +533 -0
- package/src/codex/catalog/model-hints.ts +691 -0
- package/src/codex/catalog/model-visibility.ts +305 -0
- package/src/codex/catalog/provider-fetch.ts +52 -2942
- package/src/codex/catalog/provider-models.ts +685 -0
- package/src/codex/catalog/remote.ts +30 -0
- package/src/codex/catalog/restore.ts +132 -0
- package/src/codex/catalog/retained-sync.ts +714 -0
- package/src/codex/catalog/routed-gather.ts +895 -0
- package/src/codex/catalog/subagent-roster.ts +176 -0
- package/src/codex/catalog/sync.ts +52 -2698
- package/src/codex/cli-install-provenance.ts +7 -1
- package/src/codex/convergence.ts +7 -2
- package/src/codex/desktop-app/types.ts +11 -2
- package/src/codex/desktop-app/windows.ts +5 -5
- package/src/codex/inject/config-toml.ts +563 -0
- package/src/codex/inject/remove.ts +192 -0
- package/src/codex/inject/restore.ts +567 -0
- package/src/codex/inject/routing-classify.ts +109 -0
- package/src/codex/inject/routing-target.ts +125 -0
- package/src/codex/inject.ts +89 -1444
- package/src/codex/lineage.ts +458 -0
- package/src/codex/model-entitlements.ts +152 -15
- package/src/codex/pool-refresh-backoff.ts +161 -0
- package/src/codex/quota-rejection.ts +104 -15
- package/src/codex/routing/active-account.ts +194 -0
- package/src/codex/routing/cache-affinity.ts +70 -0
- package/src/codex/routing/cooldown-math.ts +285 -0
- package/src/codex/routing/health-store.ts +402 -0
- package/src/codex/routing/probe-lease.ts +358 -0
- package/src/codex/routing/selection.ts +780 -0
- package/src/codex/routing/thread-affinity.ts +586 -0
- package/src/codex/routing/transient-hold-dispatch.ts +141 -0
- package/src/codex/routing.ts +370 -2271
- package/src/codex/shim-fingerprint.ts +223 -0
- package/src/codex/shim-inspect.ts +175 -0
- package/src/codex/shim-probe.ts +367 -0
- package/src/codex/shim-restore-lock.ts +169 -0
- package/src/codex/shim-state-file.ts +151 -0
- package/src/codex/shim-templates.ts +265 -0
- package/src/codex/shim.ts +48 -1268
- package/src/codex/warmup.ts +1 -1
- package/src/combos/failover.ts +85 -0
- package/src/combos/request.ts +17 -10
- package/src/combos/types.ts +23 -2
- package/src/config/diagnostics.ts +705 -0
- package/src/config/feature-flags.ts +55 -0
- package/src/config/live-reconcile.ts +403 -0
- package/src/config/load-degrade.ts +880 -0
- package/src/config/mutation-lock.ts +244 -0
- package/src/config/openai-tier-backup.ts +268 -0
- package/src/config/pending-teardown.ts +31 -0
- package/src/config/persist-unlocked.ts +92 -0
- package/src/config/proxy-env.ts +188 -0
- package/src/config/salvage.ts +244 -0
- package/src/config/schema/config-schema.ts +640 -0
- package/src/config/schema/leaf-validators.ts +855 -0
- package/src/config/warn-memo.ts +28 -0
- package/src/config.ts +234 -4481
- package/src/generated/compatibility-version.json +649 -121
- package/src/images/loop.ts +1 -1
- package/src/lib/errors.ts +17 -0
- package/src/lib/request-execution-budget.ts +198 -23
- package/src/lib/spend-reservation-ledger.ts +958 -0
- package/src/lib/state-store-registrations.ts +6 -2
- package/src/lib/test-home-guard.ts +85 -1
- package/src/lib/upstream-retry.ts +132 -21
- package/src/lib/windows-elevation.ts +76 -14
- package/src/lib/workflow-budget.ts +553 -30
- package/src/oauth/index.ts +2 -2
- package/src/oauth/key-providers.ts +2 -2
- package/src/providers/kiro-models.ts +4 -3
- package/src/providers/label.ts +19 -1
- package/src/providers/model-discovery.ts +16 -0
- package/src/providers/quota/account-cache.ts +441 -0
- package/src/providers/quota/antigravity.ts +295 -0
- package/src/providers/quota/report-cache.ts +320 -0
- package/src/providers/quota/vendor-probes-key.ts +1243 -0
- package/src/providers/quota/vendor-probes-oauth.ts +590 -0
- package/src/providers/quota.ts +324 -3079
- package/src/providers/registry/entries-core.ts +1228 -0
- package/src/providers/registry/entries-extended.ts +1213 -0
- package/src/providers/registry/model-seeds.ts +912 -0
- package/src/providers/registry/types.ts +352 -0
- package/src/providers/registry.ts +24 -3536
- package/src/responses/continuation-ownership.ts +29 -0
- package/src/responses/reasoning-envelope.ts +6 -3
- package/src/responses/state/replay-fingerprint.ts +80 -0
- package/src/responses/state/snapshot-codec.ts +104 -0
- package/src/responses/state/spill-failure.ts +118 -0
- package/src/responses/state/spill-queue.ts +665 -0
- package/src/responses/state/temp-recovery.ts +257 -0
- package/src/responses/state.ts +82 -1143
- package/src/routing/identity-domains.ts +456 -0
- package/src/routing/probe-lease.ts +613 -0
- package/src/server/chat-completions.ts +3 -1
- package/src/server/chat-native.ts +37 -9
- package/src/server/index/bounded-request.ts +88 -0
- package/src/server/index/live-sideband.ts +601 -0
- package/src/server/index/serve-options.ts +1766 -0
- package/src/server/index/startup-warnings.ts +213 -0
- package/src/server/index/websocket-handler.ts +339 -0
- package/src/server/index.ts +45 -2552
- package/src/server/inspection-tee.ts +107 -0
- package/src/server/live.ts +46 -1
- package/src/server/management/combo-routes.ts +10 -1
- package/src/server/management/route-registry.ts +26 -23
- package/src/server/management/shared.ts +8 -5
- package/src/server/management/workflow-budget-routes.ts +133 -0
- package/src/server/management-api.ts +12 -0
- package/src/server/relay-eager.ts +2 -0
- package/src/server/relay.ts +14 -19
- package/src/server/request-log-conversation.ts +9 -7
- package/src/server/request-log.ts +372 -4
- package/src/server/response-log-body.ts +153 -0
- package/src/server/responses/account-change-state.ts +307 -0
- package/src/server/responses/adapter-continuation.ts +540 -0
- package/src/server/responses/adapter-delivery.ts +208 -0
- package/src/server/responses/adapter-dispatch.ts +1042 -0
- package/src/server/responses/codex-ws-wire.ts +5 -0
- package/src/server/responses/collaboration.ts +74 -4
- package/src/server/responses/combo-session-recall.ts +68 -8
- package/src/server/responses/compact.ts +113 -17
- package/src/server/responses/completion-policy.ts +33 -0
- package/src/server/responses/core-auth.ts +529 -0
- package/src/server/responses/core-codex-account.ts +907 -0
- package/src/server/responses/core-combo-failure.ts +210 -0
- package/src/server/responses/core-combo.ts +787 -0
- package/src/server/responses/core-errors.ts +170 -0
- package/src/server/responses/core-lifetime.ts +95 -0
- package/src/server/responses/core-normalize.ts +350 -0
- package/src/server/responses/core-opaque-recovery.ts +380 -0
- package/src/server/responses/core-options.ts +159 -0
- package/src/server/responses/core-replay.ts +298 -0
- package/src/server/responses/core.ts +192 -8893
- package/src/server/responses/encrypted-payload.ts +0 -1
- package/src/server/responses/input-admission.ts +126 -6
- package/src/server/responses/passthrough-delivery.ts +869 -0
- package/src/server/responses/passthrough-dispatch.ts +1494 -0
- package/src/server/responses/passthrough-error.ts +38 -2
- package/src/server/responses/passthrough-execution.ts +54 -0
- package/src/server/responses/request-prepare.ts +1080 -0
- package/src/server/responses/request-send-budget.ts +259 -0
- package/src/server/responses/request-sidecar-auth.ts +149 -0
- package/src/server/responses/request-spend.ts +147 -0
- package/src/server/responses/request-transport.ts +803 -0
- package/src/server/responses/response-effects.ts +157 -0
- package/src/server/responses/run-turn-execution.ts +476 -0
- package/src/server/responses/sidecar-execution.ts +463 -0
- package/src/server/responses/terminal-guard.ts +65 -4
- package/src/server/responses-image-gen-repair.ts +1 -1
- package/src/server/responses-undeclared-tool-guard.ts +9 -5
- package/src/server/workflow-refusal.ts +84 -0
- package/src/service/windows-ops.ts +210 -16
- package/src/service/windows-scheduler.ts +28 -21
- package/src/service.ts +1 -1
- package/src/types/config.ts +34 -1
- package/src/types/request.ts +8 -5
- package/src/types/tools.ts +24 -0
- package/src/types.ts +2 -0
- package/src/update/index.ts +10 -0
- package/src/update/stop-contract.d.mts +1 -0
- package/src/update/stop-contract.mjs +19 -0
- package/src/update/stop-decision.d.mts +1 -1
- package/src/update/stop-decision.mjs +12 -3
- package/src/usage/log.ts +147 -1
- package/src/usage/summary.ts +171 -21
- package/src/vision/anthropic-describe.ts +1 -1
- package/src/vision/describe.ts +5 -5
- package/src/web-search/anthropic-executor.ts +1 -1
- package/src/web-search/exa-executor.ts +1 -1
- package/src/web-search/executor.ts +1 -1
- package/src/web-search/gemini-executor.ts +1 -1
- package/src/web-search/loop.ts +1 -1
- package/src/web-search/ollama-executor.ts +1 -1
- package/src/web-search/parse.ts +67 -14
- package/src/web-search/passthrough-bridge.ts +64 -31
- package/src/web-search/xai-executor.ts +1 -1
|
@@ -10,70 +10,393 @@
|
|
|
10
10
|
* header when the client supplies one. A retry is not a new user task and gets no new
|
|
11
11
|
* allowance; a genuinely new top-level request does.
|
|
12
12
|
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
13
|
+
* Two caps intersect here. The COUNT caps (concurrency, distinct children, physical sends)
|
|
14
|
+
* are process-local and in-memory. The TOKEN cap is the durable spend-reservation ledger in
|
|
15
|
+
* spend-reservation-ledger.ts: when the caller supplies a spend request, admission also
|
|
16
|
+
* reserves input + enforceable output ceiling against the root, identity and pool scopes,
|
|
17
|
+
* and that accounting survives a restart. The count caps alone remain the guarantee for a
|
|
18
|
+
* second process sharing the pool; the durable ledger's single-process topology is stated
|
|
19
|
+
* in that module's header and applies here unchanged.
|
|
16
20
|
*/
|
|
17
21
|
|
|
22
|
+
import {
|
|
23
|
+
sharedSpendLedger,
|
|
24
|
+
type SpendReservationLedger,
|
|
25
|
+
type SpendScope,
|
|
26
|
+
type SpendUsage,
|
|
27
|
+
} from "./spend-reservation-ledger";
|
|
28
|
+
|
|
18
29
|
export interface WorkflowBudgetPolicy {
|
|
19
30
|
/** Children admitted concurrently under one root. */
|
|
20
31
|
readonly maxConcurrentChildren: number;
|
|
21
|
-
/** Physical model sends charged to one root
|
|
32
|
+
/** Physical model sends charged to one root INSIDE {@link WorkflowBudgetPolicy.windowMs}. */
|
|
22
33
|
readonly maxPhysicalSends: number;
|
|
23
|
-
/** Distinct children one root may
|
|
34
|
+
/** Distinct children one root may have inside the same window. */
|
|
24
35
|
readonly maxDistinctChildren: number;
|
|
36
|
+
/**
|
|
37
|
+
* The interval both counts are measured over.
|
|
38
|
+
*
|
|
39
|
+
* These were lifetime totals, and a lifetime total is the wrong instrument. The cap was
|
|
40
|
+
* written against a fan-out that sends once per child seven hundred times, which is a RATE;
|
|
41
|
+
* a running total cannot tell that from an ordinary session spread over an afternoon and
|
|
42
|
+
* refuses both. Because the root id is the caller thread, for Codex that made the ceiling a
|
|
43
|
+
* session expiry: a session reaching it was refused for the rest of the process even after
|
|
44
|
+
* going idle for hours, and the only cure was restarting the proxy.
|
|
45
|
+
*
|
|
46
|
+
* Omitted means {@link WORKFLOW_DEFAULT_WINDOW_MS}. A count inside a window is never larger
|
|
47
|
+
* than the same count over a lifetime, so windowing can only ever admit more for identical
|
|
48
|
+
* traffic -- no install sees a refusal it would not have seen before.
|
|
49
|
+
*/
|
|
50
|
+
readonly windowMs?: number;
|
|
25
51
|
/**
|
|
26
52
|
* Concurrency slots a fan-out may never take. An interactive turn arriving into a saturated
|
|
27
53
|
* root still gets admitted; without this a worker burst starves the conversation it serves.
|
|
28
54
|
*/
|
|
29
55
|
readonly interactiveReserve: number;
|
|
30
|
-
/**
|
|
56
|
+
/**
|
|
57
|
+
* Roots tracked at once, as a hard bound rather than a hint. At the ceiling one idle,
|
|
58
|
+
* under-limit root is evicted to make room; when no root may be forgotten safely the new
|
|
59
|
+
* root is REFUSED with `workflow-tracking-exhausted`. Admitting it anyway is what made a
|
|
60
|
+
* caller minting new ids able to grow this map past the number written here.
|
|
61
|
+
*/
|
|
31
62
|
readonly maxTrackedRoots: number;
|
|
32
63
|
}
|
|
33
64
|
|
|
65
|
+
/**
|
|
66
|
+
* Ten minutes. Long enough that the burst this ceiling was written against -- seven hundred
|
|
67
|
+
* sends in a minute -- is still refused several times over, and short enough that an ordinary
|
|
68
|
+
* session, which averages far less than a send every two seconds, never approaches it.
|
|
69
|
+
*/
|
|
70
|
+
export const WORKFLOW_DEFAULT_WINDOW_MS = 10 * 60_000;
|
|
71
|
+
|
|
34
72
|
export const DEFAULT_WORKFLOW_BUDGET_POLICY: WorkflowBudgetPolicy = {
|
|
35
73
|
maxConcurrentChildren: 8,
|
|
36
74
|
maxPhysicalSends: 256,
|
|
37
75
|
maxDistinctChildren: 64,
|
|
38
76
|
interactiveReserve: 1,
|
|
39
77
|
maxTrackedRoots: 512,
|
|
78
|
+
windowMs: WORKFLOW_DEFAULT_WINDOW_MS,
|
|
40
79
|
};
|
|
41
80
|
|
|
81
|
+
/** Fixed ring size. Ten minutes over twelve slots gives fifty-second granularity. */
|
|
82
|
+
const WORKFLOW_WINDOW_SLOTS = 12;
|
|
83
|
+
|
|
84
|
+
function workflowWindowMs(policy: WorkflowBudgetPolicy): number {
|
|
85
|
+
const declared = policy.windowMs;
|
|
86
|
+
return declared !== undefined && Number.isFinite(declared) && declared > 0
|
|
87
|
+
? declared
|
|
88
|
+
: WORKFLOW_DEFAULT_WINDOW_MS;
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
/**
|
|
92
|
+
* Slot size for one root's own window.
|
|
93
|
+
*
|
|
94
|
+
* The geometry is read off the state rather than off whatever policy the current caller
|
|
95
|
+
* happens to hold. Two callers may legitimately pass different policies for the same root --
|
|
96
|
+
* the ceiling numbers are the caller's business -- but if they also disagreed about
|
|
97
|
+
* `windowMs`, the slot ids one of them wrote would be on a scale the other cannot read, and
|
|
98
|
+
* charging with a long window while reading with a short one makes every stored slot look
|
|
99
|
+
* ancient and the ceiling never fire at all.
|
|
100
|
+
*/
|
|
101
|
+
function windowSlotMs(windowMs: number): number {
|
|
102
|
+
return Math.max(1, Math.ceil(windowMs / WORKFLOW_WINDOW_SLOTS));
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Add sends to the ring, resetting a slot whose turn has come round again.
|
|
107
|
+
*
|
|
108
|
+
* A ring rather than a list of timestamps because the storage has to be bounded: a root that
|
|
109
|
+
* sends forever would otherwise grow forever, and this ledger exists to bound a fan-out.
|
|
110
|
+
*/
|
|
111
|
+
function recordWindowedSends(state: WorkflowState, now: number, sends: number): void {
|
|
112
|
+
const slotMs = windowSlotMs(state.windowMs);
|
|
113
|
+
const slot = Math.floor(now / slotMs);
|
|
114
|
+
const index = ((slot % WORKFLOW_WINDOW_SLOTS) + WORKFLOW_WINDOW_SLOTS) % WORKFLOW_WINDOW_SLOTS;
|
|
115
|
+
if (state.sendSlotAt[index] !== slot) {
|
|
116
|
+
state.sendSlotAt[index] = slot;
|
|
117
|
+
state.sendSlotCount[index] = 0;
|
|
118
|
+
}
|
|
119
|
+
state.sendSlotCount[index] = (state.sendSlotCount[index] ?? 0) + sends;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
/** Sends inside the window. A slot older than the window contributes nothing. */
|
|
123
|
+
function windowedSends(state: WorkflowState, now: number): number {
|
|
124
|
+
const slotMs = windowSlotMs(state.windowMs);
|
|
125
|
+
const oldest = Math.floor(now / slotMs) - (WORKFLOW_WINDOW_SLOTS - 1);
|
|
126
|
+
let total = 0;
|
|
127
|
+
for (let index = 0; index < WORKFLOW_WINDOW_SLOTS; index += 1) {
|
|
128
|
+
if ((state.sendSlotAt[index] ?? Number.NEGATIVE_INFINITY) >= oldest) {
|
|
129
|
+
total += state.sendSlotCount[index] ?? 0;
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
return total;
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/**
|
|
136
|
+
* Forget children last seen before the window opened, and report how many remain.
|
|
137
|
+
*
|
|
138
|
+
* Pruning on read keeps the map bounded without a timer: every admission pays for the children
|
|
139
|
+
* it can still see, and a root that goes quiet is cleaned up the next time it speaks.
|
|
140
|
+
*/
|
|
141
|
+
function windowedChildren(state: WorkflowState, now: number): number {
|
|
142
|
+
const cutoff = now - state.windowMs;
|
|
143
|
+
for (const [childId, lastSeenMs] of state.children) {
|
|
144
|
+
if (lastSeenMs <= cutoff) state.children.delete(childId);
|
|
145
|
+
}
|
|
146
|
+
return state.children.size;
|
|
147
|
+
}
|
|
148
|
+
|
|
42
149
|
export type WorkflowDenial =
|
|
43
150
|
| "workflow-concurrency-exhausted"
|
|
44
151
|
| "workflow-sends-exhausted"
|
|
45
|
-
| "workflow-children-exhausted"
|
|
152
|
+
| "workflow-children-exhausted"
|
|
153
|
+
| "workflow-spend-exhausted"
|
|
154
|
+
/**
|
|
155
|
+
* The root table is full and every entry is active or exhausted, so admitting this root
|
|
156
|
+
* would mean evicting one whose ceiling has already fired. Refusing is the honest answer:
|
|
157
|
+
* `maxTrackedRoots` is a bound, and inserting anyway made it a suggestion.
|
|
158
|
+
*/
|
|
159
|
+
| "workflow-tracking-exhausted"
|
|
160
|
+
/** This send id was already reserved once; a repeat buys no second dispatch. */
|
|
161
|
+
| "workflow-send-replayed"
|
|
162
|
+
/** The reservation could not be made durable, and a configured ceiling requires it. */
|
|
163
|
+
| "workflow-spend-undurable";
|
|
164
|
+
|
|
165
|
+
/**
|
|
166
|
+
* The sentence an operator reads, plus the machine-readable name of the ceiling that fired.
|
|
167
|
+
*
|
|
168
|
+
* All four count denials used to share one sentence about a "concurrent-work limit", which was
|
|
169
|
+
* accurate for exactly one of them. Worse, the wire cannot carry the distinction on its own:
|
|
170
|
+
* `classifyError` rewrites every 429 to `rate_limit_error` / `rate_limit_exceeded`, so the body
|
|
171
|
+
* of a refusal this proxy made is shaped exactly like a provider rate limit. Each sentence
|
|
172
|
+
* therefore says which ceiling fired AND that no provider was contacted, because that is the
|
|
173
|
+
* first thing an operator needs and the only place left to put it.
|
|
174
|
+
*/
|
|
175
|
+
export function workflowDenialSummary(reason: WorkflowDenial): { code: string; message: string } {
|
|
176
|
+
switch (reason) {
|
|
177
|
+
case "workflow-sends-exhausted":
|
|
178
|
+
return {
|
|
179
|
+
code: "workflow_sends_exhausted",
|
|
180
|
+
message: "This proxy refused the request locally: the task reached its send ceiling for"
|
|
181
|
+
+ " the current window, so no provider was contacted. The window rolls forward on its"
|
|
182
|
+
+ " own; work already in flight settles as it finishes.",
|
|
183
|
+
};
|
|
184
|
+
case "workflow-children-exhausted":
|
|
185
|
+
return {
|
|
186
|
+
code: "workflow_children_exhausted",
|
|
187
|
+
message: "This proxy refused the request locally: the task reached its ceiling on"
|
|
188
|
+
+ " distinct child threads for the current window, so no provider was contacted."
|
|
189
|
+
+ " A child that goes quiet ages out of the count.",
|
|
190
|
+
};
|
|
191
|
+
case "workflow-concurrency-exhausted":
|
|
192
|
+
return {
|
|
193
|
+
code: "workflow_concurrency_exhausted",
|
|
194
|
+
message: "This proxy refused the request locally: the task has no free concurrency slot,"
|
|
195
|
+
+ " so no provider was contacted. Slots are released as the turns holding them finish.",
|
|
196
|
+
};
|
|
197
|
+
case "workflow-spend-exhausted":
|
|
198
|
+
return {
|
|
199
|
+
code: "workflow_spend_exhausted",
|
|
200
|
+
message: "This proxy refused the request locally: the task reached a configured token"
|
|
201
|
+
+ " ceiling, so no provider was contacted.",
|
|
202
|
+
};
|
|
203
|
+
case "workflow-tracking-exhausted":
|
|
204
|
+
return {
|
|
205
|
+
code: "workflow_tracking_exhausted",
|
|
206
|
+
message: "This proxy refused the request locally: it is already tracking as many tasks as"
|
|
207
|
+
+ " it may, and every one of them is busy or over its own ceiling, so no provider was"
|
|
208
|
+
+ " contacted.",
|
|
209
|
+
};
|
|
210
|
+
case "workflow-send-replayed":
|
|
211
|
+
return {
|
|
212
|
+
code: "workflow_send_replayed",
|
|
213
|
+
message: "This proxy refused the request locally: this send was already reserved once, and"
|
|
214
|
+
+ " a repeat buys no second dispatch.",
|
|
215
|
+
};
|
|
216
|
+
case "workflow-spend-undurable":
|
|
217
|
+
return {
|
|
218
|
+
code: "workflow_spend_undurable",
|
|
219
|
+
message: "This proxy refused the request locally: the token reservation could not be made"
|
|
220
|
+
+ " durable and a configured ceiling requires it, so no provider was contacted.",
|
|
221
|
+
};
|
|
222
|
+
}
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
/**
|
|
226
|
+
* Response header naming the ceiling that refused, on a refusal this proxy made itself.
|
|
227
|
+
*
|
|
228
|
+
* It exists because the body cannot carry it: `classifyError` rewrites every 429 to
|
|
229
|
+
* `rate_limit_error` / `rate_limit_exceeded`, so a local refusal and a provider rate limit are
|
|
230
|
+
* byte-identical in shape. Changing that classification would change how every client retries,
|
|
231
|
+
* so the name goes beside the body instead. No upstream sets this header, which is precisely
|
|
232
|
+
* what makes its presence conclusive.
|
|
233
|
+
*/
|
|
234
|
+
export const WORKFLOW_LOCAL_REFUSAL_HEADER = "x-opencodex-local-refusal";
|
|
235
|
+
|
|
236
|
+
export type WorkflowBudgetEventKind = "refused" | "cleared";
|
|
237
|
+
|
|
238
|
+
export interface WorkflowBudgetEvent {
|
|
239
|
+
readonly at: number;
|
|
240
|
+
readonly kind: WorkflowBudgetEventKind;
|
|
241
|
+
readonly rootId: string;
|
|
242
|
+
/** The ceiling that fired. Present for `refused`, absent for `cleared`. */
|
|
243
|
+
readonly reason?: WorkflowDenial;
|
|
244
|
+
/** Windowed sends at the moment of the event. */
|
|
245
|
+
readonly sends: number;
|
|
246
|
+
/** Windowed distinct children at the moment of the event. */
|
|
247
|
+
readonly children: number;
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
/**
|
|
251
|
+
* How many events are kept. Small on purpose: this is an operator's recent-history view, not an
|
|
252
|
+
* audit log, and it lives in the same process memory the ceilings do.
|
|
253
|
+
*/
|
|
254
|
+
export const WORKFLOW_EVENT_CAPACITY = 64;
|
|
255
|
+
|
|
256
|
+
const budgetEvents: WorkflowBudgetEvent[] = [];
|
|
257
|
+
|
|
258
|
+
/**
|
|
259
|
+
* Record a local budget decision.
|
|
260
|
+
*
|
|
261
|
+
* This exists because the refusal has nowhere else to go. The HTTP admission check runs before
|
|
262
|
+
* the body is parsed, so there is no model, no provider and no request-log context to attach to;
|
|
263
|
+
* writing a usage row there would mean inventing both. Every entry here is by construction a
|
|
264
|
+
* decision this proxy made without contacting anyone, which is a stronger statement than a flag
|
|
265
|
+
* on a row shared with upstream results.
|
|
266
|
+
*/
|
|
267
|
+
function recordBudgetEvent(event: WorkflowBudgetEvent): void {
|
|
268
|
+
budgetEvents.push(event);
|
|
269
|
+
while (budgetEvents.length > WORKFLOW_EVENT_CAPACITY) budgetEvents.shift();
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
/** Newest first. `limit` is clamped to what is actually kept. */
|
|
273
|
+
export function listWorkflowBudgetEvents(limit: number = WORKFLOW_EVENT_CAPACITY): WorkflowBudgetEvent[] {
|
|
274
|
+
const wanted = Number.isFinite(limit) && limit > 0
|
|
275
|
+
? Math.min(Math.floor(limit), WORKFLOW_EVENT_CAPACITY)
|
|
276
|
+
: 0;
|
|
277
|
+
if (wanted === 0) return [];
|
|
278
|
+
return budgetEvents.slice(-wanted).reverse();
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
/**
|
|
282
|
+
* Record a refusal decided outside `admitWorkflowTurn`.
|
|
283
|
+
*
|
|
284
|
+
* The pre-dispatch ceiling check in the responses path is a second refusal, taken after
|
|
285
|
+
* admission already succeeded, so nothing in this module sees it. Without this it was the one
|
|
286
|
+
* refusal an operator could hit that left no event behind.
|
|
287
|
+
*/
|
|
288
|
+
export function recordWorkflowRefusalEvent(
|
|
289
|
+
rootId: string | undefined,
|
|
290
|
+
reason: WorkflowDenial,
|
|
291
|
+
now: number = Date.now(),
|
|
292
|
+
): void {
|
|
293
|
+
if (!rootId) return;
|
|
294
|
+
const state = roots.get(rootId);
|
|
295
|
+
recordBudgetEvent({
|
|
296
|
+
at: now,
|
|
297
|
+
kind: "refused",
|
|
298
|
+
rootId,
|
|
299
|
+
reason,
|
|
300
|
+
sends: state ? windowedSends(state, now) : 0,
|
|
301
|
+
children: state ? windowedChildren(state, now) : 0,
|
|
302
|
+
});
|
|
303
|
+
}
|
|
46
304
|
|
|
47
305
|
export type WorkflowLane = "interactive" | "worker";
|
|
48
306
|
|
|
49
307
|
export interface WorkflowAdmission {
|
|
50
308
|
readonly rootId: string;
|
|
309
|
+
/**
|
|
310
|
+
* The request is about to leave for upstream. Call this at the dispatch boundary: until it
|
|
311
|
+
* runs, releasing the lease costs nothing, and after it a missing usage frame is booked as
|
|
312
|
+
* unresolved spend.
|
|
313
|
+
*/
|
|
314
|
+
markDispatched(): void;
|
|
51
315
|
release(): void;
|
|
52
316
|
}
|
|
53
317
|
|
|
54
318
|
export type WorkflowDecision =
|
|
55
319
|
| { admitted: true; lease: WorkflowAdmission }
|
|
56
|
-
| {
|
|
320
|
+
| {
|
|
321
|
+
admitted: false;
|
|
322
|
+
reason: WorkflowDenial;
|
|
323
|
+
rootId: string;
|
|
324
|
+
/** Which spend scope refused, when the denial came from the token ledger. */
|
|
325
|
+
spendScope?: SpendScope;
|
|
326
|
+
};
|
|
327
|
+
|
|
328
|
+
/**
|
|
329
|
+
* Token reservation attached to an admission. `outputCeilingTokens` is the ENFORCEABLE
|
|
330
|
+
* ceiling -- the caller's max_output_tokens or the model's documented cap, never an
|
|
331
|
+
* optimistic estimate and never shrunk by a cache-hit expectation. Omitting `spend`
|
|
332
|
+
* entirely keeps the historical count-only admission, which is also what an unconfigured
|
|
333
|
+
* install gets: token accounting is observed by default and refuses nothing until an
|
|
334
|
+
* operator sets real limits.
|
|
335
|
+
*/
|
|
336
|
+
export interface WorkflowSpendRequest {
|
|
337
|
+
/** Stable id of the physical send; settlement is idempotent on this key. */
|
|
338
|
+
readonly sendId: string;
|
|
339
|
+
readonly identityId?: string;
|
|
340
|
+
readonly poolId?: string;
|
|
341
|
+
readonly inputTokens: number;
|
|
342
|
+
readonly outputCeilingTokens: number;
|
|
343
|
+
}
|
|
57
344
|
|
|
58
345
|
interface WorkflowState {
|
|
59
346
|
active: number;
|
|
347
|
+
/** Lifetime total, kept for diagnostics only. The ceiling reads the window instead. */
|
|
60
348
|
sends: number;
|
|
61
|
-
|
|
349
|
+
/** Ring of per-slot send counts; sendSlotAt[i] names the slot that bucket holds. */
|
|
350
|
+
sendSlotCount: number[];
|
|
351
|
+
sendSlotAt: number[];
|
|
352
|
+
/** Child id to the last time it was admitted, so a child that stops ages out of the count. */
|
|
353
|
+
children: Map<string, number>;
|
|
62
354
|
lastSeenMs: number;
|
|
355
|
+
/** Window this root's ring and child map are measured over, fixed when the root appeared. */
|
|
356
|
+
windowMs: number;
|
|
357
|
+
}
|
|
358
|
+
|
|
359
|
+
function newWorkflowState(now: number, policy: WorkflowBudgetPolicy): WorkflowState {
|
|
360
|
+
return {
|
|
361
|
+
active: 0,
|
|
362
|
+
sends: 0,
|
|
363
|
+
sendSlotCount: new Array<number>(WORKFLOW_WINDOW_SLOTS).fill(0),
|
|
364
|
+
sendSlotAt: new Array<number>(WORKFLOW_WINDOW_SLOTS).fill(Number.NEGATIVE_INFINITY),
|
|
365
|
+
children: new Map<string, number>(),
|
|
366
|
+
lastSeenMs: now,
|
|
367
|
+
windowMs: workflowWindowMs(policy),
|
|
368
|
+
};
|
|
63
369
|
}
|
|
64
370
|
|
|
65
371
|
const roots = new Map<string, WorkflowState>();
|
|
66
372
|
|
|
67
|
-
|
|
373
|
+
/**
|
|
374
|
+
* Evict the oldest root that is safe to forget, and report whether one was found.
|
|
375
|
+
*
|
|
376
|
+
* The return value is the point. An earlier version returned void and the caller inserted
|
|
377
|
+
* the new root regardless, so `maxTrackedRoots` bounded nothing whenever every candidate
|
|
378
|
+
* was active or exhausted -- which is precisely the fan-out this file exists to bound.
|
|
379
|
+
*/
|
|
380
|
+
function evictOneRoot(
|
|
381
|
+
policy: WorkflowBudgetPolicy,
|
|
382
|
+
spendLedger?: SpendReservationLedger,
|
|
383
|
+
now: number = Date.now(),
|
|
384
|
+
): boolean {
|
|
68
385
|
let oldestKey: string | undefined;
|
|
69
386
|
let oldestAt = Number.POSITIVE_INFINITY;
|
|
70
387
|
for (const [key, state] of roots) {
|
|
71
388
|
// An active root is never evicted: dropping it would hand its fan-out a fresh allowance,
|
|
72
|
-
// which is the exact laundering this ledger exists to prevent.
|
|
389
|
+
// which is the exact laundering this ledger exists to prevent. The same holds for an
|
|
390
|
+
// EXHAUSTED-but-idle root -- count-exhausted or spend-exhausted -- because recreating it
|
|
391
|
+
// fresh under the same id resets the very ceiling that already fired.
|
|
73
392
|
if (state.active > 0) continue;
|
|
393
|
+
if (windowedSends(state, now) >= policy.maxPhysicalSends) continue;
|
|
394
|
+
if (spendLedger?.exhausted("root", key) === true) continue;
|
|
74
395
|
if (state.lastSeenMs < oldestAt) { oldestAt = state.lastSeenMs; oldestKey = key; }
|
|
75
396
|
}
|
|
76
|
-
if (oldestKey
|
|
397
|
+
if (oldestKey === undefined) return false;
|
|
398
|
+
roots.delete(oldestKey);
|
|
399
|
+
return true;
|
|
77
400
|
}
|
|
78
401
|
|
|
79
402
|
/**
|
|
@@ -81,6 +404,14 @@ function pruneOldestRoot(): void {
|
|
|
81
404
|
*
|
|
82
405
|
* `childId` distinguishes the members of a fan-out; omit it for the root's own turns.
|
|
83
406
|
* An interactive lane may use the reserved slots a worker lane may not.
|
|
407
|
+
*
|
|
408
|
+
* When `spend` is given, admission also reserves its tokens on the spend ledger -- at the
|
|
409
|
+
* root, identity and pool scopes at once -- before a concurrency slot is taken. A turn
|
|
410
|
+
* released without settlement is resolved by whether it was ever DISPATCHED: an undispatched
|
|
411
|
+
* turn gives its tokens back, and a dispatched one keeps them as unresolved spend, because a
|
|
412
|
+
* send whose usage never arrived may still have been billed. Call `lease.markDispatched()`
|
|
413
|
+
* at the point the request leaves for upstream; without it, admission followed by a local
|
|
414
|
+
* validation or routing failure would book spend that never happened.
|
|
84
415
|
*/
|
|
85
416
|
export function admitWorkflowTurn(
|
|
86
417
|
rootId: string | undefined,
|
|
@@ -88,44 +419,111 @@ export function admitWorkflowTurn(
|
|
|
88
419
|
policy: WorkflowBudgetPolicy = DEFAULT_WORKFLOW_BUDGET_POLICY,
|
|
89
420
|
childId?: string,
|
|
90
421
|
now: number = Date.now(),
|
|
422
|
+
spend?: WorkflowSpendRequest,
|
|
423
|
+
spendLedger?: SpendReservationLedger,
|
|
91
424
|
): WorkflowDecision | undefined {
|
|
92
425
|
if (!rootId) return undefined;
|
|
426
|
+
// An explicit ledger is consulted even without a spend request, so root eviction can
|
|
427
|
+
// still see spend-exhausted entries. With neither, no token tracking is in play.
|
|
428
|
+
const ledger = spendLedger ?? (spend ? sharedSpendLedger() : undefined);
|
|
93
429
|
let state = roots.get(rootId);
|
|
430
|
+
// Every refusal below goes on the record through this one seam. Recording at each return
|
|
431
|
+
// site instead of at the HTTP caller is what makes the record complete: the spend denials
|
|
432
|
+
// are decided inside the ledger branch and never surface as a distinct reason to the caller
|
|
433
|
+
// that formats the response.
|
|
434
|
+
const refuse = (reason: WorkflowDenial, spendScope?: SpendScope): WorkflowDecision => {
|
|
435
|
+
const current = roots.get(rootId);
|
|
436
|
+
recordBudgetEvent({
|
|
437
|
+
at: now,
|
|
438
|
+
kind: "refused",
|
|
439
|
+
rootId,
|
|
440
|
+
reason,
|
|
441
|
+
sends: current ? windowedSends(current, now) : 0,
|
|
442
|
+
children: current ? windowedChildren(current, now) : 0,
|
|
443
|
+
});
|
|
444
|
+
return { admitted: false, reason, rootId, ...(spendScope ? { spendScope } : {}) };
|
|
445
|
+
};
|
|
94
446
|
if (!state) {
|
|
95
|
-
if (roots.size >= policy.maxTrackedRoots
|
|
96
|
-
|
|
447
|
+
if (roots.size >= policy.maxTrackedRoots && !evictOneRoot(policy, ledger, now)) {
|
|
448
|
+
// Nothing may be forgotten, so the new root is refused instead of admitted over the
|
|
449
|
+
// bound. The alternative -- evicting an exhausted root -- resets the ceiling that
|
|
450
|
+
// already fired, and a caller minting fresh ids would get unlimited budget from it.
|
|
451
|
+
return refuse("workflow-tracking-exhausted");
|
|
452
|
+
}
|
|
453
|
+
state = newWorkflowState(now, policy);
|
|
97
454
|
roots.set(rootId, state);
|
|
98
455
|
}
|
|
99
456
|
state.lastSeenMs = now;
|
|
100
457
|
|
|
101
|
-
if (state
|
|
102
|
-
return
|
|
458
|
+
if (windowedSends(state, now) >= policy.maxPhysicalSends) {
|
|
459
|
+
return refuse("workflow-sends-exhausted");
|
|
103
460
|
}
|
|
104
461
|
if (childId !== undefined && !state.children.has(childId)
|
|
105
|
-
&& state
|
|
106
|
-
return
|
|
462
|
+
&& windowedChildren(state, now) >= policy.maxDistinctChildren) {
|
|
463
|
+
return refuse("workflow-children-exhausted");
|
|
107
464
|
}
|
|
108
465
|
const ceiling = lane === "worker"
|
|
109
466
|
? Math.max(0, policy.maxConcurrentChildren - policy.interactiveReserve)
|
|
110
467
|
: policy.maxConcurrentChildren;
|
|
111
468
|
if (state.active >= ceiling) {
|
|
112
|
-
return
|
|
469
|
+
return refuse("workflow-concurrency-exhausted");
|
|
470
|
+
}
|
|
471
|
+
|
|
472
|
+
if (spend && ledger) {
|
|
473
|
+
const decision = ledger.reserve({
|
|
474
|
+
sendId: spend.sendId,
|
|
475
|
+
scopes: { rootId, identityId: spend.identityId, poolId: spend.poolId },
|
|
476
|
+
inputTokens: spend.inputTokens,
|
|
477
|
+
outputCeilingTokens: spend.outputCeilingTokens,
|
|
478
|
+
at: now,
|
|
479
|
+
});
|
|
480
|
+
if (!decision.reserved) {
|
|
481
|
+
const denial = decision.denial;
|
|
482
|
+
// Every ledger refusal denies a DISPATCH. A duplicate send id and an undurable
|
|
483
|
+
// reservation are reported as themselves rather than folded into "exhausted", because
|
|
484
|
+
// an operator reading a 429 needs to know which of the three happened.
|
|
485
|
+
const reason: WorkflowDenial = denial.reason === "duplicate-send-id"
|
|
486
|
+
? "workflow-send-replayed"
|
|
487
|
+
: denial.reason === "reserve-not-durable" || denial.reason === "journal-corrupt"
|
|
488
|
+
? "workflow-spend-undurable"
|
|
489
|
+
: denial.reason === "tracking-capacity-exhausted"
|
|
490
|
+
? "workflow-tracking-exhausted"
|
|
491
|
+
: "workflow-spend-exhausted";
|
|
492
|
+
return refuse(
|
|
493
|
+
reason,
|
|
494
|
+
denial.reason === "spend-limit-exceeded" ? denial.scope : undefined,
|
|
495
|
+
);
|
|
496
|
+
}
|
|
113
497
|
}
|
|
114
498
|
|
|
115
499
|
state.active += 1;
|
|
116
|
-
if (childId !== undefined) state.children.
|
|
500
|
+
if (childId !== undefined) state.children.set(childId, now);
|
|
117
501
|
let released = false;
|
|
118
502
|
return {
|
|
119
503
|
admitted: true,
|
|
120
504
|
lease: {
|
|
121
505
|
rootId,
|
|
506
|
+
markDispatched(): void {
|
|
507
|
+
if (spend && ledger) ledger.markDispatched(spend.sendId);
|
|
508
|
+
},
|
|
122
509
|
release(): void {
|
|
123
510
|
if (released) return;
|
|
124
511
|
released = true;
|
|
125
512
|
const current = roots.get(rootId);
|
|
126
|
-
if (
|
|
127
|
-
|
|
128
|
-
|
|
513
|
+
if (current) {
|
|
514
|
+
current.active = Math.max(0, current.active - 1);
|
|
515
|
+
// Eviction ordering only; no ceiling reads lastSeenMs, so the wall clock is the
|
|
516
|
+
// right source here and a caller does not need to inject one.
|
|
517
|
+
current.lastSeenMs = Date.now();
|
|
518
|
+
}
|
|
519
|
+
// Which of the two applies depends on whether the send ever left this process.
|
|
520
|
+
// `abandon` succeeds only while the reservation is undispatched -- a turn refused by
|
|
521
|
+
// local validation or routing releases its tokens and books nothing, because
|
|
522
|
+
// inventing debt the account never incurred breaks the budget in the other
|
|
523
|
+
// direction. Once dispatched, abandon refuses and markLost keeps the cost as
|
|
524
|
+
// unresolved spend, since a send whose usage frame never arrived may still have been
|
|
525
|
+
// billed. Both are no-ops once settleWorkflowSpend already ran.
|
|
526
|
+
if (spend && ledger && !ledger.abandon(spend.sendId)) ledger.markLost(spend.sendId);
|
|
129
527
|
},
|
|
130
528
|
},
|
|
131
529
|
};
|
|
@@ -135,12 +533,55 @@ export function admitWorkflowTurn(
|
|
|
135
533
|
* Charge physical sends to a root. Called from the send budget's own accounting so a retry
|
|
136
534
|
* inside one request counts toward the workflow total, not only the request total.
|
|
137
535
|
*/
|
|
138
|
-
export function chargeWorkflowSends(
|
|
536
|
+
export function chargeWorkflowSends(
|
|
537
|
+
rootId: string | undefined,
|
|
538
|
+
sends: number,
|
|
539
|
+
now: number = Date.now(),
|
|
540
|
+
): void {
|
|
139
541
|
if (!rootId || sends <= 0) return;
|
|
140
542
|
const state = roots.get(rootId);
|
|
141
543
|
if (!state) return;
|
|
142
544
|
state.sends += sends;
|
|
143
|
-
|
|
545
|
+
// Geometry comes off the root itself, so no caller can charge on one scale and read on
|
|
546
|
+
// another. This function does not take a policy at all any more: it has no ceiling to
|
|
547
|
+
// compare, and the only thing a policy could have supplied here was that scale.
|
|
548
|
+
recordWindowedSends(state, now, sends);
|
|
549
|
+
state.lastSeenMs = now;
|
|
550
|
+
}
|
|
551
|
+
|
|
552
|
+
/**
|
|
553
|
+
* Settle a send's reservation with the usage the response actually reported. Idempotent
|
|
554
|
+
* per send id -- a second call returns false and books nothing. When the usage frame was
|
|
555
|
+
* lost, call this never and let the lease's release move the reservation to unresolved
|
|
556
|
+
* spend, or call the ledger's markLost directly.
|
|
557
|
+
*/
|
|
558
|
+
export function settleWorkflowSpend(
|
|
559
|
+
sendId: string,
|
|
560
|
+
usage: SpendUsage,
|
|
561
|
+
spendLedger?: SpendReservationLedger,
|
|
562
|
+
): boolean {
|
|
563
|
+
return (spendLedger ?? sharedSpendLedger()).settle(sendId, usage);
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
/**
|
|
567
|
+
* Record that the send left for upstream.
|
|
568
|
+
*
|
|
569
|
+
* This is the line between "may be released for free" and "may have been billed". Admission
|
|
570
|
+
* alone is not dispatch: a turn can be admitted and then fail request validation, provider
|
|
571
|
+
* routing, or a local guard without a single byte reaching a model. Booking those as spend
|
|
572
|
+
* invents debt the account never incurred, so the reservation only becomes unresolvable
|
|
573
|
+
* after this call.
|
|
574
|
+
*/
|
|
575
|
+
export function dispatchWorkflowSpend(sendId: string, spendLedger?: SpendReservationLedger): boolean {
|
|
576
|
+
return (spendLedger ?? sharedSpendLedger()).markDispatched(sendId);
|
|
577
|
+
}
|
|
578
|
+
|
|
579
|
+
/**
|
|
580
|
+
* Give a reservation back because the send never happened. Refused once dispatched, where
|
|
581
|
+
* settle or markLost is the only honest outcome.
|
|
582
|
+
*/
|
|
583
|
+
export function abandonWorkflowSpend(sendId: string, spendLedger?: SpendReservationLedger): boolean {
|
|
584
|
+
return (spendLedger ?? sharedSpendLedger()).abandon(sendId);
|
|
144
585
|
}
|
|
145
586
|
|
|
146
587
|
/**
|
|
@@ -152,21 +593,103 @@ export function chargeWorkflowSends(rootId: string | undefined, sends: number):
|
|
|
152
593
|
export function workflowSendCeilingReached(
|
|
153
594
|
rootId: string | undefined,
|
|
154
595
|
policy: WorkflowBudgetPolicy = DEFAULT_WORKFLOW_BUDGET_POLICY,
|
|
596
|
+
now: number = Date.now(),
|
|
155
597
|
): boolean {
|
|
156
598
|
if (!rootId) return false;
|
|
157
599
|
const state = roots.get(rootId);
|
|
158
|
-
return state !== undefined && state
|
|
600
|
+
return state !== undefined && windowedSends(state, now) >= policy.maxPhysicalSends;
|
|
601
|
+
}
|
|
602
|
+
|
|
603
|
+
export interface WorkflowBudgetSnapshot {
|
|
604
|
+
active: number;
|
|
605
|
+
/** Sends inside the window. This is the number the ceiling compares. */
|
|
606
|
+
sends: number;
|
|
607
|
+
/** Children inside the window, which is likewise what the ceiling compares. */
|
|
608
|
+
children: number;
|
|
609
|
+
/** Everything the root has ever sent, for diagnostics; no ceiling reads it. */
|
|
610
|
+
lifetimeSends: number;
|
|
611
|
+
windowMs: number;
|
|
612
|
+
maxPhysicalSends: number;
|
|
613
|
+
maxDistinctChildren: number;
|
|
614
|
+
}
|
|
615
|
+
|
|
616
|
+
export function workflowBudgetSnapshot(
|
|
617
|
+
rootId: string,
|
|
618
|
+
policy: WorkflowBudgetPolicy = DEFAULT_WORKFLOW_BUDGET_POLICY,
|
|
619
|
+
now: number = Date.now(),
|
|
620
|
+
): WorkflowBudgetSnapshot | undefined {
|
|
621
|
+
const state = roots.get(rootId);
|
|
622
|
+
if (!state) return undefined;
|
|
623
|
+
return {
|
|
624
|
+
active: state.active,
|
|
625
|
+
sends: windowedSends(state, now),
|
|
626
|
+
children: windowedChildren(state, now),
|
|
627
|
+
lifetimeSends: state.sends,
|
|
628
|
+
windowMs: state.windowMs,
|
|
629
|
+
maxPhysicalSends: policy.maxPhysicalSends,
|
|
630
|
+
maxDistinctChildren: policy.maxDistinctChildren,
|
|
631
|
+
};
|
|
159
632
|
}
|
|
160
633
|
|
|
161
|
-
|
|
634
|
+
/**
|
|
635
|
+
* Roots this process is currently tracking, most recently active first.
|
|
636
|
+
*
|
|
637
|
+
* Bounded by `limit` because `maxTrackedRoots` is 512 and an operator asking what is going on
|
|
638
|
+
* wants the busy end of that, not a dump.
|
|
639
|
+
*/
|
|
640
|
+
export function listTrackedWorkflowRoots(
|
|
641
|
+
limit = 64,
|
|
642
|
+
policy: WorkflowBudgetPolicy = DEFAULT_WORKFLOW_BUDGET_POLICY,
|
|
643
|
+
now: number = Date.now(),
|
|
644
|
+
): Array<{ rootId: string } & WorkflowBudgetSnapshot> {
|
|
645
|
+
const wanted = Number.isFinite(limit) && limit > 0 ? Math.floor(limit) : 0;
|
|
646
|
+
if (wanted === 0) return [];
|
|
647
|
+
return [...roots.entries()]
|
|
648
|
+
.sort((left, right) => right[1].lastSeenMs - left[1].lastSeenMs)
|
|
649
|
+
.slice(0, wanted)
|
|
650
|
+
.flatMap(([rootId]) => {
|
|
651
|
+
const snapshot = workflowBudgetSnapshot(rootId, policy, now);
|
|
652
|
+
return snapshot ? [{ rootId, ...snapshot }] : [];
|
|
653
|
+
});
|
|
654
|
+
}
|
|
162
655
|
|
|
163
|
-
|
|
164
|
-
|
|
656
|
+
/**
|
|
657
|
+
* Clear ONE root's windowed count ceilings, and report what they were.
|
|
658
|
+
*
|
|
659
|
+
* Three things are deliberately left alone. `active` belongs to turns still in flight, and
|
|
660
|
+
* zeroing it would let their releases drive the count negative and hand out concurrency slots
|
|
661
|
+
* that are already taken. The spend ledger is a token budget an operator did not ask to
|
|
662
|
+
* forgive, and a count ceiling is not a licence to reset it. `sends` -- the lifetime total --
|
|
663
|
+
* survives too, so the record of what this root actually did cannot be laundered by clearing
|
|
664
|
+
* it; only the ceilings move.
|
|
665
|
+
*
|
|
666
|
+
* Returns the snapshot taken immediately before the clear, so the caller can put on the record
|
|
667
|
+
* what it forgave, or `undefined` when the root is not tracked at all.
|
|
668
|
+
*/
|
|
669
|
+
export function clearWorkflowBudgetForRoot(
|
|
670
|
+
rootId: string,
|
|
671
|
+
policy: WorkflowBudgetPolicy = DEFAULT_WORKFLOW_BUDGET_POLICY,
|
|
672
|
+
now: number = Date.now(),
|
|
673
|
+
): WorkflowBudgetSnapshot | undefined {
|
|
165
674
|
const state = roots.get(rootId);
|
|
166
|
-
|
|
675
|
+
if (!state) return undefined;
|
|
676
|
+
const before = workflowBudgetSnapshot(rootId, policy, now);
|
|
677
|
+
state.sendSlotCount.fill(0);
|
|
678
|
+
state.sendSlotAt.fill(Number.NEGATIVE_INFINITY);
|
|
679
|
+
state.children.clear();
|
|
680
|
+
state.lastSeenMs = now;
|
|
681
|
+
recordBudgetEvent({
|
|
682
|
+
at: now,
|
|
683
|
+
kind: "cleared",
|
|
684
|
+
rootId,
|
|
685
|
+
sends: before?.sends ?? 0,
|
|
686
|
+
children: before?.children ?? 0,
|
|
687
|
+
});
|
|
688
|
+
return before;
|
|
167
689
|
}
|
|
168
690
|
|
|
169
691
|
/** Test seam. Production never clears a live ledger: that would reset a spent budget. */
|
|
170
692
|
export function resetWorkflowBudgetsForTest(): void {
|
|
171
693
|
roots.clear();
|
|
694
|
+
budgetEvents.length = 0;
|
|
172
695
|
}
|