@kal-elsam/kairo-runtime 0.23.0 → 0.23.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +39 -0
- package/package.json +1 -1
- package/src/global/cockpit/app.js +33 -2
- package/src/global/conversation/service.js +19 -5
- package/src/global/intelligence/execution-router.js +14 -2
- package/src/global/intelligence/quick-ask.js +30 -4
- package/src/global/observability/codex-models.js +1 -1
- package/src/global/observability/codex-usage.js +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -5,6 +5,45 @@ Historical entries below may reference the legacy `@kal-elsam/harness` package n
|
|
|
5
5
|
|
|
6
6
|
## Unreleased
|
|
7
7
|
|
|
8
|
+
## 0.23.2 — 2026-09-18 (Kairo Runtime)
|
|
9
|
+
|
|
10
|
+
Patch release.
|
|
11
|
+
|
|
12
|
+
### Fixed
|
|
13
|
+
|
|
14
|
+
- Routing eligibility (`checkCandidate`, shared by real execution
|
|
15
|
+
routing, ASK routing, and the FIT widget) and ASK's own ordering
|
|
16
|
+
only ever read the primary (5h) quota window — a provider whose
|
|
17
|
+
weekly window was nearly exhausted still got picked first and was
|
|
18
|
+
never excluded, as long as its 5h window looked healthy. Now takes
|
|
19
|
+
the worse of the two windows everywhere, fail-closed, matching the
|
|
20
|
+
same principle already applied to OpenCode Go's own windows.
|
|
21
|
+
- `askCodex`/`askClaude` (the ASK-mode question path) used a fixed 30s
|
|
22
|
+
deadline from process start. Codex's ASK path always uses the
|
|
23
|
+
provider's single default model regardless of question complexity
|
|
24
|
+
(no effort-based tiering for Codex today), so a heavier default
|
|
25
|
+
model plus a cold sandboxed `codex exec` start can genuinely exceed
|
|
26
|
+
30s with real quota to spare — a real, slow answer, not a hang. Both
|
|
27
|
+
ask calls now reset their timeout on every real stdout/stderr chunk
|
|
28
|
+
instead, never an absolute one, so only a genuine hang still times
|
|
29
|
+
out.
|
|
30
|
+
|
|
31
|
+
## 0.23.1 — 2026-09-18 (Kairo Runtime)
|
|
32
|
+
|
|
33
|
+
Patch release.
|
|
34
|
+
|
|
35
|
+
### Fixed
|
|
36
|
+
|
|
37
|
+
- A plain chat question's action label and any failure message said
|
|
38
|
+
generic "Asking Kairo" throughout the wait, even though one specific
|
|
39
|
+
real adapter (Codex or Claude) is what's actually being asked and
|
|
40
|
+
can actually fail — Kairo is the system, never the one answering.
|
|
41
|
+
Adds `service.planAsk`, a read-only preview of ASK mode's routing
|
|
42
|
+
decision (reusing the same cached probes `submitTask` already uses),
|
|
43
|
+
so the cockpit now names the real provider/model from the start of
|
|
44
|
+
the wait instead of only leaking it incidentally through a raw error
|
|
45
|
+
message on failure.
|
|
46
|
+
|
|
8
47
|
## 0.23.0 — 2026-09-18 (Kairo Runtime)
|
|
9
48
|
|
|
10
49
|
Minor release. First slice of PROJECT TEAM's automatic model-fallback
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@kal-elsam/kairo-runtime",
|
|
3
|
-
"version": "0.23.
|
|
3
|
+
"version": "0.23.2",
|
|
4
4
|
"description": "Kairo Runtime — local agent operating system for Codex, Cursor, Claude, Pi, Engram, and Graphify.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"homepage": "https://github.com/Kal-elSam/harness#readme",
|
|
@@ -248,7 +248,7 @@ export async function runCockpitApp({
|
|
|
248
248
|
});
|
|
249
249
|
}
|
|
250
250
|
|
|
251
|
-
editor.onSubmit = (text) => {
|
|
251
|
+
editor.onSubmit = async (text) => {
|
|
252
252
|
const task = text.trim();
|
|
253
253
|
if (!task) return;
|
|
254
254
|
if (task.startsWith("/")) {
|
|
@@ -435,7 +435,38 @@ export async function runCockpitApp({
|
|
|
435
435
|
// PLAN/AGENT always create a plan (see service.submitTask) — replacing
|
|
436
436
|
// the old isLikelyQuestion guess with what the user explicitly told
|
|
437
437
|
// Kairo they're doing (Shift+Tab / /plan).
|
|
438
|
-
|
|
438
|
+
//
|
|
439
|
+
// Kairo is the system, never the thing actually answering — a real
|
|
440
|
+
// adapter (Codex, Claude, …) always is, and that real name should be
|
|
441
|
+
// visible from the moment the wait starts, not just in a successful
|
|
442
|
+
// result or leaked incidentally through an error message. PLAN/AGENT
|
|
443
|
+
// is deterministically Codex (see submitArchitecture/createPlan), so
|
|
444
|
+
// that label needs no extra call; ASK's real provider depends on
|
|
445
|
+
// current quota/eligibility, so a cheap planAsk preview (the same
|
|
446
|
+
// routing submitTask itself will use, its probes cached — see
|
|
447
|
+
// service.js's planAsk doc) resolves it first. A caller that predates
|
|
448
|
+
// planAsk (or a preview that itself fails) falls back to the old
|
|
449
|
+
// generic label rather than ever blocking the real submit on it.
|
|
450
|
+
const mode = view.workMode;
|
|
451
|
+
let askLabel = "Asking Kairo";
|
|
452
|
+
if (mode === "plan" || mode === "agent") {
|
|
453
|
+
askLabel = "Asking Codex to plan";
|
|
454
|
+
} else {
|
|
455
|
+
try {
|
|
456
|
+
const preview = await service.planAsk?.({ cwd, task });
|
|
457
|
+
if (preview?.decision?.decision === "ROUTED") {
|
|
458
|
+
const provider = preview.decision.provider;
|
|
459
|
+
const providerLabel = provider ? provider.charAt(0).toUpperCase() + provider.slice(1) : null;
|
|
460
|
+
if (providerLabel) {
|
|
461
|
+
askLabel = `Asking ${providerLabel}${preview.decision.model ? ` · ${preview.decision.model}` : ""}`;
|
|
462
|
+
}
|
|
463
|
+
}
|
|
464
|
+
} catch {
|
|
465
|
+
// Best-effort label only — a real routing failure still surfaces
|
|
466
|
+
// through submitTask itself below, never swallowed here.
|
|
467
|
+
}
|
|
468
|
+
}
|
|
469
|
+
return runAction(askLabel, async () => {
|
|
439
470
|
const result = await service.submitTask({ cwd, task, mode: view.workMode });
|
|
440
471
|
if (result.kind === "answer") {
|
|
441
472
|
pushTranscript("kairo", `${result.provider}${result.model ? ` · ${result.model}` : ""}: ${result.answer}`);
|
|
@@ -651,12 +651,16 @@ export function createConversationService(deps = {}) {
|
|
|
651
651
|
return { ...publicPlan(result.status), reused: result.reused === true, projectRoot };
|
|
652
652
|
},
|
|
653
653
|
/**
|
|
654
|
-
*
|
|
655
|
-
*
|
|
656
|
-
*
|
|
657
|
-
*
|
|
654
|
+
* Read-only preview of who ASK mode would actually ask right now —
|
|
655
|
+
* never calls a provider. Lets a caller (the cockpit's action label)
|
|
656
|
+
* show the real provider/model BEFORE the potentially slow real call
|
|
657
|
+
* starts, instead of a generic "Asking Kairo" that stays true no
|
|
658
|
+
* matter which real provider ends up answering (or timing out).
|
|
659
|
+
* Cheap to call again right after: the same underlying usage/catalog
|
|
660
|
+
* probes askQuestion itself uses are cached (see readCodexUsageCached
|
|
661
|
+
* etc.), so there's no real duplicate provider I/O.
|
|
658
662
|
*/
|
|
659
|
-
async
|
|
663
|
+
async planAsk({ cwd, task }) {
|
|
660
664
|
const projectRoot = await root(cwd);
|
|
661
665
|
const adapters = inspectAdapters({ cwd: projectRoot });
|
|
662
666
|
let codexUsage = null;
|
|
@@ -672,6 +676,16 @@ export function createConversationService(deps = {}) {
|
|
|
672
676
|
claudeCatalog = readClaudeModelsImpl();
|
|
673
677
|
}
|
|
674
678
|
const decision = routeAsk({ adapters, codexUsage, claudeUsage, catalogs: { codex: codexCatalog, claude: claudeCatalog }, taskText: task });
|
|
679
|
+
return { decision, projectRoot };
|
|
680
|
+
},
|
|
681
|
+
/**
|
|
682
|
+
* Real read-only question -> real answer, via whichever provider is
|
|
683
|
+
* actually available/quota-healthy — no task, no plan, no approval
|
|
684
|
+
* gate. Throws (never returns a fabricated answer) if no provider can
|
|
685
|
+
* answer or the call itself fails.
|
|
686
|
+
*/
|
|
687
|
+
async askQuestion({ cwd, task }) {
|
|
688
|
+
const { decision, projectRoot } = await this.planAsk({ cwd, task });
|
|
675
689
|
if (decision.decision !== "ROUTED") throw new Error(`Cannot answer: ${decision.why}`);
|
|
676
690
|
const result = await askProviderImpl({ provider: decision.provider, question: task, model: decision.model, cwd: projectRoot });
|
|
677
691
|
if (result.status !== "answered") throw new Error(result.error ?? `${decision.provider} gave no answer.`);
|
|
@@ -118,9 +118,21 @@ function findAdapter(adapterId, adapters) {
|
|
|
118
118
|
return adapters.find((adapter) => adapter.id === baseId) ?? null;
|
|
119
119
|
}
|
|
120
120
|
|
|
121
|
-
/**
|
|
121
|
+
/**
|
|
122
|
+
* The WORST remaining headroom across both real windows (5h "primary" and
|
|
123
|
+
* weekly "secondary") — never just the primary one. A provider whose
|
|
124
|
+
* weekly quota is nearly gone must be treated that way everywhere
|
|
125
|
+
* (exclusion AND ordering) even while its 5h window still looks healthy;
|
|
126
|
+
* otherwise Kairo keeps routing to it, burning through the one budget
|
|
127
|
+
* that's actually about to run out. Mirrors the same fail-closed
|
|
128
|
+
* principle checkCandidate already applies to OpenCode Go's windows (any
|
|
129
|
+
* one window being real trouble is real trouble, full stop).
|
|
130
|
+
* @param {object|null} usageEntry - a codex/claude usage-probe result (primary/secondary windows)
|
|
131
|
+
*/
|
|
122
132
|
function remainingPercent(usageEntry) {
|
|
123
|
-
|
|
133
|
+
const known = [usageEntry?.primary?.remainingPercent, usageEntry?.secondary?.remainingPercent]
|
|
134
|
+
.filter((value) => typeof value === "number");
|
|
135
|
+
return known.length > 0 ? Math.min(...known) : null;
|
|
124
136
|
}
|
|
125
137
|
|
|
126
138
|
// Below this real remaining-quota percentage, a provider is treated as
|
|
@@ -37,6 +37,32 @@ function unknown(error) {
|
|
|
37
37
|
return { status: "error", answer: null, error: String(error) };
|
|
38
38
|
}
|
|
39
39
|
|
|
40
|
+
/**
|
|
41
|
+
* Resets on every real stdout/stderr chunk from the child, never an
|
|
42
|
+
* absolute deadline from process start — the same real distinction
|
|
43
|
+
* execution-adapters/opencode.js's own idle timeout draws: a real, live
|
|
44
|
+
* answer that's just taking a while (a heavier reasoning model, a cold
|
|
45
|
+
* sandbox start) must never be killed for merely being slow, only a
|
|
46
|
+
* process that's produced nothing at all for `timeoutMs` really looks
|
|
47
|
+
* hung. A single-shot ask call, so this stays local rather than reusing
|
|
48
|
+
* run-supervisor.js's own detached-run mechanism.
|
|
49
|
+
* @param {import("node:child_process").ChildProcess} child
|
|
50
|
+
* @param {number} timeoutMs
|
|
51
|
+
* @param {() => void} onIdle
|
|
52
|
+
* @returns {() => void} call to clear the timer once the call finishes
|
|
53
|
+
*/
|
|
54
|
+
function armIdleTimeout(child, timeoutMs, onIdle) {
|
|
55
|
+
let handle = null;
|
|
56
|
+
const reset = () => {
|
|
57
|
+
if (handle) clearTimeout(handle);
|
|
58
|
+
handle = setTimeout(onIdle, timeoutMs);
|
|
59
|
+
};
|
|
60
|
+
child.stdout?.on("data", reset);
|
|
61
|
+
child.stderr?.on("data", reset);
|
|
62
|
+
reset();
|
|
63
|
+
return () => { if (handle) clearTimeout(handle); };
|
|
64
|
+
}
|
|
65
|
+
|
|
40
66
|
/** @param {{question:string, model:string|null, cwd:string, spawn:Function, timeoutMs:number, env:object}} args */
|
|
41
67
|
function askClaude({ question, model, cwd, spawn, timeoutMs, env }) {
|
|
42
68
|
// --restricted: removes Bash/code-execution tools and WebFetch, ignores
|
|
@@ -57,11 +83,11 @@ function askClaude({ question, model, cwd, spawn, timeoutMs, env }) {
|
|
|
57
83
|
}
|
|
58
84
|
let stdout = "";
|
|
59
85
|
let finished = false;
|
|
60
|
-
const
|
|
86
|
+
const clearIdleTimer = armIdleTimeout(child, timeoutMs, () => finish(unknown(`claude -p idle-timed out after ${timeoutMs}ms with no output`)));
|
|
61
87
|
function finish(result) {
|
|
62
88
|
if (finished) return;
|
|
63
89
|
finished = true;
|
|
64
|
-
|
|
90
|
+
clearIdleTimer();
|
|
65
91
|
try { child.kill?.(); } catch { /* best effort */ }
|
|
66
92
|
resolve(result);
|
|
67
93
|
}
|
|
@@ -106,11 +132,11 @@ async function askCodex({ question, model, cwd, spawn, timeoutMs, env }) {
|
|
|
106
132
|
return;
|
|
107
133
|
}
|
|
108
134
|
let finished = false;
|
|
109
|
-
const
|
|
135
|
+
const clearIdleTimer = armIdleTimeout(child, timeoutMs, () => finish(unknown(`codex exec idle-timed out after ${timeoutMs}ms with no output`)));
|
|
110
136
|
function finish(result) {
|
|
111
137
|
if (finished) return;
|
|
112
138
|
finished = true;
|
|
113
|
-
|
|
139
|
+
clearIdleTimer();
|
|
114
140
|
try { child.kill?.(); } catch { /* best effort */ }
|
|
115
141
|
resolve(result);
|
|
116
142
|
}
|
|
@@ -89,7 +89,7 @@ export async function readCodexModels({
|
|
|
89
89
|
child.once?.("close", () => { if (!finished) finish(unknown("codex app-server closed before model list")); });
|
|
90
90
|
|
|
91
91
|
writeRequest(child, 1, "initialize", {
|
|
92
|
-
clientInfo: { name: "kairo", title: "Kairo", version: "0.23.
|
|
92
|
+
clientInfo: { name: "kairo", title: "Kairo", version: "0.23.2" },
|
|
93
93
|
capabilities: {}
|
|
94
94
|
});
|
|
95
95
|
});
|
|
@@ -151,7 +151,7 @@ export async function readCodexUsage({
|
|
|
151
151
|
child.once?.("close", () => { if (!finished) finish(unknown("codex app-server closed before rate limits")); });
|
|
152
152
|
|
|
153
153
|
writeRequest(child, 1, "initialize", {
|
|
154
|
-
clientInfo: { name: "kairo", title: "Kairo", version: "0.23.
|
|
154
|
+
clientInfo: { name: "kairo", title: "Kairo", version: "0.23.2" },
|
|
155
155
|
capabilities: {}
|
|
156
156
|
});
|
|
157
157
|
});
|