karajan-code 4.32.0 → 4.34.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/src/agents/aider-agent.js +2 -12
- package/src/agents/base-agent.js +9 -20
- package/src/agents/claude-agent.js +2 -12
- package/src/agents/codex-agent.js +2 -12
- package/src/agents/dead-models.js +74 -0
- package/src/agents/gemini-agent.js +2 -12
- package/src/agents/model-errors.js +35 -0
- package/src/agents/opencode-agent.js +2 -12
- package/src/brain/agent-error-classifier.js +19 -1
- package/src/brain/one-shot-policy.js +26 -0
- package/src/brain/role-fallback-chain.js +100 -0
- package/src/brain/with-brain-recovery.js +63 -4
- package/src/checks/action-pins.js +131 -0
- package/src/checks/project-checks.js +6 -0
- package/src/checks/release-check.js +51 -5
- package/src/cli/register-meta.js +42 -2
- package/src/cli/register-pipeline.js +11 -1
- package/src/commands/check.js +5 -0
- package/src/commands/code.js +59 -4
- package/src/commands/env.js +2 -1
- package/src/commands/harden.js +12 -3
- package/src/commands/policy.js +21 -14
- package/src/commands/report.js +28 -0
- package/src/commands/review-gate.js +25 -15
- package/src/environment/panel.js +85 -0
- package/src/environment/playbook.js +7 -6
- package/src/harden/config-templates.js +7 -1
- package/src/harden/guidelines-engine.js +14 -4
- package/src/harden/guidelines-templates.js +69 -10
- package/src/harden/harness-hooks.js +6 -0
- package/src/harden/hook-templates.js +4 -1
- package/src/harden/sentinel-hooks.js +227 -24
- package/src/harden/supervisor-commit.js +41 -1
- package/src/harden/workflow-templates.js +9 -3
- package/src/policy/supervisor-verify.js +45 -11
- package/src/privacy/diff-scope.js +58 -0
- package/src/privacy/scan.js +18 -0
- package/src/prompts/card-context.js +57 -0
- package/src/prompts/session-context.js +68 -0
- package/src/review/board-pending.js +68 -0
- package/src/review/loc-budget.js +97 -0
- package/src/roles/agent-role.js +11 -2
- package/src/roles/audit-role.js +28 -2
package/package.json
CHANGED
|
@@ -45,23 +45,13 @@ export class AiderAgent extends BaseAgent {
|
|
|
45
45
|
async runTask(task) {
|
|
46
46
|
const role = task.role || "coder";
|
|
47
47
|
const model = this.getRoleModel(role);
|
|
48
|
-
|
|
49
|
-
if (!result.ok && model && this.isModelNotSupportedError(result)) {
|
|
50
|
-
this.logger?.warn(`Aider model "${model}" not supported — retrying with agent default`);
|
|
51
|
-
return this._exec(task, null);
|
|
52
|
-
}
|
|
53
|
-
return result;
|
|
48
|
+
return this._exec(task, model);
|
|
54
49
|
}
|
|
55
50
|
|
|
56
51
|
async reviewTask(task) {
|
|
57
52
|
const role = task.role || "reviewer";
|
|
58
53
|
const model = this.getRoleModel(role);
|
|
59
|
-
|
|
60
|
-
if (!result.ok && model && this.isModelNotSupportedError(result)) {
|
|
61
|
-
this.logger?.warn(`Aider model "${model}" not supported — retrying with agent default`);
|
|
62
|
-
return this._exec(task, null);
|
|
63
|
-
}
|
|
64
|
-
return result;
|
|
54
|
+
return this._exec(task, model);
|
|
65
55
|
}
|
|
66
56
|
|
|
67
57
|
async _exec(task, model) {
|
package/src/agents/base-agent.js
CHANGED
|
@@ -13,15 +13,8 @@
|
|
|
13
13
|
import { defaultEnvironment } from "../infrastructure/environment.js";
|
|
14
14
|
import { buildAgentEnv } from "../utils/role-env.js";
|
|
15
15
|
import { isModelCompatible } from "../config/role-resolver.js";
|
|
16
|
+
import { deadModelRecord } from "./dead-models.js";
|
|
16
17
|
|
|
17
|
-
const MODEL_NOT_SUPPORTED_PATTERNS = [
|
|
18
|
-
/model.{0,30}is not supported/i,
|
|
19
|
-
/model.{0,30}not available/i,
|
|
20
|
-
/model.{0,30}does not exist/i,
|
|
21
|
-
/unsupported model/i,
|
|
22
|
-
/invalid model/i,
|
|
23
|
-
/model_not_found/i
|
|
24
|
-
];
|
|
25
18
|
|
|
26
19
|
export class BaseAgent {
|
|
27
20
|
/**
|
|
@@ -108,6 +101,14 @@ export class BaseAgent {
|
|
|
108
101
|
this.logger?.warn?.(`model "${roleModel}" belongs to another family — dropping it for ${this.name} (role ${role})`);
|
|
109
102
|
return null;
|
|
110
103
|
}
|
|
104
|
+
// KJC-TSK-0827: a model kj has watched this provider retire is not worth
|
|
105
|
+
// a call. The pin stays the user's to change, so the warning names the
|
|
106
|
+
// exact line instead of kj editing their config behind their back.
|
|
107
|
+
const dead = roleModel ? deadModelRecord(this.name, roleModel) : null;
|
|
108
|
+
if (dead) {
|
|
109
|
+
this.logger?.warn?.(`model "${roleModel}" was retired by ${this.name} (seen ${dead.at}) — using the provider default. Change it with roles.${role}.model in kj.config.yml`);
|
|
110
|
+
return null;
|
|
111
|
+
}
|
|
111
112
|
return roleModel;
|
|
112
113
|
}
|
|
113
114
|
|
|
@@ -119,16 +120,4 @@ export class BaseAgent {
|
|
|
119
120
|
if (role === "reviewer") return false;
|
|
120
121
|
return Boolean(this.config?.coder_options?.auto_approve);
|
|
121
122
|
}
|
|
122
|
-
|
|
123
|
-
/**
|
|
124
|
-
* Heuristic: does the agent's error look like "model not supported"?
|
|
125
|
-
* Used by the retry/fallback path.
|
|
126
|
-
* @param {Partial<AgentResult>} result
|
|
127
|
-
* @returns {boolean}
|
|
128
|
-
*/
|
|
129
|
-
isModelNotSupportedError(result) {
|
|
130
|
-
const text = [result?.error, result?.output, result?.stderr, result?.stdout]
|
|
131
|
-
.filter(Boolean).join("\n");
|
|
132
|
-
return MODEL_NOT_SUPPORTED_PATTERNS.some(re => re.test(text));
|
|
133
|
-
}
|
|
134
123
|
}
|
|
@@ -343,23 +343,13 @@ export class ClaudeAgent extends BaseAgent {
|
|
|
343
343
|
async runTask(task) {
|
|
344
344
|
const role = task.role || "coder";
|
|
345
345
|
const model = this.getRoleModel(role);
|
|
346
|
-
|
|
347
|
-
if (!result.ok && model && this.isModelNotSupportedError(result)) {
|
|
348
|
-
this.logger?.warn(`Claude model "${model}" not supported — retrying with agent default`);
|
|
349
|
-
return this._runTaskExec(task, null, role);
|
|
350
|
-
}
|
|
351
|
-
return result;
|
|
346
|
+
return this._runTaskExec(task, model, role);
|
|
352
347
|
}
|
|
353
348
|
|
|
354
349
|
async reviewTask(task) {
|
|
355
350
|
const role = task.role || "reviewer";
|
|
356
351
|
const model = this.getRoleModel(role);
|
|
357
|
-
|
|
358
|
-
if (!result.ok && model && this.isModelNotSupportedError(result)) {
|
|
359
|
-
this.logger?.warn(`Claude model "${model}" not supported — retrying with agent default`);
|
|
360
|
-
return this._reviewTaskExec(task, null);
|
|
361
|
-
}
|
|
362
|
-
return result;
|
|
352
|
+
return this._reviewTaskExec(task, model);
|
|
363
353
|
}
|
|
364
354
|
|
|
365
355
|
async _runTaskExec(task, model, _role) {
|
|
@@ -57,23 +57,13 @@ export class CodexAgent extends BaseAgent {
|
|
|
57
57
|
async runTask(task) {
|
|
58
58
|
const role = task.role || "coder";
|
|
59
59
|
const model = this.getRoleModel(role);
|
|
60
|
-
|
|
61
|
-
if (!result.ok && model && this.isModelNotSupportedError(result)) {
|
|
62
|
-
this.logger?.warn(`Codex model "${model}" not supported — retrying with agent default`);
|
|
63
|
-
return this._exec(task, null, role);
|
|
64
|
-
}
|
|
65
|
-
return result;
|
|
60
|
+
return this._exec(task, model, role);
|
|
66
61
|
}
|
|
67
62
|
|
|
68
63
|
async reviewTask(task) {
|
|
69
64
|
const role = task.role || "reviewer";
|
|
70
65
|
const model = this.getRoleModel(role);
|
|
71
|
-
|
|
72
|
-
if (!result.ok && model && this.isModelNotSupportedError(result)) {
|
|
73
|
-
this.logger?.warn(`Codex model "${model}" not supported — retrying with agent default`);
|
|
74
|
-
return this._exec(task, null, role);
|
|
75
|
-
}
|
|
76
|
-
return result;
|
|
66
|
+
return this._exec(task, model, role);
|
|
77
67
|
}
|
|
78
68
|
|
|
79
69
|
async _exec(task, model, role) {
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Models the provider has retired, remembered (KJC-TSK-0827).
|
|
3
|
+
*
|
|
4
|
+
* KJC-TSK-0859 made a dead model recoverable: the chain moves on. What it does
|
|
5
|
+
* not do is stop the SECOND run from paying for the same corpse, and the third,
|
|
6
|
+
* and every one after, because the dead model is still what the config pins.
|
|
7
|
+
* The field case: codex retired gpt-5.4 and gpt-5.4-mini on 2026-09-09 and the
|
|
8
|
+
* reviewer kept calling a dead model on every single run.
|
|
9
|
+
*
|
|
10
|
+
* So kj remembers, per account (this is a global fact, not a project one), and
|
|
11
|
+
* skips a model it has watched die. Two deliberate limits:
|
|
12
|
+
*
|
|
13
|
+
* - kj NEVER edits the user's declaration. The pin is theirs; kj says which
|
|
14
|
+
* line to change and keeps working meanwhile. (The card asked for the
|
|
15
|
+
* config to be rewritten; that part is the user's call, and it is asked in
|
|
16
|
+
* the card rather than decided here.)
|
|
17
|
+
* - an entry expires. A wrong verdict must heal by itself, and a provider
|
|
18
|
+
* bringing a model back must not need a cache flush nobody would think of.
|
|
19
|
+
*/
|
|
20
|
+
import { existsSync, mkdirSync, readFileSync, writeFileSync } from "node:fs";
|
|
21
|
+
import { dirname, join } from "node:path";
|
|
22
|
+
|
|
23
|
+
import { getKarajanHome } from "../utils/paths.js";
|
|
24
|
+
|
|
25
|
+
/** A month is long enough to stop paying for it, short enough to be wrong. */
|
|
26
|
+
export const DEAD_MODEL_TTL_MS = 30 * 24 * 60 * 60 * 1000;
|
|
27
|
+
|
|
28
|
+
const storePath = () => join(getKarajanHome(), "dead-models.json");
|
|
29
|
+
const keyOf = (provider, model) => `${provider}::${model}`;
|
|
30
|
+
|
|
31
|
+
function load() {
|
|
32
|
+
const file = storePath();
|
|
33
|
+
if (!existsSync(file)) return {};
|
|
34
|
+
try {
|
|
35
|
+
const parsed = JSON.parse(readFileSync(file, "utf8"));
|
|
36
|
+
return parsed && typeof parsed === "object" ? parsed : {};
|
|
37
|
+
} catch {
|
|
38
|
+
// A corrupt store is not a reason to fail a run: kj forgets and relearns.
|
|
39
|
+
return {};
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/** @returns {boolean} true when the record was written. */
|
|
44
|
+
export function recordDeadModel({ provider, model, reason = null, now = Date.now() }) {
|
|
45
|
+
if (!provider || !model) return false;
|
|
46
|
+
const store = load();
|
|
47
|
+
store[keyOf(provider, model)] = { provider, model, reason, at: new Date(now).toISOString(), expiresAt: now + DEAD_MODEL_TTL_MS };
|
|
48
|
+
try {
|
|
49
|
+
const file = storePath();
|
|
50
|
+
mkdirSync(dirname(file), { recursive: true });
|
|
51
|
+
writeFileSync(file, JSON.stringify(store, null, 2));
|
|
52
|
+
return true;
|
|
53
|
+
} catch {
|
|
54
|
+
// Remembering is an optimisation, never a gate: if the home is read-only
|
|
55
|
+
// the run still works, it just pays the dead call again.
|
|
56
|
+
return false;
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/** @returns {null|{provider: string, model: string, reason: string|null, at: string}} */
|
|
61
|
+
export function deadModelRecord(provider, model, now = Date.now()) {
|
|
62
|
+
if (!provider || !model) return null;
|
|
63
|
+
const hit = load()[keyOf(provider, model)];
|
|
64
|
+
if (!hit) return null;
|
|
65
|
+
if (!hit.expiresAt || hit.expiresAt <= now) return null; // expired: try it again
|
|
66
|
+
return hit;
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
/** Live entries only, newest first — for `kj doctor` and friends. */
|
|
70
|
+
export function listDeadModels(now = Date.now()) {
|
|
71
|
+
return Object.values(load())
|
|
72
|
+
.filter((e) => e.expiresAt > now)
|
|
73
|
+
.sort((a, b) => String(b.at).localeCompare(String(a.at)));
|
|
74
|
+
}
|
|
@@ -44,23 +44,13 @@ export class GeminiAgent extends BaseAgent {
|
|
|
44
44
|
async runTask(task) {
|
|
45
45
|
const role = task.role || "coder";
|
|
46
46
|
const model = this.getRoleModel(role);
|
|
47
|
-
|
|
48
|
-
if (!result.ok && model && this.isModelNotSupportedError(result)) {
|
|
49
|
-
this.logger?.warn(`${this.cliBin} model "${model}" not supported — retrying with agent default`);
|
|
50
|
-
return this._exec(task, null, "run");
|
|
51
|
-
}
|
|
52
|
-
return result;
|
|
47
|
+
return this._exec(task, model, "run");
|
|
53
48
|
}
|
|
54
49
|
|
|
55
50
|
async reviewTask(task) {
|
|
56
51
|
const role = task.role || "reviewer";
|
|
57
52
|
const model = this.getRoleModel(role);
|
|
58
|
-
|
|
59
|
-
if (!result.ok && model && this.isModelNotSupportedError(result)) {
|
|
60
|
-
this.logger?.warn(`${this.cliBin} model "${model}" not supported — retrying with agent default`);
|
|
61
|
-
return this._exec(task, null, "review");
|
|
62
|
-
}
|
|
63
|
-
return result;
|
|
53
|
+
return this._exec(task, model, "review");
|
|
64
54
|
}
|
|
65
55
|
|
|
66
56
|
async _exec(task, model, mode) {
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* "That model is gone" — one definition, two readers (KJC-TSK-0859).
|
|
3
|
+
*
|
|
4
|
+
* A dead model announced itself in two unrelated places: each agent matched
|
|
5
|
+
* these patterns privately to retry with its provider default, and the brain's
|
|
6
|
+
* error classifier had no idea the class existed, so it filed the failure as
|
|
7
|
+
* UNKNOWN_FATAL and aborted the run. Same event, two verdicts.
|
|
8
|
+
*
|
|
9
|
+
* The patterns live here so the agent and the brain read the SAME definition,
|
|
10
|
+
* and a model that a provider retired becomes a reason to move down the
|
|
11
|
+
* declared chain instead of a reason to stop.
|
|
12
|
+
*/
|
|
13
|
+
export const MODEL_NOT_SUPPORTED_PATTERNS = [
|
|
14
|
+
/model.{0,30}is not supported/i,
|
|
15
|
+
/model.{0,30}not available/i,
|
|
16
|
+
/model.{0,30}does not exist/i,
|
|
17
|
+
/unsupported model/i,
|
|
18
|
+
/invalid model/i,
|
|
19
|
+
/model_not_found/i,
|
|
20
|
+
];
|
|
21
|
+
|
|
22
|
+
/** @param {string} text @returns {boolean} */
|
|
23
|
+
export function isModelUnavailableText(text) {
|
|
24
|
+
if (!text) return false;
|
|
25
|
+
return MODEL_NOT_SUPPORTED_PATTERNS.some((re) => re.test(text));
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** The matched sentence, for a message that names what died. */
|
|
29
|
+
export function pickModelUnavailableMessage(text) {
|
|
30
|
+
for (const re of MODEL_NOT_SUPPORTED_PATTERNS) {
|
|
31
|
+
const line = String(text).split("\n").find((l) => re.test(l));
|
|
32
|
+
if (line) return line.trim();
|
|
33
|
+
}
|
|
34
|
+
return null;
|
|
35
|
+
}
|
|
@@ -37,23 +37,13 @@ export class OpenCodeAgent extends BaseAgent {
|
|
|
37
37
|
async runTask(task) {
|
|
38
38
|
const role = task.role || "coder";
|
|
39
39
|
const model = this.getRoleModel(role);
|
|
40
|
-
|
|
41
|
-
if (!result.ok && model && this.isModelNotSupportedError(result)) {
|
|
42
|
-
this.logger?.warn(`OpenCode model "${model}" not supported — retrying with agent default`);
|
|
43
|
-
return this._exec(task, null, false);
|
|
44
|
-
}
|
|
45
|
-
return result;
|
|
40
|
+
return this._exec(task, model, false);
|
|
46
41
|
}
|
|
47
42
|
|
|
48
43
|
async reviewTask(task) {
|
|
49
44
|
const role = task.role || "reviewer";
|
|
50
45
|
const model = this.getRoleModel(role);
|
|
51
|
-
|
|
52
|
-
if (!result.ok && model && this.isModelNotSupportedError(result)) {
|
|
53
|
-
this.logger?.warn(`OpenCode model "${model}" not supported — retrying with agent default`);
|
|
54
|
-
return this._exec(task, null, true);
|
|
55
|
-
}
|
|
56
|
-
return result;
|
|
46
|
+
return this._exec(task, model, true);
|
|
57
47
|
}
|
|
58
48
|
|
|
59
49
|
async _exec(task, model, jsonFormat) {
|
|
@@ -13,6 +13,7 @@
|
|
|
13
13
|
// - NETWORK_TIMEOUT por ECONN*/socket hang up sin cooldown.
|
|
14
14
|
|
|
15
15
|
import { parseCooldown } from "../utils/rate-limit-detector.js";
|
|
16
|
+
import { isModelUnavailableText, pickModelUnavailableMessage } from "../agents/model-errors.js";
|
|
16
17
|
|
|
17
18
|
export const ERROR_CLASS = Object.freeze({
|
|
18
19
|
RATE_LIMIT_SHORT: "RATE_LIMIT_SHORT",
|
|
@@ -21,6 +22,10 @@ export const ERROR_CLASS = Object.freeze({
|
|
|
21
22
|
// 15-jun-2026. Cuando llegues al cap mensual, el reset es 1-mes —
|
|
22
23
|
// muy distinto del daily de Claude Pro. Esta clase distingue ambos.
|
|
23
24
|
QUOTA_EXHAUSTED_MONTHLY: "QUOTA_EXHAUSTED_MONTHLY",
|
|
25
|
+
// KJC-TSK-0859: un modelo que el proveedor retiró no es un error fatal ni
|
|
26
|
+
// un problema de cuota — es motivo para tomar el siguiente eslabón de la
|
|
27
|
+
// cadena declarada. No tiene cooldown: esperar no lo va a resucitar.
|
|
28
|
+
MODEL_UNAVAILABLE: "MODEL_UNAVAILABLE",
|
|
24
29
|
API_DOWN: "API_DOWN",
|
|
25
30
|
AUTH_FAILED: "AUTH_FAILED",
|
|
26
31
|
NETWORK_TIMEOUT: "NETWORK_TIMEOUT",
|
|
@@ -80,7 +85,20 @@ export function classifyAgentError({ provider = "unknown", stdout = "", stderr =
|
|
|
80
85
|
return { ...base, class: ERROR_CLASS.SILENCED, message: pickMessage(combined, SILENCED_PATTERNS) || "Agent silenciado por timeout", recoverable: true };
|
|
81
86
|
}
|
|
82
87
|
|
|
83
|
-
// 3.
|
|
88
|
+
// 3. MODEL_UNAVAILABLE: el modelo fijado ya no existe para esta cuenta.
|
|
89
|
+
// Antes caía en UNKNOWN_FATAL y abortaba el run: el agente lo detectaba
|
|
90
|
+
// por su cuenta y reintentaba con su default, pero el brain no se
|
|
91
|
+
// enteraba. Va antes del rate limit porque no tiene cooldown alguno.
|
|
92
|
+
if (isModelUnavailableText(combined)) {
|
|
93
|
+
return {
|
|
94
|
+
...base,
|
|
95
|
+
class: ERROR_CLASS.MODEL_UNAVAILABLE,
|
|
96
|
+
message: pickModelUnavailableMessage(combined) || "el modelo configurado ya no está disponible",
|
|
97
|
+
recoverable: true,
|
|
98
|
+
};
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
// 4. RATE_LIMIT con cooldown. Thresholds:
|
|
84
102
|
// cooldown > 7d → MONTHLY (Anthropic Agent SDK $200/mes desde jun-2026)
|
|
85
103
|
// cooldown > 1h → DAILY (Claude Pro daily, OpenAI rate-limit-by-day)
|
|
86
104
|
// cooldown <= 1h → SHORT (rate limit transitorio)
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Recovery for a command that runs once (KJC-BUG-0194).
|
|
3
|
+
*
|
|
4
|
+
* The pipeline can afford to wait: a quota wall means hibernate, persist the
|
|
5
|
+
* session and resume when the window reopens. A one-shot command cannot.
|
|
6
|
+
* Sleeping five hours inside `kj code` or `kj audit` is worse than the failure
|
|
7
|
+
* it is handling, and an MCP caller gets no answer at all.
|
|
8
|
+
*
|
|
9
|
+
* So here a quota wall takes the next declared candidate AT ONCE, and with
|
|
10
|
+
* none left it stops and says what it tried. Everything else keeps the
|
|
11
|
+
* pipeline's behaviour: a retired model still moves on immediately, provider
|
|
12
|
+
* 500s still back off, an auth failure still aborts.
|
|
13
|
+
*/
|
|
14
|
+
import { DEFAULT_RECOVERY_POLICY } from "./with-brain-recovery.js";
|
|
15
|
+
import { ERROR_CLASS } from "./agent-error-classifier.js";
|
|
16
|
+
|
|
17
|
+
const takeTheNextOne = { mode: "abort", maxRetries: 0, fallbackEligible: true, fallbackImmediate: true };
|
|
18
|
+
|
|
19
|
+
export const ONE_SHOT_POLICY = Object.freeze({
|
|
20
|
+
...DEFAULT_RECOVERY_POLICY,
|
|
21
|
+
classes: {
|
|
22
|
+
...DEFAULT_RECOVERY_POLICY.classes,
|
|
23
|
+
[ERROR_CLASS.QUOTA_EXHAUSTED_DAILY]: takeTheNextOne,
|
|
24
|
+
[ERROR_CLASS.QUOTA_EXHAUSTED_MONTHLY]: takeTheNextOne,
|
|
25
|
+
},
|
|
26
|
+
});
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The declared chain, finally connected (KJC-TSK-0859).
|
|
3
|
+
*
|
|
4
|
+
* `roles.<role>.fallback` has been in the config schema since KJC-TSK-0415,
|
|
5
|
+
* recursive and carrying a `model`, and `withBrainRecovery` has always known
|
|
6
|
+
* how to walk it. Nothing ever built it: of the call sites that recover an
|
|
7
|
+
* agent run, not one passed `fallback`. A wire laid and never connected.
|
|
8
|
+
*
|
|
9
|
+
* This builds it from config, one instantiated agent per link, so the user's
|
|
10
|
+
* declared order is what runs. Each link may change the model, the provider or
|
|
11
|
+
* both: a chain within one provider is the case the card asks for, and
|
|
12
|
+
* crossing providers falls out for free.
|
|
13
|
+
*/
|
|
14
|
+
import { createAgent as defaultCreateAgent } from "../agents/index.js";
|
|
15
|
+
import { resolveRole } from "../config/role-resolver.js";
|
|
16
|
+
|
|
17
|
+
/** A chain longer than this is a config smell, not a plan. */
|
|
18
|
+
export const MAX_CHAIN_LINKS = 5;
|
|
19
|
+
|
|
20
|
+
/** The config a link runs with: same project, this link's provider and model. */
|
|
21
|
+
export function withRoleModel(config, role, provider, model) {
|
|
22
|
+
return {
|
|
23
|
+
...config,
|
|
24
|
+
roles: { ...config?.roles, [role]: { ...config?.roles?.[role], provider, model: model ?? null } },
|
|
25
|
+
};
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
const keyOf = (provider, model) => `${provider}::${model ?? ""}`;
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* @param {{config: object, role: string, agentMethod?: string,
|
|
32
|
+
* createAgentFn?: Function, logger?: object}} args
|
|
33
|
+
* @returns {null|{agent: object, provider: string, model: string|null,
|
|
34
|
+
* maxWaitHours: number|undefined, fallback: object|null}}
|
|
35
|
+
*/
|
|
36
|
+
export function buildRoleFallbackChain({ config, role, agentMethod = "runTask", createAgentFn = defaultCreateAgent, logger = null }) {
|
|
37
|
+
const primary = resolveRole(config, role);
|
|
38
|
+
// The primary is already in play: a link repeating it would retry the same
|
|
39
|
+
// thing and look like progress.
|
|
40
|
+
const seen = new Set([keyOf(primary.provider, primary.model)]);
|
|
41
|
+
|
|
42
|
+
const build = (link, depth) => {
|
|
43
|
+
if (!link?.provider) return null;
|
|
44
|
+
if (depth > MAX_CHAIN_LINKS) {
|
|
45
|
+
logger?.warn?.(`[brain] ${role}: fallback chain longer than ${MAX_CHAIN_LINKS} links — the rest is ignored`);
|
|
46
|
+
return null;
|
|
47
|
+
}
|
|
48
|
+
const key = keyOf(link.provider, link.model);
|
|
49
|
+
if (seen.has(key)) {
|
|
50
|
+
logger?.warn?.(`[brain] ${role}: fallback link ${key} repeats one already in the chain — skipped`);
|
|
51
|
+
return build(link.fallback, depth + 1);
|
|
52
|
+
}
|
|
53
|
+
seen.add(key);
|
|
54
|
+
let agent;
|
|
55
|
+
try {
|
|
56
|
+
agent = createAgentFn(link.provider, withRoleModel(config, role, link.provider, link.model), logger);
|
|
57
|
+
} catch (err) {
|
|
58
|
+
// An unknown provider in the chain is the user's typo, not a reason to
|
|
59
|
+
// lose the links behind it. Said out loud, then skipped.
|
|
60
|
+
logger?.warn?.(`[brain] ${role}: fallback link ${key} unusable (${err.message}) — skipped`);
|
|
61
|
+
return build(link.fallback, depth + 1);
|
|
62
|
+
}
|
|
63
|
+
return {
|
|
64
|
+
agent: { runTask: (args) => agent[agentMethod](args), provider: link.provider, model: link.model ?? null },
|
|
65
|
+
provider: link.provider,
|
|
66
|
+
model: link.model ?? null,
|
|
67
|
+
maxWaitHours: link.max_wait_hours,
|
|
68
|
+
fallback: build(link.fallback, depth + 1),
|
|
69
|
+
};
|
|
70
|
+
};
|
|
71
|
+
|
|
72
|
+
const declared = build(config?.roles?.[role]?.fallback, 1);
|
|
73
|
+
|
|
74
|
+
// KJC-TSK-0859: the provider's own default model is the last resort, and it
|
|
75
|
+
// always was — every agent kept a private retry that dropped the pinned
|
|
76
|
+
// model and tried again. Making it the chain's tail puts that step where the
|
|
77
|
+
// user can see it, after the candidates they declared and never before.
|
|
78
|
+
if (!primary.model) return declared;
|
|
79
|
+
const tailKey = keyOf(primary.provider, null);
|
|
80
|
+
if (seen.has(tailKey)) return declared;
|
|
81
|
+
seen.add(tailKey);
|
|
82
|
+
let tailAgent;
|
|
83
|
+
try {
|
|
84
|
+
tailAgent = createAgentFn(primary.provider, withRoleModel(config, role, primary.provider, null), logger);
|
|
85
|
+
} catch {
|
|
86
|
+
return declared; // the primary provider is already in use; if it cannot be
|
|
87
|
+
} // built again, the declared chain is what we have.
|
|
88
|
+
const tail = {
|
|
89
|
+
agent: { runTask: (args) => tailAgent[agentMethod](args), provider: primary.provider, model: null },
|
|
90
|
+
provider: primary.provider,
|
|
91
|
+
model: null,
|
|
92
|
+
maxWaitHours: undefined,
|
|
93
|
+
fallback: null,
|
|
94
|
+
};
|
|
95
|
+
if (!declared) return tail;
|
|
96
|
+
let last = declared;
|
|
97
|
+
while (last.fallback) last = last.fallback;
|
|
98
|
+
last.fallback = tail;
|
|
99
|
+
return declared;
|
|
100
|
+
}
|
|
@@ -7,6 +7,7 @@
|
|
|
7
7
|
// puede ignorarlo (fallback a long standby capped) hasta que llegue 0414.
|
|
8
8
|
|
|
9
9
|
import { classifyAgentError, ERROR_CLASS } from "./agent-error-classifier.js";
|
|
10
|
+
import { recordDeadModel } from "../agents/dead-models.js";
|
|
10
11
|
import { persistStandby, markStandbyDone } from "./standby-store.js";
|
|
11
12
|
|
|
12
13
|
const ONE_MIN = 60 * 1000;
|
|
@@ -31,6 +32,10 @@ export const DEFAULT_RECOVERY_POLICY = Object.freeze({
|
|
|
31
32
|
[ERROR_CLASS.RATE_LIMIT_SHORT]: { mode: "standby", maxRetries: 3 },
|
|
32
33
|
[ERROR_CLASS.QUOTA_EXHAUSTED_DAILY]: { mode: "hibernate", maxRetries: 1, fallbackEligible: true },
|
|
33
34
|
[ERROR_CLASS.QUOTA_EXHAUSTED_MONTHLY]: { mode: "hibernate", maxRetries: 1, fallbackEligible: true },
|
|
35
|
+
// KJC-TSK-0859: a retired model has no cooldown to wait out, so it takes
|
|
36
|
+
// the next link straight away. Without a chain it aborts, because
|
|
37
|
+
// retrying the same dead model is how a run burns quota saying nothing.
|
|
38
|
+
[ERROR_CLASS.MODEL_UNAVAILABLE]: { mode: "abort", maxRetries: 0, fallbackEligible: true, fallbackImmediate: true },
|
|
34
39
|
[ERROR_CLASS.API_DOWN]: { mode: "backoff", maxRetries: 3, baseMs: 5_000, factor: 3, jitterPct: 0.2 },
|
|
35
40
|
[ERROR_CLASS.NETWORK_TIMEOUT]: { mode: "backoff", maxRetries: 3, baseMs: 5_000, factor: 3, jitterPct: 0.2 },
|
|
36
41
|
[ERROR_CLASS.SILENCED]: { mode: "backoff", maxRetries: 2, baseMs: 30_000, factor: 2, jitterPct: 0.15 },
|
|
@@ -88,14 +93,38 @@ export async function withBrainRecovery({
|
|
|
88
93
|
// > maxWaitHours (default 12h) y hay fallback, switch en vez de hibernar.
|
|
89
94
|
fallback = null,
|
|
90
95
|
}) {
|
|
96
|
+
// The substitution is only useful if it names WHICH model died and which
|
|
97
|
+
// one took over, and agents expose it under different keys.
|
|
98
|
+
const modelOf = (a) => a?.model || a?.config?.model || null;
|
|
99
|
+
// Retrying the same agent is one entry with a count, not four lines: the
|
|
100
|
+
// order the user needs is the order of CANDIDATES, not of attempts.
|
|
101
|
+
const noteAttempt = (list, entry) => {
|
|
102
|
+
const last = list.at(-1);
|
|
103
|
+
if (last && last.provider === entry.provider && last.model === entry.model && last.class === entry.class) {
|
|
104
|
+
last.attempts += 1;
|
|
105
|
+
return;
|
|
106
|
+
}
|
|
107
|
+
list.push({ ...entry, attempts: 1 });
|
|
108
|
+
};
|
|
109
|
+
const formatTried = (list) =>
|
|
110
|
+
list
|
|
111
|
+
.map((t) => `${t.provider}${t.model ? ` (${t.model})` : ""} → ${t.class}${t.attempts > 1 ? ` x${t.attempts}` : ""}`)
|
|
112
|
+
.join("; ");
|
|
91
113
|
let effectiveAgent = agent;
|
|
92
114
|
let effectiveProvider = provider || agent?.provider || "unknown";
|
|
93
115
|
let effectiveFallback = fallback;
|
|
94
116
|
const attemptsByClass = {};
|
|
117
|
+
// KJC-TSK-0859: attemptsByClass counts, it does not remember. With a chain
|
|
118
|
+
// exhausted the user got "the work failed" and no idea what had been tried,
|
|
119
|
+
// so every failure also lands here, in order.
|
|
120
|
+
const tried = [];
|
|
95
121
|
|
|
96
122
|
while (true) {
|
|
97
123
|
const result = await effectiveAgent.runTask(taskArgs);
|
|
98
|
-
|
|
124
|
+
// KJC-TSK-0873: quien HA CORRIDO viaja con el resultado. Tras una caída al
|
|
125
|
+
// fallback, el llamante creía que había escrito el provider declarado, y
|
|
126
|
+
// con eso el recuento del panel contaba trabajo ajeno como propio.
|
|
127
|
+
if (result?.ok) return { ...result, provider: effectiveProvider };
|
|
99
128
|
|
|
100
129
|
const cls = classifyAgentError({
|
|
101
130
|
provider: effectiveProvider,
|
|
@@ -104,16 +133,46 @@ export async function withBrainRecovery({
|
|
|
104
133
|
exitCode: result?.exitCode ?? null,
|
|
105
134
|
});
|
|
106
135
|
const classPolicy = policy.classes[cls.class] || { mode: "abort", maxRetries: 0 };
|
|
136
|
+
const failedModel = modelOf(effectiveAgent);
|
|
137
|
+
noteAttempt(tried, { provider: effectiveProvider, model: failedModel, class: cls.class, message: cls.message });
|
|
138
|
+
// KJC-TSK-0827: remember the corpse, or the next run pays for it again.
|
|
139
|
+
// Recording is an optimisation: a read-only home just means kj forgets.
|
|
140
|
+
if (cls.class === ERROR_CLASS.MODEL_UNAVAILABLE && failedModel) {
|
|
141
|
+
recordDeadModel({ provider: effectiveProvider, model: failedModel, reason: cls.message });
|
|
142
|
+
}
|
|
107
143
|
attemptsByClass[cls.class] = (attemptsByClass[cls.class] || 0) + 1;
|
|
108
144
|
const attempt = attemptsByClass[cls.class];
|
|
109
145
|
|
|
110
146
|
emit(emitter, "brain:agent-error", eventBase, { role, class: cls.class, attempt, message: cls.message, provider: effectiveProvider });
|
|
111
147
|
logger?.warn?.(`[brain] ${role} (${effectiveProvider}) → ${cls.class} attempt ${attempt}/${classPolicy.maxRetries}: ${cls.message}`);
|
|
112
148
|
|
|
149
|
+
// KJC-TSK-0859: un modelo retirado se resuelve ANTES de abortar. No hay
|
|
150
|
+
// cooldown que esperar, así que la condición de la cadena por cuota
|
|
151
|
+
// (retryAfter > maxWait) no aplica: si hay eslabón siguiente, se toma.
|
|
152
|
+
if (classPolicy.fallbackImmediate && effectiveFallback?.agent) {
|
|
153
|
+
const from = `${effectiveProvider}${modelOf(effectiveAgent) ? ` (${modelOf(effectiveAgent)})` : ""}`;
|
|
154
|
+
const to = effectiveFallback.provider || effectiveFallback.agent.provider || "unknown";
|
|
155
|
+
const toModel = effectiveFallback.model || modelOf(effectiveFallback.agent);
|
|
156
|
+
emit(emitter, "brain:fallback-switched", eventBase, { role, from: effectiveProvider, to, class: cls.class, immediate: true });
|
|
157
|
+
logger?.warn?.(`[brain] ${role}: ${cls.message} — substituting ${from} with ${to}${toModel ? ` (${toModel})` : " (provider default)"}. Pin it in kj.config.yml under roles.${role} to stop the substitution.`);
|
|
158
|
+
effectiveAgent = effectiveFallback.agent;
|
|
159
|
+
effectiveProvider = to;
|
|
160
|
+
effectiveFallback = effectiveFallback.fallback || null;
|
|
161
|
+
for (const k of Object.keys(attemptsByClass)) attemptsByClass[k] = 0;
|
|
162
|
+
continue;
|
|
163
|
+
}
|
|
164
|
+
|
|
113
165
|
// ABORT: no recuperable.
|
|
114
166
|
if (classPolicy.mode === "abort" || attempt > classPolicy.maxRetries) {
|
|
115
|
-
emit(emitter, "brain:fatal", eventBase, { role, class: cls.class, message: cls.message });
|
|
116
|
-
return {
|
|
167
|
+
emit(emitter, "brain:fatal", eventBase, { role, class: cls.class, message: cls.message, tried });
|
|
168
|
+
return {
|
|
169
|
+
ok: false, action: "abort", recovery: cls, tried,
|
|
170
|
+
// With a chain behind us the LAST error explains nothing: the user
|
|
171
|
+
// needs the order, which is also what tells them where to look.
|
|
172
|
+
error: tried.length > 1
|
|
173
|
+
? `Brain aborted for ${role} after trying, in order: ${formatTried(tried)} — last error: ${cls.message}`
|
|
174
|
+
: `Brain aborted: ${cls.class} — ${cls.message}`,
|
|
175
|
+
};
|
|
117
176
|
}
|
|
118
177
|
|
|
119
178
|
// KJC-TSK-0415: ¿switch a fallback antes de hibernar?
|
|
@@ -170,7 +229,7 @@ export async function withBrainRecovery({
|
|
|
170
229
|
if (tooLongToWait) {
|
|
171
230
|
logger?.info?.(`[brain] hibernated session ${sessionState?.sessionId ?? "(no id)"} → ${standbyFile} (resume at ${cls.retryUntil}; wait > ${Math.round(maxWaitMs/ONE_HOUR)}h, exiting)`);
|
|
172
231
|
emit(emitter, "brain:hibernate-request", eventBase, { role, class: cls.class, retryUntil: cls.retryUntil, retryAfter: cls.retryAfter, standbyFile });
|
|
173
|
-
return { ok: false, action: "hibernate", recovery: cls, standbyFile };
|
|
232
|
+
return { ok: false, action: "hibernate", recovery: cls, standbyFile, tried };
|
|
174
233
|
}
|
|
175
234
|
|
|
176
235
|
// Standby-in-process: keep kj alive, sleep until the cooldown,
|