@sagentlab/navarch-runtime 0.1.38 → 0.1.39

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -473,6 +473,39 @@ account's selection.
473
473
  Machine-wide extra arguments still configure other CLI behavior; dispatched
474
474
  model policy wins.
475
475
 
476
+ Codex defaults to `gpt-6-astra`; Claude Code defaults to `claude-fable-5-1`.
477
+ Explicit project and `steering/navarch.yaml` model choices still win. A task's
478
+ explicit execution profile wins over the repository's per-task-type profile.
479
+ Otherwise, claim-time classification uses the task summary and metadata:
480
+
481
+ - Clearly scoped typo, spelling, link, wording, or label fixes use `fast` / `low`.
482
+ - Planning, retries, video work, tasks with at least three dependencies,
483
+ production operations, and descriptions indicating architectural or sensitive
484
+ changes use `complex` / `high`.
485
+ - Ambiguous work uses `standard` / the project effort baseline (`medium` by default).
486
+
487
+ The classifier is deterministic and conservative; it never chooses `critical`
488
+ or `deep` automatically. Set an explicit profile to override its assessment.
489
+ Existing explicit task profiles remain explicit.
490
+
491
+ Fable sessions pass `--fallback-model claude-opus-5` for Claude Code's native
492
+ [availability fallback](https://code.claude.com/docs/en/model-config#fallback-model-chains).
493
+ Because native fallback excludes usage limits, a structured 429 usage-limit
494
+ rejection gets one additional Opus attempt within the original timeout. It
495
+ resumes the saved conversation, or starts afresh only for a confirmed zero-turn
496
+ rejection. Missing transcripts after possible work, authentication/policy
497
+ errors, cancellation, and explicit fallback/session arguments do not trigger
498
+ that retry. If Opus is also limited, the usual machine cooldown applies.
499
+ The runtime records the fallback in the transcript and reports Opus for the
500
+ retry and subsequent turns. Native multi-model turns still report the requested
501
+ primary model; their aggregate provider-reported usage includes both models.
502
+ Requires a Claude Code version supporting Fable 5.1 (v2.1.255 or later).
503
+
504
+ Astra token cost reporting uses a standard short-context API-equivalent
505
+ estimate from [OpenAI pricing](https://developers.openai.com/api/docs/pricing),
506
+ not the account's subscription charge or long-context/fast-mode uplifts.
507
+
508
+
476
509
  The Codex CLI invocation was verified against `codex-cli 0.144.1` on
477
510
  2026-07-18. The runtime uses `codex exec "<prompt>" --json` and translates
478
511
  the existing per-session MCP JSON into one-off `-c mcp_servers.*` overrides.
@@ -4,6 +4,12 @@ exports.claudeCodeAdapter = void 0;
4
4
  exports.runClaudeCodeAdapter = runClaudeCodeAdapter;
5
5
  const node_child_process_1 = require("node:child_process");
6
6
  const exit_conditions_cjs_1 = require("../exit-conditions.cjs");
7
+ const adapter_capacity_cjs_1 = require("../adapter-capacity.cjs");
8
+ const DEFAULT_CLAUDE_MODEL = "claude-fable-5-1";
9
+ const FALLBACK_CLAUDE_MODEL = "claude-opus-5";
10
+ function hasFlag(args, flag) {
11
+ return args.some((arg) => arg === flag || arg.startsWith(`${flag}=`));
12
+ }
7
13
  /**
8
14
  * Headless Claude Code adapter (project-plan.md §3.9 / implementation-plan.md
9
15
  * WP-07): `claude -p "<context bundle>" --mcp-config platform-mcp.json
@@ -25,6 +31,65 @@ const exit_conditions_cjs_1 = require("../exit-conditions.cjs");
25
31
  * default) rather than throwing -- see runtime/README.md.
26
32
  */
27
33
  async function runClaudeCodeAdapter(options) {
34
+ if (options.signal?.aborted) {
35
+ return { exitCode: null, timedOut: false, killedByLeaseLoss: true, stdout: "", stderr: "" };
36
+ }
37
+ const model = options.model ?? DEFAULT_CLAUDE_MODEL;
38
+ const automaticFallback = [DEFAULT_CLAUDE_MODEL, "fable"].includes(model) &&
39
+ !hasFlag(options.extraArgs, "--fallback-model");
40
+ const startedAt = Date.now();
41
+ const primary = await runClaudeAttempt({ ...options, model }, automaticFallback);
42
+ const remainingMs = options.timeoutMs - (Date.now() - startedAt);
43
+ if (!automaticFallback || primary.timedOut || primary.killedByLeaseLoss ||
44
+ options.signal?.aborted || remainingMs <= 0)
45
+ return primary;
46
+ const parsed = (0, exit_conditions_cjs_1.parseClaudeJsonResult)(primary.stdout) ?? (0, exit_conditions_cjs_1.parseClaudeJsonResult)(primary.stderr);
47
+ // Native fallback handles overload/unavailability, but explicitly excludes
48
+ // 429 usage limits. Try Opus once; if the limit is account-wide, the ordinary
49
+ // capacity cooldown still sees the fallback's failure.
50
+ if (!parsed?.is_error || !(0, adapter_capacity_cjs_1.detectAdapterCapacityLimit)(primary))
51
+ return primary;
52
+ const canResume = typeof parsed.session_id === "string" &&
53
+ /^[0-9a-f]{8}(?:-[0-9a-f]{4}){3}-[0-9a-f]{12}$/i.test(parsed.session_id) &&
54
+ !hasFlag(options.extraArgs, "--no-session-persistence");
55
+ // Never replay a task that may already have side effects without its saved
56
+ // conversation. A structured zero-turn rejection is safe to start afresh.
57
+ if (!canResume && parsed.num_turns !== 0)
58
+ return primary;
59
+ if (["--resume", "-r", "--continue", "-c", "--session-id", "--fork-session"].some((flag) => hasFlag(options.extraArgs, flag)))
60
+ return primary;
61
+ const fallback = await runClaudeAttempt({
62
+ ...options,
63
+ model: FALLBACK_CLAUDE_MODEL,
64
+ timeoutMs: remainingMs,
65
+ prompt: canResume
66
+ ? "Continue the interrupted task from the saved conversation. Check existing work before taking further action."
67
+ : options.prompt,
68
+ extraArgs: canResume
69
+ ? [...options.extraArgs, "--resume", parsed.session_id]
70
+ : options.extraArgs,
71
+ }, false);
72
+ return {
73
+ ...fallback,
74
+ model: FALLBACK_CLAUDE_MODEL,
75
+ // Keep the provider rejection in the transcript without letting the old
76
+ // JSON result override the final attempt's exit condition/capacity signal.
77
+ stderr: [
78
+ `Navarch model fallback: ${model} -> ${FALLBACK_CLAUDE_MODEL}`,
79
+ ...primary.stderr.split("\n").map((line) => `Primary stderr: ${line}`),
80
+ ...primary.stdout.split("\n").map((line) => `Primary stdout: ${line}`),
81
+ fallback.stderr,
82
+ ].join("\n"),
83
+ tokensIn: addUsage(primary.tokensIn, fallback.tokensIn),
84
+ tokensOut: addUsage(primary.tokensOut, fallback.tokensOut),
85
+ cacheHitTokensIn: addUsage(primary.cacheHitTokensIn, fallback.cacheHitTokensIn),
86
+ costUsd: addUsage(primary.costUsd, fallback.costUsd),
87
+ };
88
+ }
89
+ function addUsage(first, second) {
90
+ return first === undefined && second === undefined ? undefined : (first ?? 0) + (second ?? 0);
91
+ }
92
+ async function runClaudeAttempt(options, automaticFallback) {
28
93
  const args = ["-p", options.prompt];
29
94
  const hasSettingSources = options.extraArgs.some((arg) => arg === "--setting-sources" || arg.startsWith("--setting-sources="));
30
95
  const hasExplicitPermissionMode = options.extraArgs.some((arg) => ["--permission-mode", "--permission-prompt-tool", "--dangerously-skip-permissions"].some((flag) => arg === flag || arg.startsWith(`${flag}=`)));
@@ -76,6 +141,8 @@ async function runClaudeCodeAdapter(options) {
76
141
  // dispatched session consistently uses the settings recorded by Navarch.
77
142
  if (options.model)
78
143
  args.push("--model", options.model);
144
+ if (automaticFallback)
145
+ args.push("--fallback-model", FALLBACK_CLAUDE_MODEL);
79
146
  if (options.reasoningEffort)
80
147
  args.push("--effort", options.reasoningEffort);
81
148
  const runOptions = options.reasoningEffort
@@ -7,14 +7,23 @@ exports.estimateCodexCostUsd = estimateCodexCostUsd;
7
7
  * usage into project spend. Keep this deliberately limited to models Navarch
8
8
  * offers rather than silently applying the wrong price to custom/gateway ids.
9
9
  *
10
- * Source (checked 2026-08-05): https://developers.openai.com/api/docs/pricing
10
+ * Source (checked 2026-09-07): https://developers.openai.com/api/docs/pricing
11
11
  */
12
12
  const CODEX_RATES = {
13
+ // Standard short-context estimate, verified 2026-09-07 against the pricing
14
+ // source above. As with other runtime estimates, this excludes tier uplifts.
15
+ "gpt-6-astra": {
16
+ input: 10,
17
+ cachedInput: 1,
18
+ cacheWrite: 12.5,
19
+ output: 50,
20
+ },
21
+ // Promotional standard rates, available at least through 2026-11-21.
13
22
  "gpt-5.6-sol": {
14
- input: 5,
15
- cachedInput: 0.5,
16
- cacheWrite: 6.25,
17
- output: 30,
23
+ input: 4,
24
+ cachedInput: 0.4,
25
+ cacheWrite: 5,
26
+ output: 20,
18
27
  },
19
28
  };
20
29
  function nonnegative(value) {
package/dist/session.cjs CHANGED
@@ -64,12 +64,12 @@ async function runSession(deps, claimed, sessionId) {
64
64
  const execution = bundle.execution ?? {
65
65
  profile: task.execution_profile ?? "standard",
66
66
  model: runtime === "codex"
67
- ? "gpt-5.6-sol"
67
+ ? "gpt-6-astra"
68
68
  : runtime === "gemini"
69
69
  ? "auto"
70
70
  : runtime === "opencode"
71
71
  ? "default"
72
- : "best",
72
+ : "claude-fable-5-1",
73
73
  reasoning_effort: "medium",
74
74
  };
75
75
  const crashDetail = describeCompletionError(err).slice(0, 1800);
@@ -102,12 +102,12 @@ async function runClaimedSession(deps, claimed, sessionId, lifecycle) {
102
102
  const execution = bundle.execution ?? {
103
103
  profile: task.execution_profile ?? "standard",
104
104
  model: runtime === "codex"
105
- ? "gpt-5.6-sol"
105
+ ? "gpt-6-astra"
106
106
  : runtime === "gemini"
107
107
  ? "auto"
108
108
  : runtime === "opencode"
109
109
  ? "default"
110
- : "best",
110
+ : "claude-fable-5-1",
111
111
  reasoning_effort: "medium",
112
112
  };
113
113
  // Isolation posture reported on every completion so `sessions` records what
@@ -118,7 +118,7 @@ async function runClaimedSession(deps, claimed, sessionId, lifecycle) {
118
118
  sandbox_profile: config.sandboxMode === "docker" ? config.sandboxProfile : "host",
119
119
  ...(config.sandboxMode === "docker" ? { sandbox_image: config.dockerImage } : {}),
120
120
  };
121
- const executionReport = {
121
+ let executionReport = {
122
122
  model: execution.model,
123
123
  execution_profile: execution.profile,
124
124
  reasoning_effort: execution.reasoning_effort,
@@ -397,6 +397,10 @@ async function runClaimedSession(deps, claimed, sessionId, lifecycle) {
397
397
  signal: activeAbortController.signal,
398
398
  });
399
399
  activeAbortController = null;
400
+ if (turnResult.model) {
401
+ execution.model = turnResult.model;
402
+ executionReport = { ...executionReport, model: turnResult.model };
403
+ }
400
404
  attempts.push(turnResult);
401
405
  const capacityLimit = (0, adapter_capacity_cjs_1.detectAdapterCapacityLimit)(turnResult);
402
406
  if (capacityLimit) {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@sagentlab/navarch-runtime",
3
- "version": "0.1.38",
3
+ "version": "0.1.39",
4
4
  "description": "Navarch machine-side session manager: claims delivery tasks and runs them through Claude Code, Codex, Gemini, or OpenCode.",
5
5
  "type": "commonjs",
6
6
  "license": "MIT",