loki-mode 8.6.1 → 8.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/autonomy/run.sh CHANGED
@@ -535,6 +535,43 @@ if [ -n "${LOKI_SESSION_ID:-}" ]; then
535
535
  unset _loki_sid_raw _loki_sid_safe
536
536
  fi
537
537
 
538
+ # GENERIC TIER VOCABULARY (small|medium|high). A user should be able to ask for
539
+ # a capability class without naming a vendor model, and get that provider's
540
+ # latest model in the class. LOKI_SESSION_MODEL is the knob that already does
541
+ # this -- it accepts the raw tier names planning|development|fast alongside the
542
+ # Claude aliases -- so the generic words are normalized ONTO it here rather than
543
+ # becoming a fourth spelling. LOKI_MAX_TIER (a cost CEILING) and LOKI_TIER (the
544
+ # OSS/enterprise licensing seam) mean different things and are left alone.
545
+ #
546
+ # WHY NORMALIZE AT THE ENTRY POINT: the session-pin case block is byte-mirrored
547
+ # in the estimator (autonomy/loki) and the dashboard (dashboard/server.py), and
548
+ # every one of those mirrors is locked by a parity test. Translating here means
549
+ # they keep seeing only the three canonical tier names and none of them change.
550
+ #
551
+ # The mapping is loki_tier_alias() in providers/models.sh -- the single source
552
+ # of truth, not a second copy. Inlined as a case because run.sh must not source
553
+ # a provider file this early in startup. Kept in lockstep by
554
+ # tests/test-generic-tiers.sh.
555
+ #
556
+ # ONLY the three new words are translated. sonnet/haiku/opus/fable and the raw
557
+ # tier names pass through untouched, so an unset LOKI_SESSION_MODEL still
558
+ # defaults to sonnet and "medium" resolves to the same development-tier model
559
+ # today's builds already use. This changes no existing run's model.
560
+ # NORMALIZATION IS TRIM-ONLY + LOWERCASE, matching the estimator and dashboard
561
+ # mirrors exactly. Interior whitespace is deliberately PRESERVED, so " med ium "
562
+ # stays junk here just as it does there. Stripping interior spaces would make
563
+ # this reader accept a value the other two reject, which is the precise kind of
564
+ # divergence the session-pin parity tests exist to catch.
565
+ _loki_generic_tier="${LOKI_SESSION_MODEL:-}"
566
+ _loki_generic_tier="${_loki_generic_tier#"${_loki_generic_tier%%[![:space:]]*}"}"
567
+ _loki_generic_tier="${_loki_generic_tier%"${_loki_generic_tier##*[![:space:]]}"}"
568
+ case "$(printf '%s' "$_loki_generic_tier" | tr '[:upper:]' '[:lower:]')" in
569
+ small) LOKI_SESSION_MODEL="fast" ; export LOKI_SESSION_MODEL ;;
570
+ medium) LOKI_SESSION_MODEL="development" ; export LOKI_SESSION_MODEL ;;
571
+ high) LOKI_SESSION_MODEL="planning" ; export LOKI_SESSION_MODEL ;;
572
+ esac
573
+ unset _loki_generic_tier
574
+
538
575
  # Process Supervision (opt-in)
539
576
  WATCHDOG_ENABLED=${LOKI_WATCHDOG:-"false"} # Enable process health monitoring
540
577
  WATCHDOG_INTERVAL=${LOKI_WATCHDOG_INTERVAL:-30} # Check interval in seconds
@@ -736,6 +773,35 @@ print(catalog["providers"]["claude"]["cli_aliases"].get(os.environ["_LOKI_SELECT
736
773
  }
737
774
  loki_apply_build_profile
738
775
 
776
+ # Default hang guard for EVERY build, not just simple-web.
777
+ #
778
+ # The two timeouts above are set inside loki_apply_build_profile(), which
779
+ # returns immediately unless LOKI_BUILD_PROFILE=simple-web. So on a normal
780
+ # build both resolved to 0, and 0 means no guard at all -- verified by running
781
+ # the deadline helper directly: `deadline.py 0 0 3 -- sleep 5` runs to
782
+ # completion unkilled. A provider that hung had nothing to stop it.
783
+ #
784
+ # IDLE only, and no retry. That is what keeps this compatible with the standing
785
+ # objection recorded above (search: "former invoke_with_timeout"), whose two
786
+ # reasons remain correct:
787
+ #
788
+ # 1. "No safe generous default" applies to a fixed TOTAL timeout, which
789
+ # cannot tell a long legitimate iteration from a hang. An idle timeout
790
+ # can: it measures silence, not duration. Verified both directions --
791
+ # `sleep 600` under a 120s idle cap dies, while a process emitting output
792
+ # every second survives indefinitely. A coding agent streams constantly;
793
+ # one silent for two minutes is not working.
794
+ # 2. "Wrong retry semantics" stands, so nothing here retries. The call is
795
+ # killed, and the existing failure path handles it. Re-running an agent
796
+ # that may have already edited files remains off the table.
797
+ #
798
+ # 7200s hard ceiling is a backstop against a process that streams forever
799
+ # without converging; the idle cap is the load-bearing guard. Both are
800
+ # overridable, and setting either to 0 restores the old unguarded behaviour.
801
+ : "${LOKI_PROVIDER_IDLE_TIMEOUT:=120}"
802
+ : "${LOKI_PROVIDER_CALL_TIMEOUT:=7200}"
803
+ export LOKI_PROVIDER_IDLE_TIMEOUT LOKI_PROVIDER_CALL_TIMEOUT
804
+
739
805
  loki_background_services_enabled() {
740
806
  ! loki_is_supervised_simple_web
741
807
  }
@@ -12951,6 +13017,67 @@ with open(os.environ["LOKI_DA_PROMPT_OUT"], "w", encoding="utf-8") as handle:
12951
13017
  BUILD_DA_PROMPT
12952
13018
  }
12953
13019
 
13020
+ # Derive a review size cap (in bytes) from the active provider's context window.
13021
+ #
13022
+ # Args: $1 = env override (wins outright when set), $2 = historical default.
13023
+ # Echoes the effective cap.
13024
+ #
13025
+ # Why this exists: the caps were fixed byte counts sized for a ~200k-token model.
13026
+ # A local 12b/14b with an 8k-32k window would be handed a 425000-byte prompt and
13027
+ # fail in a way that reads as "the model is bad" rather than "we mis-sized it".
13028
+ #
13029
+ # Two stated assumptions, kept separate so they stay auditable:
13030
+ # 1. ~3 bytes per token. Real tokenizers land around 3-4 for code-heavy text;
13031
+ # 3 is the conservative end, and under-estimating capacity errs toward a
13032
+ # smaller cap, which is the safe direction for a fail-closed gate.
13033
+ # 2. ~75% of the window is available for review INPUT. The remainder is the
13034
+ # reviewer's own output and reasoning, which share the same window.
13035
+ # Neither is precise, and neither needs to be: the result is only ever used to
13036
+ # LOWER a cap below the shipped default.
13037
+ #
13038
+ # The min() is load-bearing. PROVIDER_CONTEXT_WINDOW is set on every current run
13039
+ # (LOKI_PROVIDER defaults to claude, whose window is 1000000), so deriving
13040
+ # upward would raise the cap 4-8x for every existing user. Taking the smaller of
13041
+ # derived-vs-default means a cap can only ever move DOWN. Concretely, a window
13042
+ # clamps to the shipped default whenever it is >= ~188889 tokens; every provider
13043
+ # that declares a window today (1M/400k/200k/200k) clears that, and a provider
13044
+ # declaring none takes the unset path, so no shipped provider changes behavior.
13045
+ # Only a genuinely small window (a local 12b/14b) shrinks anything.
13046
+ #
13047
+ # The shipped defaults are 425000 (prompt) and 400000 (diff). That 25000-byte gap
13048
+ # is the reviewer scaffolding wrapped around the diff, and the ordering is
13049
+ # load-bearing: if both caps derived to the SAME number, a diff sized just under
13050
+ # the diff gate would build a prompt exceeding the prompt gate, so every review
13051
+ # would block fail-closed with no operator-visible cause, on exactly the
13052
+ # small-window providers this derivation exists to support. Scaling by
13053
+ # _default/425000 preserves that gap at every window size. The 425000 denominator
13054
+ # must track the prompt-cap default passed by the caller below; if that default
13055
+ # changes, change the denominator with it or the diff-cap proportion breaks.
13056
+ review_effective_cap() {
13057
+ local _override="$1" _default="$2"
13058
+ # Operator wins outright, at any value, over both the default and the window.
13059
+ if [ -n "$_override" ]; then
13060
+ printf '%s' "$_override"
13061
+ return 0
13062
+ fi
13063
+ # Fail safe to the historical default unless the window is a clean positive
13064
+ # integer. A non-numeric result here would make the caller's `[ ... -gt ... ]`
13065
+ # exit 2, which reads as false and would dispatch an oversized review.
13066
+ case "${PROVIDER_CONTEXT_WINDOW:-}" in
13067
+ ''|*[!0-9]*) printf '%s' "$_default"; return 0 ;;
13068
+ esac
13069
+ [ "$PROVIDER_CONTEXT_WINDOW" -gt 0 ] 2>/dev/null || { printf '%s' "$_default"; return 0; }
13070
+ # Derive the INPUT budget, then scale it to this caller's cap so the
13071
+ # prompt/diff proportion (and thus the scaffolding gap) is preserved.
13072
+ local _budget=$(( PROVIDER_CONTEXT_WINDOW * 3 / 4 * 3 ))
13073
+ local _derived=$(( _budget * _default / 425000 ))
13074
+ if [ "$_derived" -lt "$_default" ]; then
13075
+ printf '%s' "$_derived"
13076
+ else
13077
+ printf '%s' "$_default"
13078
+ fi
13079
+ }
13080
+
12954
13081
  run_code_review() {
12955
13082
  local loki_dir="${TARGET_DIR:-.}/.loki"
12956
13083
  local review_dir="$loki_dir/quality/reviews"
@@ -13311,11 +13438,12 @@ ${dependency_context}"
13311
13438
  # produces an opaque block. This remains fail-closed and never truncates.
13312
13439
  local _review_diff_bytes=0
13313
13440
  _review_diff_bytes=$(printf '%s' "$diff_content" | wc -c | tr -d ' ')
13314
- local _review_max_bytes="${LOKI_REVIEW_MAX_DIFF_BYTES:-400000}"
13441
+ local _review_max_bytes
13442
+ _review_max_bytes=$(review_effective_cap "${LOKI_REVIEW_MAX_DIFF_BYTES:-}" 400000)
13315
13443
  if [ "${_review_diff_bytes:-0}" -gt "$_review_max_bytes" ] 2>/dev/null; then
13316
13444
  local _big_dirs
13317
13445
  _big_dirs=$(printf '%s\n' "$changed_files" | sed 's#/.*##' | grep -v '^$' | sort | uniq -c | sort -rn | head -3 | awk '{print $2" ("$1" files)"}' | tr '\n' ' ')
13318
- log_error "Code review: context is ${_review_diff_bytes} bytes (limit ${_review_max_bytes}); refusing to truncate or dispatch a partial review. Biggest dirs: ${_big_dirs:-unknown}. Split the change or raise LOKI_REVIEW_MAX_DIFF_BYTES."
13446
+ log_error "Code review: context is ${_review_diff_bytes} bytes (limit ${_review_max_bytes}, derived from PROVIDER_CONTEXT_WINDOW=${PROVIDER_CONTEXT_WINDOW:-unset}); refusing to truncate or dispatch a partial review. Biggest dirs: ${_big_dirs:-unknown}. Split the change or raise LOKI_REVIEW_MAX_DIFF_BYTES."
13319
13447
  emit_event_json "code_review_diff_oversized" \
13320
13448
  "review_id=$review_id" \
13321
13449
  "diff_bytes=$_review_diff_bytes" \
@@ -13847,7 +13975,8 @@ REVIEW_SELECTION_RECORD
13847
13975
  reviewer_count=$(echo "$selected_specialists" | python3 -c "import sys,json; print(len(json.load(sys.stdin)['reviewers']))")
13848
13976
  local dispatch_count
13849
13977
  dispatch_count=$(echo "$dispatch_specialists" | python3 -c "import sys,json; print(len(json.load(sys.stdin)['reviewers']))")
13850
- local _review_max_prompt_bytes="${LOKI_REVIEW_MAX_PROMPT_BYTES:-425000}"
13978
+ local _review_max_prompt_bytes
13979
+ _review_max_prompt_bytes=$(review_effective_cap "${LOKI_REVIEW_MAX_PROMPT_BYTES:-}" 425000)
13851
13980
  local _review_max_output_bytes="${LOKI_REVIEW_MAX_OUTPUT_BYTES:-1048576}"
13852
13981
  local review_pending_dir="$review_dir/$review_id/.pending"
13853
13982
  if ! mkdir -m 700 "$review_pending_dir" 2>/dev/null; then
@@ -15433,40 +15562,71 @@ start_dashboard() {
15433
15562
 
15434
15563
  # Check all required imports
15435
15564
  if ! "$python_cmd" -c "import fastapi; import sqlalchemy; import aiosqlite" 2>/dev/null; then
15436
- log_step "Setting up dashboard virtualenv..."
15437
- if ! [ -x "${dashboard_venv}/bin/python3" ]; then
15438
- # Remove broken venv if exists
15439
- [ -d "$dashboard_venv" ] && rm -rf "$dashboard_venv"
15440
- mkdir -p "$HOME/.loki"
15441
- python3 -m venv "$dashboard_venv" 2>/dev/null || python3.13 -m venv "$dashboard_venv" 2>/dev/null || {
15442
- log_warn "Failed to create virtualenv"
15443
- log_warn "You may need: sudo apt install python3-venv"
15444
- }
15445
- fi
15446
- if [ -x "${dashboard_venv}/bin/python3" ]; then
15447
- python_cmd="${dashboard_venv}/bin/python3"
15448
- log_step "Installing dashboard dependencies..."
15449
- if [ -f "$req_file" ]; then
15450
- "${dashboard_venv}/bin/pip" install -r "$req_file" 2>&1 | tail -1 || {
15451
- log_warn "Pinned deps failed, trying unpinned..."
15565
+ # The venv is HOST-GLOBAL, so concurrent runs must not rebuild it at
15566
+ # once: one run's `rm -rf` would delete the tree another is importing
15567
+ # from. Lock the venv path itself -- safe_acquire_lock appends
15568
+ # ".lockdir", giving a SIBLING mutex the teardown below cannot destroy.
15569
+ # Timeout must outlast a real cold venv create + pip install (not the 5s
15570
+ # used by the JSON read-modify-write call sites). Mirrors the same guard
15571
+ # in ensure_dashboard_venv (autonomy/loki) -- edit BOTH.
15572
+ local _venv_locked=false
15573
+ if type safe_acquire_lock >/dev/null 2>&1 \
15574
+ && safe_acquire_lock "$dashboard_venv" "${LOKI_VENV_LOCK_TIMEOUT:-300}"; then
15575
+ _venv_locked=true
15576
+ fi
15577
+ # Re-probe: the run we queued behind may have just built it for us.
15578
+ [ -x "${dashboard_venv}/bin/python3" ] && python_cmd="${dashboard_venv}/bin/python3"
15579
+ if "$python_cmd" -c "import fastapi; import sqlalchemy; import aiosqlite" 2>/dev/null; then
15580
+ [ "$_venv_locked" = true ] && safe_release_lock "$dashboard_venv"
15581
+ _venv_locked=false
15582
+ elif [ "$_venv_locked" = false ] && type safe_acquire_lock >/dev/null 2>&1; then
15583
+ # Lock timed out and the venv is still unusable. Do NOT rm -rf
15584
+ # unlocked -- that is the exact race this lock exists to prevent.
15585
+ #
15586
+ # The re-probe above may have pointed python_cmd at the OTHER run's
15587
+ # half-built venv (bin/python3 exists, pip install not finished).
15588
+ # Reset to the system interpreter so the server launch below does not
15589
+ # exec a knowingly-broken one.
15590
+ python_cmd="python3"
15591
+ log_warn "Timed out waiting for another run to build the dashboard venv"
15592
+ log_warn "Dashboard will not be available (stale lock? rm -rf ${dashboard_venv}.lockdir)"
15593
+ else
15594
+ log_step "Setting up dashboard virtualenv..."
15595
+ if ! [ -x "${dashboard_venv}/bin/python3" ]; then
15596
+ # Remove broken venv if exists
15597
+ [ -d "$dashboard_venv" ] && rm -rf "$dashboard_venv"
15598
+ mkdir -p "$HOME/.loki"
15599
+ python3 -m venv "$dashboard_venv" 2>/dev/null || python3.13 -m venv "$dashboard_venv" 2>/dev/null || {
15600
+ log_warn "Failed to create virtualenv"
15601
+ log_warn "You may need: sudo apt install python3-venv"
15602
+ }
15603
+ fi
15604
+ if [ -x "${dashboard_venv}/bin/python3" ]; then
15605
+ python_cmd="${dashboard_venv}/bin/python3"
15606
+ log_step "Installing dashboard dependencies..."
15607
+ if [ -f "$req_file" ]; then
15608
+ "${dashboard_venv}/bin/pip" install -r "$req_file" 2>&1 | tail -1 || {
15609
+ log_warn "Pinned deps failed, trying unpinned..."
15610
+ "${dashboard_venv}/bin/pip" install fastapi uvicorn pydantic websockets sqlalchemy aiosqlite 2>&1 | tail -1 || {
15611
+ log_warn "Failed to install dashboard dependencies"
15612
+ log_warn "Dashboard will not be available"
15613
+ }
15614
+ # greenlet is optional (needs C compiler on some platforms)
15615
+ "${dashboard_venv}/bin/pip" install greenlet 2>/dev/null || true
15616
+ }
15617
+ else
15452
15618
  "${dashboard_venv}/bin/pip" install fastapi uvicorn pydantic websockets sqlalchemy aiosqlite 2>&1 | tail -1 || {
15453
15619
  log_warn "Failed to install dashboard dependencies"
15454
15620
  log_warn "Dashboard will not be available"
15455
15621
  }
15456
- # greenlet is optional (needs C compiler on some platforms)
15457
15622
  "${dashboard_venv}/bin/pip" install greenlet 2>/dev/null || true
15458
- }
15623
+ fi
15459
15624
  else
15460
- "${dashboard_venv}/bin/pip" install fastapi uvicorn pydantic websockets sqlalchemy aiosqlite 2>&1 | tail -1 || {
15461
- log_warn "Failed to install dashboard dependencies"
15462
- log_warn "Dashboard will not be available"
15463
- }
15464
- "${dashboard_venv}/bin/pip" install greenlet 2>/dev/null || true
15625
+ log_warn "Failed to install dashboard dependencies"
15626
+ log_warn "Run manually: python3 -m venv ${dashboard_venv} && ${dashboard_venv}/bin/pip install fastapi uvicorn sqlalchemy aiosqlite"
15465
15627
  fi
15466
- else
15467
- log_warn "Failed to install dashboard dependencies"
15468
- log_warn "Run manually: python3 -m venv ${dashboard_venv} && ${dashboard_venv}/bin/pip install fastapi uvicorn sqlalchemy aiosqlite"
15469
15628
  fi
15629
+ [ "$_venv_locked" = true ] && safe_release_lock "$dashboard_venv"
15470
15630
  fi
15471
15631
 
15472
15632
  # Start the FastAPI dashboard server
@@ -20750,6 +20910,16 @@ except Exception as exc:
20750
20910
  # Trim to last 500KB
20751
20911
  tail -c 500000 "$agent_log" > "$agent_log.tmp" && mv "$agent_log.tmp" "$agent_log"
20752
20912
  fi
20913
+
20914
+ # Same cap on the daily log. agent.log has been trimmed since it was
20915
+ # introduced; its sibling never was, and it receives the full raw
20916
+ # stream-json of every iteration -- measured ~1.5MB per iteration, so a
20917
+ # 500-iteration run leaves ~725MB per day per build, times however many
20918
+ # builds share the machine. Same threshold, same trim, no new rotation
20919
+ # scheme.
20920
+ if [ -f "$log_file" ] && [ "$(stat -f%z "$log_file" 2>/dev/null || stat -c%s "$log_file" 2>/dev/null)" -gt 1000000 ]; then
20921
+ tail -c 500000 "$log_file" > "$log_file.tmp" && mv "$log_file.tmp" "$log_file"
20922
+ fi
20753
20923
  touch "$agent_log"
20754
20924
  echo "" >> "$agent_log"
20755
20925
  echo "════════════════════════════════════════════════════════════════" >> "$agent_log"
@@ -7,7 +7,7 @@ Modules:
7
7
  control: Session control API (start/stop/pause/resume)
8
8
  """
9
9
 
10
- __version__ = "8.6.1"
10
+ __version__ = "8.8.1"
11
11
 
12
12
  # Expose the control app for easy import
13
13
  try:
@@ -678,6 +678,37 @@ async def _push_loki_state_loop() -> None:
678
678
  except (json.JSONDecodeError, KeyError):
679
679
  pass
680
680
 
681
+ # Third source: the .loki/pids/ registry, which a
682
+ # CLI-started background run DOES write. Without this
683
+ # the dashboard reported STOPPED for a healthy build:
684
+ # `loki start` writes neither loki.pid nor session.json
685
+ # (run.sh only UPDATES session.json when it already
686
+ # exists), so both checks above failed and every such
687
+ # run fell through to "stopped" while it was actively
688
+ # working. Confirmed against a live build: STATUS.txt
689
+ # said BUILDING and iterations were advancing while the
690
+ # dashboard showed STOPPED with 0 agents.
691
+ #
692
+ # Liveness is proven with os.kill(pid, 0), never by the
693
+ # file's presence -- a stale entry from a crashed run
694
+ # must NOT read as alive, which is the same
695
+ # anti-stale rule BUG-NEW-006 established above.
696
+ if not _pid_alive:
697
+ try:
698
+ _pid_dir = loki_dir / "pids"
699
+ for _entry in _pid_dir.glob("*.json"):
700
+ _rec = _safe_json_read(_entry, {})
701
+ if _rec.get("kind") not in ("wrapper", "runner"):
702
+ continue
703
+ try:
704
+ os.kill(int(_rec.get("pid", 0)), 0)
705
+ except (ValueError, OSError, ProcessLookupError):
706
+ continue
707
+ _pid_alive = True
708
+ break
709
+ except OSError:
710
+ pass
711
+
681
712
  status_str = raw.get("mode", "autonomous")
682
713
  # Control files are the AUTHORITY, and they are checked
683
714
  # first. dashboard-state.json's "mode" is written by the
@@ -2802,16 +2833,153 @@ _SESSION_MODEL_ALLOWLIST = ("haiku", "sonnet", "opus", "fable")
2802
2833
  # allowlist is unchanged) because it is an explicit live-run control.
2803
2834
  _START_MODEL_ALLOWLIST = ("haiku", "sonnet", "opus")
2804
2835
 
2836
+ # Provider-agnostic capability tiers. These are the vocabulary the picker offers
2837
+ # on a non-Claude provider, and they resolve per-provider through
2838
+ # providers/models.sh (loki_tier_alias): small -> fast, medium -> development,
2839
+ # high -> planning. Accepting them here is what makes the start-time picker work
2840
+ # on codex at all -- the Claude aliases above are meaningless there, so before
2841
+ # this every value a codex user could pick normalized to "" and was silently
2842
+ # dropped, and the run started on the provider default with no feedback.
2843
+ _START_MODEL_GENERIC_TIERS = ("small", "medium", "high")
2844
+
2805
2845
 
2806
2846
  def _normalize_start_model(raw: str | None) -> str:
2807
- """Normalize a start-time model / advisor alias (haiku|sonnet|opus, no fable).
2847
+ """Normalize a start-time model / advisor alias.
2808
2848
 
2809
- Same trim + lowercase + exact-match rule as _normalize_session_model, but on
2810
- the narrower _START_MODEL_ALLOWLIST. Returns "" for absent/invalid/fable so
2849
+ Accepts the Claude aliases (haiku|sonnet|opus, no fable) and the generic
2850
+ capability tiers (small|medium|high). Returns "" for absent/invalid/fable so
2811
2851
  callers can treat empty as "no selection" (engine uses its own default).
2852
+
2853
+ fable stays excluded: it is advisory-only and the runner collapses it to
2854
+ opus, so offering it as a start-time execution model would be a cost
2855
+ surprise. That reasoning is unchanged by adding the generic tiers.
2812
2856
  """
2813
2857
  val = (raw or "").strip().lower()
2814
- return val if val in _START_MODEL_ALLOWLIST else ""
2858
+ if val in _START_MODEL_ALLOWLIST or val in _START_MODEL_GENERIC_TIERS:
2859
+ return val
2860
+ return ""
2861
+
2862
+
2863
+ # =============================================================================
2864
+ # Provider-aware model offer set
2865
+ # =============================================================================
2866
+ # The two allowlists above are WIRE values: what run.sh will actually honor in
2867
+ # .loki/state/model-override. They are Claude aliases because run.sh:20996 gates
2868
+ # the whole override block on PROVIDER_NAME=claude and feeds the file straight
2869
+ # into `claude --model`. They must not change.
2870
+ #
2871
+ # What the dashboard OFFERS is a separate question, and it was the bug: the
2872
+ # picker rendered those four Claude aliases on every run, so a codex session was
2873
+ # offered Haiku/Sonnet/Opus/Fable, none of which codex can dispatch. The offer
2874
+ # set below is derived from the RUNNING session's provider plus the canonical
2875
+ # providers/model_catalog.json, so the picker never names a model the active
2876
+ # provider cannot run.
2877
+
2878
+ # Generic tier -> catalog key. These three tier names are provider-independent
2879
+ # (every catalog entry carries latest_fast/development/planning), which is what
2880
+ # makes the picker portable across providers.
2881
+ _TIER_LABELS = (
2882
+ ("small", "fast"),
2883
+ ("medium", "development"),
2884
+ ("high", "planning"),
2885
+ )
2886
+
2887
+
2888
+ def _active_provider() -> str:
2889
+ """The provider the CURRENT run is executing on.
2890
+
2891
+ Resolution order mirrors the CLI (autonomy/loki:5142): the per-project state
2892
+ file run.sh writes at launch (run.sh:1458), then the environment, then the
2893
+ stock default. The state file wins because it is the only source that
2894
+ reflects the live run rather than the dashboard process's own environment.
2895
+ """
2896
+ try:
2897
+ p = _get_loki_dir() / "state" / "provider"
2898
+ if p.is_file():
2899
+ val = p.read_text().strip().lower()
2900
+ if val:
2901
+ return val
2902
+ except OSError:
2903
+ pass
2904
+ return (os.environ.get("LOKI_PROVIDER") or "claude").strip().lower() or "claude"
2905
+
2906
+
2907
+ def _load_model_catalog() -> dict:
2908
+ """Read providers/model_catalog.json, the single source of truth for model ids.
2909
+
2910
+ Same candidate paths as GET /api/providers/models. Returns {} when the
2911
+ catalog is unreadable; every caller degrades to "no model ids to show"
2912
+ rather than inventing one.
2913
+ """
2914
+ for path in (
2915
+ _Path(__file__).resolve().parent.parent / "providers" / "model_catalog.json",
2916
+ _Path("providers/model_catalog.json"),
2917
+ ):
2918
+ try:
2919
+ if path.exists():
2920
+ with path.open("r", encoding="utf-8") as fh:
2921
+ return json.load(fh)
2922
+ except (json.JSONDecodeError, OSError):
2923
+ continue
2924
+ return {}
2925
+
2926
+
2927
+ def _resolve_catalog_model(provider: str, catalog_tier: str) -> str:
2928
+ """The model id `provider` dispatches for `catalog_tier` (fast/development/planning).
2929
+
2930
+ Python mirror of loki_latest_model (providers/models.sh:22), including its
2931
+ env-override chain and its "generic" registry fallback for a provider the
2932
+ catalog does not name. Kept in Python rather than shelling out to models.sh:
2933
+ the dashboard answers this per request and a subprocess per tier per poll is
2934
+ not worth it. Model IDS still come only from the catalog, never from here.
2935
+ """
2936
+ provider_env = re.sub(r"[^A-Z0-9_]", "_", provider.upper())
2937
+ for var in (
2938
+ f"LOKI_{provider_env}_MODEL_{catalog_tier.upper()}",
2939
+ f"LOKI_{provider_env}_MODEL",
2940
+ ):
2941
+ val = (os.environ.get(var) or "").strip()
2942
+ if val:
2943
+ return val
2944
+ providers = _load_model_catalog().get("providers", {})
2945
+ entry = providers.get(provider) or providers.get("generic") or {}
2946
+ return str(entry.get(f"latest_{catalog_tier}") or "")
2947
+
2948
+
2949
+ def _provider_model_offers(provider: str) -> list[dict]:
2950
+ """The model choices to OFFER for `provider`, each with what it resolves to.
2951
+
2952
+ Claude keeps its established alias picker byte-for-byte: those aliases are
2953
+ the values run.sh honors in the override file, so changing them would break
2954
+ the one provider where mid-run switching actually works.
2955
+
2956
+ Every other provider is offered the generic tiers (small/medium/high), which
2957
+ are provider-independent, each annotated with the concrete model id the
2958
+ catalog says that provider dispatches. That is what makes the picker read
2959
+ "medium -> gpt-5.6-terra" on codex and "medium -> claude-sonnet-5" on claude
2960
+ without the frontend knowing a single model id.
2961
+ """
2962
+ if provider == "claude":
2963
+ aliases = _load_model_catalog().get("providers", {}).get("claude", {}).get("cli_aliases", {})
2964
+ return [
2965
+ {"value": alias, "tier": None, "model": aliases.get(alias, "")}
2966
+ for alias in _SESSION_MODEL_ALLOWLIST
2967
+ ]
2968
+ return [
2969
+ {"value": tier, "tier": tier, "model": _resolve_catalog_model(provider, catalog_tier)}
2970
+ for tier, catalog_tier in _TIER_LABELS
2971
+ ]
2972
+
2973
+
2974
+ def _provider_supports_model_switch(provider: str) -> bool:
2975
+ """Whether a live run on `provider` honors .loki/state/model-override.
2976
+
2977
+ Only claude does: run.sh:20996 gates the entire override-read block on
2978
+ PROVIDER_NAME=claude. On any other provider the file is written and never
2979
+ read, so the POST path rejects rather than reporting a switch that will not
2980
+ happen.
2981
+ """
2982
+ return provider == "claude"
2815
2983
 
2816
2984
 
2817
2985
  class SessionModelRequest(BaseModel):
@@ -2890,17 +3058,28 @@ def _normalize_session_model(raw: str | None) -> str:
2890
3058
  # tier names ARE valid pins.
2891
3059
  _SESSION_PIN_ALLOWLIST = _SESSION_MODEL_ALLOWLIST + ("planning", "development", "fast")
2892
3060
 
3061
+ # Generic capability vocabulary. A user picks a CLASS of model (small/medium/
3062
+ # high) and each provider supplies its own latest model in that class, so nobody
3063
+ # has to know a vendor's model names. These are translated onto the canonical
3064
+ # tier names rather than added to the allowlist, keeping this mirror in step
3065
+ # with run.sh's entry-point case and the `loki plan` estimator without widening
3066
+ # what any of the three actually route on. Mirrors loki_tier_alias() in
3067
+ # providers/models.sh.
3068
+ _GENERIC_TIERS = {"small": "fast", "medium": "development", "high": "planning"}
3069
+
2893
3070
 
2894
3071
  def _normalize_session_pin(raw: str | None) -> str:
2895
3072
  """Normalize a LOKI_SESSION_MODEL pin value (aliases + raw tier names).
2896
3073
 
2897
3074
  Mirrors run.sh's session-pin case: trim + lowercase, accept the four model
2898
- aliases and the three tier names. Interior whitespace is preserved (so
3075
+ aliases and the three tier names, and translate the generic small/medium/
3076
+ high vocabulary onto those tier names. Interior whitespace is preserved (so
2899
3077
  "fab le" stays junk and falls through to the default tier, exactly like the
2900
3078
  runner's "*" arm). Use this for the session-pin (no-override) derivation;
2901
3079
  use _normalize_session_model for the override-file / POST path.
2902
3080
  """
2903
3081
  val = (raw or "").strip().lower()
3082
+ val = _GENERIC_TIERS.get(val, val)
2904
3083
  return val if val in _SESSION_PIN_ALLOWLIST else ""
2905
3084
 
2906
3085
 
@@ -3126,11 +3305,24 @@ async def get_session_model():
3126
3305
  # the reported effective model agrees with dispatch on BOTH routes (v7.39.1).
3127
3306
  if effective == "fable":
3128
3307
  effective = "opus"
3308
+ provider = _active_provider()
3309
+ offers = _provider_model_offers(provider)
3310
+ if provider != "claude":
3311
+ # Non-claude: the claude-alias default/effective computed above describe a
3312
+ # dispatch that is not happening on this run. Report what the provider
3313
+ # actually runs, from the catalog, and drop the stale override (run.sh
3314
+ # never reads the file on this provider, so it cannot be in effect).
3315
+ override = None
3316
+ default = "medium"
3317
+ effective = _resolve_catalog_model(provider, "development")
3129
3318
  return {
3130
3319
  "override": override,
3131
3320
  "default": default,
3132
3321
  "effective": effective,
3133
- "allowed": list(_SESSION_MODEL_ALLOWLIST),
3322
+ "provider": provider,
3323
+ "switchable": _provider_supports_model_switch(provider),
3324
+ "offers": offers,
3325
+ "allowed": [o["value"] for o in offers],
3134
3326
  }
3135
3327
 
3136
3328
 
@@ -3156,6 +3348,22 @@ async def set_session_model(request: SessionModelRequest):
3156
3348
  """
3157
3349
  requested_raw = (request.model or "").strip().lower()
3158
3350
  override_path = _model_override_path()
3351
+ # Mid-run switching is a claude-only runtime capability: run.sh:20996 gates the
3352
+ # override-read block on PROVIDER_NAME=claude, so on any other provider this
3353
+ # file would be written and never read. Reject instead of writing a file that
3354
+ # does nothing and reporting success (a false affordance is worse than no
3355
+ # control). Clearing is still allowed everywhere: removing a stale file is
3356
+ # always safe and never claims a switch.
3357
+ provider = _active_provider()
3358
+ if requested_raw != "" and not _provider_supports_model_switch(provider):
3359
+ raise HTTPException(
3360
+ status_code=409,
3361
+ detail=(
3362
+ f"Mid-run model switching is not supported on provider '{provider}'. "
3363
+ f"The run dispatches {_resolve_catalog_model(provider, 'development') or 'its configured model'}; "
3364
+ "restart the run with a different model to change it."
3365
+ ),
3366
+ )
3159
3367
  if requested_raw == "":
3160
3368
  # Clear the override; revert to tier mapping.
3161
3369
  try:
@@ -3965,19 +4173,30 @@ async def start_build(request: Request, body: StartBuildRequest):
3965
4173
  popen_env["LOKI_TARGET_DIR"] = str(workspace_dir)
3966
4174
  popen_env["LOKI_DIR"] = str(loki_dir)
3967
4175
  if start_model:
3968
- # EXACT-model pin (not the session-pin tier route): set all three tier
3969
- # models to the chosen alias so resolve_model_for_tier returns the alias
3970
- # for every tier and every iteration dispatches exactly the picked model.
3971
- # This is the honest start-time equivalent of the mid-flight override
3972
- # file, which run.sh clears at iteration 0. LOKI_SESSION_MODEL is set too
3973
- # for internal coherence (the run's own tier accounting/logging), but the
3974
- # env triple is the load-bearing dispatch-honesty mechanism: on the
3975
- # v7.104.0 stock config the session pin alone would remap opus->planning->
3976
- # sonnet and haiku->fast->sonnet, dispatching sonnet for both.
3977
- popen_env["LOKI_CLAUDE_MODEL_PLANNING"] = start_model
3978
- popen_env["LOKI_CLAUDE_MODEL_DEVELOPMENT"] = start_model
3979
- popen_env["LOKI_CLAUDE_MODEL_FAST"] = start_model
3980
- popen_env["LOKI_SESSION_MODEL"] = start_model
4176
+ if start_model in _START_MODEL_GENERIC_TIERS:
4177
+ # A generic capability tier is provider-agnostic BY CONSTRUCTION --
4178
+ # it names a capability class, not a model, and each provider
4179
+ # resolves its own latest model for that class via
4180
+ # providers/models.sh. Pinning the LOKI_CLAUDE_MODEL_* triple here
4181
+ # would be actively wrong: those variables are inert on codex and
4182
+ # every other non-Claude provider, so the pin would silently do
4183
+ # nothing. LOKI_SESSION_MODEL is the correct and only lever.
4184
+ popen_env["LOKI_SESSION_MODEL"] = start_model
4185
+ else:
4186
+ # EXACT-model pin (not the session-pin tier route): set all three tier
4187
+ # models to the chosen alias so resolve_model_for_tier returns the alias
4188
+ # for every tier and every iteration dispatches exactly the picked model.
4189
+ # This is the honest start-time equivalent of the mid-flight override
4190
+ # file, which run.sh clears at iteration 0. LOKI_SESSION_MODEL is set too
4191
+ # for internal coherence (the run's own tier accounting/logging), but the
4192
+ # env triple is the load-bearing dispatch-honesty mechanism: on the
4193
+ # v7.104.0 stock config the session pin alone would remap
4194
+ # opus->planning->sonnet and haiku->fast->sonnet, dispatching sonnet
4195
+ # for both.
4196
+ popen_env["LOKI_CLAUDE_MODEL_PLANNING"] = start_model
4197
+ popen_env["LOKI_CLAUDE_MODEL_DEVELOPMENT"] = start_model
4198
+ popen_env["LOKI_CLAUDE_MODEL_FAST"] = start_model
4199
+ popen_env["LOKI_SESSION_MODEL"] = start_model
3981
4200
  if advisor_model:
3982
4201
  # Opt-in Opus (or other) judge for code review; execution model unchanged.
3983
4202
  popen_env["LOKI_ADVISOR_MODEL"] = advisor_model
@@ -7079,6 +7298,15 @@ _DEFAULT_PRICING = {
7079
7298
  "haiku": {"input": 1.00, "output": 5.00},
7080
7299
  # OpenAI Codex
7081
7300
  "gpt-5.3-codex": {"input": 1.50, "output": 12.00},
7301
+ # gpt-5.6 line: sol (high) / terra (medium, default) / luna (small).
7302
+ # UNVERIFIED RATES. The model IDs are confirmed against
7303
+ # developers.openai.com/api/docs/models, but OpenAI's published per-token
7304
+ # prices for this line were not, so these are placeholders scaled from the
7305
+ # gpt-5.3 rate. They drive a display estimate only, never a gate. Replace
7306
+ # from the pricing page; tools/probe-model-catalog.py is the refresh path.
7307
+ "gpt-5.6-sol": {"input": 2.50, "output": 20.00},
7308
+ "gpt-5.6-terra": {"input": 1.50, "output": 12.00},
7309
+ "gpt-5.6-luna": {"input": 0.50, "output": 4.00},
7082
7310
  }
7083
7311
 
7084
7312
  # Active pricing - starts with defaults, updated from .loki/pricing.json
@@ -7655,6 +7883,9 @@ _PROVIDER_LABELS = {
7655
7883
  "sonnet": "Sonnet 5",
7656
7884
  "haiku": "Haiku 4.5",
7657
7885
  "gpt-5.3-codex": "GPT-5.3 Codex",
7886
+ "gpt-5.6-sol": "GPT-5.6 Sol",
7887
+ "gpt-5.6-terra": "GPT-5.6 Terra",
7888
+ "gpt-5.6-luna": "GPT-5.6 Luna",
7658
7889
  }
7659
7890
 
7660
7891
  # Display-only pricing notes, keyed by model. These annotate the pricing table in