loki-mode 9.12.6 → 9.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/autonomy/run.sh CHANGED
@@ -2049,7 +2049,40 @@ except Exception:
2049
2049
  pass
2050
2050
  " 2>/dev/null || true)"
2051
2051
  fi
2052
- [ -n "$_cost" ] && _fields="${_fields} | \$${_cost}"
2052
+ if [ -n "$_cost" ]; then
2053
+ _fields="${_fields} | \$${_cost}"
2054
+ # PROJECTED spend at the iteration cap, shown beside the actual.
2055
+ #
2056
+ # WHY. MAX_ITERATIONS (default 25) is the ONLY backstop on a run's cost:
2057
+ # LOKI_BUDGET_LIMIT defaults to "" and LOKI_MAX_DURATION to 0, both
2058
+ # documented at run.sh:1118-1121, and the runtime says so out loud at
2059
+ # :21333. That is a defensible default -- a run killed mid-flight at a
2060
+ # dollar threshold the user never chose is worse than one that finishes.
2061
+ # But it left the user unable to SEE where the run was heading: $4.20 at
2062
+ # iteration 3 of 25 reads as cheap right up to the moment it is not.
2063
+ #
2064
+ # Linear extrapolation, and labelled "proj" rather than presented as a
2065
+ # forecast: later iterations are usually cheaper than early ones (more
2066
+ # cache hits, smaller diffs), so this is an upper bound, not a promise.
2067
+ # Shown only when it would actually tell the user something -- from
2068
+ # iteration 2 (one data point cannot extrapolate) and only when the
2069
+ # projection is meaningfully above what has already been spent.
2070
+ if [ "${_iter:-0}" -ge 2 ] && [ "${_max:-0}" -gt 0 ]; then
2071
+ local _proj
2072
+ _proj="$(LOKI_C="$_cost" LOKI_I="$_iter" LOKI_M="$_max" python3 -c "
2073
+ import os
2074
+ try:
2075
+ c = float(os.environ['LOKI_C']); i = int(os.environ['LOKI_I']); m = int(os.environ['LOKI_M'])
2076
+ if i > 0 and m > i:
2077
+ p = c / i * m
2078
+ if p >= c * 1.5:
2079
+ print('%.2f' % p)
2080
+ except Exception:
2081
+ pass
2082
+ " 2>/dev/null || true)"
2083
+ [ -n "$_proj" ] && _fields="${_fields} (proj \$${_proj} at ${_max})"
2084
+ fi
2085
+ fi
2053
2086
 
2054
2087
  # Files changed (+ins/-del and file count) vs the run start SHA. Reuse the
2055
2088
  # build_completion_summary diff approach incl. the .loki/.git exclude pathspec.
@@ -3147,13 +3180,26 @@ validate_api_keys() {
3147
3180
  if [[ "$provider" == "claude" && "${LOKI_SKIP_AUTH_PREFLIGHT:-}" != "1" && -z "${ANTHROPIC_API_KEY:-}" ]]; then
3148
3181
  local _login_state
3149
3182
  _login_state="$(_loki_claude_login_state)"
3183
+ # Both branches report the blocker before returning. This is the wall a
3184
+ # user hits AFTER answering every quickstart prompt and confirming the
3185
+ # spend, and until now it emitted nothing -- so the funnel showed a first
3186
+ # run attempted, then silence, indistinguishable from a successful build.
3187
+ # Bounded enum only (`not_logged_in`), never the login state, path or
3188
+ # credential; backgrounded and non-fatal so a diagnostic can never break
3189
+ # the refusal it is describing.
3150
3190
  if [[ "$_login_state" == "loggedout" ]]; then
3191
+ if declare -f loki_emit_first_run_blocked >/dev/null 2>&1; then
3192
+ ( loki_emit_first_run_blocked "not_logged_in" >/dev/null 2>&1 </dev/null & ) 2>/dev/null || true
3193
+ fi
3151
3194
  log_error "Claude Code is installed but not logged in -- the build would stall instead of running."
3152
3195
  log_error "Log in once, then retry:"
3153
3196
  log_error " claude login"
3154
3197
  log_error "(or set ANTHROPIC_API_KEY, or LOKI_SKIP_AUTH_PREFLIGHT=1 to bypass this check)"
3155
3198
  return 1
3156
3199
  elif [[ "$_login_state" == "expired" ]]; then
3200
+ if declare -f loki_emit_first_run_blocked >/dev/null 2>&1; then
3201
+ ( loki_emit_first_run_blocked "not_logged_in" >/dev/null 2>&1 </dev/null & ) 2>/dev/null || true
3202
+ fi
3157
3203
  log_error "Your Claude Code login has expired -- the build would stall instead of running."
3158
3204
  log_error "Fix it in one step, then retry:"
3159
3205
  log_error " claude login"
@@ -3303,10 +3349,34 @@ detect_complexity() {
3303
3349
  file_count="${file_count:-0}"
3304
3350
  file_count="${file_count//[^0-9]/}"
3305
3351
 
3306
- # Check for external integrations
3352
+ # Check for external integrations.
3353
+ #
3354
+ # THE EXCLUDES ARE LOAD-BEARING. This grep used to prune nothing while the
3355
+ # find eleven lines above it prunes node_modules/.git/vendor/dist/build --
3356
+ # same function, same intent, inconsistent implementation. With --include
3357
+ # "*.json" that meant ANY transitive dependency whose package.json mentions
3358
+ # azure, stripe or aws-sdk set has_external=true.
3359
+ #
3360
+ # And has_external does not merely block "simple": in the classifier below it
3361
+ # jumps straight to "complex", skipping "standard". So a one-liner in any
3362
+ # repo that has ever run npm install landed on the MOST expensive tier, which
3363
+ # then runs the architecture doc suite (up to 300s of silence per attempt)
3364
+ # and holds the council's forced minimum-iteration floor at 3 instead of 1.
3365
+ #
3366
+ # Reproduced from scratch before fixing: a project with ONE dependency naming
3367
+ # @azure/core classified complex; adding --exclude-dir=node_modules made the
3368
+ # identical project classify simple. It also fired TRUE on this repo.
3369
+ #
3370
+ # A prior incident matches exactly -- a coffee landing page took 1h34m over
3371
+ # 11 iterations because the simple fast-path never engaged. The fast path was
3372
+ # correctly built and correctly wired the whole time; this one missing prune
3373
+ # was what made it unreachable.
3307
3374
  local has_external=false
3308
3375
  if grep -rq "oauth\|SAML\|OIDC\|stripe\|twilio\|aws-sdk\|@google-cloud\|azure" \
3309
- "$target_dir" --include="*.json" --include="*.ts" --include="*.js" 2>/dev/null; then
3376
+ "$target_dir" --include="*.json" --include="*.ts" --include="*.js" \
3377
+ --exclude-dir=node_modules --exclude-dir=.git --exclude-dir=vendor \
3378
+ --exclude-dir=dist --exclude-dir=build --exclude-dir=__pycache__ \
3379
+ --exclude-dir=.venv --exclude-dir=venv 2>/dev/null; then
3310
3380
  has_external=true
3311
3381
  fi
3312
3382
 
@@ -7643,6 +7713,80 @@ generate_proof_of_run() {
7643
7713
  return 0
7644
7714
  }
7645
7715
 
7716
+ # capture_preedit_snapshot: freeze the agent's raw diff BEFORE anything else
7717
+ # touches the tree, so quality numbers measure the agent and not the
7718
+ # agent-plus-whoever-fixed-it (autonomy/lib/preedit_snapshot.py owns the schema
7719
+ # and the write-once rule; this is only the call site).
7720
+ #
7721
+ # WHY THIS IS NOT INSIDE generate_proof_of_run, even though the receipt is the
7722
+ # obvious neighbour. Two reasons, both measured in this file:
7723
+ #
7724
+ # 1. TOO LATE AT THE LATE PROOF SITES. commit_session_changes commits the
7725
+ # session's work, and the module's default baseline is `git diff HEAD`.
7726
+ # After that commit `git diff HEAD` is EMPTY, so a capture at the teardown
7727
+ # proof site would freeze an empty diff -- and because the snapshot is
7728
+ # write-once by design, that empty capture would be permanent and
7729
+ # unrecoverable. run.sh already documents this mutation window itself: the
7730
+ # comment above the final generate_proof_of_run call says "HANDOFF.md and
7731
+ # commit_session_changes can change the worktree after the earlier receipt".
7732
+ # The receipt can be regenerated against a later tree; the snapshot cannot.
7733
+ # 2. WRONG GATE. Every generate_proof_of_run call site is gated on
7734
+ # LOKI_PROOF!=0, and the run_id resolution only exists inside its
7735
+ # LOKI_PROVEN_PR!=0 branch. Authorship evidence and shareable proofs are
7736
+ # different concerns, so a user who turns off proofs must not silently lose
7737
+ # the ability to tell agent output from human edits.
7738
+ #
7739
+ # So the capture happens EARLIER, immediately after run_autonomous returns,
7740
+ # before any post-processing step can modify the diff.
7741
+ #
7742
+ # Baseline: _LOKI_RUN_START_SHA (exported at runner init, persisted to
7743
+ # .loki/state/start-sha) is passed when available, so the snapshot is anchored
7744
+ # to the run's own starting commit rather than to a moving HEAD. That makes the
7745
+ # capture correct even if a later caller fires after a commit. Falls back to the
7746
+ # module's `git diff HEAD` default when no baseline resolved (greenfield repos
7747
+ # with no commits write an empty file there by design).
7748
+ #
7749
+ # run_id: read-path _loki_trust_run_id ONLY, never --new (minting here would
7750
+ # clobber the trust-events id file). If it resolves empty we SKIP: a snapshot
7751
+ # filed under an id nothing else references is worse than no snapshot, because
7752
+ # verdict.py would count it as authorship evidence that no receipt can join to.
7753
+ #
7754
+ # Guarded and non-fatal throughout: a diagnostic must never break the run it is
7755
+ # diagnosing. Write-once makes repeat calls free (later ones return "exists"),
7756
+ # so the earliest caller wins and extra call sites cost nothing.
7757
+ capture_preedit_snapshot() {
7758
+ local snap="$SCRIPT_DIR/lib/preedit_snapshot.py"
7759
+ [ -f "$snap" ] || return 0
7760
+ command -v python3 >/dev/null 2>&1 || return 0
7761
+ # Match _loki_trust_run_id's dir expression, not generate_proof_of_run's:
7762
+ # with LOKI_DIR set, ${TARGET_DIR:-.}/.loki would write the snapshot beside
7763
+ # a run-id file that lives somewhere else.
7764
+ local loki_dir="${LOKI_DIR:-${TARGET_DIR:-.}/.loki}"
7765
+ [ -d "$loki_dir" ] || return 0
7766
+ local _rid=""
7767
+ if declare -f _loki_trust_run_id >/dev/null 2>&1; then
7768
+ _rid="$(_loki_trust_run_id 2>/dev/null || true)"
7769
+ fi
7770
+ [ -n "$_rid" ] || return 0
7771
+ # Resolve the baseline from the PERSISTED file when the exported variable is
7772
+ # not visible. _LOKI_RUN_START_SHA is exported inside run_autonomous, and in
7773
+ # PARALLEL_MODE run_autonomous runs in a subshell -- an export from a
7774
+ # subshell never reaches the parent, so at that call site the variable is
7775
+ # empty and only the file survives. Same read the pause path already does.
7776
+ # Without this the parallel branch would silently fall back to `git diff
7777
+ # HEAD` instead of the anchored baseline this function documents.
7778
+ local _base="${_LOKI_RUN_START_SHA:-}"
7779
+ [ -n "$_base" ] || _base="$(cat "$loki_dir/state/start-sha" 2>/dev/null || true)"
7780
+ # LOKI_PREEDIT_CWD is load-bearing: the module defaults cwd to os.getcwd(),
7781
+ # and if that is not the target repo capture returns not_a_git_repo and
7782
+ # writes nothing SILENTLY -- a call site that looks wired but never fires.
7783
+ LOKI_DIR="$loki_dir" \
7784
+ LOKI_PREEDIT_CWD="${TARGET_DIR:-.}" \
7785
+ LOKI_RUN_START_SHA="$_base" \
7786
+ python3 "$snap" capture "$_rid" >/dev/null 2>&1 || true
7787
+ return 0
7788
+ }
7789
+
7646
7790
  # print_ttfv_next_steps: R7 zero-config first-run "what next / go deeper"
7647
7791
  # message. The wording MUST match what actually ran, so it branches on the mode:
7648
7792
  # - brief: a one-line brief ran on the lightweight profile (council off,
@@ -10716,6 +10860,14 @@ LOKI_STUCK_JSON
10716
10860
 
10717
10861
  track_gate_failure() {
10718
10862
  local gate_name="$1"
10863
+ # Optional evidence for the durable failure lesson (see the failure_memory
10864
+ # block below). Either a findings-artifact PATH or a literal detail string;
10865
+ # callers pass whichever they already name on an adjacent line.
10866
+ #
10867
+ # MUST be "${2:-}", not "$2": this file runs under `set -u` (line 185) and
10868
+ # most call sites are still one-arg, so a bare $2 aborts the gate it is only
10869
+ # supposed to be observing. Caught by the end-to-end check, not by review.
10870
+ local evidence="${2:-}"
10719
10871
  local gate_file="${TARGET_DIR:-.}/.loki/quality/gate-failure-count.json"
10720
10872
  mkdir -p "$(dirname "$gate_file")"
10721
10873
 
@@ -10752,6 +10904,56 @@ print(counts[gate_name])
10752
10904
  # echoed count or any gate behavior.
10753
10905
  record_trust_event_bash "gate_failure" "gate=${gate_name}" "consecutive=${count}" >/dev/null 2>&1 || true
10754
10906
 
10907
+ # Failure memory: turn this measured failure into a durable, falsifiable
10908
+ # lesson the NEXT run is told about (read side: build_prompt, below the
10909
+ # cache breakpoint).
10910
+ #
10911
+ # EVIDENCE IS REQUIRED, and deliberately not defaulted. failure_memory.py
10912
+ # refuses to write without it, because a lesson recorded from the agent's
10913
+ # own account of why it failed is unfalsifiable -- it records what the agent
10914
+ # BELIEVED, which is exactly what was wrong. Passing "$gate_name" as its own
10915
+ # evidence would satisfy the truthiness check and defeat that, so callers
10916
+ # with nothing concrete in scope pass nothing and record nothing.
10917
+ #
10918
+ # A readable evidence PATH is reduced to its first non-blank, non-comment
10919
+ # line with ANSI colour stripped -- the same reduction _loki_gate_stuck
10920
+ # applies above, so the stored lesson matches the cause that valve compares.
10921
+ #
10922
+ # CRITICAL: this function's stdout IS its return value, so this is fully
10923
+ # stdout-suppressed and best-effort, exactly like the trust-event write
10924
+ # above. failure_memory.py exits 3 on an UNKNOWN status (an expected result,
10925
+ # not an error), hence the `|| true`.
10926
+ #
10927
+ # ponytail: failures.jsonl is append-only with no dedup, so a gate stuck for
10928
+ # N iterations writes N records and recall() reads the whole file. Counts
10929
+ # stay true, so this is a ceiling not a defect; dedup on gate+evidence if a
10930
+ # perpetual run ever makes the file big enough to matter.
10931
+ if [ -n "$evidence" ] && [ -r "${SCRIPT_DIR}/lib/failure_memory.py" ]; then
10932
+ local _fm_evidence="$evidence"
10933
+ if [ -r "$_fm_evidence" ] && [ -f "$_fm_evidence" ]; then
10934
+ # `|| true`: head closing the pipe kills grep with SIGPIPE, which is
10935
+ # nonzero under `set -o pipefail` (line 185) even though the value is
10936
+ # correct. Same discipline as the trust-event write above.
10937
+ _fm_evidence="$(grep -vE '^[[:space:]]*(#|$)' "$_fm_evidence" 2>/dev/null \
10938
+ | head -1 | sed 's/\x1b\[[0-9;]*m//g' | head -c 200 || true)"
10939
+ fi
10940
+ if [ -n "$_fm_evidence" ]; then
10941
+ # run_id via the repo's existing resolver (the same one
10942
+ # record_trust_event_bash uses above), so a lesson can be traced back
10943
+ # to the run that produced it. Resolves to "" if unavailable, which
10944
+ # the module accepts -- only EVIDENCE is mandatory.
10945
+ local _fm_run_id=""
10946
+ if declare -f _loki_trust_run_id >/dev/null 2>&1; then
10947
+ _fm_run_id="${LOKI_TRUST_RUN_ID:-$(_loki_trust_run_id 2>/dev/null || true)}"
10948
+ fi
10949
+ LOKI_DIR="${LOKI_DIR:-${TARGET_DIR:-.}/.loki}" \
10950
+ python3 "${SCRIPT_DIR}/lib/failure_memory.py" record \
10951
+ "--gate=${gate_name}" "--verdict=FAIL" \
10952
+ "--evidence=${_fm_evidence}" \
10953
+ "--run_id=${_fm_run_id}" >/dev/null 2>&1 || true
10954
+ fi
10955
+ fi
10956
+
10755
10957
  echo "$count"
10756
10958
  }
10757
10959
 
@@ -12153,6 +12355,14 @@ auto_generate_docs_if_needed() {
12153
12355
  elif command -v timeout >/dev/null 2>&1; then
12154
12356
  _doc_cmd=(timeout "${_doc_to}s")
12155
12357
  fi
12358
+ # SAY WHAT IS HAPPENING BEFORE GOING QUIET. This call discards child output
12359
+ # and can run for the full timeout (default 300s), so without this line the
12360
+ # user sees a single "Auto-documentation" header and then minutes of nothing.
12361
+ # A silent gap reads as a hang: the observed incident was a build stuck ~55
12362
+ # min here with the work committed but never pushed, and nothing on screen
12363
+ # said which step owned the time. Naming the step and its cap turns an
12364
+ # apparent freeze into a bounded wait the user can reason about.
12365
+ log_info "Auto-documentation: generating architecture suite (no output until it finishes; up to ${_doc_to}s)"
12156
12366
  if "${_doc_cmd[@]}" "$loki_bin" docs generate "$project_dir" >/dev/null 2>&1; then
12157
12367
  :
12158
12368
  else
@@ -12211,6 +12421,10 @@ run_magic_debate_gate() {
12211
12421
  # verdict on genuinely thin input, not a spurious process block.
12212
12422
  log_info "Magic Modules: running debate on '$latest_name'"
12213
12423
  local debate_out debate_rc
12424
+ # Captured to a variable, so nothing reaches the screen for up to 300s. Same
12425
+ # reasoning as the doc suite above: name the step and its cap so a bounded
12426
+ # wait does not read as a hang.
12427
+ log_info "Magic debate: reviewing $latest_name (2 rounds, output shown when it finishes; up to 300s)"
12214
12428
  debate_out=$(cd "$TARGET_DIR" && PYTHONPATH="$PROJECT_DIR" LOKI_PROVIDER="${PROVIDER_NAME:-claude}" \
12215
12429
  timeout 300 "$PROJECT_DIR/autonomy/loki" magic debate "$latest_name" --rounds 2 2>&1) \
12216
12430
  && debate_rc=0 || debate_rc=$?
@@ -17253,6 +17467,70 @@ except Exception:
17253
17467
  # tried to signal completion via state files; we now honor that.
17254
17468
  #
17255
17469
  # Output on stdout: the JSON payload (for callers that want to log it).
17470
+ # _loki_check_claim_grounding: does the completion claim name files this run
17471
+ # actually changed? Report-only, never a gate.
17472
+ #
17473
+ # READS THE SIGNAL FILE WITHOUT CONSUMING IT. check_task_completion_signal below
17474
+ # owns consumption (rm -f on read); this must run BEFORE that owner and must not
17475
+ # race it, so it only ever opens the file for reading. Both signal shapes carry
17476
+ # the text under the same key: the MCP tool writes {"statement": ...}, and the
17477
+ # COMPLETION_REQUESTED fallback is normalised into the same envelope by the
17478
+ # owner. One key covers both.
17479
+ #
17480
+ # The changed-file set is derived from _LOKI_RUN_START_SHA -- the same baseline
17481
+ # the evidence gate and the review diff use (run.sh:13817) -- so the receipt and
17482
+ # the grounding line describe ONE diff. Untracked files are included: a claim
17483
+ # naming a file the agent created but never staged is grounded, and calling it
17484
+ # ungrounded would be exactly the false positive this check must never produce.
17485
+ #
17486
+ # Passed via --files-from, never --files: --files is a comma-separated list, so
17487
+ # any path containing a comma would split into two bogus paths, and a large
17488
+ # changed set would approach ARG_MAX. The module's own comment documents that
17489
+ # flag's history.
17490
+ _loki_check_claim_grounding() {
17491
+ local lib="${SCRIPT_DIR}/lib/claim_grounding.py"
17492
+ [ -f "$lib" ] || return 0
17493
+ command -v python3 >/dev/null 2>&1 || return 0
17494
+
17495
+ local target="${TARGET_DIR:-.}"
17496
+ local sig="$target/.loki/signals/TASK_COMPLETION_CLAIMED"
17497
+ [ -f "$sig" ] || sig="$target/.loki/signals/COMPLETION_REQUESTED"
17498
+ [ -f "$sig" ] || return 0
17499
+
17500
+ local claim
17501
+ claim=$(python3 -c "
17502
+ import json, sys
17503
+ try:
17504
+ d = json.load(open(sys.argv[1]))
17505
+ sys.stdout.write(str(d.get('statement', '')) if isinstance(d, dict) else '')
17506
+ except Exception:
17507
+ pass
17508
+ " "$sig" 2>/dev/null || echo "")
17509
+ # A signal with no statement (bare touch of COMPLETION_REQUESTED) is
17510
+ # UNGROUNDABLE, not a finding. Nothing to check; leave no stale artifact.
17511
+ [ -n "$claim" ] || return 0
17512
+
17513
+ local files_tmp="$target/.loki/state/claim-grounding-files.$$"
17514
+ mkdir -p "$target/.loki/state" 2>/dev/null || return 0
17515
+ {
17516
+ if [ -n "${_LOKI_RUN_START_SHA:-}" ] \
17517
+ && git -C "$target" rev-parse --verify --quiet "${_LOKI_RUN_START_SHA}^{commit}" >/dev/null 2>&1; then
17518
+ git -C "$target" diff --name-only "${_LOKI_RUN_START_SHA}" 2>/dev/null
17519
+ else
17520
+ git -C "$target" diff --name-only HEAD 2>/dev/null
17521
+ fi
17522
+ git -C "$target" diff --name-only --cached 2>/dev/null
17523
+ git -C "$target" ls-files --others --exclude-standard 2>/dev/null
17524
+ } | sort -u > "$files_tmp" 2>/dev/null || { rm -f "$files_tmp" 2>/dev/null; return 0; }
17525
+
17526
+ # Exit 1 means "a named path is absent from the diff" -- the finding itself,
17527
+ # not an error. Swallowed: this reports, it never blocks completion.
17528
+ python3 "$lib" --claim "$claim" --files-from "$files_tmp" \
17529
+ > "$target/.loki/state/claim-grounding.json" 2>/dev/null || true
17530
+ rm -f "$files_tmp" 2>/dev/null
17531
+ return 0
17532
+ }
17533
+
17256
17534
  check_task_completion_signal() {
17257
17535
  local signal_file=".loki/signals/TASK_COMPLETION_CLAIMED"
17258
17536
  local fallback_file=".loki/signals/COMPLETION_REQUESTED"
@@ -19561,6 +19839,41 @@ if d.get('blocked'):
19561
19839
  --loki-dir ".loki" --prompt-block 2>/dev/null || true)"
19562
19840
  fi
19563
19841
 
19842
+ # Failure memory (read side; write side: track_gate_failure). Tells this
19843
+ # iteration what has actually failed in THIS repo before, so a gate the
19844
+ # agent has already lost to is not re-learned from scratch every run.
19845
+ #
19846
+ # COUNTS, NOT PROSE, and that restriction is the whole point. "the
19847
+ # mock_integrity gate has failed here 6 times" is a fact the reader can
19848
+ # check; "this repo tends to have mocking problems" is a generalization that
19849
+ # reads identically and is not falsifiable. The module renders the lines and
19850
+ # this only prints them -- no second renderer to drift, matching the
19851
+ # single-renderer discipline used for efficiency_trend above.
19852
+ #
19853
+ # Emits "" when nothing has been recorded, so a repo with no failure history
19854
+ # adds NOTHING to the prompt (and the 60 build_prompt parity fixtures, none
19855
+ # of which carry a failures.jsonl, stay byte-identical).
19856
+ #
19857
+ # Calls prompt_context() in-process rather than the CLI: the CLI prints JSON
19858
+ # and exits 3 on an UNKNOWN status, which is an expected "no lessons yet"
19859
+ # result and not an error worth parsing around.
19860
+ local failure_memory_context=""
19861
+ if [ -r "${SCRIPT_DIR}/lib/failure_memory.py" ] && [ -d ".loki" ]; then
19862
+ failure_memory_context="$(_FM_LIB="${SCRIPT_DIR}/lib" \
19863
+ _FM_DIR="${LOKI_DIR:-${TARGET_DIR:-.}/.loki}" python3 -c '
19864
+ import os, sys
19865
+ sys.path.insert(0, os.environ["_FM_LIB"])
19866
+ try:
19867
+ from failure_memory import prompt_context
19868
+ lines = prompt_context(os.environ["_FM_DIR"]).get("lines") or []
19869
+ except Exception:
19870
+ lines = []
19871
+ if lines:
19872
+ print("KNOWN FAILURE HISTORY IN THIS REPO (measured, from previous runs): "
19873
+ + "; ".join(lines) + ".")
19874
+ ' 2>/dev/null || true)"
19875
+ fi
19876
+
19564
19877
  # PRD Checklist status injection (v5.44.0)
19565
19878
  local checklist_status=""
19566
19879
  if [ -n "$prd" ] && [ ! -f ".loki/checklist/checklist.json" ]; then
@@ -19776,8 +20089,11 @@ except Exception:
19776
20089
  if [ -n "$gate_escalation_context" ]; then
19777
20090
  _legacy_priority="${_legacy_priority}${_legacy_priority:+ }${gate_escalation_context}"
19778
20091
  fi
20092
+ # Same cap, same reason, same default as the degraded path above --
20093
+ # a pasted spec has to be bounded, but the bound must not silently
20094
+ # eat requirements. 4000 bytes dropped anything past ~600 words.
19779
20095
  if [ -n "$prd" ] && [ -f "$prd" ]; then
19780
- _legacy_prd_content=$(head -c 4000 "$prd")
20096
+ _legacy_prd_content=$(head -c "${LOKI_DEGRADED_PRD_CAP:-24000}" "$prd")
19781
20097
  fi
19782
20098
  if [ $retry -eq 0 ]; then
19783
20099
  if [ -n "$prd" ]; then
@@ -19828,9 +20144,35 @@ except Exception:
19828
20144
 
19829
20145
  if [ "${PROVIDER_DEGRADED:-false}" = "true" ]; then
19830
20146
  # Degraded providers: simpler wording, but still static-first.
20147
+ #
20148
+ # THE CAP IS NOW ANNOUNCED, NOT SILENT. This path PASTES the spec text
20149
+ # (a degraded provider cannot be told "read the file at this path" the
20150
+ # way claude/cline/opencode are at :20196), so it has to be bounded. It
20151
+ # was bounded at 4000 bytes with no notice: a requirement past ~600 words
20152
+ # was dropped mid-sentence and the model never knew a spec existed beyond
20153
+ # what it saw. Demonstrated on a 4229-byte spec -- the requirement on the
20154
+ # last line was simply absent from what the model received.
20155
+ #
20156
+ # That is the same class of defect spec-expand.sh:5-7 already names for
20157
+ # OpenAPI ("a 40-operation file loses 21 of 40 ops") and fixed for
20158
+ # contracts only. Markdown specs still had it, and only for the two
20159
+ # degraded providers -- so codex and aider users silently got a worse
20160
+ # build than claude users from the identical spec.
20161
+ #
20162
+ # Raised to 24000 (a large PRD fits whole) and, when the spec still
20163
+ # exceeds it, the model is TOLD so and given the path to read the rest.
20164
+ # An unannounced truncation makes the model confidently build the wrong
20165
+ # thing; an announced one makes it go look.
20166
+ local _prd_cap="${LOKI_DEGRADED_PRD_CAP:-24000}"
19831
20167
  local prd_content=""
20168
+ local _prd_truncated=0
19832
20169
  if [ -n "$prd" ] && [ -f "$prd" ]; then
19833
- prd_content=$(head -c 4000 "$prd")
20170
+ prd_content=$(head -c "$_prd_cap" "$prd")
20171
+ local _prd_bytes
20172
+ _prd_bytes=$(wc -c < "$prd" 2>/dev/null | tr -d ' ')
20173
+ if [ -n "$_prd_bytes" ] && [ "$_prd_bytes" -gt "$_prd_cap" ] 2>/dev/null; then
20174
+ _prd_truncated=1
20175
+ fi
19834
20176
  fi
19835
20177
 
19836
20178
  local degraded_prd_anchor="Loki Mode"
@@ -19859,6 +20201,38 @@ except Exception:
19859
20201
  [ -n "$queue_tasks" ] && printf 'Tasks: %s\n' "$queue_tasks"
19860
20202
  if [ -n "$prd" ]; then
19861
20203
  printf 'PRD contents: %s\n' "$prd_content"
20204
+ # Announce the cut. Silence here is what made the old 4000-byte cap
20205
+ # dangerous: the model treated a partial spec as the whole spec and
20206
+ # built confidently against requirements it had never seen. Naming
20207
+ # the file lets it read the remainder itself.
20208
+ if [ "${_prd_truncated:-0}" = "1" ]; then
20209
+ printf 'NOTE: the spec above is TRUNCATED at %s bytes. The full spec is at %s -- read it before deciding the work is complete.\n' \
20210
+ "$_prd_cap" "$prd"
20211
+ fi
20212
+ fi
20213
+
20214
+ # FIRST-PASS EXCELLENCE FOR DEGRADED PROVIDERS.
20215
+ #
20216
+ # This directive existed only for Claude. providers/claude.sh:322 injects
20217
+ # it via --append-system-prompt, a flag codex/aider do not have, so the
20218
+ # one mechanism built specifically to make a WEAKER model land complete
20219
+ # on iteration 1 reached only the strongest one. Measured before writing
20220
+ # this: grep for FIRST_PASS_EXCELLENCE returns 0 in codex.sh, aider.sh,
20221
+ # cline.sh and opencode.sh.
20222
+ #
20223
+ # It matters most exactly where it was missing. The premise (recorded
20224
+ # when the Claude version was built) is that iteration count is a proxy
20225
+ # for how much the first pass missed, and that for a weak model context
20226
+ # quality beats iteration count. Codex is also the free on-ramp, so the
20227
+ # users least able to absorb a bad build were the ones getting no help.
20228
+ #
20229
+ # Condensed rather than byte-mirrored: the Claude text is ~4.3KB of
20230
+ # system prompt, and these providers take it inline in the user turn
20231
+ # where budget is tighter. The four load-bearing instructions are kept --
20232
+ # build fully, wire the backend, verify by RUNNING, commit to one design.
20233
+ # Same iteration-1 gate and same env var, so one switch controls both.
20234
+ if [ "${LOKI_FIRST_PASS_EXCELLENCE:-1}" != "0" ] && [ "${iteration:-1}" -le 1 ] 2>/dev/null; then
20235
+ printf '%s\n' '[FIRST-PASS EXCELLENCE] Treat THIS pass as your one shot to ship a complete, working solution. The loop is a safety net, not a plan. 1) BUILD IT FULLY: no stubs, no TODOs, no placeholder or mock data where real logic belongs. If the spec implies a backend (auth, persistence, a form that submits), WIRE IT so it actually persists -- a UI whose buttons do nothing is the most common failure. 2) VERIFY BY RUNNING each acceptance path, not by reading the code. 3) DECIDE the architecture now rather than refactoring later. 4) Commit to ONE specific design; avoid the generic purple-gradient default look.'
19862
20236
  fi
19863
20237
  printf '</dynamic_context>\n'
19864
20238
  return 0
@@ -19968,6 +20342,10 @@ except Exception:
19968
20342
  [ -n "$app_runner_info" ] && printf '%s\n' "$app_runner_info"
19969
20343
  [ -n "$playwright_info" ] && printf '%s\n' "$playwright_info"
19970
20344
  [ -n "$memory_context_section" ] && printf '%s\n' "$memory_context_section"
20345
+ # Failure memory: volatile (it changes the moment a gate fails), so it lives
20346
+ # here in the dynamic tail, never in the cache-stable <loki_system> prefix.
20347
+ # Sits with the other memory context, before the efficiency trend.
20348
+ [ -n "$failure_memory_context" ] && printf '%s\n' "$failure_memory_context"
19971
20349
  # Volatile per-iteration data: belongs below [CACHE_BREAKPOINT], never in the
19972
20350
  # cache-stable prefix. Same ordinal position as the Bun route (after the
19973
20351
  # context section, before the completion instruction).
@@ -22796,6 +23174,73 @@ if __name__ == "__main__":
22796
23174
  # costs zero extra subprocesses -- we pass the existing epoch through.
22797
23175
  emit_stage_complete "agent" "$([ "$exit_code" -eq 0 ] 2>/dev/null && echo pass || echo fail)" "$start_time"
22798
23176
 
23177
+ # LLM DECISION RECORD (autonomy/lib/decision_record.py).
23178
+ #
23179
+ # WHY HERE. This is the single point where every provider arm converges
23180
+ # after dispatch: claude, codex, cline and aider all land here with
23181
+ # $tier_param (the model actually dispatched), $exit_code and $duration
23182
+ # in scope. Recording per-arm would be four call sites that drift.
23183
+ #
23184
+ # WHY tier_param AND NOT LOKI_CURRENT_MODEL. Only the claude arm exports
23185
+ # LOKI_CURRENT_MODEL (line ~22214); on a codex/cline/aider iteration that
23186
+ # variable is either unset or a STALE value left by an earlier claude
23187
+ # iteration after a failover. tier_param is the same string the claude
23188
+ # arm exports, and it is correct on every arm. It is read AFTER every
23189
+ # mutation (opus-pin force, LOKI_MAX_TIER clamp, mid-flight override,
23190
+ # fable collapse), so it is the model that ran, not the tier alias.
23191
+ #
23192
+ # WHAT IS DELIBERATELY OMITTED. temperature: this runtime never sets one
23193
+ # on any provider (claude dispatch passes --model/--effort, never a
23194
+ # temperature), so writing a value would be inventing the exact field
23195
+ # whose whole purpose is making config drift falsifiable. The module
23196
+ # treats an absent field as absent; a guessed 0.0 would be a lie that
23197
+ # reads as a measurement. confidence: self-reported and not available at
23198
+ # this seam. Tokens come from the authoritative per-iteration result-cost
23199
+ # file when the provider wrote one, and are omitted rather than zeroed
23200
+ # when it did not (a zero claims the call was free).
23201
+ #
23202
+ # NON-FATAL AND BACKGROUNDED: a diagnostic must never be able to break
23203
+ # the iteration it is diagnosing, and this is a python3 spawn on the
23204
+ # critical path of the loop's largest stage.
23205
+ if [ -n "${tier_param:-}" ] && [ -f "${SCRIPT_DIR:-}/lib/decision_record.py" ]; then
23206
+ local _dr_args=(
23207
+ "--model_id=$tier_param"
23208
+ "--provider=${PROVIDER_NAME:-claude}"
23209
+ "--stage=iteration_${ITERATION_COUNT:-0}_${rarv_phase:-unknown}"
23210
+ "--outcome=$([ "$exit_code" -eq 0 ] 2>/dev/null && echo ok || echo error)"
23211
+ "--duration_ms=$((duration * 1000))"
23212
+ )
23213
+ # Correlation ids only when genuinely set: an empty run_id written as
23214
+ # "" is indistinguishable from a real one in a later diff, and the
23215
+ # module records whatever an allowlisted field carries.
23216
+ [ -n "${LOKI_TRUST_RUN_ID:-}" ] && _dr_args+=("--run_id=$LOKI_TRUST_RUN_ID") || true
23217
+ [ -n "${LOKI_SESSION_ID:-}" ] && _dr_args+=("--session_id=$LOKI_SESSION_ID") || true
23218
+ local _dr_cost="${TARGET_DIR:-.}/.loki/metrics/result-cost-${ITERATION_COUNT:-0}.json"
23219
+ if [ -s "$_dr_cost" ]; then
23220
+ # Read into named locals, NOT `set --`: this runs in the middle of
23221
+ # run_autonomous, and clobbering the function's positional
23222
+ # parameters to parse a diagnostic is how a metrics read turns
23223
+ # into a control-flow bug.
23224
+ local _dr_in="" _dr_out=""
23225
+ read -r _dr_in _dr_out <<EOF
23226
+ $(python3 -c 'import json,sys
23227
+ d = json.load(open(sys.argv[1]))
23228
+ # Print BOTH or neither: a half-record invites a reader to treat a missing
23229
+ # output count as zero output, which reads as "the model produced nothing".
23230
+ i, o = d.get("input_tokens"), d.get("output_tokens")
23231
+ if isinstance(i, int) and isinstance(o, int):
23232
+ print(i, o)' "$_dr_cost" 2>/dev/null)
23233
+ EOF
23234
+ case "${_dr_in}${_dr_out}" in
23235
+ ''|*[!0-9]*) ;; # unparseable -> omit rather than fabricate
23236
+ *) _dr_args+=("--tokens_in=$_dr_in" "--tokens_out=$_dr_out") ;;
23237
+ esac
23238
+ fi
23239
+ ( LOKI_DIR="${TARGET_DIR:-.}/.loki" \
23240
+ python3 "${SCRIPT_DIR}/lib/decision_record.py" record "${_dr_args[@]}" \
23241
+ >/dev/null 2>&1 </dev/null & ) 2>/dev/null || true
23242
+ fi
23243
+
22799
23244
  # AGENT PROMPT SIZE. The call this brackets is 93% of a run's wall clock
22800
23245
  # (1814s of 1941s measured), and its INPUT was never measured -- every
22801
23246
  # reviewer logs its prompt bytes, the dominant call logged nothing.
@@ -23079,7 +23524,8 @@ if __name__ == "__main__":
23079
23524
  else
23080
23525
  _stg_ok=fail
23081
23526
  local sa_count
23082
- sa_count=$(track_gate_failure "static_analysis")
23527
+ sa_count=$(track_gate_failure "static_analysis" \
23528
+ "${TARGET_DIR:-.}/.loki/quality/static-analysis.json")
23083
23529
  gate_failures="${gate_failures}static_analysis,"
23084
23530
  log_warn "Static analysis FAILED ($sa_count consecutive) - findings injected into next iteration"
23085
23531
  # F0, extended past mutation_integrity. Static analysis is
@@ -23187,7 +23633,8 @@ if __name__ == "__main__":
23187
23633
  ;;
23188
23634
  fail)
23189
23635
  local mk_count
23190
- mk_count=$(track_gate_failure "mock_integrity")
23636
+ mk_count=$(track_gate_failure "mock_integrity" \
23637
+ "${TARGET_DIR:-.}/.loki/quality/mock-findings.txt")
23191
23638
  gate_failures="${gate_failures}mock_integrity,"
23192
23639
  log_warn "Mock integrity gate FAILED ($mk_count consecutive) - CRITICAL/HIGH mock problems"
23193
23640
  # Escalation guidance was DEAD for this gate.
@@ -23240,7 +23687,8 @@ if __name__ == "__main__":
23240
23687
  else
23241
23688
  _stg_ok=fail
23242
23689
  local mt_count
23243
- mt_count=$(track_gate_failure "mutation_integrity")
23690
+ mt_count=$(track_gate_failure "mutation_integrity" \
23691
+ "${TARGET_DIR:-.}/.loki/quality/mutation-findings.txt")
23244
23692
  gate_failures="${gate_failures}mutation_integrity,"
23245
23693
  log_warn "Mutation integrity gate FAILED ($mt_count consecutive) - HIGH test-fitting detected"
23246
23694
  # Same dead-branch fix as mock_integrity above:
@@ -23306,7 +23754,8 @@ if __name__ == "__main__":
23306
23754
  _lsp_e=$(printf '%s' "${_LOKI_LSP_DIAGNOSTICS_DETAIL:-}" | awk '{print $2}')
23307
23755
  _lsp_w=$(printf '%s' "${_LOKI_LSP_DIAGNOSTICS_DETAIL:-}" | awk '{print $3}')
23308
23756
  local lsp_count
23309
- lsp_count=$(track_gate_failure "lsp_diagnostics")
23757
+ lsp_count=$(track_gate_failure "lsp_diagnostics" \
23758
+ "${_LOKI_LSP_DIAGNOSTICS_DETAIL:-}")
23310
23759
  log_warn "LSP diagnostics reported errors ($lsp_count consecutive) - ${_lsp_e} error(s), ${_lsp_w} warning(s); advisory only"
23311
23760
  ;;
23312
23761
  pass)
@@ -23350,7 +23799,8 @@ if __name__ == "__main__":
23350
23799
  clear_gate_failure "semantic_tests"
23351
23800
  else
23352
23801
  local sem_count
23353
- sem_count=$(track_gate_failure "semantic_tests")
23802
+ sem_count=$(track_gate_failure "semantic_tests" \
23803
+ "${TARGET_DIR:-.}/.loki/quality/semantic-findings.txt")
23354
23804
  if [ "${LOKI_GATE_SEMANTIC_TESTS_BLOCK:-false}" = "true" ] \
23355
23805
  || [ "${LOKI_GATE_SEMANTIC_TESTS_BLOCK:-false}" = "1" ]; then
23356
23806
  gate_failures="${gate_failures}semantic_tests,"
@@ -23376,7 +23826,8 @@ if __name__ == "__main__":
23376
23826
  clear_gate_failure "invariants"
23377
23827
  else
23378
23828
  local inv_count
23379
- inv_count=$(track_gate_failure "invariants")
23829
+ inv_count=$(track_gate_failure "invariants" \
23830
+ "${TARGET_DIR:-.}/.loki/quality/invariant-findings.txt")
23380
23831
  if [ "${LOKI_GATE_INVARIANTS_BLOCK:-false}" = "true" ] \
23381
23832
  || [ "${LOKI_GATE_INVARIANTS_BLOCK:-false}" = "1" ]; then
23382
23833
  gate_failures="${gate_failures}invariants,"
@@ -23694,6 +24145,31 @@ if __name__ == "__main__":
23694
24145
  || [ -f "${TARGET_DIR:-.}/.loki/signals/COMPLETION_REQUESTED" ]; then
23695
24146
  _loki_completion_claimed=1
23696
24147
  fi
24148
+ # CLAIM GROUNDING (report-only): does the completion claim name files
24149
+ # that are actually in this run's diff? Every existing evidence axis is
24150
+ # a REPO-level fact (diff non-empty, tests green, app boots), so an
24151
+ # agent can finish by claiming "added retry logic to the payment
24152
+ # client" while the diff shows a README edit and all six axes pass.
24153
+ # The claim itself is the one artifact nothing else reads.
24154
+ #
24155
+ # PLACED HERE, at the non-destructive peek, deliberately: this is the
24156
+ # only point where the claim signal still EXISTS. The default route
24157
+ # below (check_completion_promise -> check_task_completion_signal)
24158
+ # consumes it with rm -f on read, so reading the statement after that
24159
+ # returns nothing, and re-reading it through the consuming detector
24160
+ # would re-introduce the v7.28 claim-drop bug. We read the signal file
24161
+ # directly and never remove it -- consumption keeps its single owner.
24162
+ #
24163
+ # FAIL-OPEN AND NON-BLOCKING BY DESIGN: only a claim naming a path
24164
+ # demonstrably absent from the diff is a finding, and it is written to
24165
+ # a file, never returned into the gate chain. claim_grounding.py exits
24166
+ # 1 on exactly that case, hence `|| true` -- an ungrounded claim must
24167
+ # report, not block. A grounding check that blocked on ambiguity would
24168
+ # fire on ordinary prose and be disabled within a week.
24169
+ if [ "$_loki_completion_claimed" = 1 ] \
24170
+ && [ -f "${SCRIPT_DIR}/lib/claim_grounding.py" ]; then
24171
+ _loki_check_claim_grounding || true
24172
+ fi
23697
24173
  local _loki_completion_ready=1
23698
24174
  if loki_is_supervised_simple_web; then
23699
24175
  _loki_supervised_completion_gates_pass "${gate_failures:-}" && _loki_completion_ready=0
@@ -24809,6 +25285,17 @@ except (json.JSONDecodeError, OSError): pass
24809
25285
  _loki_write_termination_record "$signal_name" "$final_exit_code"
24810
25286
  fi
24811
25287
  emit_event_json "session_end" "result=$final_exit_code" "reason=$final_reason"
25288
+ # An interrupted run still produced agent output, and this teardown is
25289
+ # the only exit it takes -- it never reaches the post-loop capture.
25290
+ # Backgrounded here (and ONLY here) because this runs inside a signal
25291
+ # handler, where a blocking git call would stall the shutdown the user
25292
+ # just asked for. Backgrounding is safe at this site specifically
25293
+ # because the capture is ORDERING-INDEPENDENT: it baselines to the
25294
+ # run-start SHA, not to a moving HEAD, and nothing between here and
25295
+ # process exit commits -- so it records the same diff whenever the
25296
+ # subshell lands. Do NOT copy this backgrounding to the post-loop site,
25297
+ # where completing before the tree mutates is the entire point.
25298
+ ( capture_preedit_snapshot >/dev/null 2>&1 </dev/null & ) 2>/dev/null || true
24812
25299
  if [ "$supervised_signal" = "true" ] \
24813
25300
  && [ "${LOKI_PROOF:-1}" != "0" ] \
24814
25301
  && type generate_proof_of_run >/dev/null 2>&1; then
@@ -25528,6 +26015,13 @@ main() {
25528
26015
  kill $orchestrator_pid 2>/dev/null || true
25529
26016
  wait $orchestrator_pid 2>/dev/null || true
25530
26017
 
26018
+ # Same pre-edit capture as the standard branch, placed before
26019
+ # cleanup_parallel_streams because that tears down worktrees and can
26020
+ # change what the diff sees. Parallel mode never reaches the standard
26021
+ # branch's call site, so without this the whole mode would have no
26022
+ # authorship evidence. Write-once, so this is still a single snapshot.
26023
+ capture_preedit_snapshot || true
26024
+
25531
26025
  # Cleanup parallel streams
25532
26026
  cleanup_parallel_streams
25533
26027
  else
@@ -25536,6 +26030,17 @@ main() {
25536
26030
  # a stuck "Planning" state.
25537
26031
  _advance_current_phase "BUILDING"
25538
26032
  run_autonomous "$PRD_PATH" || result=$?
26033
+ # PRE-EDIT SNAPSHOT: freeze the agent's raw diff HERE, the first
26034
+ # instruction after the loop returns, because everything below this line
26035
+ # can change the tree -- commit_session_changes commits the work (after
26036
+ # which `git diff HEAD` is empty), and HANDOFF.md/learnings writers touch
26037
+ # files before that. The snapshot is write-once, so capturing it late
26038
+ # would permanently record someone else's edits as the agent's. Runs in
26039
+ # the FOREGROUND on purpose: the entire value of this position is that
26040
+ # the capture COMPLETES before any mutation, and backgrounding it would
26041
+ # reintroduce exactly the race the placement exists to remove (the
26042
+ # module bounds each git call at 60s, so the cost is bounded).
26043
+ capture_preedit_snapshot || true
25539
26044
  # ZOMBIE-RECEIPT GUARD: proof generation + the COMPLETED marker live in the
25540
26045
  # teardown far below. If the process is killed (Docker restart, OOM, worker
25541
26046
  # reap) between here and there, a genuinely finished build (real code, exit