loki-mode 9.12.6 → 9.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +81 -101
- package/SKILL.md +2 -2
- package/VERSION +1 -1
- package/autonomy/intent.sh +414 -0
- package/autonomy/issue-providers.sh +21 -0
- package/autonomy/lib/agent_readiness.py +202 -0
- package/autonomy/lib/claim_grounding.py +171 -0
- package/autonomy/lib/config-map.sh +10 -6
- package/autonomy/lib/decision_record.py +198 -0
- package/autonomy/lib/failure_memory.py +199 -0
- package/autonomy/lib/outcome_ledger.py +498 -0
- package/autonomy/lib/preedit_snapshot.py +216 -0
- package/autonomy/lib/verdict.py +204 -0
- package/autonomy/loki +358 -14
- package/autonomy/provider-offer.sh +25 -1
- package/autonomy/run.sh +516 -11
- package/autonomy/telemetry.sh +8 -1
- package/completions/_loki +4 -0
- package/completions/loki.bash +2 -1
- package/dashboard/__init__.py +1 -1
- package/dashboard/run.py +13 -2
- package/dashboard/scim.py +221 -0
- package/dashboard/server.py +22 -0
- package/docs/GATE-FAILURE-TRIAGE.md +254 -0
- package/docs/LOOP-CANDIDATE-PROPOSAL-v1.md +167 -0
- package/docs/LOOP-HARNESS-AUDIT.md +53 -0
- package/docs/VERIFICATION-COST.md +103 -0
- package/docs/WANG-PRINCIPLES-PLAN.md +1 -1
- package/loki-ts/dist/loki.js +402 -398
- package/mcp/__init__.py +1 -1
- package/mcp/_sdk_loader.py +25 -0
- package/package.json +1 -1
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
package/autonomy/run.sh
CHANGED
|
@@ -2049,7 +2049,40 @@ except Exception:
|
|
|
2049
2049
|
pass
|
|
2050
2050
|
" 2>/dev/null || true)"
|
|
2051
2051
|
fi
|
|
2052
|
-
[ -n "$_cost" ]
|
|
2052
|
+
if [ -n "$_cost" ]; then
|
|
2053
|
+
_fields="${_fields} | \$${_cost}"
|
|
2054
|
+
# PROJECTED spend at the iteration cap, shown beside the actual.
|
|
2055
|
+
#
|
|
2056
|
+
# WHY. MAX_ITERATIONS (default 25) is the ONLY backstop on a run's cost:
|
|
2057
|
+
# LOKI_BUDGET_LIMIT defaults to "" and LOKI_MAX_DURATION to 0, both
|
|
2058
|
+
# documented at run.sh:1118-1121, and the runtime says so out loud at
|
|
2059
|
+
# :21333. That is a defensible default -- a run killed mid-flight at a
|
|
2060
|
+
# dollar threshold the user never chose is worse than one that finishes.
|
|
2061
|
+
# But it left the user unable to SEE where the run was heading: $4.20 at
|
|
2062
|
+
# iteration 3 of 25 reads as cheap right up to the moment it is not.
|
|
2063
|
+
#
|
|
2064
|
+
# Linear extrapolation, and labelled "proj" rather than presented as a
|
|
2065
|
+
# forecast: later iterations are usually cheaper than early ones (more
|
|
2066
|
+
# cache hits, smaller diffs), so this is an upper bound, not a promise.
|
|
2067
|
+
# Shown only when it would actually tell the user something -- from
|
|
2068
|
+
# iteration 2 (one data point cannot extrapolate) and only when the
|
|
2069
|
+
# projection is meaningfully above what has already been spent.
|
|
2070
|
+
if [ "${_iter:-0}" -ge 2 ] && [ "${_max:-0}" -gt 0 ]; then
|
|
2071
|
+
local _proj
|
|
2072
|
+
_proj="$(LOKI_C="$_cost" LOKI_I="$_iter" LOKI_M="$_max" python3 -c "
|
|
2073
|
+
import os
|
|
2074
|
+
try:
|
|
2075
|
+
c = float(os.environ['LOKI_C']); i = int(os.environ['LOKI_I']); m = int(os.environ['LOKI_M'])
|
|
2076
|
+
if i > 0 and m > i:
|
|
2077
|
+
p = c / i * m
|
|
2078
|
+
if p >= c * 1.5:
|
|
2079
|
+
print('%.2f' % p)
|
|
2080
|
+
except Exception:
|
|
2081
|
+
pass
|
|
2082
|
+
" 2>/dev/null || true)"
|
|
2083
|
+
[ -n "$_proj" ] && _fields="${_fields} (proj \$${_proj} at ${_max})"
|
|
2084
|
+
fi
|
|
2085
|
+
fi
|
|
2053
2086
|
|
|
2054
2087
|
# Files changed (+ins/-del and file count) vs the run start SHA. Reuse the
|
|
2055
2088
|
# build_completion_summary diff approach incl. the .loki/.git exclude pathspec.
|
|
@@ -3147,13 +3180,26 @@ validate_api_keys() {
|
|
|
3147
3180
|
if [[ "$provider" == "claude" && "${LOKI_SKIP_AUTH_PREFLIGHT:-}" != "1" && -z "${ANTHROPIC_API_KEY:-}" ]]; then
|
|
3148
3181
|
local _login_state
|
|
3149
3182
|
_login_state="$(_loki_claude_login_state)"
|
|
3183
|
+
# Both branches report the blocker before returning. This is the wall a
|
|
3184
|
+
# user hits AFTER answering every quickstart prompt and confirming the
|
|
3185
|
+
# spend, and until now it emitted nothing -- so the funnel showed a first
|
|
3186
|
+
# run attempted, then silence, indistinguishable from a successful build.
|
|
3187
|
+
# Bounded enum only (`not_logged_in`), never the login state, path or
|
|
3188
|
+
# credential; backgrounded and non-fatal so a diagnostic can never break
|
|
3189
|
+
# the refusal it is describing.
|
|
3150
3190
|
if [[ "$_login_state" == "loggedout" ]]; then
|
|
3191
|
+
if declare -f loki_emit_first_run_blocked >/dev/null 2>&1; then
|
|
3192
|
+
( loki_emit_first_run_blocked "not_logged_in" >/dev/null 2>&1 </dev/null & ) 2>/dev/null || true
|
|
3193
|
+
fi
|
|
3151
3194
|
log_error "Claude Code is installed but not logged in -- the build would stall instead of running."
|
|
3152
3195
|
log_error "Log in once, then retry:"
|
|
3153
3196
|
log_error " claude login"
|
|
3154
3197
|
log_error "(or set ANTHROPIC_API_KEY, or LOKI_SKIP_AUTH_PREFLIGHT=1 to bypass this check)"
|
|
3155
3198
|
return 1
|
|
3156
3199
|
elif [[ "$_login_state" == "expired" ]]; then
|
|
3200
|
+
if declare -f loki_emit_first_run_blocked >/dev/null 2>&1; then
|
|
3201
|
+
( loki_emit_first_run_blocked "not_logged_in" >/dev/null 2>&1 </dev/null & ) 2>/dev/null || true
|
|
3202
|
+
fi
|
|
3157
3203
|
log_error "Your Claude Code login has expired -- the build would stall instead of running."
|
|
3158
3204
|
log_error "Fix it in one step, then retry:"
|
|
3159
3205
|
log_error " claude login"
|
|
@@ -3303,10 +3349,34 @@ detect_complexity() {
|
|
|
3303
3349
|
file_count="${file_count:-0}"
|
|
3304
3350
|
file_count="${file_count//[^0-9]/}"
|
|
3305
3351
|
|
|
3306
|
-
# Check for external integrations
|
|
3352
|
+
# Check for external integrations.
|
|
3353
|
+
#
|
|
3354
|
+
# THE EXCLUDES ARE LOAD-BEARING. This grep used to prune nothing while the
|
|
3355
|
+
# find eleven lines above it prunes node_modules/.git/vendor/dist/build --
|
|
3356
|
+
# same function, same intent, inconsistent implementation. With --include
|
|
3357
|
+
# "*.json" that meant ANY transitive dependency whose package.json mentions
|
|
3358
|
+
# azure, stripe or aws-sdk set has_external=true.
|
|
3359
|
+
#
|
|
3360
|
+
# And has_external does not merely block "simple": in the classifier below it
|
|
3361
|
+
# jumps straight to "complex", skipping "standard". So a one-liner in any
|
|
3362
|
+
# repo that has ever run npm install landed on the MOST expensive tier, which
|
|
3363
|
+
# then runs the architecture doc suite (up to 300s of silence per attempt)
|
|
3364
|
+
# and holds the council's forced minimum-iteration floor at 3 instead of 1.
|
|
3365
|
+
#
|
|
3366
|
+
# Reproduced from scratch before fixing: a project with ONE dependency naming
|
|
3367
|
+
# @azure/core classified complex; adding --exclude-dir=node_modules made the
|
|
3368
|
+
# identical project classify simple. It also fired TRUE on this repo.
|
|
3369
|
+
#
|
|
3370
|
+
# A prior incident matches exactly -- a coffee landing page took 1h34m over
|
|
3371
|
+
# 11 iterations because the simple fast-path never engaged. The fast path was
|
|
3372
|
+
# correctly built and correctly wired the whole time; this one missing prune
|
|
3373
|
+
# was what made it unreachable.
|
|
3307
3374
|
local has_external=false
|
|
3308
3375
|
if grep -rq "oauth\|SAML\|OIDC\|stripe\|twilio\|aws-sdk\|@google-cloud\|azure" \
|
|
3309
|
-
"$target_dir" --include="*.json" --include="*.ts" --include="*.js"
|
|
3376
|
+
"$target_dir" --include="*.json" --include="*.ts" --include="*.js" \
|
|
3377
|
+
--exclude-dir=node_modules --exclude-dir=.git --exclude-dir=vendor \
|
|
3378
|
+
--exclude-dir=dist --exclude-dir=build --exclude-dir=__pycache__ \
|
|
3379
|
+
--exclude-dir=.venv --exclude-dir=venv 2>/dev/null; then
|
|
3310
3380
|
has_external=true
|
|
3311
3381
|
fi
|
|
3312
3382
|
|
|
@@ -7643,6 +7713,80 @@ generate_proof_of_run() {
|
|
|
7643
7713
|
return 0
|
|
7644
7714
|
}
|
|
7645
7715
|
|
|
7716
|
+
# capture_preedit_snapshot: freeze the agent's raw diff BEFORE anything else
|
|
7717
|
+
# touches the tree, so quality numbers measure the agent and not the
|
|
7718
|
+
# agent-plus-whoever-fixed-it (autonomy/lib/preedit_snapshot.py owns the schema
|
|
7719
|
+
# and the write-once rule; this is only the call site).
|
|
7720
|
+
#
|
|
7721
|
+
# WHY THIS IS NOT INSIDE generate_proof_of_run, even though the receipt is the
|
|
7722
|
+
# obvious neighbour. Two reasons, both measured in this file:
|
|
7723
|
+
#
|
|
7724
|
+
# 1. TOO LATE AT THE LATE PROOF SITES. commit_session_changes commits the
|
|
7725
|
+
# session's work, and the module's default baseline is `git diff HEAD`.
|
|
7726
|
+
# After that commit `git diff HEAD` is EMPTY, so a capture at the teardown
|
|
7727
|
+
# proof site would freeze an empty diff -- and because the snapshot is
|
|
7728
|
+
# write-once by design, that empty capture would be permanent and
|
|
7729
|
+
# unrecoverable. run.sh already documents this mutation window itself: the
|
|
7730
|
+
# comment above the final generate_proof_of_run call says "HANDOFF.md and
|
|
7731
|
+
# commit_session_changes can change the worktree after the earlier receipt".
|
|
7732
|
+
# The receipt can be regenerated against a later tree; the snapshot cannot.
|
|
7733
|
+
# 2. WRONG GATE. Every generate_proof_of_run call site is gated on
|
|
7734
|
+
# LOKI_PROOF!=0, and the run_id resolution only exists inside its
|
|
7735
|
+
# LOKI_PROVEN_PR!=0 branch. Authorship evidence and shareable proofs are
|
|
7736
|
+
# different concerns, so a user who turns off proofs must not silently lose
|
|
7737
|
+
# the ability to tell agent output from human edits.
|
|
7738
|
+
#
|
|
7739
|
+
# So the capture happens EARLIER, immediately after run_autonomous returns,
|
|
7740
|
+
# before any post-processing step can modify the diff.
|
|
7741
|
+
#
|
|
7742
|
+
# Baseline: _LOKI_RUN_START_SHA (exported at runner init, persisted to
|
|
7743
|
+
# .loki/state/start-sha) is passed when available, so the snapshot is anchored
|
|
7744
|
+
# to the run's own starting commit rather than to a moving HEAD. That makes the
|
|
7745
|
+
# capture correct even if a later caller fires after a commit. Falls back to the
|
|
7746
|
+
# module's `git diff HEAD` default when no baseline resolved (greenfield repos
|
|
7747
|
+
# with no commits write an empty file there by design).
|
|
7748
|
+
#
|
|
7749
|
+
# run_id: read-path _loki_trust_run_id ONLY, never --new (minting here would
|
|
7750
|
+
# clobber the trust-events id file). If it resolves empty we SKIP: a snapshot
|
|
7751
|
+
# filed under an id nothing else references is worse than no snapshot, because
|
|
7752
|
+
# verdict.py would count it as authorship evidence that no receipt can join to.
|
|
7753
|
+
#
|
|
7754
|
+
# Guarded and non-fatal throughout: a diagnostic must never break the run it is
|
|
7755
|
+
# diagnosing. Write-once makes repeat calls free (later ones return "exists"),
|
|
7756
|
+
# so the earliest caller wins and extra call sites cost nothing.
|
|
7757
|
+
capture_preedit_snapshot() {
|
|
7758
|
+
local snap="$SCRIPT_DIR/lib/preedit_snapshot.py"
|
|
7759
|
+
[ -f "$snap" ] || return 0
|
|
7760
|
+
command -v python3 >/dev/null 2>&1 || return 0
|
|
7761
|
+
# Match _loki_trust_run_id's dir expression, not generate_proof_of_run's:
|
|
7762
|
+
# with LOKI_DIR set, ${TARGET_DIR:-.}/.loki would write the snapshot beside
|
|
7763
|
+
# a run-id file that lives somewhere else.
|
|
7764
|
+
local loki_dir="${LOKI_DIR:-${TARGET_DIR:-.}/.loki}"
|
|
7765
|
+
[ -d "$loki_dir" ] || return 0
|
|
7766
|
+
local _rid=""
|
|
7767
|
+
if declare -f _loki_trust_run_id >/dev/null 2>&1; then
|
|
7768
|
+
_rid="$(_loki_trust_run_id 2>/dev/null || true)"
|
|
7769
|
+
fi
|
|
7770
|
+
[ -n "$_rid" ] || return 0
|
|
7771
|
+
# Resolve the baseline from the PERSISTED file when the exported variable is
|
|
7772
|
+
# not visible. _LOKI_RUN_START_SHA is exported inside run_autonomous, and in
|
|
7773
|
+
# PARALLEL_MODE run_autonomous runs in a subshell -- an export from a
|
|
7774
|
+
# subshell never reaches the parent, so at that call site the variable is
|
|
7775
|
+
# empty and only the file survives. Same read the pause path already does.
|
|
7776
|
+
# Without this the parallel branch would silently fall back to `git diff
|
|
7777
|
+
# HEAD` instead of the anchored baseline this function documents.
|
|
7778
|
+
local _base="${_LOKI_RUN_START_SHA:-}"
|
|
7779
|
+
[ -n "$_base" ] || _base="$(cat "$loki_dir/state/start-sha" 2>/dev/null || true)"
|
|
7780
|
+
# LOKI_PREEDIT_CWD is load-bearing: the module defaults cwd to os.getcwd(),
|
|
7781
|
+
# and if that is not the target repo capture returns not_a_git_repo and
|
|
7782
|
+
# writes nothing SILENTLY -- a call site that looks wired but never fires.
|
|
7783
|
+
LOKI_DIR="$loki_dir" \
|
|
7784
|
+
LOKI_PREEDIT_CWD="${TARGET_DIR:-.}" \
|
|
7785
|
+
LOKI_RUN_START_SHA="$_base" \
|
|
7786
|
+
python3 "$snap" capture "$_rid" >/dev/null 2>&1 || true
|
|
7787
|
+
return 0
|
|
7788
|
+
}
|
|
7789
|
+
|
|
7646
7790
|
# print_ttfv_next_steps: R7 zero-config first-run "what next / go deeper"
|
|
7647
7791
|
# message. The wording MUST match what actually ran, so it branches on the mode:
|
|
7648
7792
|
# - brief: a one-line brief ran on the lightweight profile (council off,
|
|
@@ -10716,6 +10860,14 @@ LOKI_STUCK_JSON
|
|
|
10716
10860
|
|
|
10717
10861
|
track_gate_failure() {
|
|
10718
10862
|
local gate_name="$1"
|
|
10863
|
+
# Optional evidence for the durable failure lesson (see the failure_memory
|
|
10864
|
+
# block below). Either a findings-artifact PATH or a literal detail string;
|
|
10865
|
+
# callers pass whichever they already name on an adjacent line.
|
|
10866
|
+
#
|
|
10867
|
+
# MUST be "${2:-}", not "$2": this file runs under `set -u` (line 185) and
|
|
10868
|
+
# most call sites are still one-arg, so a bare $2 aborts the gate it is only
|
|
10869
|
+
# supposed to be observing. Caught by the end-to-end check, not by review.
|
|
10870
|
+
local evidence="${2:-}"
|
|
10719
10871
|
local gate_file="${TARGET_DIR:-.}/.loki/quality/gate-failure-count.json"
|
|
10720
10872
|
mkdir -p "$(dirname "$gate_file")"
|
|
10721
10873
|
|
|
@@ -10752,6 +10904,56 @@ print(counts[gate_name])
|
|
|
10752
10904
|
# echoed count or any gate behavior.
|
|
10753
10905
|
record_trust_event_bash "gate_failure" "gate=${gate_name}" "consecutive=${count}" >/dev/null 2>&1 || true
|
|
10754
10906
|
|
|
10907
|
+
# Failure memory: turn this measured failure into a durable, falsifiable
|
|
10908
|
+
# lesson the NEXT run is told about (read side: build_prompt, below the
|
|
10909
|
+
# cache breakpoint).
|
|
10910
|
+
#
|
|
10911
|
+
# EVIDENCE IS REQUIRED, and deliberately not defaulted. failure_memory.py
|
|
10912
|
+
# refuses to write without it, because a lesson recorded from the agent's
|
|
10913
|
+
# own account of why it failed is unfalsifiable -- it records what the agent
|
|
10914
|
+
# BELIEVED, which is exactly what was wrong. Passing "$gate_name" as its own
|
|
10915
|
+
# evidence would satisfy the truthiness check and defeat that, so callers
|
|
10916
|
+
# with nothing concrete in scope pass nothing and record nothing.
|
|
10917
|
+
#
|
|
10918
|
+
# A readable evidence PATH is reduced to its first non-blank, non-comment
|
|
10919
|
+
# line with ANSI colour stripped -- the same reduction _loki_gate_stuck
|
|
10920
|
+
# applies above, so the stored lesson matches the cause that valve compares.
|
|
10921
|
+
#
|
|
10922
|
+
# CRITICAL: this function's stdout IS its return value, so this is fully
|
|
10923
|
+
# stdout-suppressed and best-effort, exactly like the trust-event write
|
|
10924
|
+
# above. failure_memory.py exits 3 on an UNKNOWN status (an expected result,
|
|
10925
|
+
# not an error), hence the `|| true`.
|
|
10926
|
+
#
|
|
10927
|
+
# ponytail: failures.jsonl is append-only with no dedup, so a gate stuck for
|
|
10928
|
+
# N iterations writes N records and recall() reads the whole file. Counts
|
|
10929
|
+
# stay true, so this is a ceiling not a defect; dedup on gate+evidence if a
|
|
10930
|
+
# perpetual run ever makes the file big enough to matter.
|
|
10931
|
+
if [ -n "$evidence" ] && [ -r "${SCRIPT_DIR}/lib/failure_memory.py" ]; then
|
|
10932
|
+
local _fm_evidence="$evidence"
|
|
10933
|
+
if [ -r "$_fm_evidence" ] && [ -f "$_fm_evidence" ]; then
|
|
10934
|
+
# `|| true`: head closing the pipe kills grep with SIGPIPE, which is
|
|
10935
|
+
# nonzero under `set -o pipefail` (line 185) even though the value is
|
|
10936
|
+
# correct. Same discipline as the trust-event write above.
|
|
10937
|
+
_fm_evidence="$(grep -vE '^[[:space:]]*(#|$)' "$_fm_evidence" 2>/dev/null \
|
|
10938
|
+
| head -1 | sed 's/\x1b\[[0-9;]*m//g' | head -c 200 || true)"
|
|
10939
|
+
fi
|
|
10940
|
+
if [ -n "$_fm_evidence" ]; then
|
|
10941
|
+
# run_id via the repo's existing resolver (the same one
|
|
10942
|
+
# record_trust_event_bash uses above), so a lesson can be traced back
|
|
10943
|
+
# to the run that produced it. Resolves to "" if unavailable, which
|
|
10944
|
+
# the module accepts -- only EVIDENCE is mandatory.
|
|
10945
|
+
local _fm_run_id=""
|
|
10946
|
+
if declare -f _loki_trust_run_id >/dev/null 2>&1; then
|
|
10947
|
+
_fm_run_id="${LOKI_TRUST_RUN_ID:-$(_loki_trust_run_id 2>/dev/null || true)}"
|
|
10948
|
+
fi
|
|
10949
|
+
LOKI_DIR="${LOKI_DIR:-${TARGET_DIR:-.}/.loki}" \
|
|
10950
|
+
python3 "${SCRIPT_DIR}/lib/failure_memory.py" record \
|
|
10951
|
+
"--gate=${gate_name}" "--verdict=FAIL" \
|
|
10952
|
+
"--evidence=${_fm_evidence}" \
|
|
10953
|
+
"--run_id=${_fm_run_id}" >/dev/null 2>&1 || true
|
|
10954
|
+
fi
|
|
10955
|
+
fi
|
|
10956
|
+
|
|
10755
10957
|
echo "$count"
|
|
10756
10958
|
}
|
|
10757
10959
|
|
|
@@ -12153,6 +12355,14 @@ auto_generate_docs_if_needed() {
|
|
|
12153
12355
|
elif command -v timeout >/dev/null 2>&1; then
|
|
12154
12356
|
_doc_cmd=(timeout "${_doc_to}s")
|
|
12155
12357
|
fi
|
|
12358
|
+
# SAY WHAT IS HAPPENING BEFORE GOING QUIET. This call discards child output
|
|
12359
|
+
# and can run for the full timeout (default 300s), so without this line the
|
|
12360
|
+
# user sees a single "Auto-documentation" header and then minutes of nothing.
|
|
12361
|
+
# A silent gap reads as a hang: the observed incident was a build stuck ~55
|
|
12362
|
+
# min here with the work committed but never pushed, and nothing on screen
|
|
12363
|
+
# said which step owned the time. Naming the step and its cap turns an
|
|
12364
|
+
# apparent freeze into a bounded wait the user can reason about.
|
|
12365
|
+
log_info "Auto-documentation: generating architecture suite (no output until it finishes; up to ${_doc_to}s)"
|
|
12156
12366
|
if "${_doc_cmd[@]}" "$loki_bin" docs generate "$project_dir" >/dev/null 2>&1; then
|
|
12157
12367
|
:
|
|
12158
12368
|
else
|
|
@@ -12211,6 +12421,10 @@ run_magic_debate_gate() {
|
|
|
12211
12421
|
# verdict on genuinely thin input, not a spurious process block.
|
|
12212
12422
|
log_info "Magic Modules: running debate on '$latest_name'"
|
|
12213
12423
|
local debate_out debate_rc
|
|
12424
|
+
# Captured to a variable, so nothing reaches the screen for up to 300s. Same
|
|
12425
|
+
# reasoning as the doc suite above: name the step and its cap so a bounded
|
|
12426
|
+
# wait does not read as a hang.
|
|
12427
|
+
log_info "Magic debate: reviewing $latest_name (2 rounds, output shown when it finishes; up to 300s)"
|
|
12214
12428
|
debate_out=$(cd "$TARGET_DIR" && PYTHONPATH="$PROJECT_DIR" LOKI_PROVIDER="${PROVIDER_NAME:-claude}" \
|
|
12215
12429
|
timeout 300 "$PROJECT_DIR/autonomy/loki" magic debate "$latest_name" --rounds 2 2>&1) \
|
|
12216
12430
|
&& debate_rc=0 || debate_rc=$?
|
|
@@ -17253,6 +17467,70 @@ except Exception:
|
|
|
17253
17467
|
# tried to signal completion via state files; we now honor that.
|
|
17254
17468
|
#
|
|
17255
17469
|
# Output on stdout: the JSON payload (for callers that want to log it).
|
|
17470
|
+
# _loki_check_claim_grounding: does the completion claim name files this run
|
|
17471
|
+
# actually changed? Report-only, never a gate.
|
|
17472
|
+
#
|
|
17473
|
+
# READS THE SIGNAL FILE WITHOUT CONSUMING IT. check_task_completion_signal below
|
|
17474
|
+
# owns consumption (rm -f on read); this must run BEFORE that owner and must not
|
|
17475
|
+
# race it, so it only ever opens the file for reading. Both signal shapes carry
|
|
17476
|
+
# the text under the same key: the MCP tool writes {"statement": ...}, and the
|
|
17477
|
+
# COMPLETION_REQUESTED fallback is normalised into the same envelope by the
|
|
17478
|
+
# owner. One key covers both.
|
|
17479
|
+
#
|
|
17480
|
+
# The changed-file set is derived from _LOKI_RUN_START_SHA -- the same baseline
|
|
17481
|
+
# the evidence gate and the review diff use (run.sh:13817) -- so the receipt and
|
|
17482
|
+
# the grounding line describe ONE diff. Untracked files are included: a claim
|
|
17483
|
+
# naming a file the agent created but never staged is grounded, and calling it
|
|
17484
|
+
# ungrounded would be exactly the false positive this check must never produce.
|
|
17485
|
+
#
|
|
17486
|
+
# Passed via --files-from, never --files: --files is a comma-separated list, so
|
|
17487
|
+
# any path containing a comma would split into two bogus paths, and a large
|
|
17488
|
+
# changed set would approach ARG_MAX. The module's own comment documents that
|
|
17489
|
+
# flag's history.
|
|
17490
|
+
_loki_check_claim_grounding() {
|
|
17491
|
+
local lib="${SCRIPT_DIR}/lib/claim_grounding.py"
|
|
17492
|
+
[ -f "$lib" ] || return 0
|
|
17493
|
+
command -v python3 >/dev/null 2>&1 || return 0
|
|
17494
|
+
|
|
17495
|
+
local target="${TARGET_DIR:-.}"
|
|
17496
|
+
local sig="$target/.loki/signals/TASK_COMPLETION_CLAIMED"
|
|
17497
|
+
[ -f "$sig" ] || sig="$target/.loki/signals/COMPLETION_REQUESTED"
|
|
17498
|
+
[ -f "$sig" ] || return 0
|
|
17499
|
+
|
|
17500
|
+
local claim
|
|
17501
|
+
claim=$(python3 -c "
|
|
17502
|
+
import json, sys
|
|
17503
|
+
try:
|
|
17504
|
+
d = json.load(open(sys.argv[1]))
|
|
17505
|
+
sys.stdout.write(str(d.get('statement', '')) if isinstance(d, dict) else '')
|
|
17506
|
+
except Exception:
|
|
17507
|
+
pass
|
|
17508
|
+
" "$sig" 2>/dev/null || echo "")
|
|
17509
|
+
# A signal with no statement (bare touch of COMPLETION_REQUESTED) is
|
|
17510
|
+
# UNGROUNDABLE, not a finding. Nothing to check; leave no stale artifact.
|
|
17511
|
+
[ -n "$claim" ] || return 0
|
|
17512
|
+
|
|
17513
|
+
local files_tmp="$target/.loki/state/claim-grounding-files.$$"
|
|
17514
|
+
mkdir -p "$target/.loki/state" 2>/dev/null || return 0
|
|
17515
|
+
{
|
|
17516
|
+
if [ -n "${_LOKI_RUN_START_SHA:-}" ] \
|
|
17517
|
+
&& git -C "$target" rev-parse --verify --quiet "${_LOKI_RUN_START_SHA}^{commit}" >/dev/null 2>&1; then
|
|
17518
|
+
git -C "$target" diff --name-only "${_LOKI_RUN_START_SHA}" 2>/dev/null
|
|
17519
|
+
else
|
|
17520
|
+
git -C "$target" diff --name-only HEAD 2>/dev/null
|
|
17521
|
+
fi
|
|
17522
|
+
git -C "$target" diff --name-only --cached 2>/dev/null
|
|
17523
|
+
git -C "$target" ls-files --others --exclude-standard 2>/dev/null
|
|
17524
|
+
} | sort -u > "$files_tmp" 2>/dev/null || { rm -f "$files_tmp" 2>/dev/null; return 0; }
|
|
17525
|
+
|
|
17526
|
+
# Exit 1 means "a named path is absent from the diff" -- the finding itself,
|
|
17527
|
+
# not an error. Swallowed: this reports, it never blocks completion.
|
|
17528
|
+
python3 "$lib" --claim "$claim" --files-from "$files_tmp" \
|
|
17529
|
+
> "$target/.loki/state/claim-grounding.json" 2>/dev/null || true
|
|
17530
|
+
rm -f "$files_tmp" 2>/dev/null
|
|
17531
|
+
return 0
|
|
17532
|
+
}
|
|
17533
|
+
|
|
17256
17534
|
check_task_completion_signal() {
|
|
17257
17535
|
local signal_file=".loki/signals/TASK_COMPLETION_CLAIMED"
|
|
17258
17536
|
local fallback_file=".loki/signals/COMPLETION_REQUESTED"
|
|
@@ -19561,6 +19839,41 @@ if d.get('blocked'):
|
|
|
19561
19839
|
--loki-dir ".loki" --prompt-block 2>/dev/null || true)"
|
|
19562
19840
|
fi
|
|
19563
19841
|
|
|
19842
|
+
# Failure memory (read side; write side: track_gate_failure). Tells this
|
|
19843
|
+
# iteration what has actually failed in THIS repo before, so a gate the
|
|
19844
|
+
# agent has already lost to is not re-learned from scratch every run.
|
|
19845
|
+
#
|
|
19846
|
+
# COUNTS, NOT PROSE, and that restriction is the whole point. "the
|
|
19847
|
+
# mock_integrity gate has failed here 6 times" is a fact the reader can
|
|
19848
|
+
# check; "this repo tends to have mocking problems" is a generalization that
|
|
19849
|
+
# reads identically and is not falsifiable. The module renders the lines and
|
|
19850
|
+
# this only prints them -- no second renderer to drift, matching the
|
|
19851
|
+
# single-renderer discipline used for efficiency_trend above.
|
|
19852
|
+
#
|
|
19853
|
+
# Emits "" when nothing has been recorded, so a repo with no failure history
|
|
19854
|
+
# adds NOTHING to the prompt (and the 60 build_prompt parity fixtures, none
|
|
19855
|
+
# of which carry a failures.jsonl, stay byte-identical).
|
|
19856
|
+
#
|
|
19857
|
+
# Calls prompt_context() in-process rather than the CLI: the CLI prints JSON
|
|
19858
|
+
# and exits 3 on an UNKNOWN status, which is an expected "no lessons yet"
|
|
19859
|
+
# result and not an error worth parsing around.
|
|
19860
|
+
local failure_memory_context=""
|
|
19861
|
+
if [ -r "${SCRIPT_DIR}/lib/failure_memory.py" ] && [ -d ".loki" ]; then
|
|
19862
|
+
failure_memory_context="$(_FM_LIB="${SCRIPT_DIR}/lib" \
|
|
19863
|
+
_FM_DIR="${LOKI_DIR:-${TARGET_DIR:-.}/.loki}" python3 -c '
|
|
19864
|
+
import os, sys
|
|
19865
|
+
sys.path.insert(0, os.environ["_FM_LIB"])
|
|
19866
|
+
try:
|
|
19867
|
+
from failure_memory import prompt_context
|
|
19868
|
+
lines = prompt_context(os.environ["_FM_DIR"]).get("lines") or []
|
|
19869
|
+
except Exception:
|
|
19870
|
+
lines = []
|
|
19871
|
+
if lines:
|
|
19872
|
+
print("KNOWN FAILURE HISTORY IN THIS REPO (measured, from previous runs): "
|
|
19873
|
+
+ "; ".join(lines) + ".")
|
|
19874
|
+
' 2>/dev/null || true)"
|
|
19875
|
+
fi
|
|
19876
|
+
|
|
19564
19877
|
# PRD Checklist status injection (v5.44.0)
|
|
19565
19878
|
local checklist_status=""
|
|
19566
19879
|
if [ -n "$prd" ] && [ ! -f ".loki/checklist/checklist.json" ]; then
|
|
@@ -19776,8 +20089,11 @@ except Exception:
|
|
|
19776
20089
|
if [ -n "$gate_escalation_context" ]; then
|
|
19777
20090
|
_legacy_priority="${_legacy_priority}${_legacy_priority:+ }${gate_escalation_context}"
|
|
19778
20091
|
fi
|
|
20092
|
+
# Same cap, same reason, same default as the degraded path above --
|
|
20093
|
+
# a pasted spec has to be bounded, but the bound must not silently
|
|
20094
|
+
# eat requirements. 4000 bytes dropped anything past ~600 words.
|
|
19779
20095
|
if [ -n "$prd" ] && [ -f "$prd" ]; then
|
|
19780
|
-
_legacy_prd_content=$(head -c
|
|
20096
|
+
_legacy_prd_content=$(head -c "${LOKI_DEGRADED_PRD_CAP:-24000}" "$prd")
|
|
19781
20097
|
fi
|
|
19782
20098
|
if [ $retry -eq 0 ]; then
|
|
19783
20099
|
if [ -n "$prd" ]; then
|
|
@@ -19828,9 +20144,35 @@ except Exception:
|
|
|
19828
20144
|
|
|
19829
20145
|
if [ "${PROVIDER_DEGRADED:-false}" = "true" ]; then
|
|
19830
20146
|
# Degraded providers: simpler wording, but still static-first.
|
|
20147
|
+
#
|
|
20148
|
+
# THE CAP IS NOW ANNOUNCED, NOT SILENT. This path PASTES the spec text
|
|
20149
|
+
# (a degraded provider cannot be told "read the file at this path" the
|
|
20150
|
+
# way claude/cline/opencode are at :20196), so it has to be bounded. It
|
|
20151
|
+
# was bounded at 4000 bytes with no notice: a requirement past ~600 words
|
|
20152
|
+
# was dropped mid-sentence and the model never knew a spec existed beyond
|
|
20153
|
+
# what it saw. Demonstrated on a 4229-byte spec -- the requirement on the
|
|
20154
|
+
# last line was simply absent from what the model received.
|
|
20155
|
+
#
|
|
20156
|
+
# That is the same class of defect spec-expand.sh:5-7 already names for
|
|
20157
|
+
# OpenAPI ("a 40-operation file loses 21 of 40 ops") and fixed for
|
|
20158
|
+
# contracts only. Markdown specs still had it, and only for the two
|
|
20159
|
+
# degraded providers -- so codex and aider users silently got a worse
|
|
20160
|
+
# build than claude users from the identical spec.
|
|
20161
|
+
#
|
|
20162
|
+
# Raised to 24000 (a large PRD fits whole) and, when the spec still
|
|
20163
|
+
# exceeds it, the model is TOLD so and given the path to read the rest.
|
|
20164
|
+
# An unannounced truncation makes the model confidently build the wrong
|
|
20165
|
+
# thing; an announced one makes it go look.
|
|
20166
|
+
local _prd_cap="${LOKI_DEGRADED_PRD_CAP:-24000}"
|
|
19831
20167
|
local prd_content=""
|
|
20168
|
+
local _prd_truncated=0
|
|
19832
20169
|
if [ -n "$prd" ] && [ -f "$prd" ]; then
|
|
19833
|
-
prd_content=$(head -c
|
|
20170
|
+
prd_content=$(head -c "$_prd_cap" "$prd")
|
|
20171
|
+
local _prd_bytes
|
|
20172
|
+
_prd_bytes=$(wc -c < "$prd" 2>/dev/null | tr -d ' ')
|
|
20173
|
+
if [ -n "$_prd_bytes" ] && [ "$_prd_bytes" -gt "$_prd_cap" ] 2>/dev/null; then
|
|
20174
|
+
_prd_truncated=1
|
|
20175
|
+
fi
|
|
19834
20176
|
fi
|
|
19835
20177
|
|
|
19836
20178
|
local degraded_prd_anchor="Loki Mode"
|
|
@@ -19859,6 +20201,38 @@ except Exception:
|
|
|
19859
20201
|
[ -n "$queue_tasks" ] && printf 'Tasks: %s\n' "$queue_tasks"
|
|
19860
20202
|
if [ -n "$prd" ]; then
|
|
19861
20203
|
printf 'PRD contents: %s\n' "$prd_content"
|
|
20204
|
+
# Announce the cut. Silence here is what made the old 4000-byte cap
|
|
20205
|
+
# dangerous: the model treated a partial spec as the whole spec and
|
|
20206
|
+
# built confidently against requirements it had never seen. Naming
|
|
20207
|
+
# the file lets it read the remainder itself.
|
|
20208
|
+
if [ "${_prd_truncated:-0}" = "1" ]; then
|
|
20209
|
+
printf 'NOTE: the spec above is TRUNCATED at %s bytes. The full spec is at %s -- read it before deciding the work is complete.\n' \
|
|
20210
|
+
"$_prd_cap" "$prd"
|
|
20211
|
+
fi
|
|
20212
|
+
fi
|
|
20213
|
+
|
|
20214
|
+
# FIRST-PASS EXCELLENCE FOR DEGRADED PROVIDERS.
|
|
20215
|
+
#
|
|
20216
|
+
# This directive existed only for Claude. providers/claude.sh:322 injects
|
|
20217
|
+
# it via --append-system-prompt, a flag codex/aider do not have, so the
|
|
20218
|
+
# one mechanism built specifically to make a WEAKER model land complete
|
|
20219
|
+
# on iteration 1 reached only the strongest one. Measured before writing
|
|
20220
|
+
# this: grep for FIRST_PASS_EXCELLENCE returns 0 in codex.sh, aider.sh,
|
|
20221
|
+
# cline.sh and opencode.sh.
|
|
20222
|
+
#
|
|
20223
|
+
# It matters most exactly where it was missing. The premise (recorded
|
|
20224
|
+
# when the Claude version was built) is that iteration count is a proxy
|
|
20225
|
+
# for how much the first pass missed, and that for a weak model context
|
|
20226
|
+
# quality beats iteration count. Codex is also the free on-ramp, so the
|
|
20227
|
+
# users least able to absorb a bad build were the ones getting no help.
|
|
20228
|
+
#
|
|
20229
|
+
# Condensed rather than byte-mirrored: the Claude text is ~4.3KB of
|
|
20230
|
+
# system prompt, and these providers take it inline in the user turn
|
|
20231
|
+
# where budget is tighter. The four load-bearing instructions are kept --
|
|
20232
|
+
# build fully, wire the backend, verify by RUNNING, commit to one design.
|
|
20233
|
+
# Same iteration-1 gate and same env var, so one switch controls both.
|
|
20234
|
+
if [ "${LOKI_FIRST_PASS_EXCELLENCE:-1}" != "0" ] && [ "${iteration:-1}" -le 1 ] 2>/dev/null; then
|
|
20235
|
+
printf '%s\n' '[FIRST-PASS EXCELLENCE] Treat THIS pass as your one shot to ship a complete, working solution. The loop is a safety net, not a plan. 1) BUILD IT FULLY: no stubs, no TODOs, no placeholder or mock data where real logic belongs. If the spec implies a backend (auth, persistence, a form that submits), WIRE IT so it actually persists -- a UI whose buttons do nothing is the most common failure. 2) VERIFY BY RUNNING each acceptance path, not by reading the code. 3) DECIDE the architecture now rather than refactoring later. 4) Commit to ONE specific design; avoid the generic purple-gradient default look.'
|
|
19862
20236
|
fi
|
|
19863
20237
|
printf '</dynamic_context>\n'
|
|
19864
20238
|
return 0
|
|
@@ -19968,6 +20342,10 @@ except Exception:
|
|
|
19968
20342
|
[ -n "$app_runner_info" ] && printf '%s\n' "$app_runner_info"
|
|
19969
20343
|
[ -n "$playwright_info" ] && printf '%s\n' "$playwright_info"
|
|
19970
20344
|
[ -n "$memory_context_section" ] && printf '%s\n' "$memory_context_section"
|
|
20345
|
+
# Failure memory: volatile (it changes the moment a gate fails), so it lives
|
|
20346
|
+
# here in the dynamic tail, never in the cache-stable <loki_system> prefix.
|
|
20347
|
+
# Sits with the other memory context, before the efficiency trend.
|
|
20348
|
+
[ -n "$failure_memory_context" ] && printf '%s\n' "$failure_memory_context"
|
|
19971
20349
|
# Volatile per-iteration data: belongs below [CACHE_BREAKPOINT], never in the
|
|
19972
20350
|
# cache-stable prefix. Same ordinal position as the Bun route (after the
|
|
19973
20351
|
# context section, before the completion instruction).
|
|
@@ -22796,6 +23174,73 @@ if __name__ == "__main__":
|
|
|
22796
23174
|
# costs zero extra subprocesses -- we pass the existing epoch through.
|
|
22797
23175
|
emit_stage_complete "agent" "$([ "$exit_code" -eq 0 ] 2>/dev/null && echo pass || echo fail)" "$start_time"
|
|
22798
23176
|
|
|
23177
|
+
# LLM DECISION RECORD (autonomy/lib/decision_record.py).
|
|
23178
|
+
#
|
|
23179
|
+
# WHY HERE. This is the single point where every provider arm converges
|
|
23180
|
+
# after dispatch: claude, codex, cline and aider all land here with
|
|
23181
|
+
# $tier_param (the model actually dispatched), $exit_code and $duration
|
|
23182
|
+
# in scope. Recording per-arm would be four call sites that drift.
|
|
23183
|
+
#
|
|
23184
|
+
# WHY tier_param AND NOT LOKI_CURRENT_MODEL. Only the claude arm exports
|
|
23185
|
+
# LOKI_CURRENT_MODEL (line ~22214); on a codex/cline/aider iteration that
|
|
23186
|
+
# variable is either unset or a STALE value left by an earlier claude
|
|
23187
|
+
# iteration after a failover. tier_param is the same string the claude
|
|
23188
|
+
# arm exports, and it is correct on every arm. It is read AFTER every
|
|
23189
|
+
# mutation (opus-pin force, LOKI_MAX_TIER clamp, mid-flight override,
|
|
23190
|
+
# fable collapse), so it is the model that ran, not the tier alias.
|
|
23191
|
+
#
|
|
23192
|
+
# WHAT IS DELIBERATELY OMITTED. temperature: this runtime never sets one
|
|
23193
|
+
# on any provider (claude dispatch passes --model/--effort, never a
|
|
23194
|
+
# temperature), so writing a value would be inventing the exact field
|
|
23195
|
+
# whose whole purpose is making config drift falsifiable. The module
|
|
23196
|
+
# treats an absent field as absent; a guessed 0.0 would be a lie that
|
|
23197
|
+
# reads as a measurement. confidence: self-reported and not available at
|
|
23198
|
+
# this seam. Tokens come from the authoritative per-iteration result-cost
|
|
23199
|
+
# file when the provider wrote one, and are omitted rather than zeroed
|
|
23200
|
+
# when it did not (a zero claims the call was free).
|
|
23201
|
+
#
|
|
23202
|
+
# NON-FATAL AND BACKGROUNDED: a diagnostic must never be able to break
|
|
23203
|
+
# the iteration it is diagnosing, and this is a python3 spawn on the
|
|
23204
|
+
# critical path of the loop's largest stage.
|
|
23205
|
+
if [ -n "${tier_param:-}" ] && [ -f "${SCRIPT_DIR:-}/lib/decision_record.py" ]; then
|
|
23206
|
+
local _dr_args=(
|
|
23207
|
+
"--model_id=$tier_param"
|
|
23208
|
+
"--provider=${PROVIDER_NAME:-claude}"
|
|
23209
|
+
"--stage=iteration_${ITERATION_COUNT:-0}_${rarv_phase:-unknown}"
|
|
23210
|
+
"--outcome=$([ "$exit_code" -eq 0 ] 2>/dev/null && echo ok || echo error)"
|
|
23211
|
+
"--duration_ms=$((duration * 1000))"
|
|
23212
|
+
)
|
|
23213
|
+
# Correlation ids only when genuinely set: an empty run_id written as
|
|
23214
|
+
# "" is indistinguishable from a real one in a later diff, and the
|
|
23215
|
+
# module records whatever an allowlisted field carries.
|
|
23216
|
+
[ -n "${LOKI_TRUST_RUN_ID:-}" ] && _dr_args+=("--run_id=$LOKI_TRUST_RUN_ID") || true
|
|
23217
|
+
[ -n "${LOKI_SESSION_ID:-}" ] && _dr_args+=("--session_id=$LOKI_SESSION_ID") || true
|
|
23218
|
+
local _dr_cost="${TARGET_DIR:-.}/.loki/metrics/result-cost-${ITERATION_COUNT:-0}.json"
|
|
23219
|
+
if [ -s "$_dr_cost" ]; then
|
|
23220
|
+
# Read into named locals, NOT `set --`: this runs in the middle of
|
|
23221
|
+
# run_autonomous, and clobbering the function's positional
|
|
23222
|
+
# parameters to parse a diagnostic is how a metrics read turns
|
|
23223
|
+
# into a control-flow bug.
|
|
23224
|
+
local _dr_in="" _dr_out=""
|
|
23225
|
+
read -r _dr_in _dr_out <<EOF
|
|
23226
|
+
$(python3 -c 'import json,sys
|
|
23227
|
+
d = json.load(open(sys.argv[1]))
|
|
23228
|
+
# Print BOTH or neither: a half-record invites a reader to treat a missing
|
|
23229
|
+
# output count as zero output, which reads as "the model produced nothing".
|
|
23230
|
+
i, o = d.get("input_tokens"), d.get("output_tokens")
|
|
23231
|
+
if isinstance(i, int) and isinstance(o, int):
|
|
23232
|
+
print(i, o)' "$_dr_cost" 2>/dev/null)
|
|
23233
|
+
EOF
|
|
23234
|
+
case "${_dr_in}${_dr_out}" in
|
|
23235
|
+
''|*[!0-9]*) ;; # unparseable -> omit rather than fabricate
|
|
23236
|
+
*) _dr_args+=("--tokens_in=$_dr_in" "--tokens_out=$_dr_out") ;;
|
|
23237
|
+
esac
|
|
23238
|
+
fi
|
|
23239
|
+
( LOKI_DIR="${TARGET_DIR:-.}/.loki" \
|
|
23240
|
+
python3 "${SCRIPT_DIR}/lib/decision_record.py" record "${_dr_args[@]}" \
|
|
23241
|
+
>/dev/null 2>&1 </dev/null & ) 2>/dev/null || true
|
|
23242
|
+
fi
|
|
23243
|
+
|
|
22799
23244
|
# AGENT PROMPT SIZE. The call this brackets is 93% of a run's wall clock
|
|
22800
23245
|
# (1814s of 1941s measured), and its INPUT was never measured -- every
|
|
22801
23246
|
# reviewer logs its prompt bytes, the dominant call logged nothing.
|
|
@@ -23079,7 +23524,8 @@ if __name__ == "__main__":
|
|
|
23079
23524
|
else
|
|
23080
23525
|
_stg_ok=fail
|
|
23081
23526
|
local sa_count
|
|
23082
|
-
sa_count=$(track_gate_failure "static_analysis"
|
|
23527
|
+
sa_count=$(track_gate_failure "static_analysis" \
|
|
23528
|
+
"${TARGET_DIR:-.}/.loki/quality/static-analysis.json")
|
|
23083
23529
|
gate_failures="${gate_failures}static_analysis,"
|
|
23084
23530
|
log_warn "Static analysis FAILED ($sa_count consecutive) - findings injected into next iteration"
|
|
23085
23531
|
# F0, extended past mutation_integrity. Static analysis is
|
|
@@ -23187,7 +23633,8 @@ if __name__ == "__main__":
|
|
|
23187
23633
|
;;
|
|
23188
23634
|
fail)
|
|
23189
23635
|
local mk_count
|
|
23190
|
-
mk_count=$(track_gate_failure "mock_integrity"
|
|
23636
|
+
mk_count=$(track_gate_failure "mock_integrity" \
|
|
23637
|
+
"${TARGET_DIR:-.}/.loki/quality/mock-findings.txt")
|
|
23191
23638
|
gate_failures="${gate_failures}mock_integrity,"
|
|
23192
23639
|
log_warn "Mock integrity gate FAILED ($mk_count consecutive) - CRITICAL/HIGH mock problems"
|
|
23193
23640
|
# Escalation guidance was DEAD for this gate.
|
|
@@ -23240,7 +23687,8 @@ if __name__ == "__main__":
|
|
|
23240
23687
|
else
|
|
23241
23688
|
_stg_ok=fail
|
|
23242
23689
|
local mt_count
|
|
23243
|
-
mt_count=$(track_gate_failure "mutation_integrity"
|
|
23690
|
+
mt_count=$(track_gate_failure "mutation_integrity" \
|
|
23691
|
+
"${TARGET_DIR:-.}/.loki/quality/mutation-findings.txt")
|
|
23244
23692
|
gate_failures="${gate_failures}mutation_integrity,"
|
|
23245
23693
|
log_warn "Mutation integrity gate FAILED ($mt_count consecutive) - HIGH test-fitting detected"
|
|
23246
23694
|
# Same dead-branch fix as mock_integrity above:
|
|
@@ -23306,7 +23754,8 @@ if __name__ == "__main__":
|
|
|
23306
23754
|
_lsp_e=$(printf '%s' "${_LOKI_LSP_DIAGNOSTICS_DETAIL:-}" | awk '{print $2}')
|
|
23307
23755
|
_lsp_w=$(printf '%s' "${_LOKI_LSP_DIAGNOSTICS_DETAIL:-}" | awk '{print $3}')
|
|
23308
23756
|
local lsp_count
|
|
23309
|
-
lsp_count=$(track_gate_failure "lsp_diagnostics"
|
|
23757
|
+
lsp_count=$(track_gate_failure "lsp_diagnostics" \
|
|
23758
|
+
"${_LOKI_LSP_DIAGNOSTICS_DETAIL:-}")
|
|
23310
23759
|
log_warn "LSP diagnostics reported errors ($lsp_count consecutive) - ${_lsp_e} error(s), ${_lsp_w} warning(s); advisory only"
|
|
23311
23760
|
;;
|
|
23312
23761
|
pass)
|
|
@@ -23350,7 +23799,8 @@ if __name__ == "__main__":
|
|
|
23350
23799
|
clear_gate_failure "semantic_tests"
|
|
23351
23800
|
else
|
|
23352
23801
|
local sem_count
|
|
23353
|
-
sem_count=$(track_gate_failure "semantic_tests"
|
|
23802
|
+
sem_count=$(track_gate_failure "semantic_tests" \
|
|
23803
|
+
"${TARGET_DIR:-.}/.loki/quality/semantic-findings.txt")
|
|
23354
23804
|
if [ "${LOKI_GATE_SEMANTIC_TESTS_BLOCK:-false}" = "true" ] \
|
|
23355
23805
|
|| [ "${LOKI_GATE_SEMANTIC_TESTS_BLOCK:-false}" = "1" ]; then
|
|
23356
23806
|
gate_failures="${gate_failures}semantic_tests,"
|
|
@@ -23376,7 +23826,8 @@ if __name__ == "__main__":
|
|
|
23376
23826
|
clear_gate_failure "invariants"
|
|
23377
23827
|
else
|
|
23378
23828
|
local inv_count
|
|
23379
|
-
inv_count=$(track_gate_failure "invariants"
|
|
23829
|
+
inv_count=$(track_gate_failure "invariants" \
|
|
23830
|
+
"${TARGET_DIR:-.}/.loki/quality/invariant-findings.txt")
|
|
23380
23831
|
if [ "${LOKI_GATE_INVARIANTS_BLOCK:-false}" = "true" ] \
|
|
23381
23832
|
|| [ "${LOKI_GATE_INVARIANTS_BLOCK:-false}" = "1" ]; then
|
|
23382
23833
|
gate_failures="${gate_failures}invariants,"
|
|
@@ -23694,6 +24145,31 @@ if __name__ == "__main__":
|
|
|
23694
24145
|
|| [ -f "${TARGET_DIR:-.}/.loki/signals/COMPLETION_REQUESTED" ]; then
|
|
23695
24146
|
_loki_completion_claimed=1
|
|
23696
24147
|
fi
|
|
24148
|
+
# CLAIM GROUNDING (report-only): does the completion claim name files
|
|
24149
|
+
# that are actually in this run's diff? Every existing evidence axis is
|
|
24150
|
+
# a REPO-level fact (diff non-empty, tests green, app boots), so an
|
|
24151
|
+
# agent can finish by claiming "added retry logic to the payment
|
|
24152
|
+
# client" while the diff shows a README edit and all six axes pass.
|
|
24153
|
+
# The claim itself is the one artifact nothing else reads.
|
|
24154
|
+
#
|
|
24155
|
+
# PLACED HERE, at the non-destructive peek, deliberately: this is the
|
|
24156
|
+
# only point where the claim signal still EXISTS. The default route
|
|
24157
|
+
# below (check_completion_promise -> check_task_completion_signal)
|
|
24158
|
+
# consumes it with rm -f on read, so reading the statement after that
|
|
24159
|
+
# returns nothing, and re-reading it through the consuming detector
|
|
24160
|
+
# would re-introduce the v7.28 claim-drop bug. We read the signal file
|
|
24161
|
+
# directly and never remove it -- consumption keeps its single owner.
|
|
24162
|
+
#
|
|
24163
|
+
# FAIL-OPEN AND NON-BLOCKING BY DESIGN: only a claim naming a path
|
|
24164
|
+
# demonstrably absent from the diff is a finding, and it is written to
|
|
24165
|
+
# a file, never returned into the gate chain. claim_grounding.py exits
|
|
24166
|
+
# 1 on exactly that case, hence `|| true` -- an ungrounded claim must
|
|
24167
|
+
# report, not block. A grounding check that blocked on ambiguity would
|
|
24168
|
+
# fire on ordinary prose and be disabled within a week.
|
|
24169
|
+
if [ "$_loki_completion_claimed" = 1 ] \
|
|
24170
|
+
&& [ -f "${SCRIPT_DIR}/lib/claim_grounding.py" ]; then
|
|
24171
|
+
_loki_check_claim_grounding || true
|
|
24172
|
+
fi
|
|
23697
24173
|
local _loki_completion_ready=1
|
|
23698
24174
|
if loki_is_supervised_simple_web; then
|
|
23699
24175
|
_loki_supervised_completion_gates_pass "${gate_failures:-}" && _loki_completion_ready=0
|
|
@@ -24809,6 +25285,17 @@ except (json.JSONDecodeError, OSError): pass
|
|
|
24809
25285
|
_loki_write_termination_record "$signal_name" "$final_exit_code"
|
|
24810
25286
|
fi
|
|
24811
25287
|
emit_event_json "session_end" "result=$final_exit_code" "reason=$final_reason"
|
|
25288
|
+
# An interrupted run still produced agent output, and this teardown is
|
|
25289
|
+
# the only exit it takes -- it never reaches the post-loop capture.
|
|
25290
|
+
# Backgrounded here (and ONLY here) because this runs inside a signal
|
|
25291
|
+
# handler, where a blocking git call would stall the shutdown the user
|
|
25292
|
+
# just asked for. Backgrounding is safe at this site specifically
|
|
25293
|
+
# because the capture is ORDERING-INDEPENDENT: it baselines to the
|
|
25294
|
+
# run-start SHA, not to a moving HEAD, and nothing between here and
|
|
25295
|
+
# process exit commits -- so it records the same diff whenever the
|
|
25296
|
+
# subshell lands. Do NOT copy this backgrounding to the post-loop site,
|
|
25297
|
+
# where completing before the tree mutates is the entire point.
|
|
25298
|
+
( capture_preedit_snapshot >/dev/null 2>&1 </dev/null & ) 2>/dev/null || true
|
|
24812
25299
|
if [ "$supervised_signal" = "true" ] \
|
|
24813
25300
|
&& [ "${LOKI_PROOF:-1}" != "0" ] \
|
|
24814
25301
|
&& type generate_proof_of_run >/dev/null 2>&1; then
|
|
@@ -25528,6 +26015,13 @@ main() {
|
|
|
25528
26015
|
kill $orchestrator_pid 2>/dev/null || true
|
|
25529
26016
|
wait $orchestrator_pid 2>/dev/null || true
|
|
25530
26017
|
|
|
26018
|
+
# Same pre-edit capture as the standard branch, placed before
|
|
26019
|
+
# cleanup_parallel_streams because that tears down worktrees and can
|
|
26020
|
+
# change what the diff sees. Parallel mode never reaches the standard
|
|
26021
|
+
# branch's call site, so without this the whole mode would have no
|
|
26022
|
+
# authorship evidence. Write-once, so this is still a single snapshot.
|
|
26023
|
+
capture_preedit_snapshot || true
|
|
26024
|
+
|
|
25531
26025
|
# Cleanup parallel streams
|
|
25532
26026
|
cleanup_parallel_streams
|
|
25533
26027
|
else
|
|
@@ -25536,6 +26030,17 @@ main() {
|
|
|
25536
26030
|
# a stuck "Planning" state.
|
|
25537
26031
|
_advance_current_phase "BUILDING"
|
|
25538
26032
|
run_autonomous "$PRD_PATH" || result=$?
|
|
26033
|
+
# PRE-EDIT SNAPSHOT: freeze the agent's raw diff HERE, the first
|
|
26034
|
+
# instruction after the loop returns, because everything below this line
|
|
26035
|
+
# can change the tree -- commit_session_changes commits the work (after
|
|
26036
|
+
# which `git diff HEAD` is empty), and HANDOFF.md/learnings writers touch
|
|
26037
|
+
# files before that. The snapshot is write-once, so capturing it late
|
|
26038
|
+
# would permanently record someone else's edits as the agent's. Runs in
|
|
26039
|
+
# the FOREGROUND on purpose: the entire value of this position is that
|
|
26040
|
+
# the capture COMPLETES before any mutation, and backgrounding it would
|
|
26041
|
+
# reintroduce exactly the race the placement exists to remove (the
|
|
26042
|
+
# module bounds each git call at 60s, so the cost is bounded).
|
|
26043
|
+
capture_preedit_snapshot || true
|
|
25539
26044
|
# ZOMBIE-RECEIPT GUARD: proof generation + the COMPLETED marker live in the
|
|
25540
26045
|
# teardown far below. If the process is killed (Docker restart, OOM, worker
|
|
25541
26046
|
# reap) between here and there, a genuinely finished build (real code, exit
|