loki-mode 9.12.6 → 9.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. package/README.md +81 -101
  2. package/SKILL.md +2 -2
  3. package/VERSION +1 -1
  4. package/autonomy/intent.sh +414 -0
  5. package/autonomy/issue-providers.sh +24 -0
  6. package/autonomy/lib/agent_readiness.py +280 -0
  7. package/autonomy/lib/claim_grounding.py +171 -0
  8. package/autonomy/lib/config-map.sh +10 -6
  9. package/autonomy/lib/decision_record.py +198 -0
  10. package/autonomy/lib/failure_memory.py +199 -0
  11. package/autonomy/lib/gate_policy.py +166 -0
  12. package/autonomy/lib/outcome_ledger.py +620 -0
  13. package/autonomy/lib/preedit_snapshot.py +216 -0
  14. package/autonomy/lib/proof-generator.py +71 -4
  15. package/autonomy/lib/verdict.py +204 -0
  16. package/autonomy/loki +430 -15
  17. package/autonomy/notify.sh +70 -1
  18. package/autonomy/provider-offer.sh +25 -1
  19. package/autonomy/queue-consumer.sh +290 -18
  20. package/autonomy/run.sh +527 -12
  21. package/autonomy/telemetry.sh +8 -1
  22. package/completions/_loki +5 -0
  23. package/completions/loki.bash +2 -1
  24. package/dashboard/__init__.py +1 -1
  25. package/dashboard/run.py +13 -2
  26. package/dashboard/scim.py +221 -0
  27. package/dashboard/server.py +190 -1
  28. package/dashboard/static/index.html +248 -55
  29. package/docs/GATE-FAILURE-TRIAGE.md +254 -0
  30. package/docs/LOOP-CANDIDATE-PROPOSAL-v1.md +167 -0
  31. package/docs/LOOP-HARNESS-AUDIT.md +53 -0
  32. package/docs/QUEUE-OPERATIONS.md +107 -0
  33. package/docs/VERIFICATION-COST.md +273 -0
  34. package/docs/WANG-PRINCIPLES-PLAN.md +1 -1
  35. package/loki-ts/dist/loki.js +414 -416
  36. package/mcp/__init__.py +1 -1
  37. package/mcp/_sdk_loader.py +25 -0
  38. package/package.json +1 -1
  39. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
package/autonomy/loki CHANGED
@@ -1040,15 +1040,15 @@ show_help() {
1040
1040
  echo " agent analyze api assets audit bench checkpoint (cp) ci cleanup"
1041
1041
  echo " cluster cockpit code compliance completions compound config context (ctx)"
1042
1042
  echo " cost council crash dashboard demo deploy docker docs doctor dogfood"
1043
- echo " enterprise estimate explain export failover github grill heal help"
1044
- echo " import init"
1043
+ echo " enterprise estimate explain export failover gates github grill heal help"
1044
+ echo " import init intent"
1045
1045
  echo " issue kpis logs magic mcp memory metrics migrate modernize monitor"
1046
- echo " next notify onboard open optimize otel own (handoff) pause plan preview"
1047
- echo " projects proof (receipt) provider quick quickstart rc remote report reset"
1046
+ echo " next notify onboard open optimize otel outcomes own (handoff) pause plan preview"
1047
+ echo " projects proof (receipt) provider quick quickstart rc readiness remote report reset"
1048
1048
  echo " resume review rollback run sandbox secrets secure self-update sentrux"
1049
1049
  echo " serve setup-skill share ship spec start state stats status steer stop"
1050
1050
  echo " syslog telemetry template test tour trigger trust trust-metrics"
1051
- echo " ultracode update verify version voice watch watchdog web welcome why"
1051
+ echo " ultracode update verdict verify version voice watch watchdog web welcome why"
1052
1052
  echo " wiki worktree (wt)"
1053
1053
  echo ""
1054
1054
  echo "Any command: loki <command> --help"
@@ -10155,7 +10155,14 @@ _export_check_overwrite() {
10155
10155
  echo "y (auto-confirmed)"
10156
10156
  return 0
10157
10157
  fi
10158
- if [ ! -t 0 ]; then
10158
+ # A prompt is only answerable when BOTH ends are a terminal. Testing
10159
+ # stdin alone was the bug: under a PTY harness that redirects stdout
10160
+ # (the gate runs every check as `bash tests/... | tail -3`), `-t 0` is
10161
+ # true, so we fell through to `read` -- but the process is not the
10162
+ # terminal's foreground group, so reading the tty raises SIGTTIN and
10163
+ # STOPS it. A stopped process never runs its SIGALRM handler, which is
10164
+ # why `read -t` cannot rescue this; only not reading at all can.
10165
+ if [ ! -t 0 ] || [ ! -t 1 ]; then
10159
10166
  echo "Export cancelled (non-interactive: refusing to overwrite '$output'." >&2
10160
10167
  echo "Set LOKI_AUTO_CONFIRM=true to overwrite, or choose a new output path.)" >&2
10161
10168
  return 1
@@ -10165,7 +10172,7 @@ _export_check_overwrite() {
10165
10172
  read -r reply || reply=""
10166
10173
  case "$reply" in
10167
10174
  [yY]|[yY][eE][sS]) return 0 ;;
10168
- *) echo "Export cancelled."; return 1 ;;
10175
+ *) echo "Export cancelled (set LOKI_AUTO_CONFIRM=true to overwrite)."; return 1 ;;
10169
10176
  esac
10170
10177
  fi
10171
10178
  return 0
@@ -11511,7 +11518,16 @@ cmd_doctor() {
11511
11518
  # fails closed on any doubt, so when it returns false we emit today's
11512
11519
  # blocker verbatim. Mirrored in loki-ts/src/commands/doctor.ts.
11513
11520
  if declare -f detect_bundled_sdk_provider >/dev/null 2>&1 && detect_bundled_sdk_provider; then
11514
- echo -e " ${GREEN}PASS${NC} Bundled Claude Agent SDK is usable -- no separate CLI needed"
11521
+ # Say exactly which commands this covers. "No separate CLI needed" was
11522
+ # true for `loki start` and FALSE for demo/quick/quickstart, which stay
11523
+ # on the bash route and require a binary on PATH (provider-offer.sh:44
11524
+ # documents why folding the SDK predicate into detect_any_provider
11525
+ # would be a fail-open). The old wording produced the worst first-run
11526
+ # outcome we have: a green doctor ending in "Next: loki quickstart",
11527
+ # followed by quickstart, demo, quick and start-on-bash all exiting 2.
11528
+ echo -e " ${GREEN}PASS${NC} Bundled Claude Agent SDK is usable -- 'loki start' needs no separate CLI"
11529
+ echo -e " ${YELLOW}Note: loki demo/quick/quickstart still need a provider CLI on PATH${NC}"
11530
+ echo -e " ${YELLOW} Install: npm install -g @anthropic-ai/claude-code${NC}"
11515
11531
  pass_count=$((pass_count + 1))
11516
11532
  else
11517
11533
  echo -e " ${RED}FAIL${NC} No AI provider CLI installed -- at least one is required"
@@ -11581,11 +11597,23 @@ except Exception:
11581
11597
  echo -e " ${GREEN}PASS${NC} Claude CLI is logged in (subscription/OAuth login)"
11582
11598
  pass_count=$((pass_count + 1))
11583
11599
  elif [ "$_claude_login" = "no" ]; then
11584
- echo -e " ${YELLOW}WARN${NC} Claude CLI is NOT logged in -- run 'claude login' before a build (it would otherwise stall)"
11585
- warn_count=$((warn_count + 1))
11600
+ # BLOCKER, not a warning. A build cannot run without a login, and the
11601
+ # warning form let a user pass doctor, answer every quickstart prompt,
11602
+ # pick a template and CONFIRM THE SPEND before hitting the refusal at
11603
+ # run.sh's auth preflight. Discovering a missing login after the
11604
+ # consent screen is the worst placement available. Doctor is where a
11605
+ # missing credential belongs. (An ANTHROPIC_API_KEY short-circuits
11606
+ # that preflight, which is why this branch is reached only when the
11607
+ # key is absent too.)
11608
+ echo -e " ${RED}FAIL${NC} Claude CLI is NOT logged in -- a build would stall instead of running"
11609
+ echo -e " ${YELLOW}Fix: claude login${NC} (or set ANTHROPIC_API_KEY)"
11610
+ _doctor_block "Claude CLI is not logged in. Fix: claude login (or set ANTHROPIC_API_KEY)"
11611
+ fail_count=$((fail_count + 1))
11586
11612
  elif [ "$_claude_expired" = "expired" ]; then
11587
- echo -e " ${YELLOW}WARN${NC} Claude login has EXPIRED -- run 'claude login' before a build (it would otherwise stall)"
11588
- warn_count=$((warn_count + 1))
11613
+ echo -e " ${RED}FAIL${NC} Claude login has EXPIRED -- a build would stall instead of running"
11614
+ echo -e " ${YELLOW}Fix: claude login${NC} (or set ANTHROPIC_API_KEY)"
11615
+ _doctor_block "Claude login has expired. Fix: claude login (or set ANTHROPIC_API_KEY)"
11616
+ fail_count=$((fail_count + 1))
11589
11617
  else
11590
11618
  echo -e " ${DIM} -- ${NC} ANTHROPIC_API_KEY not set (Claude CLI uses its own login)"
11591
11619
  fi
@@ -11630,7 +11658,19 @@ except Exception:
11630
11658
  pass_count=$((pass_count + 1))
11631
11659
  elif [ -L "$sdir" ]; then
11632
11660
  local _target
11633
- _target=$(readlink "$sdir" 2>/dev/null || echo "unknown")
11661
+ # readlink is NOT in POSIX and is absent from minimal images and
11662
+ # from constrained PATHs (the doctor parity harness builds one).
11663
+ # When it is missing this reported the literal string "unknown"
11664
+ # while the Bun route printed the real target, so the two routes
11665
+ # diverged on the same host and bun-parity went red. ls -ld is
11666
+ # POSIX and always present; strip through the FIRST " -> " only,
11667
+ # since a link target may itself contain that sequence.
11668
+ if command -v readlink >/dev/null 2>&1; then
11669
+ _target=$(readlink "$sdir" 2>/dev/null || echo "unknown")
11670
+ else
11671
+ _target=$(ls -ld "$sdir" 2>/dev/null | sed -e 's/^[^>]*-> //')
11672
+ [ -n "$_target" ] || _target="unknown"
11673
+ fi
11634
11674
  echo -e " ${RED}FAIL${NC} $sname ${DIM}(broken symlink -> $_target)${NC}"
11635
11675
  echo -e " ${YELLOW}Fix: loki setup-skill${NC}"
11636
11676
  _doctor_block "$sname is a broken symlink. Fix: loki setup-skill"
@@ -11865,7 +11905,14 @@ STALE_DAYS = 90
11865
11905
  p = os.environ['LOKI_CATALOG_PATH']
11866
11906
  try:
11867
11907
  updated = json.load(open(p))['updated']
11868
- age = (datetime.date.today() - datetime.date.fromisoformat(updated)).days
11908
+ # UTC, matching the --json path below (loki:12337) and the Bun route
11909
+ # (doctor.ts:545). This is the TEXT-mode twin of that computation, and it was
11910
+ # missed when the JSON one was fixed: the parity gate compares BOTH surfaces,
11911
+ # so fixing only --json left doctor text-mode still diverging 5 vs 6 and the
11912
+ # gate still red. Two copies of one calculation is the actual defect here;
11913
+ # they are left as two only because the text path prints and the JSON path
11914
+ # returns a dict.
11915
+ age = (datetime.datetime.now(datetime.timezone.utc).date() - datetime.date.fromisoformat(updated)).days
11869
11916
  except Exception:
11870
11917
  print('warn|Catalog unreadable or missing an ISO \"updated\" date -- cannot determine age')
11871
11918
  else:
@@ -11989,6 +12036,17 @@ else:
11989
12036
  local _blk_key="other"
11990
12037
  case "$_doctor_blockers" in
11991
12038
  *"No AI provider CLI"*) _blk_key="no_provider" ;;
12039
+ # This doctor DETECTS the logged-out / expired wall above
12040
+ # (:11610, :11615) but had no arm for it, so the single most
12041
+ # common post-install failure was reported as `other` -- the one
12042
+ # bucket that cannot be acted on. not_logged_in is already a
12043
+ # first-class enum value (telemetry.sh:204) and the Bun doctor
12044
+ # already reports it, so without this arm the SAME host answered
12045
+ # `other` on bash and `not_logged_in` on Bun and the two routes'
12046
+ # counts could not be added together. Ordered directly after
12047
+ # no_provider because install-vs-authenticate need opposite fixes
12048
+ # and a missing provider is the earlier wall.
12049
+ *"not logged in"*|*"login has expired"*) _blk_key="not_logged_in" ;;
11992
12050
  *"Node.js is not installed"*|*"Node.js must be"*) _blk_key="node" ;;
11993
12051
  *"Python 3 is not installed"*|*"Python 3 must be"*) _blk_key="python3" ;;
11994
12052
  *"jq is not installed"*) _blk_key="jq" ;;
@@ -12172,10 +12230,60 @@ memory = {
12172
12230
  'status': 'pass' if not memory_recent_errors else 'warn',
12173
12231
  }
12174
12232
 
12233
+ # SKILL LINK INTEGRITY. The text path has always failed closed on a broken
12234
+ # skill symlink (the Skills section, _doctor_block, exit 1), but --json
12235
+ # omitted skills entirely -- so on a host with a dangling ~/.claude/skills/
12236
+ # loki-mode the two outputs gave OPPOSITE verdicts: text exited 1 while
12237
+ # --json reported failed 0 and ok true. An operator gating on the JSON got a
12238
+ # green on a host where the skill cannot load. Same class of fake-green as
12239
+ # the ai_provider block below, and fixed the same way: counted in the tally.
12240
+ #
12241
+ # Per-entry counting matches the text path, which tallies each of the four
12242
+ # entries separately. Paths are the same ones the text path already prints,
12243
+ # so this exposes nothing new.
12244
+ #
12245
+ # CAUTION: this block lives inside a DOUBLE-QUOTED python3 -c program, so no
12246
+ # apostrophes and no double quotes anywhere here, comments included.
12247
+ _SKILL_ENTRIES = (
12248
+ ('Claude Code', '.claude/skills/loki-mode'),
12249
+ ('Codex CLI', '.codex/skills/loki-mode'),
12250
+ ('Cline CLI', '.cline/skills/loki-mode'),
12251
+ ('Aider CLI', '.aider/skills/loki-mode'),
12252
+ )
12253
+ skills = []
12254
+ for _sk_name, _sk_rel in _SKILL_ENTRIES:
12255
+ _sk_dir = os.path.join(os.path.expanduser('~'), _sk_rel)
12256
+ if os.path.isfile(os.path.join(_sk_dir, 'SKILL.md')):
12257
+ skills.append({'name': _sk_name, 'path': _sk_dir, 'status': 'pass',
12258
+ 'detail': None, 'required': 'required'})
12259
+ elif os.path.islink(_sk_dir):
12260
+ # islink is true for a DANGLING link (lstat-based), which is exactly
12261
+ # the broken case the text path reports as FAIL.
12262
+ try:
12263
+ _sk_target = os.readlink(_sk_dir)
12264
+ except OSError:
12265
+ _sk_target = 'unknown'
12266
+ skills.append({'name': _sk_name, 'path': _sk_dir, 'status': 'fail',
12267
+ 'detail': 'broken symlink -> ' + _sk_target
12268
+ + '. Fix: loki setup-skill',
12269
+ 'required': 'required'})
12270
+ else:
12271
+ skills.append({'name': _sk_name, 'path': _sk_dir, 'status': 'warn',
12272
+ 'detail': 'not found - run loki setup-skill',
12273
+ 'required': 'required'})
12274
+
12175
12275
  pass_count = sum(1 for c in checks if c['status'] == 'pass')
12176
12276
  fail_count = sum(1 for c in checks if c['status'] == 'fail')
12177
12277
  warn_count = sum(1 for c in checks if c['status'] == 'warn')
12178
12278
 
12279
+ for _sk in skills:
12280
+ if _sk['status'] == 'pass':
12281
+ pass_count += 1
12282
+ elif _sk['status'] == 'fail':
12283
+ fail_count += 1
12284
+ else:
12285
+ warn_count += 1
12286
+
12179
12287
  if disk_status == 'pass': pass_count += 1
12180
12288
  elif disk_status == 'fail': fail_count += 1
12181
12289
  elif disk_status == 'warn': warn_count += 1
@@ -12222,7 +12330,18 @@ _cat_path = os.environ['LOKI_CATALOG_PATH']
12222
12330
  try:
12223
12331
  import datetime as _dt
12224
12332
  _cat_updated = json.load(open(_cat_path))['updated']
12225
- _cat_age = (_dt.date.today() - _dt.date.fromisoformat(_cat_updated)).days
12333
+ # UTC, not local. date.today() is the HOST's date, while the Bun route parses
12334
+ # both sides at UTC midnight (doctor.ts:545 says so, to keep the day count
12335
+ # from shifting with the host timezone). Anywhere west of UTC the two
12336
+ # disagree for the hours between local midnight and UTC midnight, and
12337
+ # doctor --json is compared BYTE FOR BYTE between routes by bun-parity.
12338
+ #
12339
+ # Measured on this host at 02:45 UTC / 22:45 local: bash reported age_days 5
12340
+ # and Bun reported 6, from the same file at the same instant. It reproduced
12341
+ # across two full gate runs hours apart, so it is not a midnight-rollover
12342
+ # flake. It is a real parity defect that stays invisible while the local date
12343
+ # and the UTC date agree, which is most of the day.
12344
+ _cat_age = (_dt.datetime.now(_dt.timezone.utc).date() - _dt.date.fromisoformat(_cat_updated)).days
12226
12345
  if _cat_age > CATALOG_STALE_DAYS:
12227
12346
  # Deliberately NOT counted. The block comment above states catalog age is
12228
12347
  # excluded from pass/fail/warn and from 'ok' so a stale catalog can never
@@ -12258,6 +12377,7 @@ result = {
12258
12377
  'status': disk_status
12259
12378
  },
12260
12379
  'ai_provider': ai_provider,
12380
+ 'skills': skills,
12261
12381
  'sentrux': sentrux,
12262
12382
  'receipt_signing': receipt_signing,
12263
12383
  'memory': memory,
@@ -13683,12 +13803,90 @@ set_ttfv_lightweight_profile() {
13683
13803
  # never pollutes the v7.8.1 generated-PRD-reuse signature logic. The brief text
13684
13804
  # is the project intent; the rest is a minimal scaffold the agent fills in.
13685
13805
  # Usage: synthesize_brief_prd <output_file> <brief_text>
13806
+ # _brief_acceptance_criteria <brief_text>: acceptance criteria derived from what
13807
+ # the user ACTUALLY asked for, not constants.
13808
+ #
13809
+ # WHY. Every one-liner used to get byte-identical Requirements and Success
13810
+ # Criteria: "build a todo app" and "build a Stripe billing dashboard" produced
13811
+ # the same acceptance criteria, and the user's own words appeared exactly once,
13812
+ # under Overview. So the completion council, the checklist and the evidence gate
13813
+ # were all checking generic prose rather than the request. That is the weakest
13814
+ # input shape getting the least specific help, which is backwards -- a cheap
13815
+ # model's output quality depends more on how precisely the target is stated than
13816
+ # on the model.
13817
+ #
13818
+ # DETERMINISTIC ON PURPOSE. No model call: this runs before a provider is even
13819
+ # selected, must work with no API key, and must not add latency or cost to the
13820
+ # first thing a new user does. It is keyword-to-obligation mapping, which is
13821
+ # honest about being shallow -- it turns stated nouns into checkable lines and
13822
+ # claims nothing about intent it cannot see. Anything cleverer belongs in the
13823
+ # spec-interrogation grill, which already runs after this and does call a model.
13824
+ #
13825
+ # STABLE IDs. Each criterion is emitted as "AC-<AXIS>-NNN: <text>" rather than a
13826
+ # bare bullet. An anonymous bullet cannot be referred to: a receipt can say "3 of
13827
+ # 8 gates passed" but never "AC-PERSIST-001 is satisfied by this test", drift
13828
+ # cannot be tracked per criterion, and two runs of the same spec produce lists
13829
+ # nothing can diff. The ID is what turns a criterion into a citable claim, which
13830
+ # is the whole point of shipping a receipt someone can check.
13831
+ #
13832
+ # The axis is derived from WHICH obligation fired, not from the criterion's
13833
+ # position, so IDs are stable across runs: adding a payment criterion never
13834
+ # renumbers the persistence one. Same reason we do not use a running counter.
13835
+ _brief_acceptance_criteria() {
13836
+ local t
13837
+ t="$(printf '%s' "${1:-}" | tr '[:upper:]' '[:lower:]')"
13838
+ local out=""
13839
+ # _bac <AXIS> <text>. The sequence is per-axis and always 001 today because
13840
+ # each axis fires at most once; the NNN slot exists so a second criterion on
13841
+ # the same axis can be added later without renumbering the first.
13842
+ _bac() { out="${out}- AC-${1}-001: ${2}"$'\n'; }
13843
+
13844
+ # Persistence. The single most common churn report is "I submitted the form
13845
+ # and nothing happened", so a stated store or form becomes an explicit
13846
+ # survives-a-reload obligation rather than an implied one.
13847
+ case "$t" in
13848
+ *save*|*persist*|*store*|*databas*|*crud*|*todo*|*note*|*task*|*record*)
13849
+ _bac PERSIST "Data the user creates survives a page reload and a server restart (it is written to a real store, not held in memory)." ;;
13850
+ esac
13851
+ case "$t" in
13852
+ *form*|*submit*|*signup*|*"sign up"*|*contact*|*upload*|*checkout*)
13853
+ _bac FORM "Every form actually submits: the happy path writes real data and the user sees a confirmation, and a validation failure shows an inline error." ;;
13854
+ esac
13855
+ case "$t" in
13856
+ *auth*|*login*|*"log in"*|*"sign in"*|*account*|*user*|*password*|*session*)
13857
+ _bac AUTH "Authentication works end to end: a real signup, a real login, and a protected route that returns 401 when logged out." ;;
13858
+ esac
13859
+ case "$t" in
13860
+ *api*|*endpoint*|*rest*|*graphql*|*backend*|*server*)
13861
+ _bac API "Each endpoint returns real data with correct status codes, and is callable with curl without a browser." ;;
13862
+ esac
13863
+ case "$t" in
13864
+ *payment*|*stripe*|*billing*|*subscription*|*checkout*|*invoice*)
13865
+ _bac PAY "The payment path is wired to the provider's test mode and a test transaction completes; no mocked charge stands in for the integration." ;;
13866
+ esac
13867
+ case "$t" in
13868
+ *search*|*filter*|*sort*)
13869
+ _bac SEARCH "Search or filtering queries the real dataset and returns different results for different inputs." ;;
13870
+ esac
13871
+ case "$t" in
13872
+ *dashboard*|*chart*|*graph*|*analytic*|*report*|*metric*)
13873
+ _bac DATA "Every figure shown traces to a real query. No hardcoded sample numbers." ;;
13874
+ esac
13875
+ case "$t" in
13876
+ *page*|*landing*|*site*|*website*|*ui*|*app*|*frontend*)
13877
+ _bac UI "The page renders with real content, no lorem ipsum, and no placeholder image where a real asset belongs." ;;
13878
+ esac
13879
+ printf '%s' "$out"
13880
+ }
13881
+
13686
13882
  synthesize_brief_prd() {
13687
13883
  local out_file="$1"
13688
13884
  local brief_text="$2"
13689
13885
  local out_dir
13690
13886
  out_dir="$(dirname "$out_file")"
13691
13887
  mkdir -p "$out_dir" 2>/dev/null || true
13888
+ local _derived
13889
+ _derived="$(_brief_acceptance_criteria "$brief_text")"
13692
13890
  cat > "$out_file" << BRIEFEOF
13693
13891
  # Project Brief
13694
13892
 
@@ -13704,6 +13902,7 @@ $brief_text
13704
13902
  ## Success Criteria
13705
13903
  - A user can run the result and observe the core behavior described above.
13706
13904
  - No errors on a clean start; the happy path works end to end.
13905
+ ${_derived}
13707
13906
 
13708
13907
  ## Constraints
13709
13908
  - This is a fast first pass (zero-config first run). Keep scope tight.
@@ -13805,6 +14004,10 @@ cmd_quick() {
13805
14004
  find "$LOKI_DIR" -maxdepth 1 -name 'quick-prd-*.md' -mtime +1 -delete 2>/dev/null || true
13806
14005
  find "$LOKI_DIR" -maxdepth 1 -name 'brief-prd-*.md' -mtime +1 -delete 2>/dev/null || true
13807
14006
  local quick_prd="$LOKI_DIR/quick-prd-$$.md"
14007
+ # Same spec-derived criteria as the brief path: a quick task was getting the
14008
+ # identical generic Success Criteria regardless of what was asked.
14009
+ local _derived
14010
+ _derived="$(_brief_acceptance_criteria "$task_desc")"
13808
14011
  cat > "$quick_prd" << QPRDEOF
13809
14012
  # Quick Task
13810
14013
 
@@ -13821,6 +14024,7 @@ $task_desc
13821
14024
  - Task is completed as described
13822
14025
  - No errors or regressions introduced
13823
14026
  - Code follows project conventions
14027
+ ${_derived}
13824
14028
 
13825
14029
  ## Constraints
13826
14030
  - This is a quick single-task execution
@@ -15902,6 +16106,25 @@ cmd_verify() {
15902
16106
  # and detects drift deterministically, emitting .loki/spec/drift-report.json.
15903
16107
  # Exit codes are propagated so `loki spec status` is CI-gate usable.
15904
16108
  # ---------------------------------------------------------------------------
16109
+ # loki intent: does the spec still say what was actually wanted?
16110
+ #
16111
+ # Our gates prove code matches spec. They cannot prove the spec was RIGHT --
16112
+ # 8090 AI documents a build where "the software converged with the
16113
+ # interpretation. The interpretation had diverged from the intent", which a
16114
+ # perfect verification gate passes. This measures that gap deterministically by
16115
+ # comparing the requirement hash an intent was affirmed against with the
16116
+ # requirement's hash now. No model judges anything; see autonomy/intent.sh.
16117
+ cmd_intent() {
16118
+ local intent_mod="$_LOKI_SCRIPT_DIR/intent.sh"
16119
+ if [ ! -f "$intent_mod" ]; then
16120
+ echo -e "${RED}Error: intent module not found at $intent_mod${NC}" >&2
16121
+ return 3
16122
+ fi
16123
+ # shellcheck source=/dev/null
16124
+ source "$intent_mod"
16125
+ intent_main "$@"
16126
+ }
16127
+
15905
16128
  cmd_spec() {
15906
16129
  local spec_mod="$_LOKI_SCRIPT_DIR/spec.sh"
15907
16130
  if [ ! -f "$spec_mod" ]; then
@@ -19141,6 +19364,11 @@ main() {
19141
19364
  spec)
19142
19365
  cmd_spec "$@"
19143
19366
  ;;
19367
+ intent)
19368
+ # The spec is an INTERPRETATION of what was wanted. This checks
19369
+ # whether it still matches the intent it was affirmed against.
19370
+ cmd_intent "$@"
19371
+ ;;
19144
19372
  grill)
19145
19373
  cmd_grill "$@"
19146
19374
  ;;
@@ -19256,6 +19484,25 @@ main() {
19256
19484
  # Receipt surface): same subcommands (list/show/verify/open/share).
19257
19485
  cmd_proof "$@"
19258
19486
  ;;
19487
+ outcomes)
19488
+ # The receipt says what was proven at the time. This says what
19489
+ # happened to it afterwards -- reverted, reworked, or survived.
19490
+ cmd_outcomes "$@"
19491
+ ;;
19492
+ verdict)
19493
+ # All five trust signals in one block. The individual measurements
19494
+ # existed and were unreachable; this is the surface a reviewer reads.
19495
+ cmd_verdict "$@"
19496
+ ;;
19497
+ readiness)
19498
+ # Can an agent verify its own work here? Measured, not LLM-scored.
19499
+ cmd_readiness "$@"
19500
+ ;;
19501
+ gates)
19502
+ # What blocks here vs only advises, and what promoting a gate would
19503
+ # have cost. Reports only; never promotes.
19504
+ cmd_gates "$@"
19505
+ ;;
19259
19506
  secure)
19260
19507
  # Secure-by-default gate surface: inspect findings + manage waivers.
19261
19508
  cmd_secure "$@"
@@ -30693,6 +30940,174 @@ if _result.get('error'):
30693
30940
  # this is a thin dispatcher, mirroring how cmd_proof delegates to the proof
30694
30941
  # generator. Generation is incremental (skips when the codebase is unchanged).
30695
30942
  # =============================================================================
30943
+ # loki outcomes: what happened to the work AFTER the receipt was written.
30944
+ #
30945
+ # Every competing agent reports VOLUME. Factory AI's analytics expose files
30946
+ # edited, lines modified, commits, PRs created, tokens and DAU -- and no defect
30947
+ # rate, no revert rate, no change-failure rate; their own telemetry doc leaves
30948
+ # outcome correlation to the customer. Devin's security page concedes the agent
30949
+ # "can still experience hallucinations, introduce bugs into code" and points at
30950
+ # your existing review. Both can prove the agent was BUSY. Neither shows it was
30951
+ # RIGHT. This answers the other question, from local git history alone.
30952
+ #
30953
+ # It reports UNKNOWN far more often than it reports a number, and that is the
30954
+ # feature. See autonomy/lib/outcome_ledger.py for the anchor gate and the
30955
+ # measured evidence behind it.
30956
+ cmd_outcomes() {
30957
+ local lib="${_LOKI_SCRIPT_DIR}/lib/outcome_ledger.py"
30958
+ if [ ! -f "$lib" ]; then
30959
+ echo "outcome ledger is not installed at $lib" >&2
30960
+ return 2
30961
+ fi
30962
+ case "${1:-}" in
30963
+ --help|-h|help)
30964
+ echo -e "${BOLD}loki outcomes${NC} - did the work turn out to be RIGHT, not just that it happened"
30965
+ echo ""
30966
+ echo "Usage: loki outcomes [--json] [--run-id <id>]"
30967
+ echo ""
30968
+ echo "Follows each Evidence Receipt past the moment it was written and"
30969
+ echo "reports, from local git only: was it reverted, did its lines survive,"
30970
+ echo "how much was reworked, and the change-failure rate across receipts."
30971
+ echo ""
30972
+ echo "A receipt is measured ONLY when sha algebra proves base..head is that"
30973
+ echo "change. Everything else reads UNKNOWN with a named reason -- never 0,"
30974
+ echo "never a pass. Read-only: it never writes to the repo it analyses."
30975
+ return 0
30976
+ ;;
30977
+ esac
30978
+ LOKI_DIR="${LOKI_DIR:-.loki}" python3 "$lib" "$@"
30979
+ }
30980
+
30981
+ # loki verdict: the five measured trust signals, in one block a reviewer reads
30982
+ # in ten seconds.
30983
+ #
30984
+ # WHY A COMMAND AND NOT JUST A LIBRARY. We measure five things nobody else does
30985
+ # -- did the work survive, does the spec still match intent, was this the agent
30986
+ # or a human rescue, does the completion claim name real work, which model
30987
+ # decided -- and each was correct, tested, and unreachable. A moat nobody can
30988
+ # see is not a moat. This is the surface.
30989
+ #
30990
+ # There is deliberately NO composite score. Averaging a revert count, a hash
30991
+ # comparison, a diff hash, a path match and a model id yields a number whose
30992
+ # movement nobody can explain, which is what competitors already ship. UNKNOWN
30993
+ # is PRINTED, never suppressed: a reviewer must tell "we checked and it is fine"
30994
+ # from "we could not check". See autonomy/lib/verdict.py.
30995
+ cmd_verdict() {
30996
+ local lib="${_LOKI_SCRIPT_DIR}/lib/verdict.py"
30997
+ if [ ! -f "$lib" ]; then
30998
+ echo "verdict renderer is not installed at $lib" >&2
30999
+ return 2
31000
+ fi
31001
+ case "${1:-}" in
31002
+ --help|-h|help)
31003
+ echo -e "${BOLD}loki verdict${NC} - the five measured trust signals, in one readable block"
31004
+ echo ""
31005
+ echo "Usage: loki verdict [--json]"
31006
+ echo ""
31007
+ echo "Prints one line each for outcome, intent, authorship, grounding and"
31008
+ echo "model: what survived, whether the spec still matches intent, whether"
31009
+ echo "this was the agent or a human rescue, whether the completion claim"
31010
+ echo "named real work, and which model decided."
31011
+ echo ""
31012
+ echo "Every line is either a measured fact or an explicit UNKNOWN. There is"
31013
+ echo "no composite score: a number nobody can explain is not evidence."
31014
+ echo "Read-only: it never writes to the repo it analyses."
31015
+ return 0
31016
+ ;;
31017
+ esac
31018
+ LOKI_DIR="${LOKI_DIR:-.loki}" python3 "$lib" "$@"
31019
+ }
31020
+
31021
+ # loki readiness: can an autonomous agent verify its own work in THIS repo?
31022
+ #
31023
+ # WHY THIS IS NOT A COPY OF FACTORY AI'S AGENT READINESS MODEL. Theirs is
31024
+ # LLM-scored -- their report objects record modelUsed and reasoningEffort, so
31025
+ # the number is a model's opinion and two runs can disagree about the same
31026
+ # commit. Every criterion here is a file that exists or does not, a command
31027
+ # present or absent: same commit, same answer, every machine, no key, no spend.
31028
+ #
31029
+ # No percentage and no letter grade. A composite invites ranking, ranking
31030
+ # invites gaming, and the individual signals are the actionable part -- "there
31031
+ # is no test command" tells you what to do, "readiness 62%" does not. Criteria
31032
+ # that cannot be determined report UNKNOWN by name rather than counting as
31033
+ # failures. See autonomy/lib/agent_readiness.py.
31034
+ cmd_gates() {
31035
+ local lib="${_LOKI_SCRIPT_DIR}/lib/gate_policy.py"
31036
+ if [ ! -f "$lib" ]; then
31037
+ echo "gate policy reporter is not installed at $lib" >&2
31038
+ return 2
31039
+ fi
31040
+ case "${1:-}" in
31041
+ --help|-h|help)
31042
+ echo -e "${BOLD}loki gates${NC} - what blocks here, and what only advises"
31043
+ echo ""
31044
+ echo "Usage: loki gates [.loki-dir] [--json]"
31045
+ echo ""
31046
+ echo "Ona's Veto Exec ships an audit-first ladder: start in audit mode,"
31047
+ echo "review what matched, then promote the confirmed rules to block. The"
31048
+ echo "middle step is the load-bearing one -- a policy you cannot safely"
31049
+ echo "turn on is a policy nobody turns on."
31050
+ echo ""
31051
+ echo "We had both ends and nothing between them: gates are advisory or"
31052
+ echo "blocking, three promotion knobs exist, and the failure ledger has"
31053
+ echo "counted per-gate hits all along. Nothing joined them, so deciding"
31054
+ echo "whether to promote a gate meant guessing."
31055
+ echo ""
31056
+ echo "For each gate this prints its mode, how many times it has fired,"
31057
+ echo "and -- for an advisory one -- the exact variable that promotes it."
31058
+ echo ""
31059
+ echo "Deterministic: reads two files and the environment. No model, no"
31060
+ echo "key, no spend. Counts come from"
31061
+ echo ".loki/quality/gate-failure-count.json; open it and count them"
31062
+ echo "yourself."
31063
+ echo ""
31064
+ echo "A gate with no ledger entry reports 'not measured', never 0: an"
31065
+ echo "absent measurement is not evidence a gate never fired."
31066
+ echo ""
31067
+ echo "This command NEVER promotes a gate. Promotion stays an explicit"
31068
+ echo "operator act via the named variable."
31069
+ return 0
31070
+ ;;
31071
+ esac
31072
+ python3 "$lib" "$@"
31073
+ }
31074
+
31075
+ cmd_readiness() {
31076
+ local lib="${_LOKI_SCRIPT_DIR}/lib/agent_readiness.py"
31077
+ if [ ! -f "$lib" ]; then
31078
+ echo "readiness assessor is not installed at $lib" >&2
31079
+ return 2
31080
+ fi
31081
+ case "${1:-}" in
31082
+ --help|-h|help)
31083
+ echo -e "${BOLD}loki readiness${NC} - can an agent verify its own work in this repo?"
31084
+ echo ""
31085
+ echo "Usage: loki readiness [path] [--json] [--fix]"
31086
+ echo ""
31087
+ echo "Measures whether this repo gives an agent a way to check itself: a"
31088
+ echo "test command, a build, CI config, typed sources, a lockfile. Not"
31089
+ echo "general code quality -- the narrower question every agent depends on."
31090
+ echo ""
31091
+ echo "Deterministic: no model, no key, no spend. Same commit, same answer."
31092
+ echo "Criteria that cannot be determined report UNKNOWN rather than failing."
31093
+ echo ""
31094
+ echo -e "${BOLD}--fix${NC} writes the missing files whose content can be derived"
31095
+ echo "honestly (README.md, AGENTS.md, .gitignore) as TODO stubs, then"
31096
+ echo "re-measures and reports what is true AFTER the change."
31097
+ echo ""
31098
+ echo "It deliberately REFUSES to generate a test command, a lockfile or a"
31099
+ echo "CI config. Guessing one writes a line that lies: an invented"
31100
+ echo "'npm test' in a repo with no runner fails forever, and this check"
31101
+ echo "would then report the criterion present for something that does not"
31102
+ echo "work. Those stay reported, never generated."
31103
+ echo ""
31104
+ echo "Without --fix it is read-only and never writes to the repo."
31105
+ return 0
31106
+ ;;
31107
+ esac
31108
+ python3 "$lib" "$@"
31109
+ }
31110
+
30696
31111
  cmd_wiki() {
30697
31112
  local subcmd="${1:-}"
30698
31113
  shift 2>/dev/null || true