loki-mode 9.12.5 → 9.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +81 -101
- package/SKILL.md +2 -2
- package/VERSION +1 -1
- package/autonomy/intent.sh +414 -0
- package/autonomy/issue-providers.sh +21 -0
- package/autonomy/lib/agent_readiness.py +202 -0
- package/autonomy/lib/claim_grounding.py +171 -0
- package/autonomy/lib/config-map.sh +10 -6
- package/autonomy/lib/decision_record.py +198 -0
- package/autonomy/lib/failure_memory.py +199 -0
- package/autonomy/lib/outcome_ledger.py +498 -0
- package/autonomy/lib/preedit_snapshot.py +216 -0
- package/autonomy/lib/verdict.py +204 -0
- package/autonomy/loki +378 -22
- package/autonomy/provider-offer.sh +25 -1
- package/autonomy/run.sh +516 -11
- package/autonomy/telemetry.sh +8 -1
- package/autonomy/verify.sh +10 -1
- package/completions/_loki +4 -0
- package/completions/loki.bash +2 -1
- package/dashboard/__init__.py +1 -1
- package/dashboard/api_operator.py +15 -2
- package/dashboard/api_v2.py +6 -1
- package/dashboard/control.py +62 -10
- package/dashboard/run.py +13 -2
- package/dashboard/scim.py +221 -0
- package/dashboard/server.py +48 -5
- package/docs/GATE-FAILURE-TRIAGE.md +254 -0
- package/docs/LOOP-CANDIDATE-PROPOSAL-v1.md +167 -0
- package/docs/LOOP-HARNESS-AUDIT.md +684 -0
- package/docs/VERIFICATION-COST.md +103 -0
- package/docs/WANG-PRINCIPLES-PLAN.md +1 -1
- package/loki-ts/dist/loki.js +402 -398
- package/mcp/__init__.py +1 -1
- package/mcp/_sdk_loader.py +25 -0
- package/package.json +1 -1
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
- package/tools/loop-harness-report.py +216 -0
package/autonomy/loki
CHANGED
|
@@ -1041,14 +1041,14 @@ show_help() {
|
|
|
1041
1041
|
echo " cluster cockpit code compliance completions compound config context (ctx)"
|
|
1042
1042
|
echo " cost council crash dashboard demo deploy docker docs doctor dogfood"
|
|
1043
1043
|
echo " enterprise estimate explain export failover github grill heal help"
|
|
1044
|
-
echo " import init"
|
|
1044
|
+
echo " import init intent"
|
|
1045
1045
|
echo " issue kpis logs magic mcp memory metrics migrate modernize monitor"
|
|
1046
|
-
echo " next notify onboard open optimize otel own (handoff) pause plan preview"
|
|
1047
|
-
echo " projects proof (receipt) provider quick quickstart rc remote report reset"
|
|
1046
|
+
echo " next notify onboard open optimize otel outcomes own (handoff) pause plan preview"
|
|
1047
|
+
echo " projects proof (receipt) provider quick quickstart rc readiness remote report reset"
|
|
1048
1048
|
echo " resume review rollback run sandbox secrets secure self-update sentrux"
|
|
1049
1049
|
echo " serve setup-skill share ship spec start state stats status steer stop"
|
|
1050
1050
|
echo " syslog telemetry template test tour trigger trust trust-metrics"
|
|
1051
|
-
echo " ultracode update verify version voice watch watchdog web welcome why"
|
|
1051
|
+
echo " ultracode update verdict verify version voice watch watchdog web welcome why"
|
|
1052
1052
|
echo " wiki worktree (wt)"
|
|
1053
1053
|
echo ""
|
|
1054
1054
|
echo "Any command: loki <command> --help"
|
|
@@ -10155,7 +10155,14 @@ _export_check_overwrite() {
|
|
|
10155
10155
|
echo "y (auto-confirmed)"
|
|
10156
10156
|
return 0
|
|
10157
10157
|
fi
|
|
10158
|
-
|
|
10158
|
+
# A prompt is only answerable when BOTH ends are a terminal. Testing
|
|
10159
|
+
# stdin alone was the bug: under a PTY harness that redirects stdout
|
|
10160
|
+
# (the gate runs every check as `bash tests/... | tail -3`), `-t 0` is
|
|
10161
|
+
# true, so we fell through to `read` -- but the process is not the
|
|
10162
|
+
# terminal's foreground group, so reading the tty raises SIGTTIN and
|
|
10163
|
+
# STOPS it. A stopped process never runs its SIGALRM handler, which is
|
|
10164
|
+
# why `read -t` cannot rescue this; only not reading at all can.
|
|
10165
|
+
if [ ! -t 0 ] || [ ! -t 1 ]; then
|
|
10159
10166
|
echo "Export cancelled (non-interactive: refusing to overwrite '$output'." >&2
|
|
10160
10167
|
echo "Set LOKI_AUTO_CONFIRM=true to overwrite, or choose a new output path.)" >&2
|
|
10161
10168
|
return 1
|
|
@@ -10165,7 +10172,7 @@ _export_check_overwrite() {
|
|
|
10165
10172
|
read -r reply || reply=""
|
|
10166
10173
|
case "$reply" in
|
|
10167
10174
|
[yY]|[yY][eE][sS]) return 0 ;;
|
|
10168
|
-
*) echo "Export cancelled."; return 1 ;;
|
|
10175
|
+
*) echo "Export cancelled (set LOKI_AUTO_CONFIRM=true to overwrite)."; return 1 ;;
|
|
10169
10176
|
esac
|
|
10170
10177
|
fi
|
|
10171
10178
|
return 0
|
|
@@ -11511,7 +11518,16 @@ cmd_doctor() {
|
|
|
11511
11518
|
# fails closed on any doubt, so when it returns false we emit today's
|
|
11512
11519
|
# blocker verbatim. Mirrored in loki-ts/src/commands/doctor.ts.
|
|
11513
11520
|
if declare -f detect_bundled_sdk_provider >/dev/null 2>&1 && detect_bundled_sdk_provider; then
|
|
11514
|
-
|
|
11521
|
+
# Say exactly which commands this covers. "No separate CLI needed" was
|
|
11522
|
+
# true for `loki start` and FALSE for demo/quick/quickstart, which stay
|
|
11523
|
+
# on the bash route and require a binary on PATH (provider-offer.sh:44
|
|
11524
|
+
# documents why folding the SDK predicate into detect_any_provider
|
|
11525
|
+
# would be a fail-open). The old wording produced the worst first-run
|
|
11526
|
+
# outcome we have: a green doctor ending in "Next: loki quickstart",
|
|
11527
|
+
# followed by quickstart, demo, quick and start-on-bash all exiting 2.
|
|
11528
|
+
echo -e " ${GREEN}PASS${NC} Bundled Claude Agent SDK is usable -- 'loki start' needs no separate CLI"
|
|
11529
|
+
echo -e " ${YELLOW}Note: loki demo/quick/quickstart still need a provider CLI on PATH${NC}"
|
|
11530
|
+
echo -e " ${YELLOW} Install: npm install -g @anthropic-ai/claude-code${NC}"
|
|
11515
11531
|
pass_count=$((pass_count + 1))
|
|
11516
11532
|
else
|
|
11517
11533
|
echo -e " ${RED}FAIL${NC} No AI provider CLI installed -- at least one is required"
|
|
@@ -11581,11 +11597,23 @@ except Exception:
|
|
|
11581
11597
|
echo -e " ${GREEN}PASS${NC} Claude CLI is logged in (subscription/OAuth login)"
|
|
11582
11598
|
pass_count=$((pass_count + 1))
|
|
11583
11599
|
elif [ "$_claude_login" = "no" ]; then
|
|
11584
|
-
|
|
11585
|
-
|
|
11600
|
+
# BLOCKER, not a warning. A build cannot run without a login, and the
|
|
11601
|
+
# warning form let a user pass doctor, answer every quickstart prompt,
|
|
11602
|
+
# pick a template and CONFIRM THE SPEND before hitting the refusal at
|
|
11603
|
+
# run.sh's auth preflight. Discovering a missing login after the
|
|
11604
|
+
# consent screen is the worst placement available. Doctor is where a
|
|
11605
|
+
# missing credential belongs. (An ANTHROPIC_API_KEY short-circuits
|
|
11606
|
+
# that preflight, which is why this branch is reached only when the
|
|
11607
|
+
# key is absent too.)
|
|
11608
|
+
echo -e " ${RED}FAIL${NC} Claude CLI is NOT logged in -- a build would stall instead of running"
|
|
11609
|
+
echo -e " ${YELLOW}Fix: claude login${NC} (or set ANTHROPIC_API_KEY)"
|
|
11610
|
+
_doctor_block "Claude CLI is not logged in. Fix: claude login (or set ANTHROPIC_API_KEY)"
|
|
11611
|
+
fail_count=$((fail_count + 1))
|
|
11586
11612
|
elif [ "$_claude_expired" = "expired" ]; then
|
|
11587
|
-
echo -e " ${
|
|
11588
|
-
|
|
11613
|
+
echo -e " ${RED}FAIL${NC} Claude login has EXPIRED -- a build would stall instead of running"
|
|
11614
|
+
echo -e " ${YELLOW}Fix: claude login${NC} (or set ANTHROPIC_API_KEY)"
|
|
11615
|
+
_doctor_block "Claude login has expired. Fix: claude login (or set ANTHROPIC_API_KEY)"
|
|
11616
|
+
fail_count=$((fail_count + 1))
|
|
11589
11617
|
else
|
|
11590
11618
|
echo -e " ${DIM} -- ${NC} ANTHROPIC_API_KEY not set (Claude CLI uses its own login)"
|
|
11591
11619
|
fi
|
|
@@ -11630,7 +11658,19 @@ except Exception:
|
|
|
11630
11658
|
pass_count=$((pass_count + 1))
|
|
11631
11659
|
elif [ -L "$sdir" ]; then
|
|
11632
11660
|
local _target
|
|
11633
|
-
|
|
11661
|
+
# readlink is NOT in POSIX and is absent from minimal images and
|
|
11662
|
+
# from constrained PATHs (the doctor parity harness builds one).
|
|
11663
|
+
# When it is missing this reported the literal string "unknown"
|
|
11664
|
+
# while the Bun route printed the real target, so the two routes
|
|
11665
|
+
# diverged on the same host and bun-parity went red. ls -ld is
|
|
11666
|
+
# POSIX and always present; strip through the FIRST " -> " only,
|
|
11667
|
+
# since a link target may itself contain that sequence.
|
|
11668
|
+
if command -v readlink >/dev/null 2>&1; then
|
|
11669
|
+
_target=$(readlink "$sdir" 2>/dev/null || echo "unknown")
|
|
11670
|
+
else
|
|
11671
|
+
_target=$(ls -ld "$sdir" 2>/dev/null | sed -e 's/^[^>]*-> //')
|
|
11672
|
+
[ -n "$_target" ] || _target="unknown"
|
|
11673
|
+
fi
|
|
11634
11674
|
echo -e " ${RED}FAIL${NC} $sname ${DIM}(broken symlink -> $_target)${NC}"
|
|
11635
11675
|
echo -e " ${YELLOW}Fix: loki setup-skill${NC}"
|
|
11636
11676
|
_doctor_block "$sname is a broken symlink. Fix: loki setup-skill"
|
|
@@ -11865,7 +11905,14 @@ STALE_DAYS = 90
|
|
|
11865
11905
|
p = os.environ['LOKI_CATALOG_PATH']
|
|
11866
11906
|
try:
|
|
11867
11907
|
updated = json.load(open(p))['updated']
|
|
11868
|
-
|
|
11908
|
+
# UTC, matching the --json path below (loki:12337) and the Bun route
|
|
11909
|
+
# (doctor.ts:545). This is the TEXT-mode twin of that computation, and it was
|
|
11910
|
+
# missed when the JSON one was fixed: the parity gate compares BOTH surfaces,
|
|
11911
|
+
# so fixing only --json left doctor text-mode still diverging 5 vs 6 and the
|
|
11912
|
+
# gate still red. Two copies of one calculation is the actual defect here;
|
|
11913
|
+
# they are left as two only because the text path prints and the JSON path
|
|
11914
|
+
# returns a dict.
|
|
11915
|
+
age = (datetime.datetime.now(datetime.timezone.utc).date() - datetime.date.fromisoformat(updated)).days
|
|
11869
11916
|
except Exception:
|
|
11870
11917
|
print('warn|Catalog unreadable or missing an ISO \"updated\" date -- cannot determine age')
|
|
11871
11918
|
else:
|
|
@@ -11989,6 +12036,17 @@ else:
|
|
|
11989
12036
|
local _blk_key="other"
|
|
11990
12037
|
case "$_doctor_blockers" in
|
|
11991
12038
|
*"No AI provider CLI"*) _blk_key="no_provider" ;;
|
|
12039
|
+
# This doctor DETECTS the logged-out / expired wall above
|
|
12040
|
+
# (:11610, :11615) but had no arm for it, so the single most
|
|
12041
|
+
# common post-install failure was reported as `other` -- the one
|
|
12042
|
+
# bucket that cannot be acted on. not_logged_in is already a
|
|
12043
|
+
# first-class enum value (telemetry.sh:204) and the Bun doctor
|
|
12044
|
+
# already reports it, so without this arm the SAME host answered
|
|
12045
|
+
# `other` on bash and `not_logged_in` on Bun and the two routes'
|
|
12046
|
+
# counts could not be added together. Ordered directly after
|
|
12047
|
+
# no_provider because install-vs-authenticate need opposite fixes
|
|
12048
|
+
# and a missing provider is the earlier wall.
|
|
12049
|
+
*"not logged in"*|*"login has expired"*) _blk_key="not_logged_in" ;;
|
|
11992
12050
|
*"Node.js is not installed"*|*"Node.js must be"*) _blk_key="node" ;;
|
|
11993
12051
|
*"Python 3 is not installed"*|*"Python 3 must be"*) _blk_key="python3" ;;
|
|
11994
12052
|
*"jq is not installed"*) _blk_key="jq" ;;
|
|
@@ -12172,10 +12230,60 @@ memory = {
|
|
|
12172
12230
|
'status': 'pass' if not memory_recent_errors else 'warn',
|
|
12173
12231
|
}
|
|
12174
12232
|
|
|
12233
|
+
# SKILL LINK INTEGRITY. The text path has always failed closed on a broken
|
|
12234
|
+
# skill symlink (the Skills section, _doctor_block, exit 1), but --json
|
|
12235
|
+
# omitted skills entirely -- so on a host with a dangling ~/.claude/skills/
|
|
12236
|
+
# loki-mode the two outputs gave OPPOSITE verdicts: text exited 1 while
|
|
12237
|
+
# --json reported failed 0 and ok true. An operator gating on the JSON got a
|
|
12238
|
+
# green on a host where the skill cannot load. Same class of fake-green as
|
|
12239
|
+
# the ai_provider block below, and fixed the same way: counted in the tally.
|
|
12240
|
+
#
|
|
12241
|
+
# Per-entry counting matches the text path, which tallies each of the four
|
|
12242
|
+
# entries separately. Paths are the same ones the text path already prints,
|
|
12243
|
+
# so this exposes nothing new.
|
|
12244
|
+
#
|
|
12245
|
+
# CAUTION: this block lives inside a DOUBLE-QUOTED python3 -c program, so no
|
|
12246
|
+
# apostrophes and no double quotes anywhere here, comments included.
|
|
12247
|
+
_SKILL_ENTRIES = (
|
|
12248
|
+
('Claude Code', '.claude/skills/loki-mode'),
|
|
12249
|
+
('Codex CLI', '.codex/skills/loki-mode'),
|
|
12250
|
+
('Cline CLI', '.cline/skills/loki-mode'),
|
|
12251
|
+
('Aider CLI', '.aider/skills/loki-mode'),
|
|
12252
|
+
)
|
|
12253
|
+
skills = []
|
|
12254
|
+
for _sk_name, _sk_rel in _SKILL_ENTRIES:
|
|
12255
|
+
_sk_dir = os.path.join(os.path.expanduser('~'), _sk_rel)
|
|
12256
|
+
if os.path.isfile(os.path.join(_sk_dir, 'SKILL.md')):
|
|
12257
|
+
skills.append({'name': _sk_name, 'path': _sk_dir, 'status': 'pass',
|
|
12258
|
+
'detail': None, 'required': 'required'})
|
|
12259
|
+
elif os.path.islink(_sk_dir):
|
|
12260
|
+
# islink is true for a DANGLING link (lstat-based), which is exactly
|
|
12261
|
+
# the broken case the text path reports as FAIL.
|
|
12262
|
+
try:
|
|
12263
|
+
_sk_target = os.readlink(_sk_dir)
|
|
12264
|
+
except OSError:
|
|
12265
|
+
_sk_target = 'unknown'
|
|
12266
|
+
skills.append({'name': _sk_name, 'path': _sk_dir, 'status': 'fail',
|
|
12267
|
+
'detail': 'broken symlink -> ' + _sk_target
|
|
12268
|
+
+ '. Fix: loki setup-skill',
|
|
12269
|
+
'required': 'required'})
|
|
12270
|
+
else:
|
|
12271
|
+
skills.append({'name': _sk_name, 'path': _sk_dir, 'status': 'warn',
|
|
12272
|
+
'detail': 'not found - run loki setup-skill',
|
|
12273
|
+
'required': 'required'})
|
|
12274
|
+
|
|
12175
12275
|
pass_count = sum(1 for c in checks if c['status'] == 'pass')
|
|
12176
12276
|
fail_count = sum(1 for c in checks if c['status'] == 'fail')
|
|
12177
12277
|
warn_count = sum(1 for c in checks if c['status'] == 'warn')
|
|
12178
12278
|
|
|
12279
|
+
for _sk in skills:
|
|
12280
|
+
if _sk['status'] == 'pass':
|
|
12281
|
+
pass_count += 1
|
|
12282
|
+
elif _sk['status'] == 'fail':
|
|
12283
|
+
fail_count += 1
|
|
12284
|
+
else:
|
|
12285
|
+
warn_count += 1
|
|
12286
|
+
|
|
12179
12287
|
if disk_status == 'pass': pass_count += 1
|
|
12180
12288
|
elif disk_status == 'fail': fail_count += 1
|
|
12181
12289
|
elif disk_status == 'warn': warn_count += 1
|
|
@@ -12222,7 +12330,18 @@ _cat_path = os.environ['LOKI_CATALOG_PATH']
|
|
|
12222
12330
|
try:
|
|
12223
12331
|
import datetime as _dt
|
|
12224
12332
|
_cat_updated = json.load(open(_cat_path))['updated']
|
|
12225
|
-
|
|
12333
|
+
# UTC, not local. date.today() is the HOST's date, while the Bun route parses
|
|
12334
|
+
# both sides at UTC midnight (doctor.ts:545 says so, to keep the day count
|
|
12335
|
+
# from shifting with the host timezone). Anywhere west of UTC the two
|
|
12336
|
+
# disagree for the hours between local midnight and UTC midnight, and
|
|
12337
|
+
# doctor --json is compared BYTE FOR BYTE between routes by bun-parity.
|
|
12338
|
+
#
|
|
12339
|
+
# Measured on this host at 02:45 UTC / 22:45 local: bash reported age_days 5
|
|
12340
|
+
# and Bun reported 6, from the same file at the same instant. It reproduced
|
|
12341
|
+
# across two full gate runs hours apart, so it is not a midnight-rollover
|
|
12342
|
+
# flake. It is a real parity defect that stays invisible while the local date
|
|
12343
|
+
# and the UTC date agree, which is most of the day.
|
|
12344
|
+
_cat_age = (_dt.datetime.now(_dt.timezone.utc).date() - _dt.date.fromisoformat(_cat_updated)).days
|
|
12226
12345
|
if _cat_age > CATALOG_STALE_DAYS:
|
|
12227
12346
|
# Deliberately NOT counted. The block comment above states catalog age is
|
|
12228
12347
|
# excluded from pass/fail/warn and from 'ok' so a stale catalog can never
|
|
@@ -12258,6 +12377,7 @@ result = {
|
|
|
12258
12377
|
'status': disk_status
|
|
12259
12378
|
},
|
|
12260
12379
|
'ai_provider': ai_provider,
|
|
12380
|
+
'skills': skills,
|
|
12261
12381
|
'sentrux': sentrux,
|
|
12262
12382
|
'receipt_signing': receipt_signing,
|
|
12263
12383
|
'memory': memory,
|
|
@@ -13683,12 +13803,76 @@ set_ttfv_lightweight_profile() {
|
|
|
13683
13803
|
# never pollutes the v7.8.1 generated-PRD-reuse signature logic. The brief text
|
|
13684
13804
|
# is the project intent; the rest is a minimal scaffold the agent fills in.
|
|
13685
13805
|
# Usage: synthesize_brief_prd <output_file> <brief_text>
|
|
13806
|
+
# _brief_acceptance_criteria <brief_text>: acceptance criteria derived from what
|
|
13807
|
+
# the user ACTUALLY asked for, not constants.
|
|
13808
|
+
#
|
|
13809
|
+
# WHY. Every one-liner used to get byte-identical Requirements and Success
|
|
13810
|
+
# Criteria: "build a todo app" and "build a Stripe billing dashboard" produced
|
|
13811
|
+
# the same acceptance criteria, and the user's own words appeared exactly once,
|
|
13812
|
+
# under Overview. So the completion council, the checklist and the evidence gate
|
|
13813
|
+
# were all checking generic prose rather than the request. That is the weakest
|
|
13814
|
+
# input shape getting the least specific help, which is backwards -- a cheap
|
|
13815
|
+
# model's output quality depends more on how precisely the target is stated than
|
|
13816
|
+
# on the model.
|
|
13817
|
+
#
|
|
13818
|
+
# DETERMINISTIC ON PURPOSE. No model call: this runs before a provider is even
|
|
13819
|
+
# selected, must work with no API key, and must not add latency or cost to the
|
|
13820
|
+
# first thing a new user does. It is keyword-to-obligation mapping, which is
|
|
13821
|
+
# honest about being shallow -- it turns stated nouns into checkable lines and
|
|
13822
|
+
# claims nothing about intent it cannot see. Anything cleverer belongs in the
|
|
13823
|
+
# spec-interrogation grill, which already runs after this and does call a model.
|
|
13824
|
+
_brief_acceptance_criteria() {
|
|
13825
|
+
local t
|
|
13826
|
+
t="$(printf '%s' "${1:-}" | tr '[:upper:]' '[:lower:]')"
|
|
13827
|
+
local out=""
|
|
13828
|
+
_bac() { out="${out}- ${1}"$'\n'; }
|
|
13829
|
+
|
|
13830
|
+
# Persistence. The single most common churn report is "I submitted the form
|
|
13831
|
+
# and nothing happened", so a stated store or form becomes an explicit
|
|
13832
|
+
# survives-a-reload obligation rather than an implied one.
|
|
13833
|
+
case "$t" in
|
|
13834
|
+
*save*|*persist*|*store*|*databas*|*crud*|*todo*|*note*|*task*|*record*)
|
|
13835
|
+
_bac "Data the user creates survives a page reload and a server restart (it is written to a real store, not held in memory)." ;;
|
|
13836
|
+
esac
|
|
13837
|
+
case "$t" in
|
|
13838
|
+
*form*|*submit*|*signup*|*"sign up"*|*contact*|*upload*|*checkout*)
|
|
13839
|
+
_bac "Every form actually submits: the happy path writes real data and the user sees a confirmation, and a validation failure shows an inline error." ;;
|
|
13840
|
+
esac
|
|
13841
|
+
case "$t" in
|
|
13842
|
+
*auth*|*login*|*"log in"*|*"sign in"*|*account*|*user*|*password*|*session*)
|
|
13843
|
+
_bac "Authentication works end to end: a real signup, a real login, and a protected route that returns 401 when logged out." ;;
|
|
13844
|
+
esac
|
|
13845
|
+
case "$t" in
|
|
13846
|
+
*api*|*endpoint*|*rest*|*graphql*|*backend*|*server*)
|
|
13847
|
+
_bac "Each endpoint returns real data with correct status codes, and is callable with curl without a browser." ;;
|
|
13848
|
+
esac
|
|
13849
|
+
case "$t" in
|
|
13850
|
+
*payment*|*stripe*|*billing*|*subscription*|*checkout*|*invoice*)
|
|
13851
|
+
_bac "The payment path is wired to the provider's test mode and a test transaction completes; no mocked charge stands in for the integration." ;;
|
|
13852
|
+
esac
|
|
13853
|
+
case "$t" in
|
|
13854
|
+
*search*|*filter*|*sort*)
|
|
13855
|
+
_bac "Search or filtering queries the real dataset and returns different results for different inputs." ;;
|
|
13856
|
+
esac
|
|
13857
|
+
case "$t" in
|
|
13858
|
+
*dashboard*|*chart*|*graph*|*analytic*|*report*|*metric*)
|
|
13859
|
+
_bac "Every figure shown traces to a real query. No hardcoded sample numbers." ;;
|
|
13860
|
+
esac
|
|
13861
|
+
case "$t" in
|
|
13862
|
+
*page*|*landing*|*site*|*website*|*ui*|*app*|*frontend*)
|
|
13863
|
+
_bac "The page renders with real content, no lorem ipsum, and no placeholder image where a real asset belongs." ;;
|
|
13864
|
+
esac
|
|
13865
|
+
printf '%s' "$out"
|
|
13866
|
+
}
|
|
13867
|
+
|
|
13686
13868
|
synthesize_brief_prd() {
|
|
13687
13869
|
local out_file="$1"
|
|
13688
13870
|
local brief_text="$2"
|
|
13689
13871
|
local out_dir
|
|
13690
13872
|
out_dir="$(dirname "$out_file")"
|
|
13691
13873
|
mkdir -p "$out_dir" 2>/dev/null || true
|
|
13874
|
+
local _derived
|
|
13875
|
+
_derived="$(_brief_acceptance_criteria "$brief_text")"
|
|
13692
13876
|
cat > "$out_file" << BRIEFEOF
|
|
13693
13877
|
# Project Brief
|
|
13694
13878
|
|
|
@@ -13704,6 +13888,7 @@ $brief_text
|
|
|
13704
13888
|
## Success Criteria
|
|
13705
13889
|
- A user can run the result and observe the core behavior described above.
|
|
13706
13890
|
- No errors on a clean start; the happy path works end to end.
|
|
13891
|
+
${_derived}
|
|
13707
13892
|
|
|
13708
13893
|
## Constraints
|
|
13709
13894
|
- This is a fast first pass (zero-config first run). Keep scope tight.
|
|
@@ -13805,6 +13990,10 @@ cmd_quick() {
|
|
|
13805
13990
|
find "$LOKI_DIR" -maxdepth 1 -name 'quick-prd-*.md' -mtime +1 -delete 2>/dev/null || true
|
|
13806
13991
|
find "$LOKI_DIR" -maxdepth 1 -name 'brief-prd-*.md' -mtime +1 -delete 2>/dev/null || true
|
|
13807
13992
|
local quick_prd="$LOKI_DIR/quick-prd-$$.md"
|
|
13993
|
+
# Same spec-derived criteria as the brief path: a quick task was getting the
|
|
13994
|
+
# identical generic Success Criteria regardless of what was asked.
|
|
13995
|
+
local _derived
|
|
13996
|
+
_derived="$(_brief_acceptance_criteria "$task_desc")"
|
|
13808
13997
|
cat > "$quick_prd" << QPRDEOF
|
|
13809
13998
|
# Quick Task
|
|
13810
13999
|
|
|
@@ -13821,6 +14010,7 @@ $task_desc
|
|
|
13821
14010
|
- Task is completed as described
|
|
13822
14011
|
- No errors or regressions introduced
|
|
13823
14012
|
- Code follows project conventions
|
|
14013
|
+
${_derived}
|
|
13824
14014
|
|
|
13825
14015
|
## Constraints
|
|
13826
14016
|
- This is a quick single-task execution
|
|
@@ -15902,6 +16092,25 @@ cmd_verify() {
|
|
|
15902
16092
|
# and detects drift deterministically, emitting .loki/spec/drift-report.json.
|
|
15903
16093
|
# Exit codes are propagated so `loki spec status` is CI-gate usable.
|
|
15904
16094
|
# ---------------------------------------------------------------------------
|
|
16095
|
+
# loki intent: does the spec still say what was actually wanted?
|
|
16096
|
+
#
|
|
16097
|
+
# Our gates prove code matches spec. They cannot prove the spec was RIGHT --
|
|
16098
|
+
# 8090 AI documents a build where "the software converged with the
|
|
16099
|
+
# interpretation. The interpretation had diverged from the intent", which a
|
|
16100
|
+
# perfect verification gate passes. This measures that gap deterministically by
|
|
16101
|
+
# comparing the requirement hash an intent was affirmed against with the
|
|
16102
|
+
# requirement's hash now. No model judges anything; see autonomy/intent.sh.
|
|
16103
|
+
cmd_intent() {
|
|
16104
|
+
local intent_mod="$_LOKI_SCRIPT_DIR/intent.sh"
|
|
16105
|
+
if [ ! -f "$intent_mod" ]; then
|
|
16106
|
+
echo -e "${RED}Error: intent module not found at $intent_mod${NC}" >&2
|
|
16107
|
+
return 3
|
|
16108
|
+
fi
|
|
16109
|
+
# shellcheck source=/dev/null
|
|
16110
|
+
source "$intent_mod"
|
|
16111
|
+
intent_main "$@"
|
|
16112
|
+
}
|
|
16113
|
+
|
|
15905
16114
|
cmd_spec() {
|
|
15906
16115
|
local spec_mod="$_LOKI_SCRIPT_DIR/spec.sh"
|
|
15907
16116
|
if [ ! -f "$spec_mod" ]; then
|
|
@@ -19141,6 +19350,11 @@ main() {
|
|
|
19141
19350
|
spec)
|
|
19142
19351
|
cmd_spec "$@"
|
|
19143
19352
|
;;
|
|
19353
|
+
intent)
|
|
19354
|
+
# The spec is an INTERPRETATION of what was wanted. This checks
|
|
19355
|
+
# whether it still matches the intent it was affirmed against.
|
|
19356
|
+
cmd_intent "$@"
|
|
19357
|
+
;;
|
|
19144
19358
|
grill)
|
|
19145
19359
|
cmd_grill "$@"
|
|
19146
19360
|
;;
|
|
@@ -19256,6 +19470,20 @@ main() {
|
|
|
19256
19470
|
# Receipt surface): same subcommands (list/show/verify/open/share).
|
|
19257
19471
|
cmd_proof "$@"
|
|
19258
19472
|
;;
|
|
19473
|
+
outcomes)
|
|
19474
|
+
# The receipt says what was proven at the time. This says what
|
|
19475
|
+
# happened to it afterwards -- reverted, reworked, or survived.
|
|
19476
|
+
cmd_outcomes "$@"
|
|
19477
|
+
;;
|
|
19478
|
+
verdict)
|
|
19479
|
+
# All five trust signals in one block. The individual measurements
|
|
19480
|
+
# existed and were unreachable; this is the surface a reviewer reads.
|
|
19481
|
+
cmd_verdict "$@"
|
|
19482
|
+
;;
|
|
19483
|
+
readiness)
|
|
19484
|
+
# Can an agent verify its own work here? Measured, not LLM-scored.
|
|
19485
|
+
cmd_readiness "$@"
|
|
19486
|
+
;;
|
|
19259
19487
|
secure)
|
|
19260
19488
|
# Secure-by-default gate surface: inspect findings + manage waivers.
|
|
19261
19489
|
cmd_secure "$@"
|
|
@@ -27881,11 +28109,17 @@ except: pass
|
|
|
27881
28109
|
|
|
27882
28110
|
# --- Build directory tree ---
|
|
27883
28111
|
local tree_output=""
|
|
27884
|
-
|
|
28112
|
+
# -e not -d: in a git WORKTREE, .git is a pointer FILE, not a directory.
|
|
28113
|
+
# -d sent every worktree down the find fallback below.
|
|
28114
|
+
if command -v git &>/dev/null && [ -e "$target_path/.git" ]; then
|
|
27885
28115
|
# Use git ls-files for accurate tree (respects .gitignore)
|
|
27886
28116
|
tree_output=$(cd "$target_path" && git ls-files 2>/dev/null | head -200 || true)
|
|
27887
28117
|
else
|
|
27888
|
-
# Fallback: find with common excludions
|
|
28118
|
+
# Fallback: find with common excludions.
|
|
28119
|
+
# `|| true` is load-bearing under `set -euo pipefail` (line 22): head -200
|
|
28120
|
+
# exits after 200 lines, find keeps writing, gets SIGPIPE, and the pipeline
|
|
28121
|
+
# returns 141 -- which -e turned into a silent abort producing 0 bytes on
|
|
28122
|
+
# any repo with more than 200 files. The git branch above is already guarded.
|
|
27889
28123
|
tree_output=$(find "$target_path" -maxdepth 4 -type f \
|
|
27890
28124
|
-not -path '*/node_modules/*' \
|
|
27891
28125
|
-not -path '*/.git/*' \
|
|
@@ -27895,7 +28129,7 @@ except: pass
|
|
|
27895
28129
|
-not -path '*/build/*' \
|
|
27896
28130
|
-not -path '*/.next/*' \
|
|
27897
28131
|
-not -path '*/target/*' \
|
|
27898
|
-
2>/dev/null | sed "s|$target_path/||" | sort | head -200)
|
|
28132
|
+
2>/dev/null | sed "s|$target_path/||" | sort | head -200 || true)
|
|
27899
28133
|
fi
|
|
27900
28134
|
|
|
27901
28135
|
# Categorize files
|
|
@@ -28652,8 +28886,13 @@ $devdeps_list"
|
|
|
28652
28886
|
fi
|
|
28653
28887
|
|
|
28654
28888
|
# --- File counts ---
|
|
28889
|
+
# Same SIGPIPE pipeline as cmd_onboard, but a DIFFERENT symptom here: this
|
|
28890
|
+
# assigns into a bare `tree_output=` after the local declaration, so under
|
|
28891
|
+
# -e a 141 aborts. Where a site instead writes `local x=$(...)`, `local`
|
|
28892
|
+
# resets $? and the 141 is swallowed -- silent truncation, no abort. Both
|
|
28893
|
+
# are wrong; `|| true` plus -e (worktree .git is a FILE) fixes both.
|
|
28655
28894
|
local tree_output=""
|
|
28656
|
-
if command -v git &>/dev/null && [ -
|
|
28895
|
+
if command -v git &>/dev/null && [ -e "$target_path/.git" ]; then
|
|
28657
28896
|
tree_output=$(cd "$target_path" && git ls-files 2>/dev/null | head -500 || true)
|
|
28658
28897
|
else
|
|
28659
28898
|
tree_output=$(find "$target_path" -maxdepth 4 -type f \
|
|
@@ -28661,7 +28900,7 @@ $devdeps_list"
|
|
|
28661
28900
|
-not -path '*/vendor/*' -not -path '*/__pycache__/*' \
|
|
28662
28901
|
-not -path '*/dist/*' -not -path '*/build/*' \
|
|
28663
28902
|
-not -path '*/.next/*' -not -path '*/target/*' \
|
|
28664
|
-
2>/dev/null | sed "s|$target_path/||" | sort | head -500)
|
|
28903
|
+
2>/dev/null | sed "s|$target_path/||" | sort | head -500 || true)
|
|
28665
28904
|
fi
|
|
28666
28905
|
local total_files src_count test_count doc_count config_count
|
|
28667
28906
|
total_files=$(echo "$tree_output" | { grep -c . || true; })
|
|
@@ -29065,9 +29304,10 @@ cmd_docs() {
|
|
|
29065
29304
|
_docs_scan_project() {
|
|
29066
29305
|
local target_path="$1"
|
|
29067
29306
|
|
|
29068
|
-
# Build file tree (excluding common noise)
|
|
29307
|
+
# Build file tree (excluding common noise).
|
|
29308
|
+
# Same SIGPIPE pipeline and same fix as cmd_onboard; see the comment there.
|
|
29069
29309
|
local tree_output=""
|
|
29070
|
-
if command -v git &>/dev/null && [ -
|
|
29310
|
+
if command -v git &>/dev/null && [ -e "$target_path/.git" ]; then
|
|
29071
29311
|
tree_output=$(cd "$target_path" && git ls-files 2>/dev/null | head -500 || true)
|
|
29072
29312
|
else
|
|
29073
29313
|
tree_output=$(find "$target_path" -maxdepth 4 -type f \
|
|
@@ -29075,7 +29315,7 @@ _docs_scan_project() {
|
|
|
29075
29315
|
-not -path '*/vendor/*' -not -path '*/__pycache__/*' \
|
|
29076
29316
|
-not -path '*/dist/*' -not -path '*/build/*' \
|
|
29077
29317
|
-not -path '*/.next/*' -not -path '*/target/*' \
|
|
29078
|
-
2>/dev/null | sed "s|$target_path/||" | sort | head -500)
|
|
29318
|
+
2>/dev/null | sed "s|$target_path/||" | sort | head -500 || true)
|
|
29079
29319
|
fi
|
|
29080
29320
|
echo "$tree_output"
|
|
29081
29321
|
}
|
|
@@ -30681,6 +30921,122 @@ if _result.get('error'):
|
|
|
30681
30921
|
# this is a thin dispatcher, mirroring how cmd_proof delegates to the proof
|
|
30682
30922
|
# generator. Generation is incremental (skips when the codebase is unchanged).
|
|
30683
30923
|
# =============================================================================
|
|
30924
|
+
# loki outcomes: what happened to the work AFTER the receipt was written.
|
|
30925
|
+
#
|
|
30926
|
+
# Every competing agent reports VOLUME. Factory AI's analytics expose files
|
|
30927
|
+
# edited, lines modified, commits, PRs created, tokens and DAU -- and no defect
|
|
30928
|
+
# rate, no revert rate, no change-failure rate; their own telemetry doc leaves
|
|
30929
|
+
# outcome correlation to the customer. Devin's security page concedes the agent
|
|
30930
|
+
# "can still experience hallucinations, introduce bugs into code" and points at
|
|
30931
|
+
# your existing review. Both can prove the agent was BUSY. Neither shows it was
|
|
30932
|
+
# RIGHT. This answers the other question, from local git history alone.
|
|
30933
|
+
#
|
|
30934
|
+
# It reports UNKNOWN far more often than it reports a number, and that is the
|
|
30935
|
+
# feature. See autonomy/lib/outcome_ledger.py for the anchor gate and the
|
|
30936
|
+
# measured evidence behind it.
|
|
30937
|
+
cmd_outcomes() {
|
|
30938
|
+
local lib="${_LOKI_SCRIPT_DIR}/lib/outcome_ledger.py"
|
|
30939
|
+
if [ ! -f "$lib" ]; then
|
|
30940
|
+
echo "outcome ledger is not installed at $lib" >&2
|
|
30941
|
+
return 2
|
|
30942
|
+
fi
|
|
30943
|
+
case "${1:-}" in
|
|
30944
|
+
--help|-h|help)
|
|
30945
|
+
echo -e "${BOLD}loki outcomes${NC} - did the work turn out to be RIGHT, not just that it happened"
|
|
30946
|
+
echo ""
|
|
30947
|
+
echo "Usage: loki outcomes [--json] [--run-id <id>]"
|
|
30948
|
+
echo ""
|
|
30949
|
+
echo "Follows each Evidence Receipt past the moment it was written and"
|
|
30950
|
+
echo "reports, from local git only: was it reverted, did its lines survive,"
|
|
30951
|
+
echo "how much was reworked, and the change-failure rate across receipts."
|
|
30952
|
+
echo ""
|
|
30953
|
+
echo "A receipt is measured ONLY when sha algebra proves base..head is that"
|
|
30954
|
+
echo "change. Everything else reads UNKNOWN with a named reason -- never 0,"
|
|
30955
|
+
echo "never a pass. Read-only: it never writes to the repo it analyses."
|
|
30956
|
+
return 0
|
|
30957
|
+
;;
|
|
30958
|
+
esac
|
|
30959
|
+
LOKI_DIR="${LOKI_DIR:-.loki}" python3 "$lib" "$@"
|
|
30960
|
+
}
|
|
30961
|
+
|
|
30962
|
+
# loki verdict: the five measured trust signals, in one block a reviewer reads
|
|
30963
|
+
# in ten seconds.
|
|
30964
|
+
#
|
|
30965
|
+
# WHY A COMMAND AND NOT JUST A LIBRARY. We measure five things nobody else does
|
|
30966
|
+
# -- did the work survive, does the spec still match intent, was this the agent
|
|
30967
|
+
# or a human rescue, does the completion claim name real work, which model
|
|
30968
|
+
# decided -- and each was correct, tested, and unreachable. A moat nobody can
|
|
30969
|
+
# see is not a moat. This is the surface.
|
|
30970
|
+
#
|
|
30971
|
+
# There is deliberately NO composite score. Averaging a revert count, a hash
|
|
30972
|
+
# comparison, a diff hash, a path match and a model id yields a number whose
|
|
30973
|
+
# movement nobody can explain, which is what competitors already ship. UNKNOWN
|
|
30974
|
+
# is PRINTED, never suppressed: a reviewer must tell "we checked and it is fine"
|
|
30975
|
+
# from "we could not check". See autonomy/lib/verdict.py.
|
|
30976
|
+
cmd_verdict() {
|
|
30977
|
+
local lib="${_LOKI_SCRIPT_DIR}/lib/verdict.py"
|
|
30978
|
+
if [ ! -f "$lib" ]; then
|
|
30979
|
+
echo "verdict renderer is not installed at $lib" >&2
|
|
30980
|
+
return 2
|
|
30981
|
+
fi
|
|
30982
|
+
case "${1:-}" in
|
|
30983
|
+
--help|-h|help)
|
|
30984
|
+
echo -e "${BOLD}loki verdict${NC} - the five measured trust signals, in one readable block"
|
|
30985
|
+
echo ""
|
|
30986
|
+
echo "Usage: loki verdict [--json]"
|
|
30987
|
+
echo ""
|
|
30988
|
+
echo "Prints one line each for outcome, intent, authorship, grounding and"
|
|
30989
|
+
echo "model: what survived, whether the spec still matches intent, whether"
|
|
30990
|
+
echo "this was the agent or a human rescue, whether the completion claim"
|
|
30991
|
+
echo "named real work, and which model decided."
|
|
30992
|
+
echo ""
|
|
30993
|
+
echo "Every line is either a measured fact or an explicit UNKNOWN. There is"
|
|
30994
|
+
echo "no composite score: a number nobody can explain is not evidence."
|
|
30995
|
+
echo "Read-only: it never writes to the repo it analyses."
|
|
30996
|
+
return 0
|
|
30997
|
+
;;
|
|
30998
|
+
esac
|
|
30999
|
+
LOKI_DIR="${LOKI_DIR:-.loki}" python3 "$lib" "$@"
|
|
31000
|
+
}
|
|
31001
|
+
|
|
31002
|
+
# loki readiness: can an autonomous agent verify its own work in THIS repo?
|
|
31003
|
+
#
|
|
31004
|
+
# WHY THIS IS NOT A COPY OF FACTORY AI'S AGENT READINESS MODEL. Theirs is
|
|
31005
|
+
# LLM-scored -- their report objects record modelUsed and reasoningEffort, so
|
|
31006
|
+
# the number is a model's opinion and two runs can disagree about the same
|
|
31007
|
+
# commit. Every criterion here is a file that exists or does not, a command
|
|
31008
|
+
# present or absent: same commit, same answer, every machine, no key, no spend.
|
|
31009
|
+
#
|
|
31010
|
+
# No percentage and no letter grade. A composite invites ranking, ranking
|
|
31011
|
+
# invites gaming, and the individual signals are the actionable part -- "there
|
|
31012
|
+
# is no test command" tells you what to do, "readiness 62%" does not. Criteria
|
|
31013
|
+
# that cannot be determined report UNKNOWN by name rather than counting as
|
|
31014
|
+
# failures. See autonomy/lib/agent_readiness.py.
|
|
31015
|
+
cmd_readiness() {
|
|
31016
|
+
local lib="${_LOKI_SCRIPT_DIR}/lib/agent_readiness.py"
|
|
31017
|
+
if [ ! -f "$lib" ]; then
|
|
31018
|
+
echo "readiness assessor is not installed at $lib" >&2
|
|
31019
|
+
return 2
|
|
31020
|
+
fi
|
|
31021
|
+
case "${1:-}" in
|
|
31022
|
+
--help|-h|help)
|
|
31023
|
+
echo -e "${BOLD}loki readiness${NC} - can an agent verify its own work in this repo?"
|
|
31024
|
+
echo ""
|
|
31025
|
+
echo "Usage: loki readiness [path] [--json]"
|
|
31026
|
+
echo ""
|
|
31027
|
+
echo "Measures whether this repo gives an agent a way to check itself: a"
|
|
31028
|
+
echo "test command, a build, CI config, typed sources, a lockfile. Not"
|
|
31029
|
+
echo "general code quality -- the narrower question every agent depends on."
|
|
31030
|
+
echo ""
|
|
31031
|
+
echo "Deterministic: no model, no key, no spend. Same commit, same answer."
|
|
31032
|
+
echo "Criteria that cannot be determined report UNKNOWN rather than failing."
|
|
31033
|
+
echo "Read-only: it never writes to the repo it analyses."
|
|
31034
|
+
return 0
|
|
31035
|
+
;;
|
|
31036
|
+
esac
|
|
31037
|
+
python3 "$lib" "$@"
|
|
31038
|
+
}
|
|
31039
|
+
|
|
30684
31040
|
cmd_wiki() {
|
|
30685
31041
|
local subcmd="${1:-}"
|
|
30686
31042
|
shift 2>/dev/null || true
|
|
@@ -369,12 +369,36 @@ offer_provider_install() {
|
|
|
369
369
|
# signal the caller should `exit 2` (no provider, declined or non-interactive).
|
|
370
370
|
provider_offer_gate() {
|
|
371
371
|
detect_any_provider && return 0
|
|
372
|
-
offer_provider_install gate || return 2
|
|
372
|
+
offer_provider_install gate || { _provider_gate_emit_blocked; return 2; }
|
|
373
373
|
# After an accepted install, re-detect; if still absent, fail the gate.
|
|
374
374
|
detect_any_provider && return 0
|
|
375
|
+
_provider_gate_emit_blocked
|
|
375
376
|
return 2
|
|
376
377
|
}
|
|
377
378
|
|
|
379
|
+
# Report that a first run died here. THIS is the wall most first runs hit --
|
|
380
|
+
# every one of `loki start`, `demo`, `quick` and `quickstart` funnels through
|
|
381
|
+
# provider_offer_gate and exits 2 -- and until now not one of those four exits
|
|
382
|
+
# emitted anything. `first_start_attempted` fired, so the funnel recorded that a
|
|
383
|
+
# first run was ATTEMPTED and nothing about whether it survived: an attempt that
|
|
384
|
+
# died for want of a provider CLI looked identical to one that shipped a build.
|
|
385
|
+
#
|
|
386
|
+
# Emitting from the gate rather than from each of the four callers is deliberate:
|
|
387
|
+
# one site cannot drift out of sync with the others, and a future caller of the
|
|
388
|
+
# gate is instrumented by construction rather than by remembering.
|
|
389
|
+
#
|
|
390
|
+
# Sends the bounded enum `no_provider` and nothing else -- no path, no version,
|
|
391
|
+
# no hostname, no command line -- and routes through loki_emit_first_run_blocked,
|
|
392
|
+
# so every existing opt-out and the once-per-machine marker still apply. Silent
|
|
393
|
+
# and non-fatal if telemetry is unavailable: a diagnostic must never be able to
|
|
394
|
+
# break the command it is diagnosing.
|
|
395
|
+
_provider_gate_emit_blocked() {
|
|
396
|
+
if declare -f loki_emit_first_run_blocked >/dev/null 2>&1; then
|
|
397
|
+
( loki_emit_first_run_blocked "no_provider" >/dev/null 2>&1 </dev/null & ) 2>/dev/null || true
|
|
398
|
+
fi
|
|
399
|
+
return 0
|
|
400
|
+
}
|
|
401
|
+
|
|
378
402
|
# render_provider_availability: print the "Provider Availability" doctor section.
|
|
379
403
|
#
|
|
380
404
|
# WHY IT LIVES HERE. `loki doctor` stdout is compared byte for byte between the
|