loki-mode 9.12.6 → 9.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +81 -101
- package/SKILL.md +2 -2
- package/VERSION +1 -1
- package/autonomy/intent.sh +414 -0
- package/autonomy/issue-providers.sh +21 -0
- package/autonomy/lib/agent_readiness.py +202 -0
- package/autonomy/lib/claim_grounding.py +171 -0
- package/autonomy/lib/config-map.sh +10 -6
- package/autonomy/lib/decision_record.py +198 -0
- package/autonomy/lib/failure_memory.py +199 -0
- package/autonomy/lib/outcome_ledger.py +498 -0
- package/autonomy/lib/preedit_snapshot.py +216 -0
- package/autonomy/lib/verdict.py +204 -0
- package/autonomy/loki +358 -14
- package/autonomy/provider-offer.sh +25 -1
- package/autonomy/run.sh +516 -11
- package/autonomy/telemetry.sh +8 -1
- package/completions/_loki +4 -0
- package/completions/loki.bash +2 -1
- package/dashboard/__init__.py +1 -1
- package/dashboard/run.py +13 -2
- package/dashboard/scim.py +221 -0
- package/dashboard/server.py +22 -0
- package/docs/GATE-FAILURE-TRIAGE.md +254 -0
- package/docs/LOOP-CANDIDATE-PROPOSAL-v1.md +167 -0
- package/docs/LOOP-HARNESS-AUDIT.md +53 -0
- package/docs/VERIFICATION-COST.md +103 -0
- package/docs/WANG-PRINCIPLES-PLAN.md +1 -1
- package/loki-ts/dist/loki.js +402 -398
- package/mcp/__init__.py +1 -1
- package/mcp/_sdk_loader.py +25 -0
- package/package.json +1 -1
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
|
@@ -0,0 +1,414 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Intent Ledger: does the SPEC still say what the person actually wanted?
|
|
3
|
+
#
|
|
4
|
+
# THE PROBLEM THIS ANSWERS. Our verification proves code matches spec. It cannot
|
|
5
|
+
# prove the spec was RIGHT. 8090 AI documents a real build where "the software
|
|
6
|
+
# converged with the interpretation. The interpretation had diverged from the
|
|
7
|
+
# intent." A perfect verification gate passes that build, and the build is still
|
|
8
|
+
# wrong. That gap sits upstream of every gate we have, and no competitor ships an
|
|
9
|
+
# answer -- 8090 published the argument and their own "Tests" module has zero
|
|
10
|
+
# documentation pages and zero changelog entries.
|
|
11
|
+
#
|
|
12
|
+
# WHAT WE DELIBERATELY DID NOT BUILD. The obvious feature is a semantic fidelity
|
|
13
|
+
# score: ask a model "does this spec faithfully express this intent?" and print a
|
|
14
|
+
# percentage. That is an LLM judgment wearing the costume of a measurement, and
|
|
15
|
+
# it is precisely what our receipts exist to refuse -- they already separate
|
|
16
|
+
# deterministic FACTS from AI ASSESSMENTS. There is no similarity number, no
|
|
17
|
+
# embedding distance, and no percent-aligned figure anywhere in this file, and a
|
|
18
|
+
# future contributor adding one would be removing the reason it is trustworthy.
|
|
19
|
+
#
|
|
20
|
+
# WHAT IS ACTUALLY MEASURABLE, and it is enough. autonomy/spec.sh already
|
|
21
|
+
# persists a content_hash per requirement into .loki/spec/spec.lock. So if an
|
|
22
|
+
# intent statement records WHICH requirement it was affirmed against AND that
|
|
23
|
+
# requirement's hash AT THAT MOMENT, then divergence is pure hash comparison:
|
|
24
|
+
#
|
|
25
|
+
# the intent was affirmed against requirement R at hash H;
|
|
26
|
+
# R now hashes to H';
|
|
27
|
+
# nobody re-affirmed.
|
|
28
|
+
#
|
|
29
|
+
# That is 8090's failure mode, detected deterministically, re-derivable by hand,
|
|
30
|
+
# with no model in the loop. `content_hash_at_link` is the entire design -- store
|
|
31
|
+
# only a requirement id and every verdict collapses to UNKNOWN.
|
|
32
|
+
#
|
|
33
|
+
# WHAT ALREADY EXISTED (checked before building, not assumed). Three of the four
|
|
34
|
+
# things this was scoped to do are already shipped and are NOT rebuilt here:
|
|
35
|
+
# - ASSUMED-BUT-NOT-STATED -> spec-interrogation.sh:284 (.loki/assumptions/)
|
|
36
|
+
# - spec-vs-built divergence -> spec.sh + verify.sh:2161 (spec.lock, drift gate)
|
|
37
|
+
# - pre-committed predictions -> lib/expectation-ledger.py
|
|
38
|
+
# This file references the assumption store by id. A second store would drift
|
|
39
|
+
# from the first.
|
|
40
|
+
|
|
41
|
+
set -uo pipefail
|
|
42
|
+
|
|
43
|
+
_INTENT_SCHEMA_VERSION="1.0"
|
|
44
|
+
|
|
45
|
+
# Named refusals. Mirrors the outcome ledger's ANCHOR_REASONS: a status we cannot
|
|
46
|
+
# compute is reported by NAME, never as 0 and never as a pass. The most important
|
|
47
|
+
# entry is no_distinct_intent_source -- see intent_do_status.
|
|
48
|
+
_intent_unknown_reason() {
|
|
49
|
+
case "${1:-}" in
|
|
50
|
+
no_intent_record) echo "no intent has been recorded (run: loki intent record)" ;;
|
|
51
|
+
no_spec_lock) echo "no .loki/spec/spec.lock (run: loki spec lock)" ;;
|
|
52
|
+
no_distinct_intent_source) echo "the spec IS the user's own document, so intent and interpretation are the same artifact" ;;
|
|
53
|
+
link_missing_content_hash) echo "link predates hash recording, so staleness cannot be computed" ;;
|
|
54
|
+
requirement_id_not_in_lock) echo "the linked requirement id is not in the current lock" ;;
|
|
55
|
+
*) echo "unmeasurable" ;;
|
|
56
|
+
esac
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
_intent_dir() { echo "${LOKI_DIR:-.loki}/intent"; }
|
|
60
|
+
_intent_file() { echo "$(_intent_dir)/intent.json"; }
|
|
61
|
+
_intent_lock() { echo "${LOKI_DIR:-.loki}/spec/spec.lock"; }
|
|
62
|
+
|
|
63
|
+
_intent_now() { date -u +%Y-%m-%dT%H:%M:%SZ; }
|
|
64
|
+
|
|
65
|
+
_intent_sha256() {
|
|
66
|
+
if command -v shasum >/dev/null 2>&1; then
|
|
67
|
+
printf '%s' "$1" | shasum -a 256 | awk '{print $1}'
|
|
68
|
+
else
|
|
69
|
+
printf '%s' "$1" | sha256sum | awk '{print $1}'
|
|
70
|
+
fi
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
# ---------------------------------------------------------------------------
|
|
74
|
+
# record: capture an intent statement as a first-class artifact.
|
|
75
|
+
#
|
|
76
|
+
# THE TEXT IS COPIED, NEVER REFERENCED. autonomy/loki:2079 deletes
|
|
77
|
+
# .loki/state/brief.txt at the start of a later non-brief run, deliberately, so a
|
|
78
|
+
# stale one-liner cannot be inherited. An intent record whose only evidence is a
|
|
79
|
+
# file that a later run removes would be unmeasurable by construction, so the
|
|
80
|
+
# statement text and its sha256 are stored inline.
|
|
81
|
+
# ---------------------------------------------------------------------------
|
|
82
|
+
intent_do_record() {
|
|
83
|
+
local statement=""
|
|
84
|
+
local source="explicit"
|
|
85
|
+
while [ $# -gt 0 ]; do
|
|
86
|
+
case "$1" in
|
|
87
|
+
--statement) statement="${2:-}"; shift 2 ;;
|
|
88
|
+
*) shift ;;
|
|
89
|
+
esac
|
|
90
|
+
done
|
|
91
|
+
|
|
92
|
+
if [ -z "$statement" ]; then
|
|
93
|
+
local brief="${LOKI_DIR:-.loki}/state/brief.txt"
|
|
94
|
+
if [ -f "$brief" ]; then
|
|
95
|
+
statement="$(cat "$brief" 2>/dev/null || true)"
|
|
96
|
+
source="brief"
|
|
97
|
+
fi
|
|
98
|
+
fi
|
|
99
|
+
|
|
100
|
+
if [ -z "$statement" ]; then
|
|
101
|
+
echo "No intent to record." >&2
|
|
102
|
+
echo "Give one explicitly: loki intent record --statement \"what you actually want\"" >&2
|
|
103
|
+
return 2
|
|
104
|
+
fi
|
|
105
|
+
|
|
106
|
+
local dir; dir="$(_intent_dir)"
|
|
107
|
+
mkdir -p "$dir" || return 3
|
|
108
|
+
local file; file="$(_intent_file)"
|
|
109
|
+
local sha; sha="$(_intent_sha256 "$statement")"
|
|
110
|
+
local sid="${sha:0:12}"
|
|
111
|
+
local now; now="$(_intent_now)"
|
|
112
|
+
|
|
113
|
+
python3 - "$file" "$sid" "$statement" "$sha" "$source" "$now" "$_INTENT_SCHEMA_VERSION" <<'PYEOF'
|
|
114
|
+
import json, os, sys
|
|
115
|
+
path, sid, text, sha, source, now, schema = sys.argv[1:8]
|
|
116
|
+
doc = {"schema_version": schema, "statements": []}
|
|
117
|
+
if os.path.isfile(path):
|
|
118
|
+
try:
|
|
119
|
+
with open(path, "r", encoding="utf-8") as fh:
|
|
120
|
+
doc = json.load(fh)
|
|
121
|
+
except Exception:
|
|
122
|
+
pass
|
|
123
|
+
doc.setdefault("statements", [])
|
|
124
|
+
# Idempotent on statement id, matching spec_ledger_write: recording the same
|
|
125
|
+
# intent twice must not create a second record or reset its link history.
|
|
126
|
+
for s in doc["statements"]:
|
|
127
|
+
if s.get("id") == sid:
|
|
128
|
+
print("already recorded: " + sid)
|
|
129
|
+
sys.exit(0)
|
|
130
|
+
doc["statements"].append({
|
|
131
|
+
"id": sid,
|
|
132
|
+
"text": text,
|
|
133
|
+
"text_sha256": sha,
|
|
134
|
+
"source": source,
|
|
135
|
+
"recorded_at": now,
|
|
136
|
+
"assumption_ids": [],
|
|
137
|
+
"links": [],
|
|
138
|
+
})
|
|
139
|
+
with open(path, "w", encoding="utf-8") as fh:
|
|
140
|
+
json.dump(doc, fh, indent=2)
|
|
141
|
+
fh.write("\n")
|
|
142
|
+
print("recorded: " + sid)
|
|
143
|
+
PYEOF
|
|
144
|
+
return $?
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
# ---------------------------------------------------------------------------
|
|
148
|
+
# link: bind an intent statement to a spec requirement AT ITS CURRENT HASH.
|
|
149
|
+
#
|
|
150
|
+
# Capturing content_hash_at_link is the whole measurement. Without it a link
|
|
151
|
+
# records only "these are related", which no later comparison can falsify.
|
|
152
|
+
# ---------------------------------------------------------------------------
|
|
153
|
+
intent_do_link() {
|
|
154
|
+
local sid="${1:-}" rid="${2:-}"
|
|
155
|
+
if [ -z "$sid" ] || [ -z "$rid" ]; then
|
|
156
|
+
echo "Usage: loki intent link <statement_id> <requirement_id>" >&2
|
|
157
|
+
return 2
|
|
158
|
+
fi
|
|
159
|
+
local file; file="$(_intent_file)"
|
|
160
|
+
local lock; lock="$(_intent_lock)"
|
|
161
|
+
[ -f "$file" ] || { echo "no intent record; run: loki intent record" >&2; return 2; }
|
|
162
|
+
[ -f "$lock" ] || { echo "no spec.lock; run: loki spec lock" >&2; return 2; }
|
|
163
|
+
|
|
164
|
+
python3 - "$file" "$lock" "$sid" "$rid" "$(_intent_now)" <<'PYEOF'
|
|
165
|
+
import hashlib, json, sys
|
|
166
|
+
ipath, lpath, sid, rid, now = sys.argv[1:6]
|
|
167
|
+
with open(ipath, "r", encoding="utf-8") as fh:
|
|
168
|
+
doc = json.load(fh)
|
|
169
|
+
with open(lpath, "r", encoding="utf-8") as fh:
|
|
170
|
+
raw = fh.read()
|
|
171
|
+
lock = json.loads(raw)
|
|
172
|
+
lock_hash = hashlib.sha256(raw.encode("utf-8")).hexdigest()
|
|
173
|
+
|
|
174
|
+
req = None
|
|
175
|
+
for r in lock.get("requirements", []):
|
|
176
|
+
if r.get("id") == rid:
|
|
177
|
+
req = r
|
|
178
|
+
break
|
|
179
|
+
if req is None:
|
|
180
|
+
sys.stderr.write("requirement id not in spec.lock: " + rid + "\n")
|
|
181
|
+
sys.exit(2)
|
|
182
|
+
|
|
183
|
+
stmt = None
|
|
184
|
+
for s in doc.get("statements", []):
|
|
185
|
+
if s.get("id") == sid:
|
|
186
|
+
stmt = s
|
|
187
|
+
break
|
|
188
|
+
if stmt is None:
|
|
189
|
+
sys.stderr.write("statement id not recorded: " + sid + "\n")
|
|
190
|
+
sys.exit(2)
|
|
191
|
+
|
|
192
|
+
for l in stmt.setdefault("links", []):
|
|
193
|
+
if l.get("requirement_id") == rid:
|
|
194
|
+
print("already linked: " + sid + " -> " + rid)
|
|
195
|
+
sys.exit(0)
|
|
196
|
+
|
|
197
|
+
ch = req.get("content_hash", "")
|
|
198
|
+
stmt["links"].append({
|
|
199
|
+
"requirement_id": rid,
|
|
200
|
+
"content_hash_at_link": ch,
|
|
201
|
+
"spec_lock_hash_at_link": lock_hash,
|
|
202
|
+
"affirmations": [{"content_hash": ch, "affirmed_at": now, "affirmed_by": "human"}],
|
|
203
|
+
})
|
|
204
|
+
with open(ipath, "w", encoding="utf-8") as fh:
|
|
205
|
+
json.dump(doc, fh, indent=2)
|
|
206
|
+
fh.write("\n")
|
|
207
|
+
print("linked: " + sid + " -> " + rid + " at " + ch[:12])
|
|
208
|
+
PYEOF
|
|
209
|
+
return $?
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
# ---------------------------------------------------------------------------
|
|
213
|
+
# affirm: re-affirm an intent against the requirement's CURRENT hash.
|
|
214
|
+
#
|
|
215
|
+
# WHY THIS EXISTS AND IS NOT OPTIONAL. Permanent idempotence is right for the
|
|
216
|
+
# assumption ledger and wrong here: without re-affirmation the first legitimate
|
|
217
|
+
# spec edit makes a statement STALE forever, the gate stays red, and a run grinds
|
|
218
|
+
# to max iterations against a finding no action can clear. That is the exact
|
|
219
|
+
# failure spec-interrogation.sh's own header documents for unresolved
|
|
220
|
+
# contradictions.
|
|
221
|
+
#
|
|
222
|
+
# It APPENDS rather than overwrites, so "this intent was re-affirmed across three
|
|
223
|
+
# successive versions of R" stays visible. The history is itself the useful fact.
|
|
224
|
+
# ---------------------------------------------------------------------------
|
|
225
|
+
intent_do_affirm() {
|
|
226
|
+
local sid="${1:-}"
|
|
227
|
+
[ -n "$sid" ] || { echo "Usage: loki intent affirm <statement_id>" >&2; return 2; }
|
|
228
|
+
local file; file="$(_intent_file)"
|
|
229
|
+
local lock; lock="$(_intent_lock)"
|
|
230
|
+
[ -f "$file" ] || { echo "no intent record" >&2; return 2; }
|
|
231
|
+
[ -f "$lock" ] || { echo "no spec.lock" >&2; return 2; }
|
|
232
|
+
|
|
233
|
+
python3 - "$file" "$lock" "$sid" "$(_intent_now)" <<'PYEOF'
|
|
234
|
+
import json, sys
|
|
235
|
+
ipath, lpath, sid, now = sys.argv[1:5]
|
|
236
|
+
with open(ipath, "r", encoding="utf-8") as fh:
|
|
237
|
+
doc = json.load(fh)
|
|
238
|
+
with open(lpath, "r", encoding="utf-8") as fh:
|
|
239
|
+
lock = json.load(fh)
|
|
240
|
+
by_id = {r.get("id"): r for r in lock.get("requirements", [])}
|
|
241
|
+
|
|
242
|
+
n = 0
|
|
243
|
+
for s in doc.get("statements", []):
|
|
244
|
+
if s.get("id") != sid:
|
|
245
|
+
continue
|
|
246
|
+
for l in s.get("links", []):
|
|
247
|
+
req = by_id.get(l.get("requirement_id"))
|
|
248
|
+
if req is None:
|
|
249
|
+
continue
|
|
250
|
+
ch = req.get("content_hash", "")
|
|
251
|
+
l.setdefault("affirmations", []).append(
|
|
252
|
+
{"content_hash": ch, "affirmed_at": now, "affirmed_by": "human"})
|
|
253
|
+
n += 1
|
|
254
|
+
if n == 0:
|
|
255
|
+
sys.stderr.write("nothing to affirm for: " + sid + "\n")
|
|
256
|
+
sys.exit(2)
|
|
257
|
+
with open(ipath, "w", encoding="utf-8") as fh:
|
|
258
|
+
json.dump(doc, fh, indent=2)
|
|
259
|
+
fh.write("\n")
|
|
260
|
+
print("affirmed " + str(n) + " link(s) for " + sid)
|
|
261
|
+
PYEOF
|
|
262
|
+
return $?
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
# ---------------------------------------------------------------------------
|
|
266
|
+
# status: the divergence report. Every verdict is a fact or a named refusal.
|
|
267
|
+
# ---------------------------------------------------------------------------
|
|
268
|
+
intent_do_status() {
|
|
269
|
+
local as_json=0
|
|
270
|
+
[ "${1:-}" = "--json" ] && as_json=1
|
|
271
|
+
local file; file="$(_intent_file)"
|
|
272
|
+
local lock; lock="$(_intent_lock)"
|
|
273
|
+
|
|
274
|
+
if [ ! -f "$file" ]; then
|
|
275
|
+
if [ "$as_json" = "1" ]; then
|
|
276
|
+
printf '{"schema_version":"%s","status":"UNKNOWN","reason":"no_intent_record","detail":"%s"}\n' \
|
|
277
|
+
"$_INTENT_SCHEMA_VERSION" "$(_intent_unknown_reason no_intent_record)"
|
|
278
|
+
else
|
|
279
|
+
echo "Intent Ledger: UNKNOWN"
|
|
280
|
+
echo " $(_intent_unknown_reason no_intent_record)"
|
|
281
|
+
echo ""
|
|
282
|
+
echo " Intent is not the spec. A spec you wrote yourself is your"
|
|
283
|
+
echo " INTERPRETATION already; recording intent separately is what makes"
|
|
284
|
+
echo " drift between the two measurable at all."
|
|
285
|
+
fi
|
|
286
|
+
return 0
|
|
287
|
+
fi
|
|
288
|
+
if [ ! -f "$lock" ]; then
|
|
289
|
+
if [ "$as_json" = "1" ]; then
|
|
290
|
+
printf '{"schema_version":"%s","status":"UNKNOWN","reason":"no_spec_lock","detail":"%s"}\n' \
|
|
291
|
+
"$_INTENT_SCHEMA_VERSION" "$(_intent_unknown_reason no_spec_lock)"
|
|
292
|
+
else
|
|
293
|
+
echo "Intent Ledger: UNKNOWN"
|
|
294
|
+
echo " $(_intent_unknown_reason no_spec_lock)"
|
|
295
|
+
fi
|
|
296
|
+
return 0
|
|
297
|
+
fi
|
|
298
|
+
|
|
299
|
+
python3 - "$file" "$lock" "$as_json" "$_INTENT_SCHEMA_VERSION" <<'PYEOF'
|
|
300
|
+
import json, sys
|
|
301
|
+
ipath, lpath, as_json, schema = sys.argv[1:5]
|
|
302
|
+
as_json = as_json == "1"
|
|
303
|
+
with open(ipath, "r", encoding="utf-8") as fh:
|
|
304
|
+
doc = json.load(fh)
|
|
305
|
+
with open(lpath, "r", encoding="utf-8") as fh:
|
|
306
|
+
lock = json.load(fh)
|
|
307
|
+
by_id = {r.get("id"): r for r in lock.get("requirements", [])}
|
|
308
|
+
|
|
309
|
+
rows = []
|
|
310
|
+
for s in doc.get("statements", []):
|
|
311
|
+
links = s.get("links") or []
|
|
312
|
+
if not links:
|
|
313
|
+
rows.append({"statement_id": s.get("id"), "verdict": "UNLINKED",
|
|
314
|
+
"detail": "recorded but never linked to a requirement"})
|
|
315
|
+
continue
|
|
316
|
+
for l in links:
|
|
317
|
+
rid = l.get("requirement_id")
|
|
318
|
+
row = {"statement_id": s.get("id"), "requirement_id": rid}
|
|
319
|
+
if rid not in by_id:
|
|
320
|
+
row["verdict"] = "LINKED-REMOVED"
|
|
321
|
+
row["detail"] = "the linked requirement is no longer in the spec"
|
|
322
|
+
rows.append(row); continue
|
|
323
|
+
affs = l.get("affirmations") or []
|
|
324
|
+
last = affs[-1].get("content_hash") if affs else l.get("content_hash_at_link")
|
|
325
|
+
if not last:
|
|
326
|
+
row["verdict"] = "UNKNOWN"
|
|
327
|
+
row["reason"] = "link_missing_content_hash"
|
|
328
|
+
rows.append(row); continue
|
|
329
|
+
current = by_id[rid].get("content_hash", "")
|
|
330
|
+
if current == last:
|
|
331
|
+
row["verdict"] = "LINKED-CURRENT"
|
|
332
|
+
row["affirmations"] = len(affs)
|
|
333
|
+
else:
|
|
334
|
+
row["verdict"] = "LINKED-STALE"
|
|
335
|
+
row["detail"] = ("affirmed at " + last[:12] + ", requirement is now "
|
|
336
|
+
+ current[:12] + " and nobody re-affirmed")
|
|
337
|
+
row["affirmations"] = len(affs)
|
|
338
|
+
rows.append(row)
|
|
339
|
+
|
|
340
|
+
stale = [r for r in rows if r["verdict"] in ("LINKED-STALE", "LINKED-REMOVED")]
|
|
341
|
+
summary = {
|
|
342
|
+
"statements": len(doc.get("statements", [])),
|
|
343
|
+
"linked_current": len([r for r in rows if r["verdict"] == "LINKED-CURRENT"]),
|
|
344
|
+
"linked_stale": len([r for r in rows if r["verdict"] == "LINKED-STALE"]),
|
|
345
|
+
"linked_removed": len([r for r in rows if r["verdict"] == "LINKED-REMOVED"]),
|
|
346
|
+
"unlinked": len([r for r in rows if r["verdict"] == "UNLINKED"]),
|
|
347
|
+
"unknown": len([r for r in rows if r["verdict"] == "UNKNOWN"]),
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
if as_json:
|
|
351
|
+
print(json.dumps({"schema_version": schema, "summary": summary,
|
|
352
|
+
"rows": rows}, indent=2))
|
|
353
|
+
else:
|
|
354
|
+
print("Intent Ledger -- does the spec still say what was actually wanted?")
|
|
355
|
+
print("")
|
|
356
|
+
for r in rows:
|
|
357
|
+
line = " " + str(r.get("statement_id", "?"))[:14].ljust(16)
|
|
358
|
+
line += r["verdict"].ljust(16)
|
|
359
|
+
if r.get("requirement_id"):
|
|
360
|
+
line += str(r["requirement_id"])[:24].ljust(26)
|
|
361
|
+
if r.get("detail"):
|
|
362
|
+
line += " " + r["detail"]
|
|
363
|
+
print(line)
|
|
364
|
+
print("")
|
|
365
|
+
print(" statements " + str(summary["statements"])
|
|
366
|
+
+ " current " + str(summary["linked_current"])
|
|
367
|
+
+ " stale " + str(summary["linked_stale"])
|
|
368
|
+
+ " removed " + str(summary["linked_removed"])
|
|
369
|
+
+ " unlinked " + str(summary["unlinked"]))
|
|
370
|
+
if stale:
|
|
371
|
+
print("")
|
|
372
|
+
print(" A stale link is the failure a verification gate cannot see: the")
|
|
373
|
+
print(" code still matches the spec, and the spec moved away from what")
|
|
374
|
+
print(" was wanted. Re-affirm once you agree with the change:")
|
|
375
|
+
print(" loki intent affirm <statement_id>")
|
|
376
|
+
|
|
377
|
+
sys.exit(1 if stale else 0)
|
|
378
|
+
PYEOF
|
|
379
|
+
return $?
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
intent_help() {
|
|
383
|
+
echo "loki intent - does the spec still say what was actually wanted?"
|
|
384
|
+
echo ""
|
|
385
|
+
echo "Usage: loki intent <subcommand>"
|
|
386
|
+
echo ""
|
|
387
|
+
echo " record [--statement \"...\"] record intent as a first-class artifact"
|
|
388
|
+
echo " link <stmt_id> <req_id> bind intent to a requirement at its current hash"
|
|
389
|
+
echo " affirm <stmt_id> re-affirm after an agreed spec change"
|
|
390
|
+
echo " status [--json] divergence report"
|
|
391
|
+
echo ""
|
|
392
|
+
echo "Verification proves code matches spec. It cannot prove the spec was"
|
|
393
|
+
echo "right. This measures the other half, deterministically: an intent"
|
|
394
|
+
echo "affirmed against requirement R at hash H reads LINKED-STALE once R"
|
|
395
|
+
echo "changes and nobody re-affirms. No model judges anything here, and there"
|
|
396
|
+
echo "is deliberately no similarity score."
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
intent_main() {
|
|
400
|
+
local sub="${1:-}"
|
|
401
|
+
[ $# -gt 0 ] && shift
|
|
402
|
+
case "$sub" in
|
|
403
|
+
record) intent_do_record "$@" ;;
|
|
404
|
+
link) intent_do_link "$@" ;;
|
|
405
|
+
affirm) intent_do_affirm "$@" ;;
|
|
406
|
+
status) intent_do_status "$@" ;;
|
|
407
|
+
""|--help|-h|help) intent_help ;;
|
|
408
|
+
*) echo "unknown subcommand: $sub" >&2; intent_help >&2; return 2 ;;
|
|
409
|
+
esac
|
|
410
|
+
}
|
|
411
|
+
|
|
412
|
+
if [ "${BASH_SOURCE[0]}" = "$0" ]; then
|
|
413
|
+
intent_main "$@"
|
|
414
|
+
fi
|
|
@@ -380,6 +380,26 @@ repo = data.get('repo', '')
|
|
|
380
380
|
|
|
381
381
|
labels_str = ', '.join(labels) if labels else ''
|
|
382
382
|
|
|
383
|
+
_ac_text = (title + ' ' + body).lower()
|
|
384
|
+
_ac_rules = [
|
|
385
|
+
(['save', 'persist', 'store', 'databas', 'crud'],
|
|
386
|
+
'Data the change writes survives a restart (a real store, not in-memory state).'),
|
|
387
|
+
(['auth', 'login', 'sign in', 'session', 'permission'],
|
|
388
|
+
'The auth path is exercised end to end including the denied case (401/403), not only the happy path.'),
|
|
389
|
+
(['api', 'endpoint', 'rest', 'graphql', 'route'],
|
|
390
|
+
'Each affected endpoint returns the documented status codes and is callable without a browser.'),
|
|
391
|
+
(['payment', 'stripe', 'billing', 'invoice', 'subscription'],
|
|
392
|
+
'The payment path runs against provider test mode; no mocked charge stands in for the integration.'),
|
|
393
|
+
(['bug', 'fix', 'regression', 'broken', 'crash', 'error'],
|
|
394
|
+
'A test reproduces the reported failure and FAILS before the fix, then passes after it.'),
|
|
395
|
+
(['perf', 'slow', 'latency', 'timeout', 'memory leak'],
|
|
396
|
+
'The improvement is measured before and after, and the numbers appear in the change.'),
|
|
397
|
+
(['security', 'vulnerab', 'injection', 'xss', 'csrf'],
|
|
398
|
+
'A test demonstrates the vulnerable behavior is refused after the change.'),
|
|
399
|
+
]
|
|
400
|
+
_hits = [c for kws, c in _ac_rules if any(k in _ac_text for k in kws)]
|
|
401
|
+
derived_ac = ('\n'.join('- ' + h for h in _hits) + '\n') if _hits else ''
|
|
402
|
+
|
|
383
403
|
prd = f'''# PRD: {title}
|
|
384
404
|
|
|
385
405
|
**Source:** {provider.replace('_', ' ').title()} Issue [{number}]({url})
|
|
@@ -408,6 +428,7 @@ Based on the issue description, implement the following:
|
|
|
408
428
|
2. Ensure backward compatibility (unless explicitly breaking changes are requested)
|
|
409
429
|
3. Add appropriate tests for new functionality
|
|
410
430
|
4. Update documentation as needed
|
|
431
|
+
{derived_ac}
|
|
411
432
|
|
|
412
433
|
---
|
|
413
434
|
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Agent readiness: can an autonomous agent verify its own work in THIS repo?
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS, AND WHY IT IS NOT A COPY. Factory AI's Agent Readiness Model is
|
|
5
|
+
a genuinely good idea and a category-defining artifact -- 5 levels, 9 pillars,
|
|
6
|
+
2 scopes -- and it makes competitor comparisons happen on Factory's chosen axes.
|
|
7
|
+
It is also LLM-SCORED: their report objects record `modelUsed` and
|
|
8
|
+
`reasoningEffort` per report. So the number is a model's opinion of a repo, and
|
|
9
|
+
two runs can disagree about the same commit.
|
|
10
|
+
|
|
11
|
+
Ours is a measurement. Every criterion below is a file that exists or does not,
|
|
12
|
+
a command that is present or absent. Same commit, same answer, every time, on
|
|
13
|
+
any machine, with no key and no spend. "Theirs is an opinion, ours is a
|
|
14
|
+
measurement, here is the command" is the same wedge as the receipt, applied to
|
|
15
|
+
their own differentiated concept.
|
|
16
|
+
|
|
17
|
+
WHAT IT MEASURES, AND WHY THOSE. Not general code quality -- that is what
|
|
18
|
+
`loki modernize heal --assess` already scores with its own deterministic 4-level
|
|
19
|
+
maturity rubric, and duplicating it would create two numbers that eventually
|
|
20
|
+
disagree. This asks the narrower question our product actually depends on:
|
|
21
|
+
CAN AN AGENT CHECK ITSELF HERE? Factory concedes the same dependency from the
|
|
22
|
+
other side -- their Missions docs say that without "an automated, scriptable way
|
|
23
|
+
to exercise the app... the mission cannot reliably verify its own work", and
|
|
24
|
+
recommend Level 4+ before using their flagship. A repo with no test command is
|
|
25
|
+
one where every agent, ours included, is guessing.
|
|
26
|
+
|
|
27
|
+
WHAT IT REFUSES. No percentage, no letter grade, no composite. A composite
|
|
28
|
+
invites ranking, ranking invites gaming, and the individual signals are the
|
|
29
|
+
actionable part: "there is no test command" tells you what to do, "readiness 62%"
|
|
30
|
+
does not. Criteria that cannot be determined report UNKNOWN by name rather than
|
|
31
|
+
counting as failures -- an absent measurement is not a bad score.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
from __future__ import annotations
|
|
35
|
+
|
|
36
|
+
import json
|
|
37
|
+
import os
|
|
38
|
+
import sys
|
|
39
|
+
|
|
40
|
+
SCHEMA_VERSION = "1.0"
|
|
41
|
+
|
|
42
|
+
UNKNOWN = "UNKNOWN"
|
|
43
|
+
|
|
44
|
+
# Each criterion is a pure filesystem fact plus the command a reader can run to
|
|
45
|
+
# check it themselves. The `why` is not decoration: a signal whose consequence
|
|
46
|
+
# for an agent is unstated becomes a checkbox someone games.
|
|
47
|
+
CRITERIA = [
|
|
48
|
+
{
|
|
49
|
+
"id": "test_command",
|
|
50
|
+
"why": "without a runnable test command an agent cannot verify its own change",
|
|
51
|
+
"verify": "look for a test script in package.json, a Makefile test target, pytest.ini, or tests/",
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"id": "dependency_lock",
|
|
55
|
+
"why": "unpinned dependencies make a green run unreproducible tomorrow",
|
|
56
|
+
"verify": "look for package-lock.json, bun.lockb, poetry.lock, requirements.txt, Cargo.lock, go.sum",
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
"id": "ci_config",
|
|
60
|
+
"why": "without CI, nothing re-checks the agent's work independently of the agent",
|
|
61
|
+
"verify": "look for .github/workflows, .gitlab-ci.yml, or a CI config at the repo root",
|
|
62
|
+
},
|
|
63
|
+
{
|
|
64
|
+
"id": "agent_brief",
|
|
65
|
+
"why": "without a briefing file an agent rediscovers conventions every run and gets them wrong",
|
|
66
|
+
"verify": "look for AGENTS.md, CLAUDE.md, CONTRIBUTING.md",
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
"id": "readme",
|
|
70
|
+
"why": "without a README an agent has no statement of what the project is for",
|
|
71
|
+
"verify": "look for README.md or README",
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
"id": "gitignore",
|
|
75
|
+
"why": "without ignores an agent's diff fills with build output and the real change is buried",
|
|
76
|
+
"verify": "look for .gitignore",
|
|
77
|
+
},
|
|
78
|
+
]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _any_exists(root, names):
|
|
82
|
+
for n in names:
|
|
83
|
+
if os.path.exists(os.path.join(root, n)):
|
|
84
|
+
return n
|
|
85
|
+
return None
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _has_test_command(root):
|
|
89
|
+
pkg = os.path.join(root, "package.json")
|
|
90
|
+
if os.path.isfile(pkg):
|
|
91
|
+
try:
|
|
92
|
+
with open(pkg, "r", encoding="utf-8") as fh:
|
|
93
|
+
data = json.load(fh)
|
|
94
|
+
if (data.get("scripts") or {}).get("test"):
|
|
95
|
+
return "package.json scripts.test"
|
|
96
|
+
except (OSError, ValueError):
|
|
97
|
+
# A malformed package.json is not evidence either way. Fall through
|
|
98
|
+
# to the other signals rather than scoring it as absent.
|
|
99
|
+
pass
|
|
100
|
+
found = _any_exists(root, ["pytest.ini", "tox.ini", "Makefile", "tests", "test"])
|
|
101
|
+
return found
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def assess(root):
|
|
105
|
+
"""Evaluate every criterion. Returns facts, never a score."""
|
|
106
|
+
if not os.path.isdir(root):
|
|
107
|
+
return {"status": UNKNOWN, "reason": "no_such_directory", "path": root}
|
|
108
|
+
if not os.path.isdir(os.path.join(root, ".git")):
|
|
109
|
+
# Not fatal: readiness is about the working tree. Recorded so a reader
|
|
110
|
+
# knows the repo context was absent rather than assumed.
|
|
111
|
+
git_present = False
|
|
112
|
+
else:
|
|
113
|
+
git_present = True
|
|
114
|
+
|
|
115
|
+
checks = []
|
|
116
|
+
for c in CRITERIA:
|
|
117
|
+
cid = c["id"]
|
|
118
|
+
if cid == "test_command":
|
|
119
|
+
hit = _has_test_command(root)
|
|
120
|
+
elif cid == "dependency_lock":
|
|
121
|
+
hit = _any_exists(root, ["package-lock.json", "bun.lockb", "yarn.lock",
|
|
122
|
+
"poetry.lock", "requirements.txt", "Cargo.lock",
|
|
123
|
+
"go.sum", "Pipfile.lock"])
|
|
124
|
+
elif cid == "ci_config":
|
|
125
|
+
hit = _any_exists(root, [".github/workflows", ".gitlab-ci.yml",
|
|
126
|
+
".circleci", "azure-pipelines.yml", "Jenkinsfile"])
|
|
127
|
+
elif cid == "agent_brief":
|
|
128
|
+
hit = _any_exists(root, ["AGENTS.md", "CLAUDE.md", "CONTRIBUTING.md"])
|
|
129
|
+
elif cid == "readme":
|
|
130
|
+
hit = _any_exists(root, ["README.md", "README", "README.rst"])
|
|
131
|
+
elif cid == "gitignore":
|
|
132
|
+
hit = _any_exists(root, [".gitignore"])
|
|
133
|
+
else:
|
|
134
|
+
hit = None
|
|
135
|
+
|
|
136
|
+
checks.append({
|
|
137
|
+
"id": cid,
|
|
138
|
+
"present": bool(hit),
|
|
139
|
+
"found": hit or None,
|
|
140
|
+
"why": c["why"],
|
|
141
|
+
"verify": c["verify"],
|
|
142
|
+
})
|
|
143
|
+
|
|
144
|
+
present = [c for c in checks if c["present"]]
|
|
145
|
+
missing = [c for c in checks if not c["present"]]
|
|
146
|
+
|
|
147
|
+
return {
|
|
148
|
+
"schema_version": SCHEMA_VERSION,
|
|
149
|
+
"status": "measured",
|
|
150
|
+
"path": os.path.abspath(root),
|
|
151
|
+
"git_repo": git_present,
|
|
152
|
+
# Counts, not a percentage. A composite invites ranking, ranking invites
|
|
153
|
+
# gaming, and "there is no test command" is the actionable part anyway.
|
|
154
|
+
"criteria_total": len(checks),
|
|
155
|
+
"criteria_present": len(present),
|
|
156
|
+
"checks": checks,
|
|
157
|
+
"missing": [c["id"] for c in missing],
|
|
158
|
+
# The single most consequential signal, surfaced on its own: this is the
|
|
159
|
+
# one Factory's own docs concede their flagship depends on.
|
|
160
|
+
"can_self_verify": any(c["id"] == "test_command" and c["present"]
|
|
161
|
+
for c in checks),
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def render_text(res):
|
|
166
|
+
if res.get("status") != "measured":
|
|
167
|
+
return f"Agent readiness: UNKNOWN ({res.get('reason', 'unmeasurable')})"
|
|
168
|
+
out = ["Agent readiness -- can an agent verify its own work here?", ""]
|
|
169
|
+
for c in res["checks"]:
|
|
170
|
+
mark = "yes" if c["present"] else "NO "
|
|
171
|
+
line = f" {mark} {c['id']:20}"
|
|
172
|
+
if c["present"]:
|
|
173
|
+
line += f"({c['found']})"
|
|
174
|
+
else:
|
|
175
|
+
line += c["why"]
|
|
176
|
+
out.append(line)
|
|
177
|
+
out.append("")
|
|
178
|
+
out.append(f" {res['criteria_present']} of {res['criteria_total']} present")
|
|
179
|
+
if not res["can_self_verify"]:
|
|
180
|
+
out.append("")
|
|
181
|
+
out.append(" No test command found. Every agent working here, ours")
|
|
182
|
+
out.append(" included, is guessing whether its change worked.")
|
|
183
|
+
out.append("")
|
|
184
|
+
out.append(" Every line above is a file that exists or does not. Check any of")
|
|
185
|
+
out.append(" them by hand; no model was asked for an opinion.")
|
|
186
|
+
return "\n".join(out)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def main(argv):
|
|
190
|
+
as_json = "--json" in argv
|
|
191
|
+
root = "."
|
|
192
|
+
for a in argv:
|
|
193
|
+
if not a.startswith("-"):
|
|
194
|
+
root = a
|
|
195
|
+
break
|
|
196
|
+
res = assess(root)
|
|
197
|
+
print(json.dumps(res, indent=2) if as_json else render_text(res))
|
|
198
|
+
return 0 if res.get("status") == "measured" else 3
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
if __name__ == "__main__":
|
|
202
|
+
sys.exit(main(sys.argv[1:]))
|