loki-mode 9.12.6 → 9.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,414 @@
1
+ #!/usr/bin/env bash
2
+ # Intent Ledger: does the SPEC still say what the person actually wanted?
3
+ #
4
+ # THE PROBLEM THIS ANSWERS. Our verification proves code matches spec. It cannot
5
+ # prove the spec was RIGHT. 8090 AI documents a real build where "the software
6
+ # converged with the interpretation. The interpretation had diverged from the
7
+ # intent." A perfect verification gate passes that build, and the build is still
8
+ # wrong. That gap sits upstream of every gate we have, and no competitor ships an
9
+ # answer -- 8090 published the argument and their own "Tests" module has zero
10
+ # documentation pages and zero changelog entries.
11
+ #
12
+ # WHAT WE DELIBERATELY DID NOT BUILD. The obvious feature is a semantic fidelity
13
+ # score: ask a model "does this spec faithfully express this intent?" and print a
14
+ # percentage. That is an LLM judgment wearing the costume of a measurement, and
15
+ # it is precisely what our receipts exist to refuse -- they already separate
16
+ # deterministic FACTS from AI ASSESSMENTS. There is no similarity number, no
17
+ # embedding distance, and no percent-aligned figure anywhere in this file, and a
18
+ # future contributor adding one would be removing the reason it is trustworthy.
19
+ #
20
+ # WHAT IS ACTUALLY MEASURABLE, and it is enough. autonomy/spec.sh already
21
+ # persists a content_hash per requirement into .loki/spec/spec.lock. So if an
22
+ # intent statement records WHICH requirement it was affirmed against AND that
23
+ # requirement's hash AT THAT MOMENT, then divergence is pure hash comparison:
24
+ #
25
+ # the intent was affirmed against requirement R at hash H;
26
+ # R now hashes to H';
27
+ # nobody re-affirmed.
28
+ #
29
+ # That is 8090's failure mode, detected deterministically, re-derivable by hand,
30
+ # with no model in the loop. `content_hash_at_link` is the entire design -- store
31
+ # only a requirement id and every verdict collapses to UNKNOWN.
32
+ #
33
+ # WHAT ALREADY EXISTED (checked before building, not assumed). Three of the four
34
+ # things this was scoped to do are already shipped and are NOT rebuilt here:
35
+ # - ASSUMED-BUT-NOT-STATED -> spec-interrogation.sh:284 (.loki/assumptions/)
36
+ # - spec-vs-built divergence -> spec.sh + verify.sh:2161 (spec.lock, drift gate)
37
+ # - pre-committed predictions -> lib/expectation-ledger.py
38
+ # This file references the assumption store by id. A second store would drift
39
+ # from the first.
40
+
41
+ set -uo pipefail
42
+
43
+ _INTENT_SCHEMA_VERSION="1.0"
44
+
45
+ # Named refusals. Mirrors the outcome ledger's ANCHOR_REASONS: a status we cannot
46
+ # compute is reported by NAME, never as 0 and never as a pass. The most important
47
+ # entry is no_distinct_intent_source -- see intent_do_status.
48
+ _intent_unknown_reason() {
49
+ case "${1:-}" in
50
+ no_intent_record) echo "no intent has been recorded (run: loki intent record)" ;;
51
+ no_spec_lock) echo "no .loki/spec/spec.lock (run: loki spec lock)" ;;
52
+ no_distinct_intent_source) echo "the spec IS the user's own document, so intent and interpretation are the same artifact" ;;
53
+ link_missing_content_hash) echo "link predates hash recording, so staleness cannot be computed" ;;
54
+ requirement_id_not_in_lock) echo "the linked requirement id is not in the current lock" ;;
55
+ *) echo "unmeasurable" ;;
56
+ esac
57
+ }
58
+
59
+ _intent_dir() { echo "${LOKI_DIR:-.loki}/intent"; }
60
+ _intent_file() { echo "$(_intent_dir)/intent.json"; }
61
+ _intent_lock() { echo "${LOKI_DIR:-.loki}/spec/spec.lock"; }
62
+
63
+ _intent_now() { date -u +%Y-%m-%dT%H:%M:%SZ; }
64
+
65
+ _intent_sha256() {
66
+ if command -v shasum >/dev/null 2>&1; then
67
+ printf '%s' "$1" | shasum -a 256 | awk '{print $1}'
68
+ else
69
+ printf '%s' "$1" | sha256sum | awk '{print $1}'
70
+ fi
71
+ }
72
+
73
+ # ---------------------------------------------------------------------------
74
+ # record: capture an intent statement as a first-class artifact.
75
+ #
76
+ # THE TEXT IS COPIED, NEVER REFERENCED. autonomy/loki:2079 deletes
77
+ # .loki/state/brief.txt at the start of a later non-brief run, deliberately, so a
78
+ # stale one-liner cannot be inherited. An intent record whose only evidence is a
79
+ # file that a later run removes would be unmeasurable by construction, so the
80
+ # statement text and its sha256 are stored inline.
81
+ # ---------------------------------------------------------------------------
82
+ intent_do_record() {
83
+ local statement=""
84
+ local source="explicit"
85
+ while [ $# -gt 0 ]; do
86
+ case "$1" in
87
+ --statement) statement="${2:-}"; shift 2 ;;
88
+ *) shift ;;
89
+ esac
90
+ done
91
+
92
+ if [ -z "$statement" ]; then
93
+ local brief="${LOKI_DIR:-.loki}/state/brief.txt"
94
+ if [ -f "$brief" ]; then
95
+ statement="$(cat "$brief" 2>/dev/null || true)"
96
+ source="brief"
97
+ fi
98
+ fi
99
+
100
+ if [ -z "$statement" ]; then
101
+ echo "No intent to record." >&2
102
+ echo "Give one explicitly: loki intent record --statement \"what you actually want\"" >&2
103
+ return 2
104
+ fi
105
+
106
+ local dir; dir="$(_intent_dir)"
107
+ mkdir -p "$dir" || return 3
108
+ local file; file="$(_intent_file)"
109
+ local sha; sha="$(_intent_sha256 "$statement")"
110
+ local sid="${sha:0:12}"
111
+ local now; now="$(_intent_now)"
112
+
113
+ python3 - "$file" "$sid" "$statement" "$sha" "$source" "$now" "$_INTENT_SCHEMA_VERSION" <<'PYEOF'
114
+ import json, os, sys
115
+ path, sid, text, sha, source, now, schema = sys.argv[1:8]
116
+ doc = {"schema_version": schema, "statements": []}
117
+ if os.path.isfile(path):
118
+ try:
119
+ with open(path, "r", encoding="utf-8") as fh:
120
+ doc = json.load(fh)
121
+ except Exception:
122
+ pass
123
+ doc.setdefault("statements", [])
124
+ # Idempotent on statement id, matching spec_ledger_write: recording the same
125
+ # intent twice must not create a second record or reset its link history.
126
+ for s in doc["statements"]:
127
+ if s.get("id") == sid:
128
+ print("already recorded: " + sid)
129
+ sys.exit(0)
130
+ doc["statements"].append({
131
+ "id": sid,
132
+ "text": text,
133
+ "text_sha256": sha,
134
+ "source": source,
135
+ "recorded_at": now,
136
+ "assumption_ids": [],
137
+ "links": [],
138
+ })
139
+ with open(path, "w", encoding="utf-8") as fh:
140
+ json.dump(doc, fh, indent=2)
141
+ fh.write("\n")
142
+ print("recorded: " + sid)
143
+ PYEOF
144
+ return $?
145
+ }
146
+
147
+ # ---------------------------------------------------------------------------
148
+ # link: bind an intent statement to a spec requirement AT ITS CURRENT HASH.
149
+ #
150
+ # Capturing content_hash_at_link is the whole measurement. Without it a link
151
+ # records only "these are related", which no later comparison can falsify.
152
+ # ---------------------------------------------------------------------------
153
+ intent_do_link() {
154
+ local sid="${1:-}" rid="${2:-}"
155
+ if [ -z "$sid" ] || [ -z "$rid" ]; then
156
+ echo "Usage: loki intent link <statement_id> <requirement_id>" >&2
157
+ return 2
158
+ fi
159
+ local file; file="$(_intent_file)"
160
+ local lock; lock="$(_intent_lock)"
161
+ [ -f "$file" ] || { echo "no intent record; run: loki intent record" >&2; return 2; }
162
+ [ -f "$lock" ] || { echo "no spec.lock; run: loki spec lock" >&2; return 2; }
163
+
164
+ python3 - "$file" "$lock" "$sid" "$rid" "$(_intent_now)" <<'PYEOF'
165
+ import hashlib, json, sys
166
+ ipath, lpath, sid, rid, now = sys.argv[1:6]
167
+ with open(ipath, "r", encoding="utf-8") as fh:
168
+ doc = json.load(fh)
169
+ with open(lpath, "r", encoding="utf-8") as fh:
170
+ raw = fh.read()
171
+ lock = json.loads(raw)
172
+ lock_hash = hashlib.sha256(raw.encode("utf-8")).hexdigest()
173
+
174
+ req = None
175
+ for r in lock.get("requirements", []):
176
+ if r.get("id") == rid:
177
+ req = r
178
+ break
179
+ if req is None:
180
+ sys.stderr.write("requirement id not in spec.lock: " + rid + "\n")
181
+ sys.exit(2)
182
+
183
+ stmt = None
184
+ for s in doc.get("statements", []):
185
+ if s.get("id") == sid:
186
+ stmt = s
187
+ break
188
+ if stmt is None:
189
+ sys.stderr.write("statement id not recorded: " + sid + "\n")
190
+ sys.exit(2)
191
+
192
+ for l in stmt.setdefault("links", []):
193
+ if l.get("requirement_id") == rid:
194
+ print("already linked: " + sid + " -> " + rid)
195
+ sys.exit(0)
196
+
197
+ ch = req.get("content_hash", "")
198
+ stmt["links"].append({
199
+ "requirement_id": rid,
200
+ "content_hash_at_link": ch,
201
+ "spec_lock_hash_at_link": lock_hash,
202
+ "affirmations": [{"content_hash": ch, "affirmed_at": now, "affirmed_by": "human"}],
203
+ })
204
+ with open(ipath, "w", encoding="utf-8") as fh:
205
+ json.dump(doc, fh, indent=2)
206
+ fh.write("\n")
207
+ print("linked: " + sid + " -> " + rid + " at " + ch[:12])
208
+ PYEOF
209
+ return $?
210
+ }
211
+
212
+ # ---------------------------------------------------------------------------
213
+ # affirm: re-affirm an intent against the requirement's CURRENT hash.
214
+ #
215
+ # WHY THIS EXISTS AND IS NOT OPTIONAL. Permanent idempotence is right for the
216
+ # assumption ledger and wrong here: without re-affirmation the first legitimate
217
+ # spec edit makes a statement STALE forever, the gate stays red, and a run grinds
218
+ # to max iterations against a finding no action can clear. That is the exact
219
+ # failure spec-interrogation.sh's own header documents for unresolved
220
+ # contradictions.
221
+ #
222
+ # It APPENDS rather than overwrites, so "this intent was re-affirmed across three
223
+ # successive versions of R" stays visible. The history is itself the useful fact.
224
+ # ---------------------------------------------------------------------------
225
+ intent_do_affirm() {
226
+ local sid="${1:-}"
227
+ [ -n "$sid" ] || { echo "Usage: loki intent affirm <statement_id>" >&2; return 2; }
228
+ local file; file="$(_intent_file)"
229
+ local lock; lock="$(_intent_lock)"
230
+ [ -f "$file" ] || { echo "no intent record" >&2; return 2; }
231
+ [ -f "$lock" ] || { echo "no spec.lock" >&2; return 2; }
232
+
233
+ python3 - "$file" "$lock" "$sid" "$(_intent_now)" <<'PYEOF'
234
+ import json, sys
235
+ ipath, lpath, sid, now = sys.argv[1:5]
236
+ with open(ipath, "r", encoding="utf-8") as fh:
237
+ doc = json.load(fh)
238
+ with open(lpath, "r", encoding="utf-8") as fh:
239
+ lock = json.load(fh)
240
+ by_id = {r.get("id"): r for r in lock.get("requirements", [])}
241
+
242
+ n = 0
243
+ for s in doc.get("statements", []):
244
+ if s.get("id") != sid:
245
+ continue
246
+ for l in s.get("links", []):
247
+ req = by_id.get(l.get("requirement_id"))
248
+ if req is None:
249
+ continue
250
+ ch = req.get("content_hash", "")
251
+ l.setdefault("affirmations", []).append(
252
+ {"content_hash": ch, "affirmed_at": now, "affirmed_by": "human"})
253
+ n += 1
254
+ if n == 0:
255
+ sys.stderr.write("nothing to affirm for: " + sid + "\n")
256
+ sys.exit(2)
257
+ with open(ipath, "w", encoding="utf-8") as fh:
258
+ json.dump(doc, fh, indent=2)
259
+ fh.write("\n")
260
+ print("affirmed " + str(n) + " link(s) for " + sid)
261
+ PYEOF
262
+ return $?
263
+ }
264
+
265
+ # ---------------------------------------------------------------------------
266
+ # status: the divergence report. Every verdict is a fact or a named refusal.
267
+ # ---------------------------------------------------------------------------
268
+ intent_do_status() {
269
+ local as_json=0
270
+ [ "${1:-}" = "--json" ] && as_json=1
271
+ local file; file="$(_intent_file)"
272
+ local lock; lock="$(_intent_lock)"
273
+
274
+ if [ ! -f "$file" ]; then
275
+ if [ "$as_json" = "1" ]; then
276
+ printf '{"schema_version":"%s","status":"UNKNOWN","reason":"no_intent_record","detail":"%s"}\n' \
277
+ "$_INTENT_SCHEMA_VERSION" "$(_intent_unknown_reason no_intent_record)"
278
+ else
279
+ echo "Intent Ledger: UNKNOWN"
280
+ echo " $(_intent_unknown_reason no_intent_record)"
281
+ echo ""
282
+ echo " Intent is not the spec. A spec you wrote yourself is your"
283
+ echo " INTERPRETATION already; recording intent separately is what makes"
284
+ echo " drift between the two measurable at all."
285
+ fi
286
+ return 0
287
+ fi
288
+ if [ ! -f "$lock" ]; then
289
+ if [ "$as_json" = "1" ]; then
290
+ printf '{"schema_version":"%s","status":"UNKNOWN","reason":"no_spec_lock","detail":"%s"}\n' \
291
+ "$_INTENT_SCHEMA_VERSION" "$(_intent_unknown_reason no_spec_lock)"
292
+ else
293
+ echo "Intent Ledger: UNKNOWN"
294
+ echo " $(_intent_unknown_reason no_spec_lock)"
295
+ fi
296
+ return 0
297
+ fi
298
+
299
+ python3 - "$file" "$lock" "$as_json" "$_INTENT_SCHEMA_VERSION" <<'PYEOF'
300
+ import json, sys
301
+ ipath, lpath, as_json, schema = sys.argv[1:5]
302
+ as_json = as_json == "1"
303
+ with open(ipath, "r", encoding="utf-8") as fh:
304
+ doc = json.load(fh)
305
+ with open(lpath, "r", encoding="utf-8") as fh:
306
+ lock = json.load(fh)
307
+ by_id = {r.get("id"): r for r in lock.get("requirements", [])}
308
+
309
+ rows = []
310
+ for s in doc.get("statements", []):
311
+ links = s.get("links") or []
312
+ if not links:
313
+ rows.append({"statement_id": s.get("id"), "verdict": "UNLINKED",
314
+ "detail": "recorded but never linked to a requirement"})
315
+ continue
316
+ for l in links:
317
+ rid = l.get("requirement_id")
318
+ row = {"statement_id": s.get("id"), "requirement_id": rid}
319
+ if rid not in by_id:
320
+ row["verdict"] = "LINKED-REMOVED"
321
+ row["detail"] = "the linked requirement is no longer in the spec"
322
+ rows.append(row); continue
323
+ affs = l.get("affirmations") or []
324
+ last = affs[-1].get("content_hash") if affs else l.get("content_hash_at_link")
325
+ if not last:
326
+ row["verdict"] = "UNKNOWN"
327
+ row["reason"] = "link_missing_content_hash"
328
+ rows.append(row); continue
329
+ current = by_id[rid].get("content_hash", "")
330
+ if current == last:
331
+ row["verdict"] = "LINKED-CURRENT"
332
+ row["affirmations"] = len(affs)
333
+ else:
334
+ row["verdict"] = "LINKED-STALE"
335
+ row["detail"] = ("affirmed at " + last[:12] + ", requirement is now "
336
+ + current[:12] + " and nobody re-affirmed")
337
+ row["affirmations"] = len(affs)
338
+ rows.append(row)
339
+
340
+ stale = [r for r in rows if r["verdict"] in ("LINKED-STALE", "LINKED-REMOVED")]
341
+ summary = {
342
+ "statements": len(doc.get("statements", [])),
343
+ "linked_current": len([r for r in rows if r["verdict"] == "LINKED-CURRENT"]),
344
+ "linked_stale": len([r for r in rows if r["verdict"] == "LINKED-STALE"]),
345
+ "linked_removed": len([r for r in rows if r["verdict"] == "LINKED-REMOVED"]),
346
+ "unlinked": len([r for r in rows if r["verdict"] == "UNLINKED"]),
347
+ "unknown": len([r for r in rows if r["verdict"] == "UNKNOWN"]),
348
+ }
349
+
350
+ if as_json:
351
+ print(json.dumps({"schema_version": schema, "summary": summary,
352
+ "rows": rows}, indent=2))
353
+ else:
354
+ print("Intent Ledger -- does the spec still say what was actually wanted?")
355
+ print("")
356
+ for r in rows:
357
+ line = " " + str(r.get("statement_id", "?"))[:14].ljust(16)
358
+ line += r["verdict"].ljust(16)
359
+ if r.get("requirement_id"):
360
+ line += str(r["requirement_id"])[:24].ljust(26)
361
+ if r.get("detail"):
362
+ line += " " + r["detail"]
363
+ print(line)
364
+ print("")
365
+ print(" statements " + str(summary["statements"])
366
+ + " current " + str(summary["linked_current"])
367
+ + " stale " + str(summary["linked_stale"])
368
+ + " removed " + str(summary["linked_removed"])
369
+ + " unlinked " + str(summary["unlinked"]))
370
+ if stale:
371
+ print("")
372
+ print(" A stale link is the failure a verification gate cannot see: the")
373
+ print(" code still matches the spec, and the spec moved away from what")
374
+ print(" was wanted. Re-affirm once you agree with the change:")
375
+ print(" loki intent affirm <statement_id>")
376
+
377
+ sys.exit(1 if stale else 0)
378
+ PYEOF
379
+ return $?
380
+ }
381
+
382
+ intent_help() {
383
+ echo "loki intent - does the spec still say what was actually wanted?"
384
+ echo ""
385
+ echo "Usage: loki intent <subcommand>"
386
+ echo ""
387
+ echo " record [--statement \"...\"] record intent as a first-class artifact"
388
+ echo " link <stmt_id> <req_id> bind intent to a requirement at its current hash"
389
+ echo " affirm <stmt_id> re-affirm after an agreed spec change"
390
+ echo " status [--json] divergence report"
391
+ echo ""
392
+ echo "Verification proves code matches spec. It cannot prove the spec was"
393
+ echo "right. This measures the other half, deterministically: an intent"
394
+ echo "affirmed against requirement R at hash H reads LINKED-STALE once R"
395
+ echo "changes and nobody re-affirms. No model judges anything here, and there"
396
+ echo "is deliberately no similarity score."
397
+ }
398
+
399
+ intent_main() {
400
+ local sub="${1:-}"
401
+ [ $# -gt 0 ] && shift
402
+ case "$sub" in
403
+ record) intent_do_record "$@" ;;
404
+ link) intent_do_link "$@" ;;
405
+ affirm) intent_do_affirm "$@" ;;
406
+ status) intent_do_status "$@" ;;
407
+ ""|--help|-h|help) intent_help ;;
408
+ *) echo "unknown subcommand: $sub" >&2; intent_help >&2; return 2 ;;
409
+ esac
410
+ }
411
+
412
+ if [ "${BASH_SOURCE[0]}" = "$0" ]; then
413
+ intent_main "$@"
414
+ fi
@@ -380,6 +380,26 @@ repo = data.get('repo', '')
380
380
 
381
381
  labels_str = ', '.join(labels) if labels else ''
382
382
 
383
+ _ac_text = (title + ' ' + body).lower()
384
+ _ac_rules = [
385
+ (['save', 'persist', 'store', 'databas', 'crud'],
386
+ 'Data the change writes survives a restart (a real store, not in-memory state).'),
387
+ (['auth', 'login', 'sign in', 'session', 'permission'],
388
+ 'The auth path is exercised end to end including the denied case (401/403), not only the happy path.'),
389
+ (['api', 'endpoint', 'rest', 'graphql', 'route'],
390
+ 'Each affected endpoint returns the documented status codes and is callable without a browser.'),
391
+ (['payment', 'stripe', 'billing', 'invoice', 'subscription'],
392
+ 'The payment path runs against provider test mode; no mocked charge stands in for the integration.'),
393
+ (['bug', 'fix', 'regression', 'broken', 'crash', 'error'],
394
+ 'A test reproduces the reported failure and FAILS before the fix, then passes after it.'),
395
+ (['perf', 'slow', 'latency', 'timeout', 'memory leak'],
396
+ 'The improvement is measured before and after, and the numbers appear in the change.'),
397
+ (['security', 'vulnerab', 'injection', 'xss', 'csrf'],
398
+ 'A test demonstrates the vulnerable behavior is refused after the change.'),
399
+ ]
400
+ _hits = [c for kws, c in _ac_rules if any(k in _ac_text for k in kws)]
401
+ derived_ac = ('\n'.join('- ' + h for h in _hits) + '\n') if _hits else ''
402
+
383
403
  prd = f'''# PRD: {title}
384
404
 
385
405
  **Source:** {provider.replace('_', ' ').title()} Issue [{number}]({url})
@@ -408,6 +428,7 @@ Based on the issue description, implement the following:
408
428
  2. Ensure backward compatibility (unless explicitly breaking changes are requested)
409
429
  3. Add appropriate tests for new functionality
410
430
  4. Update documentation as needed
431
+ {derived_ac}
411
432
 
412
433
  ---
413
434
 
@@ -0,0 +1,202 @@
1
+ #!/usr/bin/env python3
2
+ """Agent readiness: can an autonomous agent verify its own work in THIS repo?
3
+
4
+ WHY THIS EXISTS, AND WHY IT IS NOT A COPY. Factory AI's Agent Readiness Model is
5
+ a genuinely good idea and a category-defining artifact -- 5 levels, 9 pillars,
6
+ 2 scopes -- and it makes competitor comparisons happen on Factory's chosen axes.
7
+ It is also LLM-SCORED: their report objects record `modelUsed` and
8
+ `reasoningEffort` per report. So the number is a model's opinion of a repo, and
9
+ two runs can disagree about the same commit.
10
+
11
+ Ours is a measurement. Every criterion below is a file that exists or does not,
12
+ a command that is present or absent. Same commit, same answer, every time, on
13
+ any machine, with no key and no spend. "Theirs is an opinion, ours is a
14
+ measurement, here is the command" is the same wedge as the receipt, applied to
15
+ their own differentiated concept.
16
+
17
+ WHAT IT MEASURES, AND WHY THOSE. Not general code quality -- that is what
18
+ `loki modernize heal --assess` already scores with its own deterministic 4-level
19
+ maturity rubric, and duplicating it would create two numbers that eventually
20
+ disagree. This asks the narrower question our product actually depends on:
21
+ CAN AN AGENT CHECK ITSELF HERE? Factory concedes the same dependency from the
22
+ other side -- their Missions docs say that without "an automated, scriptable way
23
+ to exercise the app... the mission cannot reliably verify its own work", and
24
+ recommend Level 4+ before using their flagship. A repo with no test command is
25
+ one where every agent, ours included, is guessing.
26
+
27
+ WHAT IT REFUSES. No percentage, no letter grade, no composite. A composite
28
+ invites ranking, ranking invites gaming, and the individual signals are the
29
+ actionable part: "there is no test command" tells you what to do, "readiness 62%"
30
+ does not. Criteria that cannot be determined report UNKNOWN by name rather than
31
+ counting as failures -- an absent measurement is not a bad score.
32
+ """
33
+
34
+ from __future__ import annotations
35
+
36
+ import json
37
+ import os
38
+ import sys
39
+
40
+ SCHEMA_VERSION = "1.0"
41
+
42
+ UNKNOWN = "UNKNOWN"
43
+
44
+ # Each criterion is a pure filesystem fact plus the command a reader can run to
45
+ # check it themselves. The `why` is not decoration: a signal whose consequence
46
+ # for an agent is unstated becomes a checkbox someone games.
47
+ CRITERIA = [
48
+ {
49
+ "id": "test_command",
50
+ "why": "without a runnable test command an agent cannot verify its own change",
51
+ "verify": "look for a test script in package.json, a Makefile test target, pytest.ini, or tests/",
52
+ },
53
+ {
54
+ "id": "dependency_lock",
55
+ "why": "unpinned dependencies make a green run unreproducible tomorrow",
56
+ "verify": "look for package-lock.json, bun.lockb, poetry.lock, requirements.txt, Cargo.lock, go.sum",
57
+ },
58
+ {
59
+ "id": "ci_config",
60
+ "why": "without CI, nothing re-checks the agent's work independently of the agent",
61
+ "verify": "look for .github/workflows, .gitlab-ci.yml, or a CI config at the repo root",
62
+ },
63
+ {
64
+ "id": "agent_brief",
65
+ "why": "without a briefing file an agent rediscovers conventions every run and gets them wrong",
66
+ "verify": "look for AGENTS.md, CLAUDE.md, CONTRIBUTING.md",
67
+ },
68
+ {
69
+ "id": "readme",
70
+ "why": "without a README an agent has no statement of what the project is for",
71
+ "verify": "look for README.md or README",
72
+ },
73
+ {
74
+ "id": "gitignore",
75
+ "why": "without ignores an agent's diff fills with build output and the real change is buried",
76
+ "verify": "look for .gitignore",
77
+ },
78
+ ]
79
+
80
+
81
+ def _any_exists(root, names):
82
+ for n in names:
83
+ if os.path.exists(os.path.join(root, n)):
84
+ return n
85
+ return None
86
+
87
+
88
+ def _has_test_command(root):
89
+ pkg = os.path.join(root, "package.json")
90
+ if os.path.isfile(pkg):
91
+ try:
92
+ with open(pkg, "r", encoding="utf-8") as fh:
93
+ data = json.load(fh)
94
+ if (data.get("scripts") or {}).get("test"):
95
+ return "package.json scripts.test"
96
+ except (OSError, ValueError):
97
+ # A malformed package.json is not evidence either way. Fall through
98
+ # to the other signals rather than scoring it as absent.
99
+ pass
100
+ found = _any_exists(root, ["pytest.ini", "tox.ini", "Makefile", "tests", "test"])
101
+ return found
102
+
103
+
104
+ def assess(root):
105
+ """Evaluate every criterion. Returns facts, never a score."""
106
+ if not os.path.isdir(root):
107
+ return {"status": UNKNOWN, "reason": "no_such_directory", "path": root}
108
+ if not os.path.isdir(os.path.join(root, ".git")):
109
+ # Not fatal: readiness is about the working tree. Recorded so a reader
110
+ # knows the repo context was absent rather than assumed.
111
+ git_present = False
112
+ else:
113
+ git_present = True
114
+
115
+ checks = []
116
+ for c in CRITERIA:
117
+ cid = c["id"]
118
+ if cid == "test_command":
119
+ hit = _has_test_command(root)
120
+ elif cid == "dependency_lock":
121
+ hit = _any_exists(root, ["package-lock.json", "bun.lockb", "yarn.lock",
122
+ "poetry.lock", "requirements.txt", "Cargo.lock",
123
+ "go.sum", "Pipfile.lock"])
124
+ elif cid == "ci_config":
125
+ hit = _any_exists(root, [".github/workflows", ".gitlab-ci.yml",
126
+ ".circleci", "azure-pipelines.yml", "Jenkinsfile"])
127
+ elif cid == "agent_brief":
128
+ hit = _any_exists(root, ["AGENTS.md", "CLAUDE.md", "CONTRIBUTING.md"])
129
+ elif cid == "readme":
130
+ hit = _any_exists(root, ["README.md", "README", "README.rst"])
131
+ elif cid == "gitignore":
132
+ hit = _any_exists(root, [".gitignore"])
133
+ else:
134
+ hit = None
135
+
136
+ checks.append({
137
+ "id": cid,
138
+ "present": bool(hit),
139
+ "found": hit or None,
140
+ "why": c["why"],
141
+ "verify": c["verify"],
142
+ })
143
+
144
+ present = [c for c in checks if c["present"]]
145
+ missing = [c for c in checks if not c["present"]]
146
+
147
+ return {
148
+ "schema_version": SCHEMA_VERSION,
149
+ "status": "measured",
150
+ "path": os.path.abspath(root),
151
+ "git_repo": git_present,
152
+ # Counts, not a percentage. A composite invites ranking, ranking invites
153
+ # gaming, and "there is no test command" is the actionable part anyway.
154
+ "criteria_total": len(checks),
155
+ "criteria_present": len(present),
156
+ "checks": checks,
157
+ "missing": [c["id"] for c in missing],
158
+ # The single most consequential signal, surfaced on its own: this is the
159
+ # one Factory's own docs concede their flagship depends on.
160
+ "can_self_verify": any(c["id"] == "test_command" and c["present"]
161
+ for c in checks),
162
+ }
163
+
164
+
165
+ def render_text(res):
166
+ if res.get("status") != "measured":
167
+ return f"Agent readiness: UNKNOWN ({res.get('reason', 'unmeasurable')})"
168
+ out = ["Agent readiness -- can an agent verify its own work here?", ""]
169
+ for c in res["checks"]:
170
+ mark = "yes" if c["present"] else "NO "
171
+ line = f" {mark} {c['id']:20}"
172
+ if c["present"]:
173
+ line += f"({c['found']})"
174
+ else:
175
+ line += c["why"]
176
+ out.append(line)
177
+ out.append("")
178
+ out.append(f" {res['criteria_present']} of {res['criteria_total']} present")
179
+ if not res["can_self_verify"]:
180
+ out.append("")
181
+ out.append(" No test command found. Every agent working here, ours")
182
+ out.append(" included, is guessing whether its change worked.")
183
+ out.append("")
184
+ out.append(" Every line above is a file that exists or does not. Check any of")
185
+ out.append(" them by hand; no model was asked for an opinion.")
186
+ return "\n".join(out)
187
+
188
+
189
+ def main(argv):
190
+ as_json = "--json" in argv
191
+ root = "."
192
+ for a in argv:
193
+ if not a.startswith("-"):
194
+ root = a
195
+ break
196
+ res = assess(root)
197
+ print(json.dumps(res, indent=2) if as_json else render_text(res))
198
+ return 0 if res.get("status") == "measured" else 3
199
+
200
+
201
+ if __name__ == "__main__":
202
+ sys.exit(main(sys.argv[1:]))