loki-mode 9.12.6 → 9.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,171 @@
1
+ #!/usr/bin/env python3
2
+ """Claim grounding: a completion claim must name work that exists in the diff.
3
+
4
+ WHY THIS EXISTS. 8090 AI's evaluation framework blocks ungrounded output AT
5
+ GENERATION rather than catching it at review, and states the payoff plainly:
6
+
7
+ "Ungrounded claims are blocked at generation. The rubric measures the quality
8
+ of what survives the architectural filter, which is a much smaller and more
9
+ interesting space."
10
+
11
+ Our evidence gate already has six axes -- diff non-empty, tests green, runtime
12
+ boot, no-mock, authorization, secret leak. Every one is a REPO-LEVEL fact. None
13
+ reads what the agent actually CLAIMED. So an agent can finish by asserting "added
14
+ retry logic to the payment client and covered it with tests" while the diff shows
15
+ a README edit, and every axis passes: the diff is non-empty, the tests are green,
16
+ the app boots. The claim itself is the one artifact nobody checks.
17
+
18
+ WHAT THIS CHECKS, AND WHAT IT REFUSES TO CHECK. It resolves file-path-shaped
19
+ tokens in the claim against the actual changed-file set. That is a deterministic
20
+ string-to-set comparison, and it is the only part of a natural-language claim
21
+ that can be checked without a model.
22
+
23
+ It does NOT judge whether the claim is semantically true. "Added retry logic" vs
24
+ "added a retry constant" is a judgement, and asking an LLM to grade it would be
25
+ the same LLM-judge-as-measurement this codebase refuses everywhere else. A claim
26
+ naming no paths is UNGROUNDABLE, reported by name -- never scored, never failed.
27
+
28
+ THE FAIL-OPEN DIRECTION IS DELIBERATE. Only a claim that names a path which is
29
+ demonstrably NOT in the diff is a finding. A claim with no paths, a claim naming
30
+ a path that exists but was not touched by this run, and an empty claim are all
31
+ reported and none of them blocks. A grounding check that blocked on ambiguity
32
+ would fire constantly on ordinary prose and be disabled within a week, which is
33
+ worse than not having it.
34
+ """
35
+
36
+ from __future__ import annotations
37
+
38
+ import json
39
+ import os
40
+ import re
41
+ import sys
42
+
43
+ SCHEMA_VERSION = "1.0"
44
+
45
+ UNKNOWN = "UNKNOWN"
46
+
47
+ GROUNDING_REASONS = {
48
+ "no_claim": "no completion claim text was provided",
49
+ "no_diff": "no changed-file set was provided to check against",
50
+ "ungroundable": "the claim names no file paths, so it cannot be checked mechanically",
51
+ }
52
+
53
+ # A path-shaped token: at least one slash or a known source extension, and no
54
+ # spaces. Deliberately conservative -- a false "this is a path" produces a false
55
+ # finding, and a check that cries wolf gets turned off.
56
+ _PATH_RE = re.compile(
57
+ r"(?<![\w/.-])"
58
+ r"(?:[\w.-]+/)+[\w.-]+\.[A-Za-z0-9]{1,6}"
59
+ r"|(?<![\w/.-])[\w.-]+\.(?:py|ts|tsx|js|jsx|sh|go|rs|rb|java|c|h|cpp|md|json|ya?ml|toml)"
60
+ r"(?![\w/.-])"
61
+ )
62
+
63
+
64
+ def extract_paths(claim):
65
+ """Path-shaped tokens in a claim, deduped, order preserved."""
66
+ if not claim:
67
+ return []
68
+ seen, out = set(), []
69
+ for m in _PATH_RE.finditer(claim):
70
+ tok = m.group(0).strip(".,;:)(\"'`")
71
+ if tok and tok not in seen:
72
+ seen.add(tok)
73
+ out.append(tok)
74
+ return out
75
+
76
+
77
+ def check(claim, changed_files):
78
+ """Are the paths a claim names present in the diff?
79
+
80
+ Matching is suffix-based on purpose: an agent writes `run.sh` or
81
+ `autonomy/run.sh` for the same file, and demanding an exact repo-relative
82
+ string would make every informal mention a false finding.
83
+ """
84
+ if not claim or not claim.strip():
85
+ return {"status": UNKNOWN, "reason": "no_claim",
86
+ "detail": GROUNDING_REASONS["no_claim"]}
87
+ if changed_files is None:
88
+ return {"status": UNKNOWN, "reason": "no_diff",
89
+ "detail": GROUNDING_REASONS["no_diff"]}
90
+
91
+ named = extract_paths(claim)
92
+ if not named:
93
+ # The common, benign case: "fixed the login bug". Nothing to check, and
94
+ # that is not a defect -- reported so the number of unverifiable claims
95
+ # is visible, rather than silently counted as grounded.
96
+ return {"status": UNKNOWN, "reason": "ungroundable",
97
+ "detail": GROUNDING_REASONS["ungroundable"],
98
+ "paths_named": []}
99
+
100
+ changed = list(changed_files)
101
+ grounded, ungrounded = [], []
102
+ for p in named:
103
+ norm = p.lstrip("./")
104
+ # Basename fallback is load-bearing, not laxity. An agent writes `run.sh`
105
+ # for `autonomy/run.sh`, and requiring the full repo-relative string made
106
+ # every informal mention a false finding -- measured: "patched run.sh"
107
+ # against a changed autonomy/run.sh was reported UNGROUNDED. A grounding
108
+ # check that flags correct claims is worse than none, because it gets
109
+ # disabled and takes the real detections with it.
110
+ base = os.path.basename(norm)
111
+ hit = any(
112
+ c == norm
113
+ or c.endswith("/" + norm)
114
+ or norm.endswith("/" + c)
115
+ or os.path.basename(c) == base
116
+ for c in changed
117
+ )
118
+ (grounded if hit else ungrounded).append(p)
119
+
120
+ return {
121
+ "status": "measured",
122
+ "paths_named": named,
123
+ "grounded": grounded,
124
+ "ungrounded": ungrounded,
125
+ # The single actionable signal. A claim that names a file the run never
126
+ # touched is the "agent says done, diff says otherwise" failure, caught
127
+ # from the claim side instead of the repo side.
128
+ "has_ungrounded_claim": bool(ungrounded),
129
+ }
130
+
131
+
132
+ def main(argv):
133
+ claim = ""
134
+ files_arg = ""
135
+ files_from = ""
136
+ i = 0
137
+ while i < len(argv):
138
+ if argv[i] == "--claim" and i + 1 < len(argv):
139
+ claim = argv[i + 1]; i += 2; continue
140
+ if argv[i] == "--files" and i + 1 < len(argv):
141
+ files_arg = argv[i + 1]; i += 2; continue
142
+ if argv[i] == "--files-from" and i + 1 < len(argv):
143
+ files_from = argv[i + 1]; i += 2; continue
144
+ i += 1
145
+
146
+ # --files is ALWAYS a comma-separated list; --files-from is always a file to
147
+ # read. The first version overloaded one flag for both and picked with
148
+ # os.path.isfile(), which silently misread `--files autonomy/run.sh` -- a
149
+ # perfectly ordinary changed-file list -- as "open that file and treat its
150
+ # 20,000 lines as filenames". The result was a correct claim reported
151
+ # UNGROUNDED, i.e. the exact false positive this check must never produce.
152
+ # An ambiguous flag whose meaning depends on the filesystem is a bug, not a
153
+ # convenience.
154
+ if files_from and os.path.isfile(files_from):
155
+ with open(files_from, "r", encoding="utf-8") as fh:
156
+ changed = [l.strip() for l in fh if l.strip()]
157
+ elif files_arg:
158
+ changed = [c.strip() for c in files_arg.split(",") if c.strip()]
159
+ else:
160
+ changed = None
161
+
162
+ res = check(claim, changed)
163
+ res["schema_version"] = SCHEMA_VERSION
164
+ print(json.dumps(res, indent=2))
165
+ # Exit 1 ONLY on a demonstrably ungrounded path. Every UNKNOWN exits 0: a
166
+ # check that failed on ambiguity would fire on ordinary prose and be disabled.
167
+ return 1 if res.get("has_ungrounded_claim") else 0
168
+
169
+
170
+ if __name__ == "__main__":
171
+ sys.exit(main(sys.argv[1:]))
@@ -863,13 +863,17 @@ loki_config_validate_file() {
863
863
  pairs="$(
864
864
  _loki_cfg_collect_pairs() {
865
865
  local f="$1" fm="$2"
866
+ # NOTE: every case pattern below carries a leading open-paren --
867
+ # bash 3.2 (macOS /bin/bash) ends a command substitution at the
868
+ # close-paren of a case label, truncating this function body. The
869
+ # balanced form parses identically on 3.2 and 4+. Do not strip it.
866
870
  case "$fm" in
867
- env)
871
+ (env)
868
872
  local line key val
869
873
  while IFS= read -r line || [ -n "$line" ]; do
870
- case "$line" in ''|'#'*) continue ;; esac
874
+ case "$line" in (''|'#'*) continue ;; esac
871
875
  line="${line#export }"
872
- case "$line" in *=*) ;; *) continue ;; esac
876
+ case "$line" in (*=*) ;; (*) continue ;; esac
873
877
  key="${line%%=*}"; val="${line#*=}"
874
878
  key="${key#"${key%%[![:space:]]*}"}"; key="${key%"${key##*[![:space:]]}"}"
875
879
  val="${val#"${val%%[![:space:]]*}"}"
@@ -878,7 +882,7 @@ loki_config_validate_file() {
878
882
  printf '%s\t%s\n' "$key" "$val"
879
883
  done < "$f"
880
884
  ;;
881
- yaml)
885
+ (yaml)
882
886
  local mapping yaml_path env_var value have_yq=0
883
887
  if command -v yq >/dev/null 2>&1; then have_yq=1; fi
884
888
  for mapping in "${LOKI_CONFIG_MAP[@]}"; do
@@ -894,8 +898,8 @@ loki_config_validate_file() {
894
898
  printf '%s\t%s\n' "$env_var" "$value"
895
899
  done
896
900
  ;;
897
- json)
898
- # Reuse the json parser's emit path by calling a print-only variant.
901
+ (json)
902
+ # Reuse the JSON parser emit path via a print-only variant.
899
903
  _loki_cfg_json_emit "$f"
900
904
  ;;
901
905
  esac
@@ -0,0 +1,198 @@
1
+ #!/usr/bin/env python3
2
+ """LLM Decision Record: which model, at what temperature, decided what.
3
+
4
+ WHY THIS EXISTS. Factory AI's audit log has eight event types and NOT ONE
5
+ records an agent action -- they are all admin configuration (membership, API
6
+ keys, integrations, managed settings). Agent forensics exists only as
7
+ customer-built OTEL: metrics by default, message content opt-in, and the customer
8
+ must stand up and retain the pipeline. So "which agent changed this line, on
9
+ whose authority, and what did it verify" is answerable only if the buyer built
10
+ the plumbing themselves. Devin's audit log is likewise session and admin scoped.
11
+
12
+ 8090's evaluation framework records, per AI operation: session id, pipeline
13
+ stage, model id, temperature, timestamps, token counts, reasoning trace,
14
+ confidence. Their stated reason is the one that matters to a regulated buyer:
15
+ capturing model id and temperature makes a MODEL SWAP DETECTABLE. Without it,
16
+ a provider silently changing a model underneath you is invisible in your own
17
+ records, and "we used an approved configuration" becomes unfalsifiable.
18
+
19
+ WHAT THIS IS. An append-only record of agent decisions, written next to the
20
+ receipt so the chain of custody is complete: the receipt says what was proven,
21
+ the outcome ledger says what happened afterwards, and this says what made the
22
+ call. Append-only because an audit trail a later run can rewrite is not an audit
23
+ trail.
24
+
25
+ WHAT THIS DELIBERATELY DOES NOT CAPTURE. No prompt bodies, no file contents, no
26
+ credentials, no environment. 8090's own boundary applies with more force to us
27
+ than to them: an adoption tool that exfiltrates a user's environment would cost
28
+ exactly the trust this product sells. Fields are a fixed allowlist, and a test
29
+ asserts the allowlist so a future contributor cannot widen it casually.
30
+
31
+ IT IS NOT TELEMETRY. Nothing here is transmitted. It writes a local JSONL file
32
+ that the user owns and can read, diff, and delete. autonomy/telemetry.sh remains
33
+ the single egress point.
34
+ """
35
+
36
+ from __future__ import annotations
37
+
38
+ import json
39
+ import os
40
+ import sys
41
+ from datetime import datetime, timezone
42
+
43
+ SCHEMA_VERSION = "1.0"
44
+
45
+ # The complete set of fields a record may carry. Anything not on this list is
46
+ # dropped rather than written: a record that can grow a new field by accident is
47
+ # how an audit log becomes a data-exfiltration surface. The test suite asserts
48
+ # this exact set, so widening it is a deliberate, reviewed act.
49
+ ALLOWED_FIELDS = (
50
+ "schema_version",
51
+ "recorded_at",
52
+ "session_id",
53
+ "run_id",
54
+ "stage", # which part of the loop made the call
55
+ "model_id", # THE field that makes a silent model swap detectable
56
+ "temperature", # same: a config drift nobody announced
57
+ "provider",
58
+ "tokens_in",
59
+ "tokens_out",
60
+ "duration_ms",
61
+ "outcome", # coarse: ok | error | refused | timeout
62
+ "confidence", # self-reported, and labelled as such below
63
+ )
64
+
65
+ # Fields whose value is the AGENT'S OWN OPINION, never a measurement. Kept
66
+ # separate so a reader cannot mistake a self-report for an observation -- the
67
+ # same FACTS-vs-ASSESSMENTS split the receipts already enforce.
68
+ SELF_REPORTED = ("confidence",)
69
+
70
+
71
+ def _records_path(loki_dir):
72
+ return os.path.join(loki_dir, "decisions", "decisions.jsonl")
73
+
74
+
75
+ def record(loki_dir, fields):
76
+ """Append one decision record. Returns the written record, or a reason.
77
+
78
+ Append-only by construction: opened "a", never "w". A trail a later run can
79
+ rewrite is not a trail, and the whole value here is that a reader can trust
80
+ what it says about a run that already finished.
81
+ """
82
+ clean = {}
83
+ dropped = []
84
+ for k, v in (fields or {}).items():
85
+ if k in ALLOWED_FIELDS:
86
+ clean[k] = v
87
+ else:
88
+ dropped.append(k)
89
+
90
+ if not clean.get("model_id"):
91
+ # A record without a model id cannot answer the question this exists to
92
+ # answer, so it is refused rather than written as a half-record.
93
+ return {"status": "UNKNOWN", "reason": "no_model_id",
94
+ "detail": "a decision record without model_id cannot make a swap detectable"}
95
+
96
+ clean["schema_version"] = SCHEMA_VERSION
97
+ clean["recorded_at"] = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
98
+
99
+ path = _records_path(loki_dir)
100
+ try:
101
+ os.makedirs(os.path.dirname(path), exist_ok=True)
102
+ with open(path, "a", encoding="utf-8") as fh:
103
+ fh.write(json.dumps(clean, sort_keys=True) + "\n")
104
+ except OSError as exc:
105
+ return {"status": "UNKNOWN", "reason": "unwritable", "detail": str(exc)}
106
+
107
+ out = {"status": "recorded", "path": path, "record": clean}
108
+ if dropped:
109
+ # Surfaced, not silent: a caller trying to log a prompt body should see
110
+ # that it was refused rather than assume it was stored.
111
+ out["dropped_fields"] = sorted(dropped)
112
+ return out
113
+
114
+
115
+ def summarize(loki_dir):
116
+ """What models actually ran, and did the configuration change mid-flight?
117
+
118
+ The useful audit question is not "how many calls" but "did the thing that
119
+ decided change without anyone saying so". A run whose records name two model
120
+ ids is exactly the case a regulated buyer needs surfaced.
121
+ """
122
+ path = _records_path(loki_dir)
123
+ if not os.path.isfile(path):
124
+ return {"status": "UNKNOWN", "reason": "no_records",
125
+ "detail": "no decision records for this project"}
126
+
127
+ models, temps, stages, n, bad = {}, {}, {}, 0, 0
128
+ try:
129
+ with open(path, "r", encoding="utf-8") as fh:
130
+ for line in fh:
131
+ line = line.strip()
132
+ if not line:
133
+ continue
134
+ try:
135
+ rec = json.loads(line)
136
+ except ValueError:
137
+ # A corrupt line is counted, never silently skipped: an audit
138
+ # trail that quietly drops what it cannot parse is worse than
139
+ # one that admits a gap.
140
+ bad += 1
141
+ continue
142
+ n += 1
143
+ m = rec.get("model_id")
144
+ if m:
145
+ models[m] = models.get(m, 0) + 1
146
+ t = rec.get("temperature")
147
+ if t is not None:
148
+ temps[str(t)] = temps.get(str(t), 0) + 1
149
+ s = rec.get("stage")
150
+ if s:
151
+ stages[s] = stages.get(s, 0) + 1
152
+ except OSError as exc:
153
+ return {"status": "UNKNOWN", "reason": "unreadable", "detail": str(exc)}
154
+
155
+ return {
156
+ "status": "measured",
157
+ "records": n,
158
+ "unparseable_lines": bad,
159
+ "models": models,
160
+ "temperatures": temps,
161
+ "stages": stages,
162
+ # The headline fact. More than one model id, or more than one
163
+ # temperature, means the configuration changed during this project and
164
+ # the records prove it.
165
+ "model_changed": len(models) > 1,
166
+ "temperature_changed": len(temps) > 1,
167
+ "self_reported_fields": list(SELF_REPORTED),
168
+ }
169
+
170
+
171
+ def main(argv):
172
+ if not argv:
173
+ print("usage: decision_record.py record|summary [--field=value ...]",
174
+ file=sys.stderr)
175
+ return 2
176
+ action = argv[0]
177
+ cwd = os.environ.get("LOKI_DECISION_CWD") or os.getcwd()
178
+ loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
179
+
180
+ if action == "record":
181
+ fields = {}
182
+ for arg in argv[1:]:
183
+ if arg.startswith("--") and "=" in arg:
184
+ k, v = arg[2:].split("=", 1)
185
+ fields[k] = v
186
+ res = record(loki_dir, fields)
187
+ elif action == "summary":
188
+ res = summarize(loki_dir)
189
+ else:
190
+ print(f"unknown action: {action}", file=sys.stderr)
191
+ return 2
192
+
193
+ print(json.dumps(res, indent=2))
194
+ return 0 if res.get("status") in ("recorded", "measured") else 3
195
+
196
+
197
+ if __name__ == "__main__":
198
+ sys.exit(main(sys.argv[1:]))
@@ -0,0 +1,199 @@
1
+ #!/usr/bin/env python3
2
+ """Failure memory: the agent gets better at YOUR repo by having been wrong in it.
3
+
4
+ WHY THIS EXISTS. Neither competitor learns from being wrong, and one of them
5
+ shipped a UI to manage that fact.
6
+
7
+ Factory AI has no memory system at all -- verified by grep over their whole doc
8
+ corpus. AGENTS.md is hand-authored by the human, AutoWiki regenerates from CODE
9
+ rather than from outcomes, QA failure learning is `suggest_in_report` (a human
10
+ reads it and decides), and mission-generated skills live under {missionDir} and
11
+ are DISCARDED with the mission. Devin has a real memory layer, but every session
12
+ starts from a fresh snapshot copy and all changes are discarded at teardown --
13
+ and their Session Insights ships a "Misleading Knowledge" tab enumerating memory
14
+ items that led Devin astray. They built an interface for their memory poisoning
15
+ output.
16
+
17
+ So on both systems, an agent that failed a gate in your repo yesterday starts
18
+ today knowing nothing about it.
19
+
20
+ WHAT MAKES THIS DIFFERENT, AND WHY IT IS NOT THE SAME TRAP. The reason Devin
21
+ needed a "Misleading Knowledge" tab is that memory written from an agent's own
22
+ narration is unfalsifiable: it records what the agent BELIEVED, which is exactly
23
+ what was wrong when it failed. So nothing here is written from narration. A
24
+ lesson is created only from a MEASURED event -- a named gate that failed, with
25
+ its verdict -- and it carries the evidence that produced it. A lesson whose
26
+ evidence no longer holds can be retired mechanically instead of accumulating.
27
+
28
+ This is deliberately built on the outcome ledger's discipline: a lesson is
29
+ recorded only when the event that justifies it is a fact, and everything else
30
+ reports UNKNOWN with a named reason rather than being written as a weak guess.
31
+
32
+ WHAT IT DOES NOT DO. It does not summarize, generalize, or ask a model what the
33
+ lesson "means". A generalization is a judgement, and a judgement stored as memory
34
+ is indistinguishable from a fact by the next reader -- which is the poisoning
35
+ mechanism. It stores the gate, the verdict, the evidence, and a count. Deciding
36
+ what that implies stays with whoever reads it.
37
+ """
38
+
39
+ from __future__ import annotations
40
+
41
+ import json
42
+ import os
43
+ import sys
44
+ from datetime import datetime, timezone
45
+
46
+ SCHEMA_VERSION = "1.0"
47
+
48
+ UNKNOWN = "UNKNOWN"
49
+
50
+ REASONS = {
51
+ "no_gate": "no gate name was given, so there is nothing to attribute the failure to",
52
+ "no_evidence": "no evidence was given, so the lesson would be unfalsifiable",
53
+ "no_lessons": "no failure lessons recorded for this project",
54
+ "unreadable": "the lesson store exists but could not be parsed",
55
+ }
56
+
57
+
58
+ def _path(loki_dir):
59
+ return os.path.join(loki_dir, "memory", "failures.jsonl")
60
+
61
+
62
+ def record_failure(loki_dir, gate, verdict, evidence, run_id=None):
63
+ """Turn a measured gate failure into a durable, falsifiable lesson.
64
+
65
+ Requires EVIDENCE. A lesson without it is the agent's own account of why it
66
+ failed, which is precisely the unfalsifiable memory that made Devin's need a
67
+ "Misleading Knowledge" tab. If we cannot say what was observed, we do not
68
+ write anything.
69
+ """
70
+ if not gate:
71
+ return {"status": UNKNOWN, "reason": "no_gate", "detail": REASONS["no_gate"]}
72
+ if not evidence:
73
+ return {"status": UNKNOWN, "reason": "no_evidence",
74
+ "detail": REASONS["no_evidence"]}
75
+
76
+ rec = {
77
+ "schema_version": SCHEMA_VERSION,
78
+ "recorded_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
79
+ "gate": gate,
80
+ "verdict": verdict or "FAIL",
81
+ # The falsifiability anchor. A future reader can check whether this still
82
+ # holds instead of taking the lesson on faith.
83
+ "evidence": evidence,
84
+ "run_id": run_id or "",
85
+ # Deliberately absent: any summary, generalization, or "what this means".
86
+ # A judgement stored beside facts becomes indistinguishable from one.
87
+ }
88
+
89
+ p = _path(loki_dir)
90
+ try:
91
+ os.makedirs(os.path.dirname(p), exist_ok=True)
92
+ with open(p, "a", encoding="utf-8") as fh:
93
+ fh.write(json.dumps(rec, sort_keys=True) + "\n")
94
+ except OSError as exc:
95
+ return {"status": UNKNOWN, "reason": "unreadable", "detail": str(exc)}
96
+ return {"status": "recorded", "path": p, "record": rec}
97
+
98
+
99
+ def recall(loki_dir, gate=None):
100
+ """What has actually gone wrong here before.
101
+
102
+ Returns counts per gate, most-failed first. Counts, not prose: "the mock
103
+ integrity gate has failed here 6 times" is a fact a reader can act on, while
104
+ "this repo tends to have mocking problems" is a generalization that sounds
105
+ the same and is not checkable.
106
+ """
107
+ p = _path(loki_dir)
108
+ if not os.path.isfile(p):
109
+ return {"status": UNKNOWN, "reason": "no_lessons",
110
+ "detail": REASONS["no_lessons"]}
111
+
112
+ per_gate, recent, bad = {}, [], 0
113
+ try:
114
+ with open(p, "r", encoding="utf-8") as fh:
115
+ for line in fh:
116
+ line = line.strip()
117
+ if not line:
118
+ continue
119
+ try:
120
+ r = json.loads(line)
121
+ except ValueError:
122
+ # Counted, never silently skipped: a memory that quietly
123
+ # drops what it cannot read is worse than one that admits it.
124
+ bad += 1
125
+ continue
126
+ g = r.get("gate")
127
+ if not g or (gate and g != gate):
128
+ continue
129
+ per_gate[g] = per_gate.get(g, 0) + 1
130
+ recent.append(r)
131
+ except OSError as exc:
132
+ return {"status": UNKNOWN, "reason": "unreadable", "detail": str(exc)}
133
+
134
+ if not per_gate:
135
+ return {"status": UNKNOWN, "reason": "no_lessons",
136
+ "detail": REASONS["no_lessons"]}
137
+
138
+ return {
139
+ "status": "measured",
140
+ "unparseable_lines": bad,
141
+ "by_gate": dict(sorted(per_gate.items(), key=lambda kv: -kv[1])),
142
+ "total": sum(per_gate.values()),
143
+ # Newest last so a reader sees the current state at the bottom, matching
144
+ # how the file itself is ordered.
145
+ "recent": recent[-5:],
146
+ }
147
+
148
+
149
+ def prompt_context(loki_dir, limit=3):
150
+ """The lines worth putting in front of the next run.
151
+
152
+ Returns bare facts. A repo where the same gate has failed repeatedly is
153
+ information the next iteration should have; what to DO about it is left to
154
+ the agent, because prescribing the fix from a count would be inventing a
155
+ causal claim the data does not contain.
156
+ """
157
+ r = recall(loki_dir)
158
+ if r.get("status") != "measured":
159
+ return {"status": r.get("status"), "reason": r.get("reason"), "lines": []}
160
+ lines = []
161
+ for gate, n in list(r["by_gate"].items())[:limit]:
162
+ if n > 1:
163
+ lines.append(f"the {gate} gate has failed {n} times in this repo before")
164
+ else:
165
+ lines.append(f"the {gate} gate has failed here before")
166
+ return {"status": "measured", "lines": lines}
167
+
168
+
169
+ def main(argv):
170
+ if not argv:
171
+ print("usage: failure_memory.py record|recall|context [...]", file=sys.stderr)
172
+ return 2
173
+ action = argv[0]
174
+ cwd = os.environ.get("LOKI_FAILMEM_CWD") or os.getcwd()
175
+ loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
176
+
177
+ kv = {}
178
+ for a in argv[1:]:
179
+ if a.startswith("--") and "=" in a:
180
+ k, v = a[2:].split("=", 1)
181
+ kv[k] = v
182
+
183
+ if action == "record":
184
+ res = record_failure(loki_dir, kv.get("gate"), kv.get("verdict"),
185
+ kv.get("evidence"), kv.get("run_id"))
186
+ elif action == "recall":
187
+ res = recall(loki_dir, kv.get("gate"))
188
+ elif action == "context":
189
+ res = prompt_context(loki_dir)
190
+ else:
191
+ print(f"unknown action: {action}", file=sys.stderr)
192
+ return 2
193
+
194
+ print(json.dumps(res, indent=2))
195
+ return 0 if res.get("status") in ("recorded", "measured") else 3
196
+
197
+
198
+ if __name__ == "__main__":
199
+ sys.exit(main(sys.argv[1:]))