loki-mode 9.12.6 → 9.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +81 -101
- package/SKILL.md +2 -2
- package/VERSION +1 -1
- package/autonomy/intent.sh +414 -0
- package/autonomy/issue-providers.sh +24 -0
- package/autonomy/lib/agent_readiness.py +280 -0
- package/autonomy/lib/claim_grounding.py +171 -0
- package/autonomy/lib/config-map.sh +10 -6
- package/autonomy/lib/decision_record.py +198 -0
- package/autonomy/lib/failure_memory.py +199 -0
- package/autonomy/lib/gate_policy.py +166 -0
- package/autonomy/lib/outcome_ledger.py +620 -0
- package/autonomy/lib/preedit_snapshot.py +216 -0
- package/autonomy/lib/proof-generator.py +71 -4
- package/autonomy/lib/verdict.py +204 -0
- package/autonomy/loki +430 -15
- package/autonomy/notify.sh +70 -1
- package/autonomy/provider-offer.sh +25 -1
- package/autonomy/queue-consumer.sh +290 -18
- package/autonomy/run.sh +527 -12
- package/autonomy/telemetry.sh +8 -1
- package/completions/_loki +5 -0
- package/completions/loki.bash +2 -1
- package/dashboard/__init__.py +1 -1
- package/dashboard/run.py +13 -2
- package/dashboard/scim.py +221 -0
- package/dashboard/server.py +190 -1
- package/dashboard/static/index.html +248 -55
- package/docs/GATE-FAILURE-TRIAGE.md +254 -0
- package/docs/LOOP-CANDIDATE-PROPOSAL-v1.md +167 -0
- package/docs/LOOP-HARNESS-AUDIT.md +53 -0
- package/docs/QUEUE-OPERATIONS.md +107 -0
- package/docs/VERIFICATION-COST.md +273 -0
- package/docs/WANG-PRINCIPLES-PLAN.md +1 -1
- package/loki-ts/dist/loki.js +414 -416
- package/mcp/__init__.py +1 -1
- package/mcp/_sdk_loader.py +25 -0
- package/package.json +1 -1
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""LLM Decision Record: which model, at what temperature, decided what.
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. Factory AI's audit log has eight event types and NOT ONE
|
|
5
|
+
records an agent action -- they are all admin configuration (membership, API
|
|
6
|
+
keys, integrations, managed settings). Agent forensics exists only as
|
|
7
|
+
customer-built OTEL: metrics by default, message content opt-in, and the customer
|
|
8
|
+
must stand up and retain the pipeline. So "which agent changed this line, on
|
|
9
|
+
whose authority, and what did it verify" is answerable only if the buyer built
|
|
10
|
+
the plumbing themselves. Devin's audit log is likewise session and admin scoped.
|
|
11
|
+
|
|
12
|
+
8090's evaluation framework records, per AI operation: session id, pipeline
|
|
13
|
+
stage, model id, temperature, timestamps, token counts, reasoning trace,
|
|
14
|
+
confidence. Their stated reason is the one that matters to a regulated buyer:
|
|
15
|
+
capturing model id and temperature makes a MODEL SWAP DETECTABLE. Without it,
|
|
16
|
+
a provider silently changing a model underneath you is invisible in your own
|
|
17
|
+
records, and "we used an approved configuration" becomes unfalsifiable.
|
|
18
|
+
|
|
19
|
+
WHAT THIS IS. An append-only record of agent decisions, written next to the
|
|
20
|
+
receipt so the chain of custody is complete: the receipt says what was proven,
|
|
21
|
+
the outcome ledger says what happened afterwards, and this says what made the
|
|
22
|
+
call. Append-only because an audit trail a later run can rewrite is not an audit
|
|
23
|
+
trail.
|
|
24
|
+
|
|
25
|
+
WHAT THIS DELIBERATELY DOES NOT CAPTURE. No prompt bodies, no file contents, no
|
|
26
|
+
credentials, no environment. 8090's own boundary applies with more force to us
|
|
27
|
+
than to them: an adoption tool that exfiltrates a user's environment would cost
|
|
28
|
+
exactly the trust this product sells. Fields are a fixed allowlist, and a test
|
|
29
|
+
asserts the allowlist so a future contributor cannot widen it casually.
|
|
30
|
+
|
|
31
|
+
IT IS NOT TELEMETRY. Nothing here is transmitted. It writes a local JSONL file
|
|
32
|
+
that the user owns and can read, diff, and delete. autonomy/telemetry.sh remains
|
|
33
|
+
the single egress point.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
import json
|
|
39
|
+
import os
|
|
40
|
+
import sys
|
|
41
|
+
from datetime import datetime, timezone
|
|
42
|
+
|
|
43
|
+
SCHEMA_VERSION = "1.0"
|
|
44
|
+
|
|
45
|
+
# The complete set of fields a record may carry. Anything not on this list is
|
|
46
|
+
# dropped rather than written: a record that can grow a new field by accident is
|
|
47
|
+
# how an audit log becomes a data-exfiltration surface. The test suite asserts
|
|
48
|
+
# this exact set, so widening it is a deliberate, reviewed act.
|
|
49
|
+
ALLOWED_FIELDS = (
|
|
50
|
+
"schema_version",
|
|
51
|
+
"recorded_at",
|
|
52
|
+
"session_id",
|
|
53
|
+
"run_id",
|
|
54
|
+
"stage", # which part of the loop made the call
|
|
55
|
+
"model_id", # THE field that makes a silent model swap detectable
|
|
56
|
+
"temperature", # same: a config drift nobody announced
|
|
57
|
+
"provider",
|
|
58
|
+
"tokens_in",
|
|
59
|
+
"tokens_out",
|
|
60
|
+
"duration_ms",
|
|
61
|
+
"outcome", # coarse: ok | error | refused | timeout
|
|
62
|
+
"confidence", # self-reported, and labelled as such below
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
# Fields whose value is the AGENT'S OWN OPINION, never a measurement. Kept
|
|
66
|
+
# separate so a reader cannot mistake a self-report for an observation -- the
|
|
67
|
+
# same FACTS-vs-ASSESSMENTS split the receipts already enforce.
|
|
68
|
+
SELF_REPORTED = ("confidence",)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _records_path(loki_dir):
|
|
72
|
+
return os.path.join(loki_dir, "decisions", "decisions.jsonl")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def record(loki_dir, fields):
|
|
76
|
+
"""Append one decision record. Returns the written record, or a reason.
|
|
77
|
+
|
|
78
|
+
Append-only by construction: opened "a", never "w". A trail a later run can
|
|
79
|
+
rewrite is not a trail, and the whole value here is that a reader can trust
|
|
80
|
+
what it says about a run that already finished.
|
|
81
|
+
"""
|
|
82
|
+
clean = {}
|
|
83
|
+
dropped = []
|
|
84
|
+
for k, v in (fields or {}).items():
|
|
85
|
+
if k in ALLOWED_FIELDS:
|
|
86
|
+
clean[k] = v
|
|
87
|
+
else:
|
|
88
|
+
dropped.append(k)
|
|
89
|
+
|
|
90
|
+
if not clean.get("model_id"):
|
|
91
|
+
# A record without a model id cannot answer the question this exists to
|
|
92
|
+
# answer, so it is refused rather than written as a half-record.
|
|
93
|
+
return {"status": "UNKNOWN", "reason": "no_model_id",
|
|
94
|
+
"detail": "a decision record without model_id cannot make a swap detectable"}
|
|
95
|
+
|
|
96
|
+
clean["schema_version"] = SCHEMA_VERSION
|
|
97
|
+
clean["recorded_at"] = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
98
|
+
|
|
99
|
+
path = _records_path(loki_dir)
|
|
100
|
+
try:
|
|
101
|
+
os.makedirs(os.path.dirname(path), exist_ok=True)
|
|
102
|
+
with open(path, "a", encoding="utf-8") as fh:
|
|
103
|
+
fh.write(json.dumps(clean, sort_keys=True) + "\n")
|
|
104
|
+
except OSError as exc:
|
|
105
|
+
return {"status": "UNKNOWN", "reason": "unwritable", "detail": str(exc)}
|
|
106
|
+
|
|
107
|
+
out = {"status": "recorded", "path": path, "record": clean}
|
|
108
|
+
if dropped:
|
|
109
|
+
# Surfaced, not silent: a caller trying to log a prompt body should see
|
|
110
|
+
# that it was refused rather than assume it was stored.
|
|
111
|
+
out["dropped_fields"] = sorted(dropped)
|
|
112
|
+
return out
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def summarize(loki_dir):
|
|
116
|
+
"""What models actually ran, and did the configuration change mid-flight?
|
|
117
|
+
|
|
118
|
+
The useful audit question is not "how many calls" but "did the thing that
|
|
119
|
+
decided change without anyone saying so". A run whose records name two model
|
|
120
|
+
ids is exactly the case a regulated buyer needs surfaced.
|
|
121
|
+
"""
|
|
122
|
+
path = _records_path(loki_dir)
|
|
123
|
+
if not os.path.isfile(path):
|
|
124
|
+
return {"status": "UNKNOWN", "reason": "no_records",
|
|
125
|
+
"detail": "no decision records for this project"}
|
|
126
|
+
|
|
127
|
+
models, temps, stages, n, bad = {}, {}, {}, 0, 0
|
|
128
|
+
try:
|
|
129
|
+
with open(path, "r", encoding="utf-8") as fh:
|
|
130
|
+
for line in fh:
|
|
131
|
+
line = line.strip()
|
|
132
|
+
if not line:
|
|
133
|
+
continue
|
|
134
|
+
try:
|
|
135
|
+
rec = json.loads(line)
|
|
136
|
+
except ValueError:
|
|
137
|
+
# A corrupt line is counted, never silently skipped: an audit
|
|
138
|
+
# trail that quietly drops what it cannot parse is worse than
|
|
139
|
+
# one that admits a gap.
|
|
140
|
+
bad += 1
|
|
141
|
+
continue
|
|
142
|
+
n += 1
|
|
143
|
+
m = rec.get("model_id")
|
|
144
|
+
if m:
|
|
145
|
+
models[m] = models.get(m, 0) + 1
|
|
146
|
+
t = rec.get("temperature")
|
|
147
|
+
if t is not None:
|
|
148
|
+
temps[str(t)] = temps.get(str(t), 0) + 1
|
|
149
|
+
s = rec.get("stage")
|
|
150
|
+
if s:
|
|
151
|
+
stages[s] = stages.get(s, 0) + 1
|
|
152
|
+
except OSError as exc:
|
|
153
|
+
return {"status": "UNKNOWN", "reason": "unreadable", "detail": str(exc)}
|
|
154
|
+
|
|
155
|
+
return {
|
|
156
|
+
"status": "measured",
|
|
157
|
+
"records": n,
|
|
158
|
+
"unparseable_lines": bad,
|
|
159
|
+
"models": models,
|
|
160
|
+
"temperatures": temps,
|
|
161
|
+
"stages": stages,
|
|
162
|
+
# The headline fact. More than one model id, or more than one
|
|
163
|
+
# temperature, means the configuration changed during this project and
|
|
164
|
+
# the records prove it.
|
|
165
|
+
"model_changed": len(models) > 1,
|
|
166
|
+
"temperature_changed": len(temps) > 1,
|
|
167
|
+
"self_reported_fields": list(SELF_REPORTED),
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def main(argv):
|
|
172
|
+
if not argv:
|
|
173
|
+
print("usage: decision_record.py record|summary [--field=value ...]",
|
|
174
|
+
file=sys.stderr)
|
|
175
|
+
return 2
|
|
176
|
+
action = argv[0]
|
|
177
|
+
cwd = os.environ.get("LOKI_DECISION_CWD") or os.getcwd()
|
|
178
|
+
loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
|
|
179
|
+
|
|
180
|
+
if action == "record":
|
|
181
|
+
fields = {}
|
|
182
|
+
for arg in argv[1:]:
|
|
183
|
+
if arg.startswith("--") and "=" in arg:
|
|
184
|
+
k, v = arg[2:].split("=", 1)
|
|
185
|
+
fields[k] = v
|
|
186
|
+
res = record(loki_dir, fields)
|
|
187
|
+
elif action == "summary":
|
|
188
|
+
res = summarize(loki_dir)
|
|
189
|
+
else:
|
|
190
|
+
print(f"unknown action: {action}", file=sys.stderr)
|
|
191
|
+
return 2
|
|
192
|
+
|
|
193
|
+
print(json.dumps(res, indent=2))
|
|
194
|
+
return 0 if res.get("status") in ("recorded", "measured") else 3
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
if __name__ == "__main__":
|
|
198
|
+
sys.exit(main(sys.argv[1:]))
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Failure memory: the agent gets better at YOUR repo by having been wrong in it.
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. Neither competitor learns from being wrong, and one of them
|
|
5
|
+
shipped a UI to manage that fact.
|
|
6
|
+
|
|
7
|
+
Factory AI has no memory system at all -- verified by grep over their whole doc
|
|
8
|
+
corpus. AGENTS.md is hand-authored by the human, AutoWiki regenerates from CODE
|
|
9
|
+
rather than from outcomes, QA failure learning is `suggest_in_report` (a human
|
|
10
|
+
reads it and decides), and mission-generated skills live under {missionDir} and
|
|
11
|
+
are DISCARDED with the mission. Devin has a real memory layer, but every session
|
|
12
|
+
starts from a fresh snapshot copy and all changes are discarded at teardown --
|
|
13
|
+
and their Session Insights ships a "Misleading Knowledge" tab enumerating memory
|
|
14
|
+
items that led Devin astray. They built an interface for their memory poisoning
|
|
15
|
+
output.
|
|
16
|
+
|
|
17
|
+
So on both systems, an agent that failed a gate in your repo yesterday starts
|
|
18
|
+
today knowing nothing about it.
|
|
19
|
+
|
|
20
|
+
WHAT MAKES THIS DIFFERENT, AND WHY IT IS NOT THE SAME TRAP. The reason Devin
|
|
21
|
+
needed a "Misleading Knowledge" tab is that memory written from an agent's own
|
|
22
|
+
narration is unfalsifiable: it records what the agent BELIEVED, which is exactly
|
|
23
|
+
what was wrong when it failed. So nothing here is written from narration. A
|
|
24
|
+
lesson is created only from a MEASURED event -- a named gate that failed, with
|
|
25
|
+
its verdict -- and it carries the evidence that produced it. A lesson whose
|
|
26
|
+
evidence no longer holds can be retired mechanically instead of accumulating.
|
|
27
|
+
|
|
28
|
+
This is deliberately built on the outcome ledger's discipline: a lesson is
|
|
29
|
+
recorded only when the event that justifies it is a fact, and everything else
|
|
30
|
+
reports UNKNOWN with a named reason rather than being written as a weak guess.
|
|
31
|
+
|
|
32
|
+
WHAT IT DOES NOT DO. It does not summarize, generalize, or ask a model what the
|
|
33
|
+
lesson "means". A generalization is a judgement, and a judgement stored as memory
|
|
34
|
+
is indistinguishable from a fact by the next reader -- which is the poisoning
|
|
35
|
+
mechanism. It stores the gate, the verdict, the evidence, and a count. Deciding
|
|
36
|
+
what that implies stays with whoever reads it.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
from __future__ import annotations
|
|
40
|
+
|
|
41
|
+
import json
|
|
42
|
+
import os
|
|
43
|
+
import sys
|
|
44
|
+
from datetime import datetime, timezone
|
|
45
|
+
|
|
46
|
+
SCHEMA_VERSION = "1.0"
|
|
47
|
+
|
|
48
|
+
UNKNOWN = "UNKNOWN"
|
|
49
|
+
|
|
50
|
+
REASONS = {
|
|
51
|
+
"no_gate": "no gate name was given, so there is nothing to attribute the failure to",
|
|
52
|
+
"no_evidence": "no evidence was given, so the lesson would be unfalsifiable",
|
|
53
|
+
"no_lessons": "no failure lessons recorded for this project",
|
|
54
|
+
"unreadable": "the lesson store exists but could not be parsed",
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _path(loki_dir):
|
|
59
|
+
return os.path.join(loki_dir, "memory", "failures.jsonl")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def record_failure(loki_dir, gate, verdict, evidence, run_id=None):
|
|
63
|
+
"""Turn a measured gate failure into a durable, falsifiable lesson.
|
|
64
|
+
|
|
65
|
+
Requires EVIDENCE. A lesson without it is the agent's own account of why it
|
|
66
|
+
failed, which is precisely the unfalsifiable memory that made Devin's need a
|
|
67
|
+
"Misleading Knowledge" tab. If we cannot say what was observed, we do not
|
|
68
|
+
write anything.
|
|
69
|
+
"""
|
|
70
|
+
if not gate:
|
|
71
|
+
return {"status": UNKNOWN, "reason": "no_gate", "detail": REASONS["no_gate"]}
|
|
72
|
+
if not evidence:
|
|
73
|
+
return {"status": UNKNOWN, "reason": "no_evidence",
|
|
74
|
+
"detail": REASONS["no_evidence"]}
|
|
75
|
+
|
|
76
|
+
rec = {
|
|
77
|
+
"schema_version": SCHEMA_VERSION,
|
|
78
|
+
"recorded_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
|
|
79
|
+
"gate": gate,
|
|
80
|
+
"verdict": verdict or "FAIL",
|
|
81
|
+
# The falsifiability anchor. A future reader can check whether this still
|
|
82
|
+
# holds instead of taking the lesson on faith.
|
|
83
|
+
"evidence": evidence,
|
|
84
|
+
"run_id": run_id or "",
|
|
85
|
+
# Deliberately absent: any summary, generalization, or "what this means".
|
|
86
|
+
# A judgement stored beside facts becomes indistinguishable from one.
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
p = _path(loki_dir)
|
|
90
|
+
try:
|
|
91
|
+
os.makedirs(os.path.dirname(p), exist_ok=True)
|
|
92
|
+
with open(p, "a", encoding="utf-8") as fh:
|
|
93
|
+
fh.write(json.dumps(rec, sort_keys=True) + "\n")
|
|
94
|
+
except OSError as exc:
|
|
95
|
+
return {"status": UNKNOWN, "reason": "unreadable", "detail": str(exc)}
|
|
96
|
+
return {"status": "recorded", "path": p, "record": rec}
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def recall(loki_dir, gate=None):
|
|
100
|
+
"""What has actually gone wrong here before.
|
|
101
|
+
|
|
102
|
+
Returns counts per gate, most-failed first. Counts, not prose: "the mock
|
|
103
|
+
integrity gate has failed here 6 times" is a fact a reader can act on, while
|
|
104
|
+
"this repo tends to have mocking problems" is a generalization that sounds
|
|
105
|
+
the same and is not checkable.
|
|
106
|
+
"""
|
|
107
|
+
p = _path(loki_dir)
|
|
108
|
+
if not os.path.isfile(p):
|
|
109
|
+
return {"status": UNKNOWN, "reason": "no_lessons",
|
|
110
|
+
"detail": REASONS["no_lessons"]}
|
|
111
|
+
|
|
112
|
+
per_gate, recent, bad = {}, [], 0
|
|
113
|
+
try:
|
|
114
|
+
with open(p, "r", encoding="utf-8") as fh:
|
|
115
|
+
for line in fh:
|
|
116
|
+
line = line.strip()
|
|
117
|
+
if not line:
|
|
118
|
+
continue
|
|
119
|
+
try:
|
|
120
|
+
r = json.loads(line)
|
|
121
|
+
except ValueError:
|
|
122
|
+
# Counted, never silently skipped: a memory that quietly
|
|
123
|
+
# drops what it cannot read is worse than one that admits it.
|
|
124
|
+
bad += 1
|
|
125
|
+
continue
|
|
126
|
+
g = r.get("gate")
|
|
127
|
+
if not g or (gate and g != gate):
|
|
128
|
+
continue
|
|
129
|
+
per_gate[g] = per_gate.get(g, 0) + 1
|
|
130
|
+
recent.append(r)
|
|
131
|
+
except OSError as exc:
|
|
132
|
+
return {"status": UNKNOWN, "reason": "unreadable", "detail": str(exc)}
|
|
133
|
+
|
|
134
|
+
if not per_gate:
|
|
135
|
+
return {"status": UNKNOWN, "reason": "no_lessons",
|
|
136
|
+
"detail": REASONS["no_lessons"]}
|
|
137
|
+
|
|
138
|
+
return {
|
|
139
|
+
"status": "measured",
|
|
140
|
+
"unparseable_lines": bad,
|
|
141
|
+
"by_gate": dict(sorted(per_gate.items(), key=lambda kv: -kv[1])),
|
|
142
|
+
"total": sum(per_gate.values()),
|
|
143
|
+
# Newest last so a reader sees the current state at the bottom, matching
|
|
144
|
+
# how the file itself is ordered.
|
|
145
|
+
"recent": recent[-5:],
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def prompt_context(loki_dir, limit=3):
|
|
150
|
+
"""The lines worth putting in front of the next run.
|
|
151
|
+
|
|
152
|
+
Returns bare facts. A repo where the same gate has failed repeatedly is
|
|
153
|
+
information the next iteration should have; what to DO about it is left to
|
|
154
|
+
the agent, because prescribing the fix from a count would be inventing a
|
|
155
|
+
causal claim the data does not contain.
|
|
156
|
+
"""
|
|
157
|
+
r = recall(loki_dir)
|
|
158
|
+
if r.get("status") != "measured":
|
|
159
|
+
return {"status": r.get("status"), "reason": r.get("reason"), "lines": []}
|
|
160
|
+
lines = []
|
|
161
|
+
for gate, n in list(r["by_gate"].items())[:limit]:
|
|
162
|
+
if n > 1:
|
|
163
|
+
lines.append(f"the {gate} gate has failed {n} times in this repo before")
|
|
164
|
+
else:
|
|
165
|
+
lines.append(f"the {gate} gate has failed here before")
|
|
166
|
+
return {"status": "measured", "lines": lines}
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def main(argv):
|
|
170
|
+
if not argv:
|
|
171
|
+
print("usage: failure_memory.py record|recall|context [...]", file=sys.stderr)
|
|
172
|
+
return 2
|
|
173
|
+
action = argv[0]
|
|
174
|
+
cwd = os.environ.get("LOKI_FAILMEM_CWD") or os.getcwd()
|
|
175
|
+
loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
|
|
176
|
+
|
|
177
|
+
kv = {}
|
|
178
|
+
for a in argv[1:]:
|
|
179
|
+
if a.startswith("--") and "=" in a:
|
|
180
|
+
k, v = a[2:].split("=", 1)
|
|
181
|
+
kv[k] = v
|
|
182
|
+
|
|
183
|
+
if action == "record":
|
|
184
|
+
res = record_failure(loki_dir, kv.get("gate"), kv.get("verdict"),
|
|
185
|
+
kv.get("evidence"), kv.get("run_id"))
|
|
186
|
+
elif action == "recall":
|
|
187
|
+
res = recall(loki_dir, kv.get("gate"))
|
|
188
|
+
elif action == "context":
|
|
189
|
+
res = prompt_context(loki_dir)
|
|
190
|
+
else:
|
|
191
|
+
print(f"unknown action: {action}", file=sys.stderr)
|
|
192
|
+
return 2
|
|
193
|
+
|
|
194
|
+
print(json.dumps(res, indent=2))
|
|
195
|
+
return 0 if res.get("status") in ("recorded", "measured") else 3
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
if __name__ == "__main__":
|
|
199
|
+
sys.exit(main(sys.argv[1:]))
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Which gates block, which only advise, and what promoting one would have cost.
|
|
3
|
+
|
|
4
|
+
Ona's Veto Exec ships an audit-first ladder: "Start with audit rules, review
|
|
5
|
+
matches, then promote confirmed rules to block." The load-bearing part is the
|
|
6
|
+
MIDDLE step. A policy you cannot safely turn on is a policy nobody turns on, so
|
|
7
|
+
before flipping a gate to blocking you get to see what it WOULD have blocked.
|
|
8
|
+
|
|
9
|
+
We already had both ends and nothing in between: gates are advisory or blocking,
|
|
10
|
+
three promotion knobs exist (LOKI_GATE_MAGIC_DEBATE_BLOCKING, LOKI_COV_ENFORCE,
|
|
11
|
+
LOKI_POLICY_APPROVAL_ENFORCE), and .loki/quality/gate-failure-count.json has
|
|
12
|
+
counted per-gate failures the whole time. Nothing joined them, so an operator
|
|
13
|
+
deciding whether to promote a gate had to guess.
|
|
14
|
+
|
|
15
|
+
DETERMINISTIC. Reads two files and the environment. No model, no network, no
|
|
16
|
+
spend -- same repo state, same answer, and every number is a count a reader can
|
|
17
|
+
recompute by opening the same file.
|
|
18
|
+
|
|
19
|
+
NEVER PROMOTES ANYTHING. This reports; it does not change policy. Turning a gate
|
|
20
|
+
blocking stays an explicit operator act via the named environment variable,
|
|
21
|
+
because a tool that silently starts blocking is the thing operators most
|
|
22
|
+
reasonably fear.
|
|
23
|
+
|
|
24
|
+
Shape (assess() and --json, schema_version 1). This is the contract the
|
|
25
|
+
dashboard endpoint GET /api/gate-policy and tests/test_gate_policy_endpoint.py
|
|
26
|
+
both read against, so field names here are load-bearing:
|
|
27
|
+
|
|
28
|
+
schema_version int 1
|
|
29
|
+
status str "measured"
|
|
30
|
+
ledger str "present" | "absent" -- whether the per-gate failure
|
|
31
|
+
ledger .loki/quality/gate-failure-count.json was read
|
|
32
|
+
gates list one record per known gate, blocking gates first, each
|
|
33
|
+
group sorted by name:
|
|
34
|
+
gate str gate name (e.g. "code_review")
|
|
35
|
+
mode str "blocking" | "advisory" -- for a promotable gate this
|
|
36
|
+
depends on the ENVIRONMENT at call time
|
|
37
|
+
promotable bool True when a real promotion knob exists in run.sh
|
|
38
|
+
audit_hits int|null failures counted for this gate. null means
|
|
39
|
+
UNMEASURED -- no ledger, or no entry for this gate.
|
|
40
|
+
NEVER 0 for an unmeasured gate: 0 is the positive claim
|
|
41
|
+
that the gate ran and never fired, which is the false
|
|
42
|
+
green an absent measurement always produces.
|
|
43
|
+
why str one-line description of what the gate checks
|
|
44
|
+
promote_with str|null "VAR=value" to make an advisory gate blocking; null
|
|
45
|
+
when the gate already blocks
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
import json
|
|
49
|
+
import os
|
|
50
|
+
import sys
|
|
51
|
+
|
|
52
|
+
SCHEMA_VERSION = 1
|
|
53
|
+
|
|
54
|
+
# Gates that can be promoted from advisory to blocking, and the knob that does
|
|
55
|
+
# it. Only gates with a REAL knob in run.sh appear here -- listing an aspiration
|
|
56
|
+
# would tell an operator to set a variable nothing reads.
|
|
57
|
+
PROMOTABLE = {
|
|
58
|
+
"magic_debate": ("LOKI_GATE_MAGIC_DEBATE_BLOCKING", "true",
|
|
59
|
+
"spec-vs-implementation debate on generated modules"),
|
|
60
|
+
"test_coverage": ("LOKI_COV_ENFORCE", "1",
|
|
61
|
+
"project test runner pass/fail"),
|
|
62
|
+
"policy_approval": ("LOKI_POLICY_APPROVAL_ENFORCE", "1",
|
|
63
|
+
"staged-autonomy approval policy"),
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
# Gates that block unconditionally. Listed so the report is a complete picture
|
|
67
|
+
# rather than only the promotable subset -- an operator asking "what blocks here"
|
|
68
|
+
# should not have to read run.sh to find out.
|
|
69
|
+
ALWAYS_BLOCKING = {
|
|
70
|
+
"static_analysis": "CodeQL, ESLint/Pylint, type-checker findings on the diff",
|
|
71
|
+
"code_review": "3-reviewer blind review; Critical/High = BLOCK",
|
|
72
|
+
"mock_integrity": "tautological-assertion and mock-ratio detection",
|
|
73
|
+
"mutation_integrity": "assertion-churn (test-fitting) detection",
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _counts(loki_dir):
|
|
78
|
+
"""Per-gate failure counts, or {} when the ledger has not been written."""
|
|
79
|
+
path = os.path.join(loki_dir, "quality", "gate-failure-count.json")
|
|
80
|
+
try:
|
|
81
|
+
with open(path) as fh:
|
|
82
|
+
data = json.load(fh)
|
|
83
|
+
return data if isinstance(data, dict) else {}
|
|
84
|
+
except (OSError, ValueError):
|
|
85
|
+
return {}
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def assess(loki_dir=".loki", env=None):
|
|
89
|
+
env = os.environ if env is None else env
|
|
90
|
+
counts = _counts(loki_dir)
|
|
91
|
+
have_ledger = bool(counts)
|
|
92
|
+
|
|
93
|
+
gates = []
|
|
94
|
+
for name, why in sorted(ALWAYS_BLOCKING.items()):
|
|
95
|
+
gates.append({
|
|
96
|
+
"gate": name, "mode": "blocking", "promotable": False,
|
|
97
|
+
"audit_hits": counts.get(name) if have_ledger else None,
|
|
98
|
+
"why": why, "promote_with": None,
|
|
99
|
+
})
|
|
100
|
+
for name, (var, val, why) in sorted(PROMOTABLE.items()):
|
|
101
|
+
on = str(env.get(var, "")).lower() in ("1", "true", "yes")
|
|
102
|
+
gates.append({
|
|
103
|
+
"gate": name,
|
|
104
|
+
"mode": "blocking" if on else "advisory",
|
|
105
|
+
"promotable": True,
|
|
106
|
+
# None, not 0: an absent ledger means UNMEASURED, and reporting 0
|
|
107
|
+
# would read as "this gate never fired" -- the same false green a
|
|
108
|
+
# missing measurement always produces.
|
|
109
|
+
"audit_hits": counts.get(name) if have_ledger else None,
|
|
110
|
+
"why": why,
|
|
111
|
+
"promote_with": None if on else f"{var}={val}",
|
|
112
|
+
})
|
|
113
|
+
|
|
114
|
+
return {
|
|
115
|
+
"schema_version": SCHEMA_VERSION,
|
|
116
|
+
"status": "measured",
|
|
117
|
+
"ledger": "present" if have_ledger else "absent",
|
|
118
|
+
"gates": gates,
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def render_text(res):
|
|
123
|
+
out = ["Gate policy -- what blocks here, and what only advises", ""]
|
|
124
|
+
for g in res["gates"]:
|
|
125
|
+
hits = g["audit_hits"]
|
|
126
|
+
if hits is None:
|
|
127
|
+
hit_s = "not measured"
|
|
128
|
+
elif hits == 0:
|
|
129
|
+
hit_s = "0 hits"
|
|
130
|
+
else:
|
|
131
|
+
hit_s = f"{hits} hit{'s' if hits != 1 else ''}"
|
|
132
|
+
mode = g["mode"].upper()
|
|
133
|
+
out.append(f" {mode:9} {g['gate']:20} {hit_s}")
|
|
134
|
+
out.append(f" {g['why']}")
|
|
135
|
+
if g["promote_with"]:
|
|
136
|
+
verb = "would have blocked" if (hits or 0) > 0 else "has not fired"
|
|
137
|
+
out.append(f" advisory: {verb} -- promote with {g['promote_with']}")
|
|
138
|
+
out.append("")
|
|
139
|
+
|
|
140
|
+
if res["ledger"] == "absent":
|
|
141
|
+
out.append(" No gate ledger yet (.loki/quality/gate-failure-count.json).")
|
|
142
|
+
out.append(" Counts read 'not measured' rather than 0: an absent")
|
|
143
|
+
out.append(" measurement is not evidence a gate never fired.")
|
|
144
|
+
else:
|
|
145
|
+
out.append(" Hits come from .loki/quality/gate-failure-count.json.")
|
|
146
|
+
out.append(" Open it and count the same numbers yourself.")
|
|
147
|
+
out.append("")
|
|
148
|
+
out.append(" This command never promotes a gate. Promotion is an explicit")
|
|
149
|
+
out.append(" operator act via the variable named above.")
|
|
150
|
+
return "\n".join(out)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def main(argv):
|
|
154
|
+
as_json = "--json" in argv
|
|
155
|
+
loki_dir = ".loki"
|
|
156
|
+
for a in argv:
|
|
157
|
+
if not a.startswith("-"):
|
|
158
|
+
loki_dir = a
|
|
159
|
+
break
|
|
160
|
+
res = assess(loki_dir)
|
|
161
|
+
print(json.dumps(res, indent=2) if as_json else render_text(res))
|
|
162
|
+
return 0
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
if __name__ == "__main__":
|
|
166
|
+
sys.exit(main(sys.argv[1:]))
|