loki-mode 9.12.6 → 9.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +81 -101
- package/SKILL.md +2 -2
- package/VERSION +1 -1
- package/autonomy/intent.sh +414 -0
- package/autonomy/issue-providers.sh +21 -0
- package/autonomy/lib/agent_readiness.py +202 -0
- package/autonomy/lib/claim_grounding.py +171 -0
- package/autonomy/lib/config-map.sh +10 -6
- package/autonomy/lib/decision_record.py +198 -0
- package/autonomy/lib/failure_memory.py +199 -0
- package/autonomy/lib/outcome_ledger.py +498 -0
- package/autonomy/lib/preedit_snapshot.py +216 -0
- package/autonomy/lib/verdict.py +204 -0
- package/autonomy/loki +358 -14
- package/autonomy/provider-offer.sh +25 -1
- package/autonomy/run.sh +516 -11
- package/autonomy/telemetry.sh +8 -1
- package/completions/_loki +4 -0
- package/completions/loki.bash +2 -1
- package/dashboard/__init__.py +1 -1
- package/dashboard/run.py +13 -2
- package/dashboard/scim.py +221 -0
- package/dashboard/server.py +22 -0
- package/docs/GATE-FAILURE-TRIAGE.md +254 -0
- package/docs/LOOP-CANDIDATE-PROPOSAL-v1.md +167 -0
- package/docs/LOOP-HARNESS-AUDIT.md +53 -0
- package/docs/VERIFICATION-COST.md +103 -0
- package/docs/WANG-PRINCIPLES-PLAN.md +1 -1
- package/loki-ts/dist/loki.js +402 -398
- package/mcp/__init__.py +1 -1
- package/mcp/_sdk_loader.py +25 -0
- package/package.json +1 -1
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Claim grounding: a completion claim must name work that exists in the diff.
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. 8090 AI's evaluation framework blocks ungrounded output AT
|
|
5
|
+
GENERATION rather than catching it at review, and states the payoff plainly:
|
|
6
|
+
|
|
7
|
+
"Ungrounded claims are blocked at generation. The rubric measures the quality
|
|
8
|
+
of what survives the architectural filter, which is a much smaller and more
|
|
9
|
+
interesting space."
|
|
10
|
+
|
|
11
|
+
Our evidence gate already has six axes -- diff non-empty, tests green, runtime
|
|
12
|
+
boot, no-mock, authorization, secret leak. Every one is a REPO-LEVEL fact. None
|
|
13
|
+
reads what the agent actually CLAIMED. So an agent can finish by asserting "added
|
|
14
|
+
retry logic to the payment client and covered it with tests" while the diff shows
|
|
15
|
+
a README edit, and every axis passes: the diff is non-empty, the tests are green,
|
|
16
|
+
the app boots. The claim itself is the one artifact nobody checks.
|
|
17
|
+
|
|
18
|
+
WHAT THIS CHECKS, AND WHAT IT REFUSES TO CHECK. It resolves file-path-shaped
|
|
19
|
+
tokens in the claim against the actual changed-file set. That is a deterministic
|
|
20
|
+
string-to-set comparison, and it is the only part of a natural-language claim
|
|
21
|
+
that can be checked without a model.
|
|
22
|
+
|
|
23
|
+
It does NOT judge whether the claim is semantically true. "Added retry logic" vs
|
|
24
|
+
"added a retry constant" is a judgement, and asking an LLM to grade it would be
|
|
25
|
+
the same LLM-judge-as-measurement this codebase refuses everywhere else. A claim
|
|
26
|
+
naming no paths is UNGROUNDABLE, reported by name -- never scored, never failed.
|
|
27
|
+
|
|
28
|
+
THE FAIL-OPEN DIRECTION IS DELIBERATE. Only a claim that names a path which is
|
|
29
|
+
demonstrably NOT in the diff is a finding. A claim with no paths, a claim naming
|
|
30
|
+
a path that exists but was not touched by this run, and an empty claim are all
|
|
31
|
+
reported and none of them blocks. A grounding check that blocked on ambiguity
|
|
32
|
+
would fire constantly on ordinary prose and be disabled within a week, which is
|
|
33
|
+
worse than not having it.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
import json
|
|
39
|
+
import os
|
|
40
|
+
import re
|
|
41
|
+
import sys
|
|
42
|
+
|
|
43
|
+
SCHEMA_VERSION = "1.0"
|
|
44
|
+
|
|
45
|
+
UNKNOWN = "UNKNOWN"
|
|
46
|
+
|
|
47
|
+
GROUNDING_REASONS = {
|
|
48
|
+
"no_claim": "no completion claim text was provided",
|
|
49
|
+
"no_diff": "no changed-file set was provided to check against",
|
|
50
|
+
"ungroundable": "the claim names no file paths, so it cannot be checked mechanically",
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
# A path-shaped token: at least one slash or a known source extension, and no
|
|
54
|
+
# spaces. Deliberately conservative -- a false "this is a path" produces a false
|
|
55
|
+
# finding, and a check that cries wolf gets turned off.
|
|
56
|
+
_PATH_RE = re.compile(
|
|
57
|
+
r"(?<![\w/.-])"
|
|
58
|
+
r"(?:[\w.-]+/)+[\w.-]+\.[A-Za-z0-9]{1,6}"
|
|
59
|
+
r"|(?<![\w/.-])[\w.-]+\.(?:py|ts|tsx|js|jsx|sh|go|rs|rb|java|c|h|cpp|md|json|ya?ml|toml)"
|
|
60
|
+
r"(?![\w/.-])"
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def extract_paths(claim):
|
|
65
|
+
"""Path-shaped tokens in a claim, deduped, order preserved."""
|
|
66
|
+
if not claim:
|
|
67
|
+
return []
|
|
68
|
+
seen, out = set(), []
|
|
69
|
+
for m in _PATH_RE.finditer(claim):
|
|
70
|
+
tok = m.group(0).strip(".,;:)(\"'`")
|
|
71
|
+
if tok and tok not in seen:
|
|
72
|
+
seen.add(tok)
|
|
73
|
+
out.append(tok)
|
|
74
|
+
return out
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def check(claim, changed_files):
|
|
78
|
+
"""Are the paths a claim names present in the diff?
|
|
79
|
+
|
|
80
|
+
Matching is suffix-based on purpose: an agent writes `run.sh` or
|
|
81
|
+
`autonomy/run.sh` for the same file, and demanding an exact repo-relative
|
|
82
|
+
string would make every informal mention a false finding.
|
|
83
|
+
"""
|
|
84
|
+
if not claim or not claim.strip():
|
|
85
|
+
return {"status": UNKNOWN, "reason": "no_claim",
|
|
86
|
+
"detail": GROUNDING_REASONS["no_claim"]}
|
|
87
|
+
if changed_files is None:
|
|
88
|
+
return {"status": UNKNOWN, "reason": "no_diff",
|
|
89
|
+
"detail": GROUNDING_REASONS["no_diff"]}
|
|
90
|
+
|
|
91
|
+
named = extract_paths(claim)
|
|
92
|
+
if not named:
|
|
93
|
+
# The common, benign case: "fixed the login bug". Nothing to check, and
|
|
94
|
+
# that is not a defect -- reported so the number of unverifiable claims
|
|
95
|
+
# is visible, rather than silently counted as grounded.
|
|
96
|
+
return {"status": UNKNOWN, "reason": "ungroundable",
|
|
97
|
+
"detail": GROUNDING_REASONS["ungroundable"],
|
|
98
|
+
"paths_named": []}
|
|
99
|
+
|
|
100
|
+
changed = list(changed_files)
|
|
101
|
+
grounded, ungrounded = [], []
|
|
102
|
+
for p in named:
|
|
103
|
+
norm = p.lstrip("./")
|
|
104
|
+
# Basename fallback is load-bearing, not laxity. An agent writes `run.sh`
|
|
105
|
+
# for `autonomy/run.sh`, and requiring the full repo-relative string made
|
|
106
|
+
# every informal mention a false finding -- measured: "patched run.sh"
|
|
107
|
+
# against a changed autonomy/run.sh was reported UNGROUNDED. A grounding
|
|
108
|
+
# check that flags correct claims is worse than none, because it gets
|
|
109
|
+
# disabled and takes the real detections with it.
|
|
110
|
+
base = os.path.basename(norm)
|
|
111
|
+
hit = any(
|
|
112
|
+
c == norm
|
|
113
|
+
or c.endswith("/" + norm)
|
|
114
|
+
or norm.endswith("/" + c)
|
|
115
|
+
or os.path.basename(c) == base
|
|
116
|
+
for c in changed
|
|
117
|
+
)
|
|
118
|
+
(grounded if hit else ungrounded).append(p)
|
|
119
|
+
|
|
120
|
+
return {
|
|
121
|
+
"status": "measured",
|
|
122
|
+
"paths_named": named,
|
|
123
|
+
"grounded": grounded,
|
|
124
|
+
"ungrounded": ungrounded,
|
|
125
|
+
# The single actionable signal. A claim that names a file the run never
|
|
126
|
+
# touched is the "agent says done, diff says otherwise" failure, caught
|
|
127
|
+
# from the claim side instead of the repo side.
|
|
128
|
+
"has_ungrounded_claim": bool(ungrounded),
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def main(argv):
|
|
133
|
+
claim = ""
|
|
134
|
+
files_arg = ""
|
|
135
|
+
files_from = ""
|
|
136
|
+
i = 0
|
|
137
|
+
while i < len(argv):
|
|
138
|
+
if argv[i] == "--claim" and i + 1 < len(argv):
|
|
139
|
+
claim = argv[i + 1]; i += 2; continue
|
|
140
|
+
if argv[i] == "--files" and i + 1 < len(argv):
|
|
141
|
+
files_arg = argv[i + 1]; i += 2; continue
|
|
142
|
+
if argv[i] == "--files-from" and i + 1 < len(argv):
|
|
143
|
+
files_from = argv[i + 1]; i += 2; continue
|
|
144
|
+
i += 1
|
|
145
|
+
|
|
146
|
+
# --files is ALWAYS a comma-separated list; --files-from is always a file to
|
|
147
|
+
# read. The first version overloaded one flag for both and picked with
|
|
148
|
+
# os.path.isfile(), which silently misread `--files autonomy/run.sh` -- a
|
|
149
|
+
# perfectly ordinary changed-file list -- as "open that file and treat its
|
|
150
|
+
# 20,000 lines as filenames". The result was a correct claim reported
|
|
151
|
+
# UNGROUNDED, i.e. the exact false positive this check must never produce.
|
|
152
|
+
# An ambiguous flag whose meaning depends on the filesystem is a bug, not a
|
|
153
|
+
# convenience.
|
|
154
|
+
if files_from and os.path.isfile(files_from):
|
|
155
|
+
with open(files_from, "r", encoding="utf-8") as fh:
|
|
156
|
+
changed = [l.strip() for l in fh if l.strip()]
|
|
157
|
+
elif files_arg:
|
|
158
|
+
changed = [c.strip() for c in files_arg.split(",") if c.strip()]
|
|
159
|
+
else:
|
|
160
|
+
changed = None
|
|
161
|
+
|
|
162
|
+
res = check(claim, changed)
|
|
163
|
+
res["schema_version"] = SCHEMA_VERSION
|
|
164
|
+
print(json.dumps(res, indent=2))
|
|
165
|
+
# Exit 1 ONLY on a demonstrably ungrounded path. Every UNKNOWN exits 0: a
|
|
166
|
+
# check that failed on ambiguity would fire on ordinary prose and be disabled.
|
|
167
|
+
return 1 if res.get("has_ungrounded_claim") else 0
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
if __name__ == "__main__":
|
|
171
|
+
sys.exit(main(sys.argv[1:]))
|
|
@@ -863,13 +863,17 @@ loki_config_validate_file() {
|
|
|
863
863
|
pairs="$(
|
|
864
864
|
_loki_cfg_collect_pairs() {
|
|
865
865
|
local f="$1" fm="$2"
|
|
866
|
+
# NOTE: every case pattern below carries a leading open-paren --
|
|
867
|
+
# bash 3.2 (macOS /bin/bash) ends a command substitution at the
|
|
868
|
+
# close-paren of a case label, truncating this function body. The
|
|
869
|
+
# balanced form parses identically on 3.2 and 4+. Do not strip it.
|
|
866
870
|
case "$fm" in
|
|
867
|
-
env)
|
|
871
|
+
(env)
|
|
868
872
|
local line key val
|
|
869
873
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
870
|
-
case "$line" in ''|'#'*) continue ;; esac
|
|
874
|
+
case "$line" in (''|'#'*) continue ;; esac
|
|
871
875
|
line="${line#export }"
|
|
872
|
-
case "$line" in *=*) ;; *) continue ;; esac
|
|
876
|
+
case "$line" in (*=*) ;; (*) continue ;; esac
|
|
873
877
|
key="${line%%=*}"; val="${line#*=}"
|
|
874
878
|
key="${key#"${key%%[![:space:]]*}"}"; key="${key%"${key##*[![:space:]]}"}"
|
|
875
879
|
val="${val#"${val%%[![:space:]]*}"}"
|
|
@@ -878,7 +882,7 @@ loki_config_validate_file() {
|
|
|
878
882
|
printf '%s\t%s\n' "$key" "$val"
|
|
879
883
|
done < "$f"
|
|
880
884
|
;;
|
|
881
|
-
yaml)
|
|
885
|
+
(yaml)
|
|
882
886
|
local mapping yaml_path env_var value have_yq=0
|
|
883
887
|
if command -v yq >/dev/null 2>&1; then have_yq=1; fi
|
|
884
888
|
for mapping in "${LOKI_CONFIG_MAP[@]}"; do
|
|
@@ -894,8 +898,8 @@ loki_config_validate_file() {
|
|
|
894
898
|
printf '%s\t%s\n' "$env_var" "$value"
|
|
895
899
|
done
|
|
896
900
|
;;
|
|
897
|
-
json)
|
|
898
|
-
# Reuse the
|
|
901
|
+
(json)
|
|
902
|
+
# Reuse the JSON parser emit path via a print-only variant.
|
|
899
903
|
_loki_cfg_json_emit "$f"
|
|
900
904
|
;;
|
|
901
905
|
esac
|
|
@@ -0,0 +1,198 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""LLM Decision Record: which model, at what temperature, decided what.
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. Factory AI's audit log has eight event types and NOT ONE
|
|
5
|
+
records an agent action -- they are all admin configuration (membership, API
|
|
6
|
+
keys, integrations, managed settings). Agent forensics exists only as
|
|
7
|
+
customer-built OTEL: metrics by default, message content opt-in, and the customer
|
|
8
|
+
must stand up and retain the pipeline. So "which agent changed this line, on
|
|
9
|
+
whose authority, and what did it verify" is answerable only if the buyer built
|
|
10
|
+
the plumbing themselves. Devin's audit log is likewise session and admin scoped.
|
|
11
|
+
|
|
12
|
+
8090's evaluation framework records, per AI operation: session id, pipeline
|
|
13
|
+
stage, model id, temperature, timestamps, token counts, reasoning trace,
|
|
14
|
+
confidence. Their stated reason is the one that matters to a regulated buyer:
|
|
15
|
+
capturing model id and temperature makes a MODEL SWAP DETECTABLE. Without it,
|
|
16
|
+
a provider silently changing a model underneath you is invisible in your own
|
|
17
|
+
records, and "we used an approved configuration" becomes unfalsifiable.
|
|
18
|
+
|
|
19
|
+
WHAT THIS IS. An append-only record of agent decisions, written next to the
|
|
20
|
+
receipt so the chain of custody is complete: the receipt says what was proven,
|
|
21
|
+
the outcome ledger says what happened afterwards, and this says what made the
|
|
22
|
+
call. Append-only because an audit trail a later run can rewrite is not an audit
|
|
23
|
+
trail.
|
|
24
|
+
|
|
25
|
+
WHAT THIS DELIBERATELY DOES NOT CAPTURE. No prompt bodies, no file contents, no
|
|
26
|
+
credentials, no environment. 8090's own boundary applies with more force to us
|
|
27
|
+
than to them: an adoption tool that exfiltrates a user's environment would cost
|
|
28
|
+
exactly the trust this product sells. Fields are a fixed allowlist, and a test
|
|
29
|
+
asserts the allowlist so a future contributor cannot widen it casually.
|
|
30
|
+
|
|
31
|
+
IT IS NOT TELEMETRY. Nothing here is transmitted. It writes a local JSONL file
|
|
32
|
+
that the user owns and can read, diff, and delete. autonomy/telemetry.sh remains
|
|
33
|
+
the single egress point.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
import json
|
|
39
|
+
import os
|
|
40
|
+
import sys
|
|
41
|
+
from datetime import datetime, timezone
|
|
42
|
+
|
|
43
|
+
SCHEMA_VERSION = "1.0"
|
|
44
|
+
|
|
45
|
+
# The complete set of fields a record may carry. Anything not on this list is
|
|
46
|
+
# dropped rather than written: a record that can grow a new field by accident is
|
|
47
|
+
# how an audit log becomes a data-exfiltration surface. The test suite asserts
|
|
48
|
+
# this exact set, so widening it is a deliberate, reviewed act.
|
|
49
|
+
ALLOWED_FIELDS = (
|
|
50
|
+
"schema_version",
|
|
51
|
+
"recorded_at",
|
|
52
|
+
"session_id",
|
|
53
|
+
"run_id",
|
|
54
|
+
"stage", # which part of the loop made the call
|
|
55
|
+
"model_id", # THE field that makes a silent model swap detectable
|
|
56
|
+
"temperature", # same: a config drift nobody announced
|
|
57
|
+
"provider",
|
|
58
|
+
"tokens_in",
|
|
59
|
+
"tokens_out",
|
|
60
|
+
"duration_ms",
|
|
61
|
+
"outcome", # coarse: ok | error | refused | timeout
|
|
62
|
+
"confidence", # self-reported, and labelled as such below
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
# Fields whose value is the AGENT'S OWN OPINION, never a measurement. Kept
|
|
66
|
+
# separate so a reader cannot mistake a self-report for an observation -- the
|
|
67
|
+
# same FACTS-vs-ASSESSMENTS split the receipts already enforce.
|
|
68
|
+
SELF_REPORTED = ("confidence",)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _records_path(loki_dir):
|
|
72
|
+
return os.path.join(loki_dir, "decisions", "decisions.jsonl")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def record(loki_dir, fields):
|
|
76
|
+
"""Append one decision record. Returns the written record, or a reason.
|
|
77
|
+
|
|
78
|
+
Append-only by construction: opened "a", never "w". A trail a later run can
|
|
79
|
+
rewrite is not a trail, and the whole value here is that a reader can trust
|
|
80
|
+
what it says about a run that already finished.
|
|
81
|
+
"""
|
|
82
|
+
clean = {}
|
|
83
|
+
dropped = []
|
|
84
|
+
for k, v in (fields or {}).items():
|
|
85
|
+
if k in ALLOWED_FIELDS:
|
|
86
|
+
clean[k] = v
|
|
87
|
+
else:
|
|
88
|
+
dropped.append(k)
|
|
89
|
+
|
|
90
|
+
if not clean.get("model_id"):
|
|
91
|
+
# A record without a model id cannot answer the question this exists to
|
|
92
|
+
# answer, so it is refused rather than written as a half-record.
|
|
93
|
+
return {"status": "UNKNOWN", "reason": "no_model_id",
|
|
94
|
+
"detail": "a decision record without model_id cannot make a swap detectable"}
|
|
95
|
+
|
|
96
|
+
clean["schema_version"] = SCHEMA_VERSION
|
|
97
|
+
clean["recorded_at"] = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
98
|
+
|
|
99
|
+
path = _records_path(loki_dir)
|
|
100
|
+
try:
|
|
101
|
+
os.makedirs(os.path.dirname(path), exist_ok=True)
|
|
102
|
+
with open(path, "a", encoding="utf-8") as fh:
|
|
103
|
+
fh.write(json.dumps(clean, sort_keys=True) + "\n")
|
|
104
|
+
except OSError as exc:
|
|
105
|
+
return {"status": "UNKNOWN", "reason": "unwritable", "detail": str(exc)}
|
|
106
|
+
|
|
107
|
+
out = {"status": "recorded", "path": path, "record": clean}
|
|
108
|
+
if dropped:
|
|
109
|
+
# Surfaced, not silent: a caller trying to log a prompt body should see
|
|
110
|
+
# that it was refused rather than assume it was stored.
|
|
111
|
+
out["dropped_fields"] = sorted(dropped)
|
|
112
|
+
return out
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def summarize(loki_dir):
|
|
116
|
+
"""What models actually ran, and did the configuration change mid-flight?
|
|
117
|
+
|
|
118
|
+
The useful audit question is not "how many calls" but "did the thing that
|
|
119
|
+
decided change without anyone saying so". A run whose records name two model
|
|
120
|
+
ids is exactly the case a regulated buyer needs surfaced.
|
|
121
|
+
"""
|
|
122
|
+
path = _records_path(loki_dir)
|
|
123
|
+
if not os.path.isfile(path):
|
|
124
|
+
return {"status": "UNKNOWN", "reason": "no_records",
|
|
125
|
+
"detail": "no decision records for this project"}
|
|
126
|
+
|
|
127
|
+
models, temps, stages, n, bad = {}, {}, {}, 0, 0
|
|
128
|
+
try:
|
|
129
|
+
with open(path, "r", encoding="utf-8") as fh:
|
|
130
|
+
for line in fh:
|
|
131
|
+
line = line.strip()
|
|
132
|
+
if not line:
|
|
133
|
+
continue
|
|
134
|
+
try:
|
|
135
|
+
rec = json.loads(line)
|
|
136
|
+
except ValueError:
|
|
137
|
+
# A corrupt line is counted, never silently skipped: an audit
|
|
138
|
+
# trail that quietly drops what it cannot parse is worse than
|
|
139
|
+
# one that admits a gap.
|
|
140
|
+
bad += 1
|
|
141
|
+
continue
|
|
142
|
+
n += 1
|
|
143
|
+
m = rec.get("model_id")
|
|
144
|
+
if m:
|
|
145
|
+
models[m] = models.get(m, 0) + 1
|
|
146
|
+
t = rec.get("temperature")
|
|
147
|
+
if t is not None:
|
|
148
|
+
temps[str(t)] = temps.get(str(t), 0) + 1
|
|
149
|
+
s = rec.get("stage")
|
|
150
|
+
if s:
|
|
151
|
+
stages[s] = stages.get(s, 0) + 1
|
|
152
|
+
except OSError as exc:
|
|
153
|
+
return {"status": "UNKNOWN", "reason": "unreadable", "detail": str(exc)}
|
|
154
|
+
|
|
155
|
+
return {
|
|
156
|
+
"status": "measured",
|
|
157
|
+
"records": n,
|
|
158
|
+
"unparseable_lines": bad,
|
|
159
|
+
"models": models,
|
|
160
|
+
"temperatures": temps,
|
|
161
|
+
"stages": stages,
|
|
162
|
+
# The headline fact. More than one model id, or more than one
|
|
163
|
+
# temperature, means the configuration changed during this project and
|
|
164
|
+
# the records prove it.
|
|
165
|
+
"model_changed": len(models) > 1,
|
|
166
|
+
"temperature_changed": len(temps) > 1,
|
|
167
|
+
"self_reported_fields": list(SELF_REPORTED),
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def main(argv):
|
|
172
|
+
if not argv:
|
|
173
|
+
print("usage: decision_record.py record|summary [--field=value ...]",
|
|
174
|
+
file=sys.stderr)
|
|
175
|
+
return 2
|
|
176
|
+
action = argv[0]
|
|
177
|
+
cwd = os.environ.get("LOKI_DECISION_CWD") or os.getcwd()
|
|
178
|
+
loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
|
|
179
|
+
|
|
180
|
+
if action == "record":
|
|
181
|
+
fields = {}
|
|
182
|
+
for arg in argv[1:]:
|
|
183
|
+
if arg.startswith("--") and "=" in arg:
|
|
184
|
+
k, v = arg[2:].split("=", 1)
|
|
185
|
+
fields[k] = v
|
|
186
|
+
res = record(loki_dir, fields)
|
|
187
|
+
elif action == "summary":
|
|
188
|
+
res = summarize(loki_dir)
|
|
189
|
+
else:
|
|
190
|
+
print(f"unknown action: {action}", file=sys.stderr)
|
|
191
|
+
return 2
|
|
192
|
+
|
|
193
|
+
print(json.dumps(res, indent=2))
|
|
194
|
+
return 0 if res.get("status") in ("recorded", "measured") else 3
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
if __name__ == "__main__":
|
|
198
|
+
sys.exit(main(sys.argv[1:]))
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Failure memory: the agent gets better at YOUR repo by having been wrong in it.
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. Neither competitor learns from being wrong, and one of them
|
|
5
|
+
shipped a UI to manage that fact.
|
|
6
|
+
|
|
7
|
+
Factory AI has no memory system at all -- verified by grep over their whole doc
|
|
8
|
+
corpus. AGENTS.md is hand-authored by the human, AutoWiki regenerates from CODE
|
|
9
|
+
rather than from outcomes, QA failure learning is `suggest_in_report` (a human
|
|
10
|
+
reads it and decides), and mission-generated skills live under {missionDir} and
|
|
11
|
+
are DISCARDED with the mission. Devin has a real memory layer, but every session
|
|
12
|
+
starts from a fresh snapshot copy and all changes are discarded at teardown --
|
|
13
|
+
and their Session Insights ships a "Misleading Knowledge" tab enumerating memory
|
|
14
|
+
items that led Devin astray. They built an interface for their memory poisoning
|
|
15
|
+
output.
|
|
16
|
+
|
|
17
|
+
So on both systems, an agent that failed a gate in your repo yesterday starts
|
|
18
|
+
today knowing nothing about it.
|
|
19
|
+
|
|
20
|
+
WHAT MAKES THIS DIFFERENT, AND WHY IT IS NOT THE SAME TRAP. The reason Devin
|
|
21
|
+
needed a "Misleading Knowledge" tab is that memory written from an agent's own
|
|
22
|
+
narration is unfalsifiable: it records what the agent BELIEVED, which is exactly
|
|
23
|
+
what was wrong when it failed. So nothing here is written from narration. A
|
|
24
|
+
lesson is created only from a MEASURED event -- a named gate that failed, with
|
|
25
|
+
its verdict -- and it carries the evidence that produced it. A lesson whose
|
|
26
|
+
evidence no longer holds can be retired mechanically instead of accumulating.
|
|
27
|
+
|
|
28
|
+
This is deliberately built on the outcome ledger's discipline: a lesson is
|
|
29
|
+
recorded only when the event that justifies it is a fact, and everything else
|
|
30
|
+
reports UNKNOWN with a named reason rather than being written as a weak guess.
|
|
31
|
+
|
|
32
|
+
WHAT IT DOES NOT DO. It does not summarize, generalize, or ask a model what the
|
|
33
|
+
lesson "means". A generalization is a judgement, and a judgement stored as memory
|
|
34
|
+
is indistinguishable from a fact by the next reader -- which is the poisoning
|
|
35
|
+
mechanism. It stores the gate, the verdict, the evidence, and a count. Deciding
|
|
36
|
+
what that implies stays with whoever reads it.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
from __future__ import annotations
|
|
40
|
+
|
|
41
|
+
import json
|
|
42
|
+
import os
|
|
43
|
+
import sys
|
|
44
|
+
from datetime import datetime, timezone
|
|
45
|
+
|
|
46
|
+
SCHEMA_VERSION = "1.0"
|
|
47
|
+
|
|
48
|
+
UNKNOWN = "UNKNOWN"
|
|
49
|
+
|
|
50
|
+
REASONS = {
|
|
51
|
+
"no_gate": "no gate name was given, so there is nothing to attribute the failure to",
|
|
52
|
+
"no_evidence": "no evidence was given, so the lesson would be unfalsifiable",
|
|
53
|
+
"no_lessons": "no failure lessons recorded for this project",
|
|
54
|
+
"unreadable": "the lesson store exists but could not be parsed",
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _path(loki_dir):
|
|
59
|
+
return os.path.join(loki_dir, "memory", "failures.jsonl")
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def record_failure(loki_dir, gate, verdict, evidence, run_id=None):
|
|
63
|
+
"""Turn a measured gate failure into a durable, falsifiable lesson.
|
|
64
|
+
|
|
65
|
+
Requires EVIDENCE. A lesson without it is the agent's own account of why it
|
|
66
|
+
failed, which is precisely the unfalsifiable memory that made Devin's need a
|
|
67
|
+
"Misleading Knowledge" tab. If we cannot say what was observed, we do not
|
|
68
|
+
write anything.
|
|
69
|
+
"""
|
|
70
|
+
if not gate:
|
|
71
|
+
return {"status": UNKNOWN, "reason": "no_gate", "detail": REASONS["no_gate"]}
|
|
72
|
+
if not evidence:
|
|
73
|
+
return {"status": UNKNOWN, "reason": "no_evidence",
|
|
74
|
+
"detail": REASONS["no_evidence"]}
|
|
75
|
+
|
|
76
|
+
rec = {
|
|
77
|
+
"schema_version": SCHEMA_VERSION,
|
|
78
|
+
"recorded_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
|
|
79
|
+
"gate": gate,
|
|
80
|
+
"verdict": verdict or "FAIL",
|
|
81
|
+
# The falsifiability anchor. A future reader can check whether this still
|
|
82
|
+
# holds instead of taking the lesson on faith.
|
|
83
|
+
"evidence": evidence,
|
|
84
|
+
"run_id": run_id or "",
|
|
85
|
+
# Deliberately absent: any summary, generalization, or "what this means".
|
|
86
|
+
# A judgement stored beside facts becomes indistinguishable from one.
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
p = _path(loki_dir)
|
|
90
|
+
try:
|
|
91
|
+
os.makedirs(os.path.dirname(p), exist_ok=True)
|
|
92
|
+
with open(p, "a", encoding="utf-8") as fh:
|
|
93
|
+
fh.write(json.dumps(rec, sort_keys=True) + "\n")
|
|
94
|
+
except OSError as exc:
|
|
95
|
+
return {"status": UNKNOWN, "reason": "unreadable", "detail": str(exc)}
|
|
96
|
+
return {"status": "recorded", "path": p, "record": rec}
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def recall(loki_dir, gate=None):
|
|
100
|
+
"""What has actually gone wrong here before.
|
|
101
|
+
|
|
102
|
+
Returns counts per gate, most-failed first. Counts, not prose: "the mock
|
|
103
|
+
integrity gate has failed here 6 times" is a fact a reader can act on, while
|
|
104
|
+
"this repo tends to have mocking problems" is a generalization that sounds
|
|
105
|
+
the same and is not checkable.
|
|
106
|
+
"""
|
|
107
|
+
p = _path(loki_dir)
|
|
108
|
+
if not os.path.isfile(p):
|
|
109
|
+
return {"status": UNKNOWN, "reason": "no_lessons",
|
|
110
|
+
"detail": REASONS["no_lessons"]}
|
|
111
|
+
|
|
112
|
+
per_gate, recent, bad = {}, [], 0
|
|
113
|
+
try:
|
|
114
|
+
with open(p, "r", encoding="utf-8") as fh:
|
|
115
|
+
for line in fh:
|
|
116
|
+
line = line.strip()
|
|
117
|
+
if not line:
|
|
118
|
+
continue
|
|
119
|
+
try:
|
|
120
|
+
r = json.loads(line)
|
|
121
|
+
except ValueError:
|
|
122
|
+
# Counted, never silently skipped: a memory that quietly
|
|
123
|
+
# drops what it cannot read is worse than one that admits it.
|
|
124
|
+
bad += 1
|
|
125
|
+
continue
|
|
126
|
+
g = r.get("gate")
|
|
127
|
+
if not g or (gate and g != gate):
|
|
128
|
+
continue
|
|
129
|
+
per_gate[g] = per_gate.get(g, 0) + 1
|
|
130
|
+
recent.append(r)
|
|
131
|
+
except OSError as exc:
|
|
132
|
+
return {"status": UNKNOWN, "reason": "unreadable", "detail": str(exc)}
|
|
133
|
+
|
|
134
|
+
if not per_gate:
|
|
135
|
+
return {"status": UNKNOWN, "reason": "no_lessons",
|
|
136
|
+
"detail": REASONS["no_lessons"]}
|
|
137
|
+
|
|
138
|
+
return {
|
|
139
|
+
"status": "measured",
|
|
140
|
+
"unparseable_lines": bad,
|
|
141
|
+
"by_gate": dict(sorted(per_gate.items(), key=lambda kv: -kv[1])),
|
|
142
|
+
"total": sum(per_gate.values()),
|
|
143
|
+
# Newest last so a reader sees the current state at the bottom, matching
|
|
144
|
+
# how the file itself is ordered.
|
|
145
|
+
"recent": recent[-5:],
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def prompt_context(loki_dir, limit=3):
|
|
150
|
+
"""The lines worth putting in front of the next run.
|
|
151
|
+
|
|
152
|
+
Returns bare facts. A repo where the same gate has failed repeatedly is
|
|
153
|
+
information the next iteration should have; what to DO about it is left to
|
|
154
|
+
the agent, because prescribing the fix from a count would be inventing a
|
|
155
|
+
causal claim the data does not contain.
|
|
156
|
+
"""
|
|
157
|
+
r = recall(loki_dir)
|
|
158
|
+
if r.get("status") != "measured":
|
|
159
|
+
return {"status": r.get("status"), "reason": r.get("reason"), "lines": []}
|
|
160
|
+
lines = []
|
|
161
|
+
for gate, n in list(r["by_gate"].items())[:limit]:
|
|
162
|
+
if n > 1:
|
|
163
|
+
lines.append(f"the {gate} gate has failed {n} times in this repo before")
|
|
164
|
+
else:
|
|
165
|
+
lines.append(f"the {gate} gate has failed here before")
|
|
166
|
+
return {"status": "measured", "lines": lines}
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def main(argv):
|
|
170
|
+
if not argv:
|
|
171
|
+
print("usage: failure_memory.py record|recall|context [...]", file=sys.stderr)
|
|
172
|
+
return 2
|
|
173
|
+
action = argv[0]
|
|
174
|
+
cwd = os.environ.get("LOKI_FAILMEM_CWD") or os.getcwd()
|
|
175
|
+
loki_dir = os.environ.get("LOKI_DIR") or os.path.join(cwd, ".loki")
|
|
176
|
+
|
|
177
|
+
kv = {}
|
|
178
|
+
for a in argv[1:]:
|
|
179
|
+
if a.startswith("--") and "=" in a:
|
|
180
|
+
k, v = a[2:].split("=", 1)
|
|
181
|
+
kv[k] = v
|
|
182
|
+
|
|
183
|
+
if action == "record":
|
|
184
|
+
res = record_failure(loki_dir, kv.get("gate"), kv.get("verdict"),
|
|
185
|
+
kv.get("evidence"), kv.get("run_id"))
|
|
186
|
+
elif action == "recall":
|
|
187
|
+
res = recall(loki_dir, kv.get("gate"))
|
|
188
|
+
elif action == "context":
|
|
189
|
+
res = prompt_context(loki_dir)
|
|
190
|
+
else:
|
|
191
|
+
print(f"unknown action: {action}", file=sys.stderr)
|
|
192
|
+
return 2
|
|
193
|
+
|
|
194
|
+
print(json.dumps(res, indent=2))
|
|
195
|
+
return 0 if res.get("status") in ("recorded", "measured") else 3
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
if __name__ == "__main__":
|
|
199
|
+
sys.exit(main(sys.argv[1:]))
|