loki-mode 9.12.6 → 9.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +81 -101
- package/SKILL.md +2 -2
- package/VERSION +1 -1
- package/autonomy/intent.sh +414 -0
- package/autonomy/issue-providers.sh +24 -0
- package/autonomy/lib/agent_readiness.py +280 -0
- package/autonomy/lib/claim_grounding.py +171 -0
- package/autonomy/lib/config-map.sh +10 -6
- package/autonomy/lib/decision_record.py +198 -0
- package/autonomy/lib/failure_memory.py +199 -0
- package/autonomy/lib/gate_policy.py +166 -0
- package/autonomy/lib/outcome_ledger.py +620 -0
- package/autonomy/lib/preedit_snapshot.py +216 -0
- package/autonomy/lib/proof-generator.py +71 -4
- package/autonomy/lib/verdict.py +204 -0
- package/autonomy/loki +430 -15
- package/autonomy/notify.sh +70 -1
- package/autonomy/provider-offer.sh +25 -1
- package/autonomy/queue-consumer.sh +290 -18
- package/autonomy/run.sh +527 -12
- package/autonomy/telemetry.sh +8 -1
- package/completions/_loki +5 -0
- package/completions/loki.bash +2 -1
- package/dashboard/__init__.py +1 -1
- package/dashboard/run.py +13 -2
- package/dashboard/scim.py +221 -0
- package/dashboard/server.py +190 -1
- package/dashboard/static/index.html +248 -55
- package/docs/GATE-FAILURE-TRIAGE.md +254 -0
- package/docs/LOOP-CANDIDATE-PROPOSAL-v1.md +167 -0
- package/docs/LOOP-HARNESS-AUDIT.md +53 -0
- package/docs/QUEUE-OPERATIONS.md +107 -0
- package/docs/VERIFICATION-COST.md +273 -0
- package/docs/WANG-PRINCIPLES-PLAN.md +1 -1
- package/loki-ts/dist/loki.js +414 -416
- package/mcp/__init__.py +1 -1
- package/mcp/_sdk_loader.py +25 -0
- package/package.json +1 -1
- package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Agent readiness: can an autonomous agent verify its own work in THIS repo?
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS, AND WHY IT IS NOT A COPY. Factory AI's Agent Readiness Model is
|
|
5
|
+
a genuinely good idea and a category-defining artifact -- 5 levels, 9 pillars,
|
|
6
|
+
2 scopes -- and it makes competitor comparisons happen on Factory's chosen axes.
|
|
7
|
+
It is also LLM-SCORED: their report objects record `modelUsed` and
|
|
8
|
+
`reasoningEffort` per report. So the number is a model's opinion of a repo, and
|
|
9
|
+
two runs can disagree about the same commit.
|
|
10
|
+
|
|
11
|
+
Ours is a measurement. Every criterion below is a file that exists or does not,
|
|
12
|
+
a command that is present or absent. Same commit, same answer, every time, on
|
|
13
|
+
any machine, with no key and no spend. "Theirs is an opinion, ours is a
|
|
14
|
+
measurement, here is the command" is the same wedge as the receipt, applied to
|
|
15
|
+
their own differentiated concept.
|
|
16
|
+
|
|
17
|
+
WHAT IT MEASURES, AND WHY THOSE. Not general code quality -- that is what
|
|
18
|
+
`loki modernize heal --assess` already scores with its own deterministic 4-level
|
|
19
|
+
maturity rubric, and duplicating it would create two numbers that eventually
|
|
20
|
+
disagree. This asks the narrower question our product actually depends on:
|
|
21
|
+
CAN AN AGENT CHECK ITSELF HERE? Factory concedes the same dependency from the
|
|
22
|
+
other side -- their Missions docs say that without "an automated, scriptable way
|
|
23
|
+
to exercise the app... the mission cannot reliably verify its own work", and
|
|
24
|
+
recommend Level 4+ before using their flagship. A repo with no test command is
|
|
25
|
+
one where every agent, ours included, is guessing.
|
|
26
|
+
|
|
27
|
+
WHAT IT REFUSES. No percentage, no letter grade, no composite. A composite
|
|
28
|
+
invites ranking, ranking invites gaming, and the individual signals are the
|
|
29
|
+
actionable part: "there is no test command" tells you what to do, "readiness 62%"
|
|
30
|
+
does not. Criteria that cannot be determined report UNKNOWN by name rather than
|
|
31
|
+
counting as failures -- an absent measurement is not a bad score.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
from __future__ import annotations
|
|
35
|
+
|
|
36
|
+
import json
|
|
37
|
+
import os
|
|
38
|
+
import sys
|
|
39
|
+
|
|
40
|
+
SCHEMA_VERSION = "1.0"
|
|
41
|
+
|
|
42
|
+
UNKNOWN = "UNKNOWN"
|
|
43
|
+
|
|
44
|
+
# Each criterion is a pure filesystem fact plus the command a reader can run to
|
|
45
|
+
# check it themselves. The `why` is not decoration: a signal whose consequence
|
|
46
|
+
# for an agent is unstated becomes a checkbox someone games.
|
|
47
|
+
CRITERIA = [
|
|
48
|
+
{
|
|
49
|
+
"id": "test_command",
|
|
50
|
+
"why": "without a runnable test command an agent cannot verify its own change",
|
|
51
|
+
"verify": "look for a test script in package.json, a Makefile test target, pytest.ini, or tests/",
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"id": "dependency_lock",
|
|
55
|
+
"why": "unpinned dependencies make a green run unreproducible tomorrow",
|
|
56
|
+
"verify": "look for package-lock.json, bun.lockb, poetry.lock, requirements.txt, Cargo.lock, go.sum",
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
"id": "ci_config",
|
|
60
|
+
"why": "without CI, nothing re-checks the agent's work independently of the agent",
|
|
61
|
+
"verify": "look for .github/workflows, .gitlab-ci.yml, or a CI config at the repo root",
|
|
62
|
+
},
|
|
63
|
+
{
|
|
64
|
+
"id": "agent_brief",
|
|
65
|
+
"why": "without a briefing file an agent rediscovers conventions every run and gets them wrong",
|
|
66
|
+
"verify": "look for AGENTS.md, CLAUDE.md, CONTRIBUTING.md",
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
"id": "readme",
|
|
70
|
+
"why": "without a README an agent has no statement of what the project is for",
|
|
71
|
+
"verify": "look for README.md or README",
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
"id": "gitignore",
|
|
75
|
+
"why": "without ignores an agent's diff fills with build output and the real change is buried",
|
|
76
|
+
"verify": "look for .gitignore",
|
|
77
|
+
},
|
|
78
|
+
]
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _any_exists(root, names):
|
|
82
|
+
for n in names:
|
|
83
|
+
if os.path.exists(os.path.join(root, n)):
|
|
84
|
+
return n
|
|
85
|
+
return None
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _has_test_command(root):
|
|
89
|
+
pkg = os.path.join(root, "package.json")
|
|
90
|
+
if os.path.isfile(pkg):
|
|
91
|
+
try:
|
|
92
|
+
with open(pkg, "r", encoding="utf-8") as fh:
|
|
93
|
+
data = json.load(fh)
|
|
94
|
+
if (data.get("scripts") or {}).get("test"):
|
|
95
|
+
return "package.json scripts.test"
|
|
96
|
+
except (OSError, ValueError):
|
|
97
|
+
# A malformed package.json is not evidence either way. Fall through
|
|
98
|
+
# to the other signals rather than scoring it as absent.
|
|
99
|
+
pass
|
|
100
|
+
found = _any_exists(root, ["pytest.ini", "tox.ini", "Makefile", "tests", "test"])
|
|
101
|
+
return found
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def assess(root):
|
|
105
|
+
"""Evaluate every criterion. Returns facts, never a score."""
|
|
106
|
+
if not os.path.isdir(root):
|
|
107
|
+
return {"status": UNKNOWN, "reason": "no_such_directory", "path": root}
|
|
108
|
+
if not os.path.isdir(os.path.join(root, ".git")):
|
|
109
|
+
# Not fatal: readiness is about the working tree. Recorded so a reader
|
|
110
|
+
# knows the repo context was absent rather than assumed.
|
|
111
|
+
git_present = False
|
|
112
|
+
else:
|
|
113
|
+
git_present = True
|
|
114
|
+
|
|
115
|
+
checks = []
|
|
116
|
+
for c in CRITERIA:
|
|
117
|
+
cid = c["id"]
|
|
118
|
+
if cid == "test_command":
|
|
119
|
+
hit = _has_test_command(root)
|
|
120
|
+
elif cid == "dependency_lock":
|
|
121
|
+
hit = _any_exists(root, ["package-lock.json", "bun.lockb", "yarn.lock",
|
|
122
|
+
"poetry.lock", "requirements.txt", "Cargo.lock",
|
|
123
|
+
"go.sum", "Pipfile.lock"])
|
|
124
|
+
elif cid == "ci_config":
|
|
125
|
+
hit = _any_exists(root, [".github/workflows", ".gitlab-ci.yml",
|
|
126
|
+
".circleci", "azure-pipelines.yml", "Jenkinsfile"])
|
|
127
|
+
elif cid == "agent_brief":
|
|
128
|
+
hit = _any_exists(root, ["AGENTS.md", "CLAUDE.md", "CONTRIBUTING.md"])
|
|
129
|
+
elif cid == "readme":
|
|
130
|
+
hit = _any_exists(root, ["README.md", "README", "README.rst"])
|
|
131
|
+
elif cid == "gitignore":
|
|
132
|
+
hit = _any_exists(root, [".gitignore"])
|
|
133
|
+
else:
|
|
134
|
+
hit = None
|
|
135
|
+
|
|
136
|
+
checks.append({
|
|
137
|
+
"id": cid,
|
|
138
|
+
"present": bool(hit),
|
|
139
|
+
"found": hit or None,
|
|
140
|
+
"why": c["why"],
|
|
141
|
+
"verify": c["verify"],
|
|
142
|
+
})
|
|
143
|
+
|
|
144
|
+
present = [c for c in checks if c["present"]]
|
|
145
|
+
missing = [c for c in checks if not c["present"]]
|
|
146
|
+
|
|
147
|
+
return {
|
|
148
|
+
"schema_version": SCHEMA_VERSION,
|
|
149
|
+
"status": "measured",
|
|
150
|
+
"path": os.path.abspath(root),
|
|
151
|
+
"git_repo": git_present,
|
|
152
|
+
# Counts, not a percentage. A composite invites ranking, ranking invites
|
|
153
|
+
# gaming, and "there is no test command" is the actionable part anyway.
|
|
154
|
+
"criteria_total": len(checks),
|
|
155
|
+
"criteria_present": len(present),
|
|
156
|
+
"checks": checks,
|
|
157
|
+
"missing": [c["id"] for c in missing],
|
|
158
|
+
# The single most consequential signal, surfaced on its own: this is the
|
|
159
|
+
# one Factory's own docs concede their flagship depends on.
|
|
160
|
+
"can_self_verify": any(c["id"] == "test_command" and c["present"]
|
|
161
|
+
for c in checks),
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def render_text(res):
|
|
166
|
+
if res.get("status") != "measured":
|
|
167
|
+
return f"Agent readiness: UNKNOWN ({res.get('reason', 'unmeasurable')})"
|
|
168
|
+
out = ["Agent readiness -- can an agent verify its own work here?", ""]
|
|
169
|
+
for c in res["checks"]:
|
|
170
|
+
mark = "yes" if c["present"] else "NO "
|
|
171
|
+
line = f" {mark} {c['id']:20}"
|
|
172
|
+
if c["present"]:
|
|
173
|
+
line += f"({c['found']})"
|
|
174
|
+
else:
|
|
175
|
+
line += c["why"]
|
|
176
|
+
out.append(line)
|
|
177
|
+
out.append("")
|
|
178
|
+
out.append(f" {res['criteria_present']} of {res['criteria_total']} present")
|
|
179
|
+
if not res["can_self_verify"]:
|
|
180
|
+
out.append("")
|
|
181
|
+
out.append(" No test command found. Every agent working here, ours")
|
|
182
|
+
out.append(" included, is guessing whether its change worked.")
|
|
183
|
+
out.append("")
|
|
184
|
+
out.append(" Every line above is a file that exists or does not. Check any of")
|
|
185
|
+
out.append(" them by hand; no model was asked for an opinion.")
|
|
186
|
+
return "\n".join(out)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
# What --fix writes for each missing criterion. Only files whose CORRECT content
|
|
190
|
+
# can be derived from the repo itself appear here.
|
|
191
|
+
#
|
|
192
|
+
# test_command and dependency_lock are deliberately absent: guessing a test
|
|
193
|
+
# command writes a line that lies (`npm test` in a repo with no runner exits
|
|
194
|
+
# non-zero forever, and the readiness check would then report "present" for
|
|
195
|
+
# something that does not work), and a lockfile must come from the real package
|
|
196
|
+
# manager or it is worse than none. Those stay REPORTED, never generated.
|
|
197
|
+
#
|
|
198
|
+
# This is the difference between a scorecard and an executable one -- Factory's
|
|
199
|
+
# /readiness-report -> /readiness-fix -- without the failure mode where the fix
|
|
200
|
+
# makes the score green while the underlying capability is still missing.
|
|
201
|
+
FIXABLE = {
|
|
202
|
+
"gitignore": (".gitignore", "node_modules/\n__pycache__/\n.env\n.venv/\ndist/\n*.log\n"),
|
|
203
|
+
"readme": ("README.md", None), # content derived below
|
|
204
|
+
"agent_brief": ("AGENTS.md", None), # content derived below
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _fix(root, res):
|
|
209
|
+
"""Create the missing files whose content can be derived honestly.
|
|
210
|
+
|
|
211
|
+
Returns (written, skipped) where skipped names criteria that a generated
|
|
212
|
+
file could not honestly satisfy.
|
|
213
|
+
"""
|
|
214
|
+
written, skipped = [], []
|
|
215
|
+
project = os.path.basename(os.path.abspath(root)) or "this project"
|
|
216
|
+
# assess() reports `missing` as a list of criterion IDs (strings), not the
|
|
217
|
+
# full check dicts. Tolerate both so a later shape change does not silently
|
|
218
|
+
# fix nothing.
|
|
219
|
+
for entry in res.get("missing", []):
|
|
220
|
+
cid = entry["id"] if isinstance(entry, dict) else entry
|
|
221
|
+
if cid not in FIXABLE:
|
|
222
|
+
skipped.append((cid, "must come from the real toolchain, not a guess"))
|
|
223
|
+
continue
|
|
224
|
+
name, body = FIXABLE[cid]
|
|
225
|
+
path = os.path.join(root, name)
|
|
226
|
+
if os.path.exists(path):
|
|
227
|
+
continue
|
|
228
|
+
if cid == "readme":
|
|
229
|
+
body = (f"# {project}\n\n"
|
|
230
|
+
"## What this is\n\n_TODO: one paragraph._\n\n"
|
|
231
|
+
"## Run it\n\n```sh\n# TODO: the command that starts this project\n```\n\n"
|
|
232
|
+
"## Test it\n\n```sh\n# TODO: the command that runs the tests\n```\n")
|
|
233
|
+
elif cid == "agent_brief":
|
|
234
|
+
body = (f"# AGENTS.md\n\nBriefing for coding agents working in {project}.\n\n"
|
|
235
|
+
"## Commands\n\n- Build: _TODO_\n- Test: _TODO_\n- Lint: _TODO_\n\n"
|
|
236
|
+
"## Conventions\n\n_TODO: what a reviewer would flag._\n\n"
|
|
237
|
+
"## Do not\n\n_TODO: the things that break this repo._\n")
|
|
238
|
+
try:
|
|
239
|
+
with open(path, "w") as fh:
|
|
240
|
+
fh.write(body)
|
|
241
|
+
written.append(name)
|
|
242
|
+
except OSError as exc:
|
|
243
|
+
skipped.append((cid, f"could not write {name}: {exc}"))
|
|
244
|
+
return written, skipped
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def main(argv):
|
|
248
|
+
as_json = "--json" in argv
|
|
249
|
+
do_fix = "--fix" in argv
|
|
250
|
+
root = "."
|
|
251
|
+
for a in argv:
|
|
252
|
+
if not a.startswith("-"):
|
|
253
|
+
root = a
|
|
254
|
+
break
|
|
255
|
+
res = assess(root)
|
|
256
|
+
|
|
257
|
+
if do_fix and res.get("status") == "measured":
|
|
258
|
+
written, skipped = _fix(root, res)
|
|
259
|
+
res = assess(root) # re-measure: report what is true AFTER the fix
|
|
260
|
+
res["fix"] = {"written": written,
|
|
261
|
+
"skipped": [{"id": i, "reason": r} for i, r in skipped]}
|
|
262
|
+
|
|
263
|
+
if as_json:
|
|
264
|
+
print(json.dumps(res, indent=2))
|
|
265
|
+
else:
|
|
266
|
+
print(render_text(res))
|
|
267
|
+
if do_fix:
|
|
268
|
+
fx = res.get("fix", {})
|
|
269
|
+
print("")
|
|
270
|
+
for name in fx.get("written", []):
|
|
271
|
+
print(f" wrote {name} (a stub -- fill in the TODOs)")
|
|
272
|
+
for s in fx.get("skipped", []):
|
|
273
|
+
print(f" skipped {s['id']}: {s['reason']}")
|
|
274
|
+
if not fx.get("written") and not fx.get("skipped"):
|
|
275
|
+
print(" nothing to fix.")
|
|
276
|
+
return 0 if res.get("status") == "measured" else 3
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
if __name__ == "__main__":
|
|
280
|
+
sys.exit(main(sys.argv[1:]))
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Claim grounding: a completion claim must name work that exists in the diff.
|
|
3
|
+
|
|
4
|
+
WHY THIS EXISTS. 8090 AI's evaluation framework blocks ungrounded output AT
|
|
5
|
+
GENERATION rather than catching it at review, and states the payoff plainly:
|
|
6
|
+
|
|
7
|
+
"Ungrounded claims are blocked at generation. The rubric measures the quality
|
|
8
|
+
of what survives the architectural filter, which is a much smaller and more
|
|
9
|
+
interesting space."
|
|
10
|
+
|
|
11
|
+
Our evidence gate already has six axes -- diff non-empty, tests green, runtime
|
|
12
|
+
boot, no-mock, authorization, secret leak. Every one is a REPO-LEVEL fact. None
|
|
13
|
+
reads what the agent actually CLAIMED. So an agent can finish by asserting "added
|
|
14
|
+
retry logic to the payment client and covered it with tests" while the diff shows
|
|
15
|
+
a README edit, and every axis passes: the diff is non-empty, the tests are green,
|
|
16
|
+
the app boots. The claim itself is the one artifact nobody checks.
|
|
17
|
+
|
|
18
|
+
WHAT THIS CHECKS, AND WHAT IT REFUSES TO CHECK. It resolves file-path-shaped
|
|
19
|
+
tokens in the claim against the actual changed-file set. That is a deterministic
|
|
20
|
+
string-to-set comparison, and it is the only part of a natural-language claim
|
|
21
|
+
that can be checked without a model.
|
|
22
|
+
|
|
23
|
+
It does NOT judge whether the claim is semantically true. "Added retry logic" vs
|
|
24
|
+
"added a retry constant" is a judgement, and asking an LLM to grade it would be
|
|
25
|
+
the same LLM-judge-as-measurement this codebase refuses everywhere else. A claim
|
|
26
|
+
naming no paths is UNGROUNDABLE, reported by name -- never scored, never failed.
|
|
27
|
+
|
|
28
|
+
THE FAIL-OPEN DIRECTION IS DELIBERATE. Only a claim that names a path which is
|
|
29
|
+
demonstrably NOT in the diff is a finding. A claim with no paths, a claim naming
|
|
30
|
+
a path that exists but was not touched by this run, and an empty claim are all
|
|
31
|
+
reported and none of them blocks. A grounding check that blocked on ambiguity
|
|
32
|
+
would fire constantly on ordinary prose and be disabled within a week, which is
|
|
33
|
+
worse than not having it.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
import json
|
|
39
|
+
import os
|
|
40
|
+
import re
|
|
41
|
+
import sys
|
|
42
|
+
|
|
43
|
+
SCHEMA_VERSION = "1.0"
|
|
44
|
+
|
|
45
|
+
UNKNOWN = "UNKNOWN"
|
|
46
|
+
|
|
47
|
+
GROUNDING_REASONS = {
|
|
48
|
+
"no_claim": "no completion claim text was provided",
|
|
49
|
+
"no_diff": "no changed-file set was provided to check against",
|
|
50
|
+
"ungroundable": "the claim names no file paths, so it cannot be checked mechanically",
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
# A path-shaped token: at least one slash or a known source extension, and no
|
|
54
|
+
# spaces. Deliberately conservative -- a false "this is a path" produces a false
|
|
55
|
+
# finding, and a check that cries wolf gets turned off.
|
|
56
|
+
_PATH_RE = re.compile(
|
|
57
|
+
r"(?<![\w/.-])"
|
|
58
|
+
r"(?:[\w.-]+/)+[\w.-]+\.[A-Za-z0-9]{1,6}"
|
|
59
|
+
r"|(?<![\w/.-])[\w.-]+\.(?:py|ts|tsx|js|jsx|sh|go|rs|rb|java|c|h|cpp|md|json|ya?ml|toml)"
|
|
60
|
+
r"(?![\w/.-])"
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def extract_paths(claim):
|
|
65
|
+
"""Path-shaped tokens in a claim, deduped, order preserved."""
|
|
66
|
+
if not claim:
|
|
67
|
+
return []
|
|
68
|
+
seen, out = set(), []
|
|
69
|
+
for m in _PATH_RE.finditer(claim):
|
|
70
|
+
tok = m.group(0).strip(".,;:)(\"'`")
|
|
71
|
+
if tok and tok not in seen:
|
|
72
|
+
seen.add(tok)
|
|
73
|
+
out.append(tok)
|
|
74
|
+
return out
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def check(claim, changed_files):
|
|
78
|
+
"""Are the paths a claim names present in the diff?
|
|
79
|
+
|
|
80
|
+
Matching is suffix-based on purpose: an agent writes `run.sh` or
|
|
81
|
+
`autonomy/run.sh` for the same file, and demanding an exact repo-relative
|
|
82
|
+
string would make every informal mention a false finding.
|
|
83
|
+
"""
|
|
84
|
+
if not claim or not claim.strip():
|
|
85
|
+
return {"status": UNKNOWN, "reason": "no_claim",
|
|
86
|
+
"detail": GROUNDING_REASONS["no_claim"]}
|
|
87
|
+
if changed_files is None:
|
|
88
|
+
return {"status": UNKNOWN, "reason": "no_diff",
|
|
89
|
+
"detail": GROUNDING_REASONS["no_diff"]}
|
|
90
|
+
|
|
91
|
+
named = extract_paths(claim)
|
|
92
|
+
if not named:
|
|
93
|
+
# The common, benign case: "fixed the login bug". Nothing to check, and
|
|
94
|
+
# that is not a defect -- reported so the number of unverifiable claims
|
|
95
|
+
# is visible, rather than silently counted as grounded.
|
|
96
|
+
return {"status": UNKNOWN, "reason": "ungroundable",
|
|
97
|
+
"detail": GROUNDING_REASONS["ungroundable"],
|
|
98
|
+
"paths_named": []}
|
|
99
|
+
|
|
100
|
+
changed = list(changed_files)
|
|
101
|
+
grounded, ungrounded = [], []
|
|
102
|
+
for p in named:
|
|
103
|
+
norm = p.lstrip("./")
|
|
104
|
+
# Basename fallback is load-bearing, not laxity. An agent writes `run.sh`
|
|
105
|
+
# for `autonomy/run.sh`, and requiring the full repo-relative string made
|
|
106
|
+
# every informal mention a false finding -- measured: "patched run.sh"
|
|
107
|
+
# against a changed autonomy/run.sh was reported UNGROUNDED. A grounding
|
|
108
|
+
# check that flags correct claims is worse than none, because it gets
|
|
109
|
+
# disabled and takes the real detections with it.
|
|
110
|
+
base = os.path.basename(norm)
|
|
111
|
+
hit = any(
|
|
112
|
+
c == norm
|
|
113
|
+
or c.endswith("/" + norm)
|
|
114
|
+
or norm.endswith("/" + c)
|
|
115
|
+
or os.path.basename(c) == base
|
|
116
|
+
for c in changed
|
|
117
|
+
)
|
|
118
|
+
(grounded if hit else ungrounded).append(p)
|
|
119
|
+
|
|
120
|
+
return {
|
|
121
|
+
"status": "measured",
|
|
122
|
+
"paths_named": named,
|
|
123
|
+
"grounded": grounded,
|
|
124
|
+
"ungrounded": ungrounded,
|
|
125
|
+
# The single actionable signal. A claim that names a file the run never
|
|
126
|
+
# touched is the "agent says done, diff says otherwise" failure, caught
|
|
127
|
+
# from the claim side instead of the repo side.
|
|
128
|
+
"has_ungrounded_claim": bool(ungrounded),
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def main(argv):
|
|
133
|
+
claim = ""
|
|
134
|
+
files_arg = ""
|
|
135
|
+
files_from = ""
|
|
136
|
+
i = 0
|
|
137
|
+
while i < len(argv):
|
|
138
|
+
if argv[i] == "--claim" and i + 1 < len(argv):
|
|
139
|
+
claim = argv[i + 1]; i += 2; continue
|
|
140
|
+
if argv[i] == "--files" and i + 1 < len(argv):
|
|
141
|
+
files_arg = argv[i + 1]; i += 2; continue
|
|
142
|
+
if argv[i] == "--files-from" and i + 1 < len(argv):
|
|
143
|
+
files_from = argv[i + 1]; i += 2; continue
|
|
144
|
+
i += 1
|
|
145
|
+
|
|
146
|
+
# --files is ALWAYS a comma-separated list; --files-from is always a file to
|
|
147
|
+
# read. The first version overloaded one flag for both and picked with
|
|
148
|
+
# os.path.isfile(), which silently misread `--files autonomy/run.sh` -- a
|
|
149
|
+
# perfectly ordinary changed-file list -- as "open that file and treat its
|
|
150
|
+
# 20,000 lines as filenames". The result was a correct claim reported
|
|
151
|
+
# UNGROUNDED, i.e. the exact false positive this check must never produce.
|
|
152
|
+
# An ambiguous flag whose meaning depends on the filesystem is a bug, not a
|
|
153
|
+
# convenience.
|
|
154
|
+
if files_from and os.path.isfile(files_from):
|
|
155
|
+
with open(files_from, "r", encoding="utf-8") as fh:
|
|
156
|
+
changed = [l.strip() for l in fh if l.strip()]
|
|
157
|
+
elif files_arg:
|
|
158
|
+
changed = [c.strip() for c in files_arg.split(",") if c.strip()]
|
|
159
|
+
else:
|
|
160
|
+
changed = None
|
|
161
|
+
|
|
162
|
+
res = check(claim, changed)
|
|
163
|
+
res["schema_version"] = SCHEMA_VERSION
|
|
164
|
+
print(json.dumps(res, indent=2))
|
|
165
|
+
# Exit 1 ONLY on a demonstrably ungrounded path. Every UNKNOWN exits 0: a
|
|
166
|
+
# check that failed on ambiguity would fire on ordinary prose and be disabled.
|
|
167
|
+
return 1 if res.get("has_ungrounded_claim") else 0
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
if __name__ == "__main__":
|
|
171
|
+
sys.exit(main(sys.argv[1:]))
|
|
@@ -863,13 +863,17 @@ loki_config_validate_file() {
|
|
|
863
863
|
pairs="$(
|
|
864
864
|
_loki_cfg_collect_pairs() {
|
|
865
865
|
local f="$1" fm="$2"
|
|
866
|
+
# NOTE: every case pattern below carries a leading open-paren --
|
|
867
|
+
# bash 3.2 (macOS /bin/bash) ends a command substitution at the
|
|
868
|
+
# close-paren of a case label, truncating this function body. The
|
|
869
|
+
# balanced form parses identically on 3.2 and 4+. Do not strip it.
|
|
866
870
|
case "$fm" in
|
|
867
|
-
env)
|
|
871
|
+
(env)
|
|
868
872
|
local line key val
|
|
869
873
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
870
|
-
case "$line" in ''|'#'*) continue ;; esac
|
|
874
|
+
case "$line" in (''|'#'*) continue ;; esac
|
|
871
875
|
line="${line#export }"
|
|
872
|
-
case "$line" in *=*) ;; *) continue ;; esac
|
|
876
|
+
case "$line" in (*=*) ;; (*) continue ;; esac
|
|
873
877
|
key="${line%%=*}"; val="${line#*=}"
|
|
874
878
|
key="${key#"${key%%[![:space:]]*}"}"; key="${key%"${key##*[![:space:]]}"}"
|
|
875
879
|
val="${val#"${val%%[![:space:]]*}"}"
|
|
@@ -878,7 +882,7 @@ loki_config_validate_file() {
|
|
|
878
882
|
printf '%s\t%s\n' "$key" "$val"
|
|
879
883
|
done < "$f"
|
|
880
884
|
;;
|
|
881
|
-
yaml)
|
|
885
|
+
(yaml)
|
|
882
886
|
local mapping yaml_path env_var value have_yq=0
|
|
883
887
|
if command -v yq >/dev/null 2>&1; then have_yq=1; fi
|
|
884
888
|
for mapping in "${LOKI_CONFIG_MAP[@]}"; do
|
|
@@ -894,8 +898,8 @@ loki_config_validate_file() {
|
|
|
894
898
|
printf '%s\t%s\n' "$env_var" "$value"
|
|
895
899
|
done
|
|
896
900
|
;;
|
|
897
|
-
json)
|
|
898
|
-
# Reuse the
|
|
901
|
+
(json)
|
|
902
|
+
# Reuse the JSON parser emit path via a print-only variant.
|
|
899
903
|
_loki_cfg_json_emit "$f"
|
|
900
904
|
;;
|
|
901
905
|
esac
|