@christang/keel 5.54.0 → 5.56.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +17 -0
- package/assets/bootstrap/AGENTS.md +1 -1
- package/bin/keel.js +23 -1
- package/package.json +1 -1
- package/plugins/keel/.claude-plugin/plugin.json +1 -1
- package/plugins/keel/.codex-plugin/plugin.json +1 -1
- package/scripts/validate_plugin.py +358 -13
- package/src/core/context.js +26 -1
- package/src/core/gates.js +53 -3
package/README.md
CHANGED
|
@@ -333,6 +333,23 @@ keel lenses add web # copy the web template into keel/lenses/web.md, the
|
|
|
333
333
|
keel lenses add web --force # overwrite an existing lens
|
|
334
334
|
```
|
|
335
335
|
|
|
336
|
+
## Re-recording a contract
|
|
337
|
+
|
|
338
|
+
Changing a task's contract after work has started moves its fingerprint, and Keel reports that the
|
|
339
|
+
evidence produced under the old one is stale. Sometimes that is too broad — a classification tag
|
|
340
|
+
added to a check leaves its assertion untouched, and re-running a three-minute experiment for it
|
|
341
|
+
buys nothing. Say so:
|
|
342
|
+
|
|
343
|
+
```bash
|
|
344
|
+
keel gate task-start --change <c> --task <t> --record --keep-evidence M1,M3
|
|
345
|
+
```
|
|
346
|
+
|
|
347
|
+
The report then names only the checks still stale, names the ones you declared unaffected, and says
|
|
348
|
+
the narrowing came from your declaration. **Keel does not verify the claim** — it keeps only the
|
|
349
|
+
previous fingerprint, not the capsule behind it, so it cannot compare a check's former text to its
|
|
350
|
+
current one. State your reason in the task's `Reauthorizations` line, where a reviewer can disagree
|
|
351
|
+
with it. Nothing about completion changes: every check still needs its Evidence.
|
|
352
|
+
|
|
336
353
|
## Pausing a change
|
|
337
354
|
|
|
338
355
|
A change you have deliberately stopped — waiting on something outside the repository, or simply
|
package/bin/keel.js
CHANGED
|
@@ -94,7 +94,8 @@ Usage:
|
|
|
94
94
|
keel capabilities [repo] [--target claude|codex|opencode] [--json]
|
|
95
95
|
keel project [repo] --target claude|codex|opencode --event startup|resume|compaction|goal|task-view|worktree|subagent-start|subagent-stop [--authorize goal|task-view|subagent] [--expected-owner owner] [--native-complete] [--change name] [--task id] [--json]
|
|
96
96
|
keel project tasks [repo] --target claude [--change name] [--json]
|
|
97
|
-
keel gate task-start|task-complete
|
|
97
|
+
keel gate task-start|task-complete [repo] [--change name] [--task id] [--base git-ref] [--no-guard] [--record] [--keep-evidence M1,M3] [--json]
|
|
98
|
+
keel gate change-close [repo] [--change name] --action sync|archive [--base git-ref] [--json]
|
|
98
99
|
keel guard start|status|clear [repo] [--change name] [--task id] [--force] [--json]
|
|
99
100
|
keel lenses list|add [name] [repo] [--force]
|
|
100
101
|
keel triage [repo] [--labels <l1,l2>] [--issue <n>] [--json]
|
|
@@ -186,6 +187,7 @@ function parseArgs(argv) {
|
|
|
186
187
|
base: null,
|
|
187
188
|
noGuard: false,
|
|
188
189
|
record: false,
|
|
190
|
+
keepEvidence: null,
|
|
189
191
|
guardSubcommand: null,
|
|
190
192
|
lensesSubcommand: null,
|
|
191
193
|
lensName: null,
|
|
@@ -296,6 +298,20 @@ function parseArgs(argv) {
|
|
|
296
298
|
parsed.record = true;
|
|
297
299
|
continue;
|
|
298
300
|
}
|
|
301
|
+
if (arg === "--keep-evidence") {
|
|
302
|
+
index += 1;
|
|
303
|
+
if (index >= argv.length) {
|
|
304
|
+
fail("--keep-evidence requires a comma-separated list of M<n> labels");
|
|
305
|
+
}
|
|
306
|
+
if (parsed.keepEvidence !== null) {
|
|
307
|
+
fail("--keep-evidence was provided more than once");
|
|
308
|
+
}
|
|
309
|
+
parsed.keepEvidence = argv[index]
|
|
310
|
+
.split(",")
|
|
311
|
+
.map((entry) => entry.trim())
|
|
312
|
+
.filter(Boolean);
|
|
313
|
+
continue;
|
|
314
|
+
}
|
|
299
315
|
if (arg === "--change" || arg === "--task") {
|
|
300
316
|
index += 1;
|
|
301
317
|
if (index >= argv.length) {
|
|
@@ -1179,6 +1195,8 @@ function keelOpenSpecOverlay(action) {
|
|
|
1179
1195
|
"",
|
|
1180
1196
|
"Keel rules below take precedence over conflicting generic OpenSpec instructions in this file.",
|
|
1181
1197
|
"",
|
|
1198
|
+
"- Invoke the OpenSpec CLI as `keel openspec …` throughout this file. The commands below are written as a bare `openspec`, which resolves only where OpenSpec is separately installed on PATH; `keel openspec` resolves either way, and `keel --doctor` reports which case this repository is.",
|
|
1199
|
+
"",
|
|
1182
1200
|
"### Expectation alignment before specs and tasks finalize",
|
|
1183
1201
|
"",
|
|
1184
1202
|
"- Before specs and executable tasks are finalized, run `keel-align-expectations`: quick path for complete low-risk requests, deep path when a material choice can change user-visible behavior, an external interface, acceptance, security/privacy/permission boundaries, data migration, protocol/state/timing/reset semantics, generated equivalence, irreversible cost, or a dependency commitment.",
|
|
@@ -1214,6 +1232,8 @@ function keelOpenSpecOverlay(action) {
|
|
|
1214
1232
|
"",
|
|
1215
1233
|
"Keel rules below take precedence over conflicting generic OpenSpec instructions in this file.",
|
|
1216
1234
|
"",
|
|
1235
|
+
"- Invoke the OpenSpec CLI as `keel openspec …` throughout this file. The commands below are written as a bare `openspec`, which resolves only where OpenSpec is separately installed on PATH; `keel openspec` resolves either way, and `keel --doctor` reports which case this repository is.",
|
|
1236
|
+
"",
|
|
1217
1237
|
...syncBody,
|
|
1218
1238
|
OPENSPEC_SURFACE_OVERLAY_END,
|
|
1219
1239
|
"",
|
|
@@ -1254,6 +1274,8 @@ function keelOpenSpecOverlay(action) {
|
|
|
1254
1274
|
"",
|
|
1255
1275
|
"Keel rules below take precedence over conflicting generic OpenSpec instructions in this file.",
|
|
1256
1276
|
"",
|
|
1277
|
+
"- Invoke the OpenSpec CLI as `keel openspec …` throughout this file. The commands below are written as a bare `openspec`, which resolves only where OpenSpec is separately installed on PATH; `keel openspec` resolves either way, and `keel --doctor` reports which case this repository is.",
|
|
1278
|
+
"",
|
|
1257
1279
|
"### Target-native subagent gate",
|
|
1258
1280
|
"",
|
|
1259
1281
|
"- The current agent remains responsible for Keel ownership, task/archive decisions, scope control, and final reporting.",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "keel",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.56.0",
|
|
4
4
|
"description": "Keel OpenSpec execution discipline: stateless continuity, task capsules, deterministic gates, and expectation alignment for Codex and Claude Code.",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "TanglmChris",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "keel",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.56.0",
|
|
4
4
|
"description": "Keel OpenSpec execution discipline: stateless continuity, task capsules, deterministic gates, and expectation alignment for Codex and Claude Code.",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "TanglmChris",
|
|
@@ -37,8 +37,8 @@ REQUIRED_SCRIPTS = [
|
|
|
37
37
|
"scripts/validate_plugin.py",
|
|
38
38
|
]
|
|
39
39
|
|
|
40
|
-
PACKAGE_VERSION = "5.
|
|
41
|
-
PROTOCOL_VERSION = "5.
|
|
40
|
+
PACKAGE_VERSION = "5.56.0"
|
|
41
|
+
PROTOCOL_VERSION = "5.56.0"
|
|
42
42
|
LEGACY_MANAGED_START = "<!-- keel:start version=2.1 -->"
|
|
43
43
|
OPENSPEC_SCHEMA_NAME = "keel-spec-driven"
|
|
44
44
|
# Mirrors KEEL_PACKAGE_NAME in scripts/install_to_repo.py, one of the two
|
|
@@ -2163,7 +2163,7 @@ def validate_authoring_continuity_scenario() -> int:
|
|
|
2163
2163
|
or scaffold_payload.get("status") != "ready"
|
|
2164
2164
|
or scaffold_payload.get("selection")
|
|
2165
2165
|
!= {"source": "inferred", "change": "draft", "task": None}
|
|
2166
|
-
or scaffold_payload.get("nextAction"
|
|
2166
|
+
or scaffold_payload.get("nextAction", {}).get("kind") != "author"
|
|
2167
2167
|
):
|
|
2168
2168
|
report(
|
|
2169
2169
|
"authoring-continuity scenario did not keep an incomplete proposal actionable."
|
|
@@ -3747,8 +3747,14 @@ def validate_stateless_continuity_scenario() -> int:
|
|
|
3747
3747
|
"change": "demo",
|
|
3748
3748
|
"task": "1.1",
|
|
3749
3749
|
},
|
|
3750
|
-
"nextAction": {"kind": "task-start"},
|
|
3751
3750
|
}
|
|
3751
|
+
expected_kind = "task-start"
|
|
3752
|
+
if payload.get("nextAction", {}).get("kind") != expected_kind:
|
|
3753
|
+
report(
|
|
3754
|
+
"stateless-continuity scenario explicit selection reported "
|
|
3755
|
+
f"next action {payload.get('nextAction')!r}."
|
|
3756
|
+
)
|
|
3757
|
+
return 1
|
|
3752
3758
|
for key, value in expected.items():
|
|
3753
3759
|
if payload.get(key) != value:
|
|
3754
3760
|
report(
|
|
@@ -3808,7 +3814,7 @@ def validate_stateless_continuity_scenario() -> int:
|
|
|
3808
3814
|
inferred_payload.get("status") != "ready"
|
|
3809
3815
|
or inferred_payload.get("selection")
|
|
3810
3816
|
!= {"source": "inferred", "change": "only", "task": "1.1"}
|
|
3811
|
-
or inferred_payload.get("nextAction"
|
|
3817
|
+
or inferred_payload.get("nextAction", {}).get("kind") != "task-start"
|
|
3812
3818
|
):
|
|
3813
3819
|
report("stateless-continuity scenario unique inference mismatch.")
|
|
3814
3820
|
report(inferred.stdout.strip())
|
|
@@ -3830,7 +3836,8 @@ def validate_stateless_continuity_scenario() -> int:
|
|
|
3830
3836
|
or recomputed_payload.get("status") != "ready"
|
|
3831
3837
|
or recomputed_payload.get("selection")
|
|
3832
3838
|
!= {"source": "inferred", "change": "only", "task": None}
|
|
3833
|
-
or recomputed_payload.get("nextAction"
|
|
3839
|
+
or recomputed_payload.get("nextAction", {}).get("kind")
|
|
3840
|
+
!= "change-close"
|
|
3834
3841
|
):
|
|
3835
3842
|
report("stateless-continuity scenario did not recompute completed state.")
|
|
3836
3843
|
report((recomputed.stderr or recomputed.stdout).strip())
|
|
@@ -3849,7 +3856,7 @@ def validate_stateless_continuity_scenario() -> int:
|
|
|
3849
3856
|
evidence_payload = json.loads(evidence_ready.stdout)
|
|
3850
3857
|
if (
|
|
3851
3858
|
evidence_ready.returncode != 0
|
|
3852
|
-
or evidence_payload.get("nextAction"
|
|
3859
|
+
or evidence_payload.get("nextAction", {}).get("kind") != "task-complete"
|
|
3853
3860
|
):
|
|
3854
3861
|
report("stateless-continuity scenario missed evidence-ready completion.")
|
|
3855
3862
|
report((evidence_ready.stderr or evidence_ready.stdout).strip())
|
|
@@ -3902,7 +3909,7 @@ def validate_stateless_continuity_scenario() -> int:
|
|
|
3902
3909
|
or handoff_payload.get("status") != "ready"
|
|
3903
3910
|
or handoff_payload.get("selection")
|
|
3904
3911
|
!= {"source": "handoff", "change": "beta", "task": "1.1"}
|
|
3905
|
-
or handoff_payload.get("nextAction"
|
|
3912
|
+
or handoff_payload.get("nextAction", {}).get("kind") != "task-start"
|
|
3906
3913
|
):
|
|
3907
3914
|
report("stateless-continuity scenario did not prioritize valid HANDOFF.")
|
|
3908
3915
|
report((handoff.stderr or handoff.stdout).strip())
|
|
@@ -4036,7 +4043,7 @@ def validate_stateless_continuity_scenario() -> int:
|
|
|
4036
4043
|
discuss = run_keel(discuss_repo, "context", "--json")
|
|
4037
4044
|
if (
|
|
4038
4045
|
discuss.returncode != 0
|
|
4039
|
-
or json.loads(discuss.stdout).get("nextAction"
|
|
4046
|
+
or json.loads(discuss.stdout).get("nextAction", {}).get("kind") != "discuss"
|
|
4040
4047
|
):
|
|
4041
4048
|
report("stateless-continuity scenario missed discuss transition.")
|
|
4042
4049
|
report((discuss.stderr or discuss.stdout).strip())
|
|
@@ -4051,7 +4058,7 @@ def validate_stateless_continuity_scenario() -> int:
|
|
|
4051
4058
|
author = run_keel(author_repo, "context", "--json")
|
|
4052
4059
|
if (
|
|
4053
4060
|
author.returncode != 0
|
|
4054
|
-
or json.loads(author.stdout).get("nextAction"
|
|
4061
|
+
or json.loads(author.stdout).get("nextAction", {}).get("kind") != "author"
|
|
4055
4062
|
):
|
|
4056
4063
|
report("stateless-continuity scenario missed author transition.")
|
|
4057
4064
|
report((author.stderr or author.stdout).strip())
|
|
@@ -4069,7 +4076,7 @@ def validate_stateless_continuity_scenario() -> int:
|
|
|
4069
4076
|
idle.returncode != 0
|
|
4070
4077
|
or idle_payload.get("status") != "idle"
|
|
4071
4078
|
or idle_payload.get("selection") is not None
|
|
4072
|
-
or idle_payload.get("nextAction"
|
|
4079
|
+
or idle_payload.get("nextAction", {}).get("kind") != "none"
|
|
4073
4080
|
):
|
|
4074
4081
|
report("stateless-continuity scenario no-work state was not idle.")
|
|
4075
4082
|
report((idle.stderr or idle.stdout).strip())
|
|
@@ -4227,7 +4234,7 @@ def validate_stateless_continuity_scenario() -> int:
|
|
|
4227
4234
|
if (
|
|
4228
4235
|
matching.returncode != 0
|
|
4229
4236
|
or matching_payload.get("status") != "ready"
|
|
4230
|
-
or matching_payload.get("nextAction"
|
|
4237
|
+
or matching_payload.get("nextAction", {}).get("kind") != "task-start"
|
|
4231
4238
|
):
|
|
4232
4239
|
report("stateless-continuity scenario did not resume a matching anchor.")
|
|
4233
4240
|
report((matching.stderr or matching.stdout).strip())
|
|
@@ -4315,7 +4322,7 @@ def validate_stateless_continuity_scenario() -> int:
|
|
|
4315
4322
|
or matching_handoff_payload.get("status") != "ready"
|
|
4316
4323
|
or matching_handoff_payload.get("selection")
|
|
4317
4324
|
!= {"source": "handoff", "change": "demo", "task": "1.1"}
|
|
4318
|
-
or matching_handoff_payload.get("nextAction"
|
|
4325
|
+
or matching_handoff_payload.get("nextAction", {}).get("kind") != "task-start"
|
|
4319
4326
|
):
|
|
4320
4327
|
report("stateless-continuity scenario did not resume a matching HANDOFF anchor.")
|
|
4321
4328
|
report((matching_handoff.stderr or matching_handoff.stdout).strip())
|
|
@@ -25968,6 +25975,341 @@ def validate_paused_change_is_not_the_next_action_scenario() -> int:
|
|
|
25968
25975
|
return 0
|
|
25969
25976
|
|
|
25970
25977
|
|
|
25978
|
+
# Issue #112 recorded four re-verifications in one session from contract
|
|
25979
|
+
# changes that could not affect evidence — renaming `M2:` to `M2 (regression):`
|
|
25980
|
+
# among them, with the assertion unchanged by a character. One of the four meant
|
|
25981
|
+
# breaking a testbench, re-running, and restoring it. The gate genuinely cannot
|
|
25982
|
+
# judge which evidence survives; what it could do is stop saying "all of it".
|
|
25983
|
+
def validate_evidence_survives_what_did_not_change_scenario() -> int:
|
|
25984
|
+
label = "evidence-survives-what-did-not-change"
|
|
25985
|
+
|
|
25986
|
+
def fixture(root: Path, name: str, strategy: str = "evidence-first") -> Path:
|
|
25987
|
+
repo = root / name
|
|
25988
|
+
task = strategy_probe_task(
|
|
25989
|
+
strategy=strategy,
|
|
25990
|
+
reason="fixture; nothing here can fail first",
|
|
25991
|
+
commands=(
|
|
25992
|
+
"M1: the first check asserts the public behavior",
|
|
25993
|
+
"M2: the second check asserts the public behavior",
|
|
25994
|
+
"M3: the third check asserts the public behavior",
|
|
25995
|
+
),
|
|
25996
|
+
)
|
|
25997
|
+
# A recorded anchor that is not the compiled one, so --record re-records.
|
|
25998
|
+
task = task.replace(
|
|
25999
|
+
" - Contract: pending",
|
|
26000
|
+
" - Contract: keel-task-capsule/v1 sha256:" + "0" * 64,
|
|
26001
|
+
)
|
|
26002
|
+
write_gate_fixture(repo, tasks=task)
|
|
26003
|
+
return repo
|
|
26004
|
+
|
|
26005
|
+
def start(repo: Path, *args):
|
|
26006
|
+
result = run_keel(
|
|
26007
|
+
repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
|
|
26008
|
+
"--json", "--no-guard", *args,
|
|
26009
|
+
)
|
|
26010
|
+
try:
|
|
26011
|
+
return json.loads(result.stdout)
|
|
26012
|
+
except json.JSONDecodeError:
|
|
26013
|
+
return {"status": "unparsed", "problems": [{"message": result.stdout[:300]}]}
|
|
26014
|
+
|
|
26015
|
+
def said(payload: dict) -> str:
|
|
26016
|
+
return " ".join(str(x) for x in (payload.get("warnings") or []))
|
|
26017
|
+
|
|
26018
|
+
with tempfile.TemporaryDirectory(prefix="keel-keep-evidence-") as raw:
|
|
26019
|
+
root = Path(raw)
|
|
26020
|
+
|
|
26021
|
+
# Control: without a declaration the report covers every check.
|
|
26022
|
+
blanket = start(fixture(root, "blanket"), "--record")
|
|
26023
|
+
if blanket.get("status") != "pass":
|
|
26024
|
+
report(
|
|
26025
|
+
f"{label}: the re-record fixture did not pass; "
|
|
26026
|
+
f"{problem_text(blanket)!r}."
|
|
26027
|
+
)
|
|
26028
|
+
return 1
|
|
26029
|
+
if "stale" not in said(blanket):
|
|
26030
|
+
report(
|
|
26031
|
+
f"{label}: the fixture did not produce a stale-evidence "
|
|
26032
|
+
f"report to narrow; {said(blanket)!r}."
|
|
26033
|
+
)
|
|
26034
|
+
return 1
|
|
26035
|
+
|
|
26036
|
+
narrowed = start(
|
|
26037
|
+
fixture(root, "narrowed"), "--record", "--keep-evidence", "M1,M3"
|
|
26038
|
+
)
|
|
26039
|
+
if narrowed.get("status") != "pass":
|
|
26040
|
+
report(
|
|
26041
|
+
f"{label}: a declaration refused a valid re-record; "
|
|
26042
|
+
f"{problem_text(narrowed)!r}."
|
|
26043
|
+
)
|
|
26044
|
+
return 1
|
|
26045
|
+
spoken = said(narrowed)
|
|
26046
|
+
stale_line = next(
|
|
26047
|
+
(w for w in narrowed["warnings"] if "stale" in str(w)), ""
|
|
26048
|
+
)
|
|
26049
|
+
if "M2" not in stale_line:
|
|
26050
|
+
report(
|
|
26051
|
+
f"{label}: the narrowed report does not name the check that is "
|
|
26052
|
+
f"still stale; {stale_line!r}."
|
|
26053
|
+
)
|
|
26054
|
+
return 1
|
|
26055
|
+
for kept in ("M1", "M3"):
|
|
26056
|
+
if kept not in stale_line:
|
|
26057
|
+
report(
|
|
26058
|
+
f"{label}: the narrowed report does not name {kept} as "
|
|
26059
|
+
f"declared unaffected; {stale_line!r}."
|
|
26060
|
+
)
|
|
26061
|
+
return 1
|
|
26062
|
+
if "declar" not in stale_line.lower():
|
|
26063
|
+
report(
|
|
26064
|
+
f"{label}: the narrowed report does not attribute the "
|
|
26065
|
+
f"narrowing to the declaration; {stale_line!r}."
|
|
26066
|
+
)
|
|
26067
|
+
return 1
|
|
26068
|
+
|
|
26069
|
+
unknown = start(
|
|
26070
|
+
fixture(root, "unknown"), "--record", "--keep-evidence", "M9"
|
|
26071
|
+
)
|
|
26072
|
+
if unknown.get("status") != "fail":
|
|
26073
|
+
report(
|
|
26074
|
+
f"{label}: a declaration naming a check the contract does not "
|
|
26075
|
+
"declare was accepted; task-start returned "
|
|
26076
|
+
f"{unknown.get('status')!r}."
|
|
26077
|
+
)
|
|
26078
|
+
return 1
|
|
26079
|
+
if "M9" not in problem_text(unknown):
|
|
26080
|
+
report(
|
|
26081
|
+
f"{label}: the refusal does not name the label it could not "
|
|
26082
|
+
f"resolve; {problem_text(unknown)!r}."
|
|
26083
|
+
)
|
|
26084
|
+
return 1
|
|
26085
|
+
|
|
26086
|
+
stray = start(fixture(root, "stray"), "--keep-evidence", "M1")
|
|
26087
|
+
if stray.get("status") != "fail":
|
|
26088
|
+
report(
|
|
26089
|
+
f"{label}: --keep-evidence without --record was accepted, so "
|
|
26090
|
+
"an author could believe they declared something nothing read."
|
|
26091
|
+
)
|
|
26092
|
+
return 1
|
|
26093
|
+
|
|
26094
|
+
# Completion is unchanged: the declaration is about the past.
|
|
26095
|
+
completing = fixture(root, "completing", strategy="vertical-tdd")
|
|
26096
|
+
start(completing, "--record", "--keep-evidence", "M1,M2,M3")
|
|
26097
|
+
tasks_path = completing / "openspec/changes/demo/tasks.md"
|
|
26098
|
+
tasks_path.write_text(
|
|
26099
|
+
tasks_path.read_text(encoding="utf-8")
|
|
26100
|
+
.replace("- [ ] 1.1", "- [x] 1.1")
|
|
26101
|
+
.replace(" - M1: pending", " - M1: pass. ran it.")
|
|
26102
|
+
.replace(" - M2: pending", " - M2: pass. ran it.")
|
|
26103
|
+
.replace(" - M3: pending", " - M3: pass. ran it."),
|
|
26104
|
+
encoding="utf-8",
|
|
26105
|
+
)
|
|
26106
|
+
done = json.loads(
|
|
26107
|
+
run_keel(
|
|
26108
|
+
completing, "gate", "task-complete", "--change", "demo",
|
|
26109
|
+
"--task", "1.1", "--json",
|
|
26110
|
+
).stdout
|
|
26111
|
+
)
|
|
26112
|
+
if done.get("status") != "fail":
|
|
26113
|
+
report(
|
|
26114
|
+
f"{label}: a declaration let a red-green task complete without "
|
|
26115
|
+
"its .red/.green Evidence."
|
|
26116
|
+
)
|
|
26117
|
+
return 1
|
|
26118
|
+
if "missing-strategy-evidence" not in problem_codes(done):
|
|
26119
|
+
report(
|
|
26120
|
+
f"{label}: completion stopped requiring red-green Evidence; "
|
|
26121
|
+
f"{problem_codes(done)!r}."
|
|
26122
|
+
)
|
|
26123
|
+
return 1
|
|
26124
|
+
|
|
26125
|
+
report(f"{label} scenario passed.")
|
|
26126
|
+
return 0
|
|
26127
|
+
|
|
26128
|
+
|
|
26129
|
+
# `keel context` exists to answer "what now" and answered with a noun. Issue
|
|
26130
|
+
# #112: `Next action: change-close` followed by `keel gate change-close
|
|
26131
|
+
# --change x` failing on the argument that stage requires — a wrong attempt
|
|
26132
|
+
# Keel had everything in the same result to prevent.
|
|
26133
|
+
def validate_next_action_is_a_command_scenario() -> int:
|
|
26134
|
+
label = "the-next-action-is-a-command"
|
|
26135
|
+
|
|
26136
|
+
def repo_with(root: Path, name: str, *, checked: bool, evidence: bool):
|
|
26137
|
+
repo = root / name
|
|
26138
|
+
task = strategy_probe_task(
|
|
26139
|
+
strategy="evidence-first",
|
|
26140
|
+
reason="fixture; nothing here can fail first",
|
|
26141
|
+
)
|
|
26142
|
+
if checked:
|
|
26143
|
+
task = task.replace("- [ ] 1.1", "- [x] 1.1")
|
|
26144
|
+
if evidence:
|
|
26145
|
+
task = task.replace(" - M1: pending", " - M1: pass. ran it.")
|
|
26146
|
+
write_gate_fixture(repo, tasks=task)
|
|
26147
|
+
started = run_keel(
|
|
26148
|
+
repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
|
|
26149
|
+
"--json", "--no-guard",
|
|
26150
|
+
)
|
|
26151
|
+
value = json.loads(started.stdout)["contract"]["fingerprint"]["value"]
|
|
26152
|
+
tasks_path = repo / "openspec/changes/demo/tasks.md"
|
|
26153
|
+
tasks_path.write_text(
|
|
26154
|
+
tasks_path.read_text(encoding="utf-8").replace(
|
|
26155
|
+
" - Contract: pending",
|
|
26156
|
+
f" - Contract: keel-task-capsule/v1 sha256:{value}",
|
|
26157
|
+
),
|
|
26158
|
+
encoding="utf-8",
|
|
26159
|
+
)
|
|
26160
|
+
return repo
|
|
26161
|
+
|
|
26162
|
+
def context(repo: Path):
|
|
26163
|
+
text = run_keel(repo, "context", "--change", "demo").stdout
|
|
26164
|
+
payload = json.loads(
|
|
26165
|
+
run_keel(repo, "context", "--change", "demo", "--json").stdout
|
|
26166
|
+
)
|
|
26167
|
+
return text, payload
|
|
26168
|
+
|
|
26169
|
+
with tempfile.TemporaryDirectory(prefix="keel-next-command-") as raw:
|
|
26170
|
+
root = Path(raw)
|
|
26171
|
+
|
|
26172
|
+
pending = repo_with(root, "pending", checked=False, evidence=False)
|
|
26173
|
+
text, payload = context(pending)
|
|
26174
|
+
command = (payload.get("nextAction") or {}).get("command")
|
|
26175
|
+
if not command:
|
|
26176
|
+
report(
|
|
26177
|
+
f"{label}: the next action carries no command; "
|
|
26178
|
+
f"{payload.get('nextAction')!r}."
|
|
26179
|
+
)
|
|
26180
|
+
return 1
|
|
26181
|
+
for needed in ("task-start", "--change demo", "--task 1.1"):
|
|
26182
|
+
if needed not in command:
|
|
26183
|
+
report(
|
|
26184
|
+
f"{label}: the reported command omits {needed!r}; "
|
|
26185
|
+
f"{command!r}."
|
|
26186
|
+
)
|
|
26187
|
+
return 1
|
|
26188
|
+
if command not in text:
|
|
26189
|
+
report(
|
|
26190
|
+
f"{label}: the text surface does not carry the same command as "
|
|
26191
|
+
f"--json; {command!r} not in {text!r}."
|
|
26192
|
+
)
|
|
26193
|
+
return 1
|
|
26194
|
+
|
|
26195
|
+
done = repo_with(root, "done", checked=False, evidence=True)
|
|
26196
|
+
_, payload = context(done)
|
|
26197
|
+
if "task-complete" not in str((payload.get("nextAction") or {}).get("command")):
|
|
26198
|
+
report(
|
|
26199
|
+
f"{label}: a task with completion evidence did not report the "
|
|
26200
|
+
f"task-complete invocation; {payload.get('nextAction')!r}."
|
|
26201
|
+
)
|
|
26202
|
+
return 1
|
|
26203
|
+
|
|
26204
|
+
closing = repo_with(root, "closing", checked=True, evidence=True)
|
|
26205
|
+
_, payload = context(closing)
|
|
26206
|
+
close_command = str((payload.get("nextAction") or {}).get("command") or "")
|
|
26207
|
+
if "change-close" not in close_command:
|
|
26208
|
+
report(
|
|
26209
|
+
f"{label}: a change whose tasks are complete did not report the "
|
|
26210
|
+
f"change-close invocation; {payload.get('nextAction')!r}."
|
|
26211
|
+
)
|
|
26212
|
+
return 1
|
|
26213
|
+
if "--action" not in close_command:
|
|
26214
|
+
report(
|
|
26215
|
+
f"{label}: the change-close command omits the argument that "
|
|
26216
|
+
f"stage requires; {close_command!r}."
|
|
26217
|
+
)
|
|
26218
|
+
return 1
|
|
26219
|
+
# Running exactly what was printed must not hit the input error the
|
|
26220
|
+
# report describes.
|
|
26221
|
+
printed = close_command.split()
|
|
26222
|
+
if printed and printed[0] == "keel":
|
|
26223
|
+
printed = printed[1:]
|
|
26224
|
+
ran = run_keel(closing, *printed)
|
|
26225
|
+
if "requires --action" in (ran.stdout + ran.stderr):
|
|
26226
|
+
report(
|
|
26227
|
+
f"{label}: the reported command still fails on the argument it "
|
|
26228
|
+
f"was supposed to supply; {(ran.stdout + ran.stderr)[:200]!r}."
|
|
26229
|
+
)
|
|
26230
|
+
return 1
|
|
26231
|
+
|
|
26232
|
+
idle = root / "idle"
|
|
26233
|
+
write_text(idle / "README.md", "# fixture\n")
|
|
26234
|
+
payload = json.loads(run_keel(idle, "context", "--json").stdout)
|
|
26235
|
+
if (payload.get("nextAction") or {}).get("command"):
|
|
26236
|
+
report(
|
|
26237
|
+
f"{label}: an action with nothing to run reported a command; "
|
|
26238
|
+
f"{payload.get('nextAction')!r}."
|
|
26239
|
+
)
|
|
26240
|
+
return 1
|
|
26241
|
+
|
|
26242
|
+
report(f"{label} scenario passed.")
|
|
26243
|
+
return 0
|
|
26244
|
+
|
|
26245
|
+
|
|
26246
|
+
# `keel --init` writes OpenSpec's command and skill surfaces, whose text says
|
|
26247
|
+
# `openspec new change "<name>"`. With only the Keel package installed globally
|
|
26248
|
+
# a bare `openspec` is not on PATH, which `keel --doctor` reports and names the
|
|
26249
|
+
# working invocation for. Those files belong to OpenSpec and Keel appends a
|
|
26250
|
+
# block rather than editing their body — so the block is where Keel says it.
|
|
26251
|
+
def validate_overlay_names_the_invocation_scenario() -> int:
|
|
26252
|
+
label = "the-overlay-names-the-invocation"
|
|
26253
|
+
start = "<!-- keel:openspec-surface-overlay"
|
|
26254
|
+
end = "<!-- keel:openspec-surface-overlay:end -->"
|
|
26255
|
+
with tempfile.TemporaryDirectory(prefix="keel-overlay-invocation-") as raw:
|
|
26256
|
+
repo = Path(raw)
|
|
26257
|
+
# The surfaces OpenSpec writes, as OpenSpec writes them: the overlay is
|
|
26258
|
+
# merged into files that already exist, and their body is the text this
|
|
26259
|
+
# scenario asserts Keel does not touch.
|
|
26260
|
+
upstream = 'Run `openspec new change "<name>"` to begin.\n'
|
|
26261
|
+
for action in ("propose", "apply", "sync", "archive"):
|
|
26262
|
+
write_text(repo / f".claude/commands/opsx/{action}.md", upstream)
|
|
26263
|
+
for skill in (
|
|
26264
|
+
"openspec-propose",
|
|
26265
|
+
"openspec-apply-change",
|
|
26266
|
+
"openspec-sync-specs",
|
|
26267
|
+
"openspec-archive-change",
|
|
26268
|
+
):
|
|
26269
|
+
write_text(repo / f".claude/skills/{skill}/SKILL.md", upstream)
|
|
26270
|
+
installed = run_keel(repo, "--install", "--target", "claude")
|
|
26271
|
+
if installed.returncode != 0:
|
|
26272
|
+
report(
|
|
26273
|
+
f"{label}: the fixture install failed; "
|
|
26274
|
+
f"{(installed.stderr or installed.stdout).strip()[:300]}"
|
|
26275
|
+
)
|
|
26276
|
+
return 1
|
|
26277
|
+
|
|
26278
|
+
carrying = [
|
|
26279
|
+
path
|
|
26280
|
+
for path in sorted(repo.rglob("*.md"))
|
|
26281
|
+
if start in path.read_text(encoding="utf-8", errors="replace")
|
|
26282
|
+
]
|
|
26283
|
+
if not carrying:
|
|
26284
|
+
report(f"{label}: the install wrote no file carrying an overlay.")
|
|
26285
|
+
return 1
|
|
26286
|
+
for path in carrying:
|
|
26287
|
+
content = path.read_text(encoding="utf-8")
|
|
26288
|
+
block = content[content.index(start):content.index(end) + len(end)]
|
|
26289
|
+
for needed in ("keel openspec", "keel --doctor"):
|
|
26290
|
+
if needed not in block:
|
|
26291
|
+
report(
|
|
26292
|
+
f"{label}: the overlay in {path.name} does not name "
|
|
26293
|
+
f"{needed!r}."
|
|
26294
|
+
)
|
|
26295
|
+
return 1
|
|
26296
|
+
# The boundary: everything outside the block is OpenSpec's.
|
|
26297
|
+
outside = (
|
|
26298
|
+
content[: content.index(start)]
|
|
26299
|
+
+ content[content.index(end) + len(end):]
|
|
26300
|
+
)
|
|
26301
|
+
if start in outside or "keel openspec" in outside:
|
|
26302
|
+
report(
|
|
26303
|
+
f"{label}: {path.name} carries Keel invocation text outside "
|
|
26304
|
+
"the overlay block, so the OpenSpec-authored body was "
|
|
26305
|
+
"edited."
|
|
26306
|
+
)
|
|
26307
|
+
return 1
|
|
26308
|
+
|
|
26309
|
+
report(f"{label} scenario passed.")
|
|
26310
|
+
return 0
|
|
26311
|
+
|
|
26312
|
+
|
|
25971
26313
|
# A scenario name, as the registry spells one. Two registered names carry no
|
|
25972
26314
|
# hyphen — `cli` and `uninstall` — so requiring one would leave exactly those
|
|
25973
26315
|
# two unchecked, and allowing single words was measured to add no false
|
|
@@ -26207,6 +26549,9 @@ SCENARIOS: tuple = (
|
|
|
26207
26549
|
("the-obligation-is-stated-early", validate_obligation_is_stated_early_scenario),
|
|
26208
26550
|
("an-explanation-is-printed-once", validate_explanation_is_printed_once_scenario),
|
|
26209
26551
|
("a-paused-change-is-not-the-next-action", validate_paused_change_is_not_the_next_action_scenario),
|
|
26552
|
+
("evidence-survives-what-did-not-change", validate_evidence_survives_what_did_not_change_scenario),
|
|
26553
|
+
("the-next-action-is-a-command", validate_next_action_is_a_command_scenario),
|
|
26554
|
+
("the-overlay-names-the-invocation", validate_overlay_names_the_invocation_scenario),
|
|
26210
26555
|
(
|
|
26211
26556
|
"authored-scenario-names-are-registered",
|
|
26212
26557
|
validate_authored_scenario_names_scenario,
|
package/src/core/context.js
CHANGED
|
@@ -29,12 +29,34 @@ function taskRecords(tasksPath) {
|
|
|
29
29
|
}));
|
|
30
30
|
}
|
|
31
31
|
|
|
32
|
+
// The invocation the action's name stands for. Everything it needs is in the
|
|
33
|
+
// same result, and the failure it prevents is the one issue #112 reports: an
|
|
34
|
+
// action name followed by an attempt that fails on an argument Keel knew was
|
|
35
|
+
// required. Change and task are named explicitly rather than left to
|
|
36
|
+
// inference, because a reader may run the printed line later or elsewhere,
|
|
37
|
+
// where inference would answer about a different repository state.
|
|
38
|
+
function nextActionCommand(kind, selection) {
|
|
39
|
+
if (!selection || !selection.change) return null;
|
|
40
|
+
const scope = `--change ${selection.change}`;
|
|
41
|
+
if (kind === "task-start" || kind === "task-complete") {
|
|
42
|
+
if (!selection.task) return null;
|
|
43
|
+
return `keel gate ${kind} ${scope} --task ${selection.task}`;
|
|
44
|
+
}
|
|
45
|
+
if (kind === "change-close") {
|
|
46
|
+
// `--action` is not optional for this stage, so a command printed without
|
|
47
|
+
// it is a command that fails.
|
|
48
|
+
return `keel gate change-close ${scope} --action archive`;
|
|
49
|
+
}
|
|
50
|
+
return null;
|
|
51
|
+
}
|
|
52
|
+
|
|
32
53
|
function result(status, selection, nextAction, read, reasons = [], contract = null) {
|
|
54
|
+
const command = nextActionCommand(nextAction, selection);
|
|
33
55
|
const context = {
|
|
34
56
|
schemaVersion: 1,
|
|
35
57
|
status,
|
|
36
58
|
selection,
|
|
37
|
-
nextAction: { kind: nextAction },
|
|
59
|
+
nextAction: command ? { kind: nextAction, command } : { kind: nextAction },
|
|
38
60
|
read,
|
|
39
61
|
reasons,
|
|
40
62
|
warnings: [],
|
|
@@ -679,6 +701,9 @@ function renderContext(result) {
|
|
|
679
701
|
`Keel context: ${result.status}`,
|
|
680
702
|
`Next action: ${result.nextAction.kind}`,
|
|
681
703
|
];
|
|
704
|
+
if (result.nextAction.command) {
|
|
705
|
+
lines.push(`Run: ${result.nextAction.command}`);
|
|
706
|
+
}
|
|
682
707
|
if (result.selection) {
|
|
683
708
|
lines.push(
|
|
684
709
|
`Selection: ${result.selection.change}`
|
package/src/core/gates.js
CHANGED
|
@@ -288,6 +288,37 @@ function taskStart(repo, options) {
|
|
|
288
288
|
// authors to — needs no manual edit. Refusal is kept only for a task with no
|
|
289
289
|
// anchor at all, which is a malformed capsule rather than a reauthorization,
|
|
290
290
|
// and it writes nothing, guard manifest included.
|
|
291
|
+
const declaredKeep = Array.isArray(options.keepEvidence)
|
|
292
|
+
? options.keepEvidence
|
|
293
|
+
: [];
|
|
294
|
+
if (declaredKeep.length > 0 && !options.record) {
|
|
295
|
+
problems.push(
|
|
296
|
+
problem(
|
|
297
|
+
"keep-evidence-without-record",
|
|
298
|
+
"--keep-evidence declares which evidence a re-record leaves standing, "
|
|
299
|
+
+ "and there is no re-record here. Pass --record, or drop the "
|
|
300
|
+
+ "declaration so nothing reads it."
|
|
301
|
+
)
|
|
302
|
+
);
|
|
303
|
+
}
|
|
304
|
+
if (declaredKeep.length > 0 && compiled.diagnostics.length === 0) {
|
|
305
|
+
const labels = compiled.capsule.verification.commands.map(
|
|
306
|
+
(item) => item.label
|
|
307
|
+
);
|
|
308
|
+
const unknown = declaredKeep.filter((label) => !labels.includes(label));
|
|
309
|
+
if (unknown.length > 0) {
|
|
310
|
+
// Refused rather than ignored: the likeliest cause is a typo or a check
|
|
311
|
+
// that was renamed, and ignoring it would leave the author believing
|
|
312
|
+
// evidence was kept that was not.
|
|
313
|
+
problems.push(
|
|
314
|
+
problem(
|
|
315
|
+
"keep-evidence-unknown-check",
|
|
316
|
+
`--keep-evidence names ${unknown.join(", ")}, which this contract `
|
|
317
|
+
+ `does not declare as a check. It declares ${labels.join(", ")}.`
|
|
318
|
+
)
|
|
319
|
+
);
|
|
320
|
+
}
|
|
321
|
+
}
|
|
291
322
|
let anchorPlan = null;
|
|
292
323
|
if (options.record && problems.length === 0) {
|
|
293
324
|
anchorPlan = contractAnchorPlan(selection, task);
|
|
@@ -364,11 +395,30 @@ function taskStart(repo, options) {
|
|
|
364
395
|
// call to the current agent's Review.
|
|
365
396
|
const replaced = anchoredFingerprint(anchorPlan.previous);
|
|
366
397
|
if (replaced && replaced !== compiled.fingerprint.value) {
|
|
398
|
+
// The gate still cannot judge which evidence survives, and this does not
|
|
399
|
+
// make it able to. What it can stop doing is saying "all of it" to an
|
|
400
|
+
// author who can see that one check's assertion did not change — because
|
|
401
|
+
// an author acting on that sentence in good faith re-runs everything,
|
|
402
|
+
// and issue #112 measured four such re-verifications in one session, one
|
|
403
|
+
// of which meant breaking a testbench and restoring it.
|
|
404
|
+
const kept = declaredKeep;
|
|
405
|
+
const labels = compiled.capsule.verification.commands.map(
|
|
406
|
+
(item) => item.label
|
|
407
|
+
);
|
|
408
|
+
const stale = labels.filter((label) => !kept.includes(label));
|
|
367
409
|
result.warnings.push(
|
|
368
410
|
`Re-recorded over a different contract: was sha256:${replaced}, now `
|
|
369
|
-
+ `sha256:${compiled.fingerprint.value}.
|
|
370
|
-
+
|
|
371
|
-
|
|
411
|
+
+ `sha256:${compiled.fingerprint.value}. `
|
|
412
|
+
+ (kept.length > 0
|
|
413
|
+
? `Evidence for ${stale.length > 0 ? stale.join(", ") : "no check"}`
|
|
414
|
+
+ " is stale; clear or re-verify it before completing this task. "
|
|
415
|
+
+ `${kept.join(", ")} ${kept.length > 1 ? "were" : "was"} `
|
|
416
|
+
+ "declared unaffected by this contract change — a declaration "
|
|
417
|
+
+ "Keel records and does not verify, since it retains only the "
|
|
418
|
+
+ "previous fingerprint and cannot compare a check's former text "
|
|
419
|
+
+ "to its current one. State the reason in Reauthorizations."
|
|
420
|
+
: "Execution evidence produced under the previous contract is "
|
|
421
|
+
+ "stale; clear or re-verify it before completing this task.")
|
|
372
422
|
);
|
|
373
423
|
}
|
|
374
424
|
}
|