@christang/keel 5.54.0 → 5.56.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -333,6 +333,23 @@ keel lenses add web # copy the web template into keel/lenses/web.md, the
333
333
  keel lenses add web --force # overwrite an existing lens
334
334
  ```
335
335
 
336
+ ## Re-recording a contract
337
+
338
+ Changing a task's contract after work has started moves its fingerprint, and Keel reports that the
339
+ evidence produced under the old one is stale. Sometimes that is too broad — a classification tag
340
+ added to a check leaves its assertion untouched, and re-running a three-minute experiment for it
341
+ buys nothing. Say so:
342
+
343
+ ```bash
344
+ keel gate task-start --change <c> --task <t> --record --keep-evidence M1,M3
345
+ ```
346
+
347
+ The report then names only the checks still stale, names the ones you declared unaffected, and says
348
+ the narrowing came from your declaration. **Keel does not verify the claim** — it keeps only the
349
+ previous fingerprint, not the capsule behind it, so it cannot compare a check's former text to its
350
+ current one. State your reason in the task's `Reauthorizations` line, where a reviewer can disagree
351
+ with it. Nothing about completion changes: every check still needs its Evidence.
352
+
336
353
  ## Pausing a change
337
354
 
338
355
  A change you have deliberately stopped — waiting on something outside the repository, or simply
@@ -1,4 +1,4 @@
1
- <!-- keel:start version=5.54.0 -->
1
+ <!-- keel:start version=5.56.0 -->
2
2
  ## Keel Bootstrap
3
3
 
4
4
  - Start every session with `keel context`; OpenSpec artifacts and Git are the only durable authority — never native memory, goals, or transcripts.
package/bin/keel.js CHANGED
@@ -94,7 +94,8 @@ Usage:
94
94
  keel capabilities [repo] [--target claude|codex|opencode] [--json]
95
95
  keel project [repo] --target claude|codex|opencode --event startup|resume|compaction|goal|task-view|worktree|subagent-start|subagent-stop [--authorize goal|task-view|subagent] [--expected-owner owner] [--native-complete] [--change name] [--task id] [--json]
96
96
  keel project tasks [repo] --target claude [--change name] [--json]
97
- keel gate task-start|task-complete|change-close [repo] [--change name] [--task id] [--action sync|archive] [--base git-ref] [--no-guard] [--record] [--json]
97
+ keel gate task-start|task-complete [repo] [--change name] [--task id] [--base git-ref] [--no-guard] [--record] [--keep-evidence M1,M3] [--json]
98
+ keel gate change-close [repo] [--change name] --action sync|archive [--base git-ref] [--json]
98
99
  keel guard start|status|clear [repo] [--change name] [--task id] [--force] [--json]
99
100
  keel lenses list|add [name] [repo] [--force]
100
101
  keel triage [repo] [--labels <l1,l2>] [--issue <n>] [--json]
@@ -186,6 +187,7 @@ function parseArgs(argv) {
186
187
  base: null,
187
188
  noGuard: false,
188
189
  record: false,
190
+ keepEvidence: null,
189
191
  guardSubcommand: null,
190
192
  lensesSubcommand: null,
191
193
  lensName: null,
@@ -296,6 +298,20 @@ function parseArgs(argv) {
296
298
  parsed.record = true;
297
299
  continue;
298
300
  }
301
+ if (arg === "--keep-evidence") {
302
+ index += 1;
303
+ if (index >= argv.length) {
304
+ fail("--keep-evidence requires a comma-separated list of M<n> labels");
305
+ }
306
+ if (parsed.keepEvidence !== null) {
307
+ fail("--keep-evidence was provided more than once");
308
+ }
309
+ parsed.keepEvidence = argv[index]
310
+ .split(",")
311
+ .map((entry) => entry.trim())
312
+ .filter(Boolean);
313
+ continue;
314
+ }
299
315
  if (arg === "--change" || arg === "--task") {
300
316
  index += 1;
301
317
  if (index >= argv.length) {
@@ -1179,6 +1195,8 @@ function keelOpenSpecOverlay(action) {
1179
1195
  "",
1180
1196
  "Keel rules below take precedence over conflicting generic OpenSpec instructions in this file.",
1181
1197
  "",
1198
+ "- Invoke the OpenSpec CLI as `keel openspec …` throughout this file. The commands below are written as a bare `openspec`, which resolves only where OpenSpec is separately installed on PATH; `keel openspec` resolves either way, and `keel --doctor` reports which case this repository is.",
1199
+ "",
1182
1200
  "### Expectation alignment before specs and tasks finalize",
1183
1201
  "",
1184
1202
  "- Before specs and executable tasks are finalized, run `keel-align-expectations`: quick path for complete low-risk requests, deep path when a material choice can change user-visible behavior, an external interface, acceptance, security/privacy/permission boundaries, data migration, protocol/state/timing/reset semantics, generated equivalence, irreversible cost, or a dependency commitment.",
@@ -1214,6 +1232,8 @@ function keelOpenSpecOverlay(action) {
1214
1232
  "",
1215
1233
  "Keel rules below take precedence over conflicting generic OpenSpec instructions in this file.",
1216
1234
  "",
1235
+ "- Invoke the OpenSpec CLI as `keel openspec …` throughout this file. The commands below are written as a bare `openspec`, which resolves only where OpenSpec is separately installed on PATH; `keel openspec` resolves either way, and `keel --doctor` reports which case this repository is.",
1236
+ "",
1217
1237
  ...syncBody,
1218
1238
  OPENSPEC_SURFACE_OVERLAY_END,
1219
1239
  "",
@@ -1254,6 +1274,8 @@ function keelOpenSpecOverlay(action) {
1254
1274
  "",
1255
1275
  "Keel rules below take precedence over conflicting generic OpenSpec instructions in this file.",
1256
1276
  "",
1277
+ "- Invoke the OpenSpec CLI as `keel openspec …` throughout this file. The commands below are written as a bare `openspec`, which resolves only where OpenSpec is separately installed on PATH; `keel openspec` resolves either way, and `keel --doctor` reports which case this repository is.",
1278
+ "",
1257
1279
  "### Target-native subagent gate",
1258
1280
  "",
1259
1281
  "- The current agent remains responsible for Keel ownership, task/archive decisions, scope control, and final reporting.",
package/package.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "name": "@christang/keel",
3
3
  "displayName": "Keel",
4
4
  "description": "Keel OpenSpec execution discipline CLI for Claude Code, Codex, and OpenCode.",
5
- "version": "5.54.0",
5
+ "version": "5.56.0",
6
6
  "license": "MIT",
7
7
  "repository": {
8
8
  "type": "git",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "keel",
3
- "version": "5.54.0",
3
+ "version": "5.56.0",
4
4
  "description": "Keel OpenSpec execution discipline: stateless continuity, task capsules, deterministic gates, and expectation alignment for Codex and Claude Code.",
5
5
  "author": {
6
6
  "name": "TanglmChris",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "keel",
3
- "version": "5.54.0",
3
+ "version": "5.56.0",
4
4
  "description": "Keel OpenSpec execution discipline: stateless continuity, task capsules, deterministic gates, and expectation alignment for Codex and Claude Code.",
5
5
  "author": {
6
6
  "name": "TanglmChris",
@@ -37,8 +37,8 @@ REQUIRED_SCRIPTS = [
37
37
  "scripts/validate_plugin.py",
38
38
  ]
39
39
 
40
- PACKAGE_VERSION = "5.54.0"
41
- PROTOCOL_VERSION = "5.54.0"
40
+ PACKAGE_VERSION = "5.56.0"
41
+ PROTOCOL_VERSION = "5.56.0"
42
42
  LEGACY_MANAGED_START = "<!-- keel:start version=2.1 -->"
43
43
  OPENSPEC_SCHEMA_NAME = "keel-spec-driven"
44
44
  # Mirrors KEEL_PACKAGE_NAME in scripts/install_to_repo.py, one of the two
@@ -2163,7 +2163,7 @@ def validate_authoring_continuity_scenario() -> int:
2163
2163
  or scaffold_payload.get("status") != "ready"
2164
2164
  or scaffold_payload.get("selection")
2165
2165
  != {"source": "inferred", "change": "draft", "task": None}
2166
- or scaffold_payload.get("nextAction") != {"kind": "author"}
2166
+ or scaffold_payload.get("nextAction", {}).get("kind") != "author"
2167
2167
  ):
2168
2168
  report(
2169
2169
  "authoring-continuity scenario did not keep an incomplete proposal actionable."
@@ -3747,8 +3747,14 @@ def validate_stateless_continuity_scenario() -> int:
3747
3747
  "change": "demo",
3748
3748
  "task": "1.1",
3749
3749
  },
3750
- "nextAction": {"kind": "task-start"},
3751
3750
  }
3751
+ expected_kind = "task-start"
3752
+ if payload.get("nextAction", {}).get("kind") != expected_kind:
3753
+ report(
3754
+ "stateless-continuity scenario explicit selection reported "
3755
+ f"next action {payload.get('nextAction')!r}."
3756
+ )
3757
+ return 1
3752
3758
  for key, value in expected.items():
3753
3759
  if payload.get(key) != value:
3754
3760
  report(
@@ -3808,7 +3814,7 @@ def validate_stateless_continuity_scenario() -> int:
3808
3814
  inferred_payload.get("status") != "ready"
3809
3815
  or inferred_payload.get("selection")
3810
3816
  != {"source": "inferred", "change": "only", "task": "1.1"}
3811
- or inferred_payload.get("nextAction") != {"kind": "task-start"}
3817
+ or inferred_payload.get("nextAction", {}).get("kind") != "task-start"
3812
3818
  ):
3813
3819
  report("stateless-continuity scenario unique inference mismatch.")
3814
3820
  report(inferred.stdout.strip())
@@ -3830,7 +3836,8 @@ def validate_stateless_continuity_scenario() -> int:
3830
3836
  or recomputed_payload.get("status") != "ready"
3831
3837
  or recomputed_payload.get("selection")
3832
3838
  != {"source": "inferred", "change": "only", "task": None}
3833
- or recomputed_payload.get("nextAction") != {"kind": "change-close"}
3839
+ or recomputed_payload.get("nextAction", {}).get("kind")
3840
+ != "change-close"
3834
3841
  ):
3835
3842
  report("stateless-continuity scenario did not recompute completed state.")
3836
3843
  report((recomputed.stderr or recomputed.stdout).strip())
@@ -3849,7 +3856,7 @@ def validate_stateless_continuity_scenario() -> int:
3849
3856
  evidence_payload = json.loads(evidence_ready.stdout)
3850
3857
  if (
3851
3858
  evidence_ready.returncode != 0
3852
- or evidence_payload.get("nextAction") != {"kind": "task-complete"}
3859
+ or evidence_payload.get("nextAction", {}).get("kind") != "task-complete"
3853
3860
  ):
3854
3861
  report("stateless-continuity scenario missed evidence-ready completion.")
3855
3862
  report((evidence_ready.stderr or evidence_ready.stdout).strip())
@@ -3902,7 +3909,7 @@ def validate_stateless_continuity_scenario() -> int:
3902
3909
  or handoff_payload.get("status") != "ready"
3903
3910
  or handoff_payload.get("selection")
3904
3911
  != {"source": "handoff", "change": "beta", "task": "1.1"}
3905
- or handoff_payload.get("nextAction") != {"kind": "task-start"}
3912
+ or handoff_payload.get("nextAction", {}).get("kind") != "task-start"
3906
3913
  ):
3907
3914
  report("stateless-continuity scenario did not prioritize valid HANDOFF.")
3908
3915
  report((handoff.stderr or handoff.stdout).strip())
@@ -4036,7 +4043,7 @@ def validate_stateless_continuity_scenario() -> int:
4036
4043
  discuss = run_keel(discuss_repo, "context", "--json")
4037
4044
  if (
4038
4045
  discuss.returncode != 0
4039
- or json.loads(discuss.stdout).get("nextAction") != {"kind": "discuss"}
4046
+ or json.loads(discuss.stdout).get("nextAction", {}).get("kind") != "discuss"
4040
4047
  ):
4041
4048
  report("stateless-continuity scenario missed discuss transition.")
4042
4049
  report((discuss.stderr or discuss.stdout).strip())
@@ -4051,7 +4058,7 @@ def validate_stateless_continuity_scenario() -> int:
4051
4058
  author = run_keel(author_repo, "context", "--json")
4052
4059
  if (
4053
4060
  author.returncode != 0
4054
- or json.loads(author.stdout).get("nextAction") != {"kind": "author"}
4061
+ or json.loads(author.stdout).get("nextAction", {}).get("kind") != "author"
4055
4062
  ):
4056
4063
  report("stateless-continuity scenario missed author transition.")
4057
4064
  report((author.stderr or author.stdout).strip())
@@ -4069,7 +4076,7 @@ def validate_stateless_continuity_scenario() -> int:
4069
4076
  idle.returncode != 0
4070
4077
  or idle_payload.get("status") != "idle"
4071
4078
  or idle_payload.get("selection") is not None
4072
- or idle_payload.get("nextAction") != {"kind": "none"}
4079
+ or idle_payload.get("nextAction", {}).get("kind") != "none"
4073
4080
  ):
4074
4081
  report("stateless-continuity scenario no-work state was not idle.")
4075
4082
  report((idle.stderr or idle.stdout).strip())
@@ -4227,7 +4234,7 @@ def validate_stateless_continuity_scenario() -> int:
4227
4234
  if (
4228
4235
  matching.returncode != 0
4229
4236
  or matching_payload.get("status") != "ready"
4230
- or matching_payload.get("nextAction") != {"kind": "task-start"}
4237
+ or matching_payload.get("nextAction", {}).get("kind") != "task-start"
4231
4238
  ):
4232
4239
  report("stateless-continuity scenario did not resume a matching anchor.")
4233
4240
  report((matching.stderr or matching.stdout).strip())
@@ -4315,7 +4322,7 @@ def validate_stateless_continuity_scenario() -> int:
4315
4322
  or matching_handoff_payload.get("status") != "ready"
4316
4323
  or matching_handoff_payload.get("selection")
4317
4324
  != {"source": "handoff", "change": "demo", "task": "1.1"}
4318
- or matching_handoff_payload.get("nextAction") != {"kind": "task-start"}
4325
+ or matching_handoff_payload.get("nextAction", {}).get("kind") != "task-start"
4319
4326
  ):
4320
4327
  report("stateless-continuity scenario did not resume a matching HANDOFF anchor.")
4321
4328
  report((matching_handoff.stderr or matching_handoff.stdout).strip())
@@ -25968,6 +25975,341 @@ def validate_paused_change_is_not_the_next_action_scenario() -> int:
25968
25975
  return 0
25969
25976
 
25970
25977
 
25978
+ # Issue #112 recorded four re-verifications in one session from contract
25979
+ # changes that could not affect evidence — renaming `M2:` to `M2 (regression):`
25980
+ # among them, with the assertion unchanged by a character. One of the four meant
25981
+ # breaking a testbench, re-running, and restoring it. The gate genuinely cannot
25982
+ # judge which evidence survives; what it could do is stop saying "all of it".
25983
+ def validate_evidence_survives_what_did_not_change_scenario() -> int:
25984
+ label = "evidence-survives-what-did-not-change"
25985
+
25986
+ def fixture(root: Path, name: str, strategy: str = "evidence-first") -> Path:
25987
+ repo = root / name
25988
+ task = strategy_probe_task(
25989
+ strategy=strategy,
25990
+ reason="fixture; nothing here can fail first",
25991
+ commands=(
25992
+ "M1: the first check asserts the public behavior",
25993
+ "M2: the second check asserts the public behavior",
25994
+ "M3: the third check asserts the public behavior",
25995
+ ),
25996
+ )
25997
+ # A recorded anchor that is not the compiled one, so --record re-records.
25998
+ task = task.replace(
25999
+ " - Contract: pending",
26000
+ " - Contract: keel-task-capsule/v1 sha256:" + "0" * 64,
26001
+ )
26002
+ write_gate_fixture(repo, tasks=task)
26003
+ return repo
26004
+
26005
+ def start(repo: Path, *args):
26006
+ result = run_keel(
26007
+ repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
26008
+ "--json", "--no-guard", *args,
26009
+ )
26010
+ try:
26011
+ return json.loads(result.stdout)
26012
+ except json.JSONDecodeError:
26013
+ return {"status": "unparsed", "problems": [{"message": result.stdout[:300]}]}
26014
+
26015
+ def said(payload: dict) -> str:
26016
+ return " ".join(str(x) for x in (payload.get("warnings") or []))
26017
+
26018
+ with tempfile.TemporaryDirectory(prefix="keel-keep-evidence-") as raw:
26019
+ root = Path(raw)
26020
+
26021
+ # Control: without a declaration the report covers every check.
26022
+ blanket = start(fixture(root, "blanket"), "--record")
26023
+ if blanket.get("status") != "pass":
26024
+ report(
26025
+ f"{label}: the re-record fixture did not pass; "
26026
+ f"{problem_text(blanket)!r}."
26027
+ )
26028
+ return 1
26029
+ if "stale" not in said(blanket):
26030
+ report(
26031
+ f"{label}: the fixture did not produce a stale-evidence "
26032
+ f"report to narrow; {said(blanket)!r}."
26033
+ )
26034
+ return 1
26035
+
26036
+ narrowed = start(
26037
+ fixture(root, "narrowed"), "--record", "--keep-evidence", "M1,M3"
26038
+ )
26039
+ if narrowed.get("status") != "pass":
26040
+ report(
26041
+ f"{label}: a declaration refused a valid re-record; "
26042
+ f"{problem_text(narrowed)!r}."
26043
+ )
26044
+ return 1
26045
+ spoken = said(narrowed)
26046
+ stale_line = next(
26047
+ (w for w in narrowed["warnings"] if "stale" in str(w)), ""
26048
+ )
26049
+ if "M2" not in stale_line:
26050
+ report(
26051
+ f"{label}: the narrowed report does not name the check that is "
26052
+ f"still stale; {stale_line!r}."
26053
+ )
26054
+ return 1
26055
+ for kept in ("M1", "M3"):
26056
+ if kept not in stale_line:
26057
+ report(
26058
+ f"{label}: the narrowed report does not name {kept} as "
26059
+ f"declared unaffected; {stale_line!r}."
26060
+ )
26061
+ return 1
26062
+ if "declar" not in stale_line.lower():
26063
+ report(
26064
+ f"{label}: the narrowed report does not attribute the "
26065
+ f"narrowing to the declaration; {stale_line!r}."
26066
+ )
26067
+ return 1
26068
+
26069
+ unknown = start(
26070
+ fixture(root, "unknown"), "--record", "--keep-evidence", "M9"
26071
+ )
26072
+ if unknown.get("status") != "fail":
26073
+ report(
26074
+ f"{label}: a declaration naming a check the contract does not "
26075
+ "declare was accepted; task-start returned "
26076
+ f"{unknown.get('status')!r}."
26077
+ )
26078
+ return 1
26079
+ if "M9" not in problem_text(unknown):
26080
+ report(
26081
+ f"{label}: the refusal does not name the label it could not "
26082
+ f"resolve; {problem_text(unknown)!r}."
26083
+ )
26084
+ return 1
26085
+
26086
+ stray = start(fixture(root, "stray"), "--keep-evidence", "M1")
26087
+ if stray.get("status") != "fail":
26088
+ report(
26089
+ f"{label}: --keep-evidence without --record was accepted, so "
26090
+ "an author could believe they declared something nothing read."
26091
+ )
26092
+ return 1
26093
+
26094
+ # Completion is unchanged: the declaration is about the past.
26095
+ completing = fixture(root, "completing", strategy="vertical-tdd")
26096
+ start(completing, "--record", "--keep-evidence", "M1,M2,M3")
26097
+ tasks_path = completing / "openspec/changes/demo/tasks.md"
26098
+ tasks_path.write_text(
26099
+ tasks_path.read_text(encoding="utf-8")
26100
+ .replace("- [ ] 1.1", "- [x] 1.1")
26101
+ .replace(" - M1: pending", " - M1: pass. ran it.")
26102
+ .replace(" - M2: pending", " - M2: pass. ran it.")
26103
+ .replace(" - M3: pending", " - M3: pass. ran it."),
26104
+ encoding="utf-8",
26105
+ )
26106
+ done = json.loads(
26107
+ run_keel(
26108
+ completing, "gate", "task-complete", "--change", "demo",
26109
+ "--task", "1.1", "--json",
26110
+ ).stdout
26111
+ )
26112
+ if done.get("status") != "fail":
26113
+ report(
26114
+ f"{label}: a declaration let a red-green task complete without "
26115
+ "its .red/.green Evidence."
26116
+ )
26117
+ return 1
26118
+ if "missing-strategy-evidence" not in problem_codes(done):
26119
+ report(
26120
+ f"{label}: completion stopped requiring red-green Evidence; "
26121
+ f"{problem_codes(done)!r}."
26122
+ )
26123
+ return 1
26124
+
26125
+ report(f"{label} scenario passed.")
26126
+ return 0
26127
+
26128
+
26129
+ # `keel context` exists to answer "what now" and answered with a noun. Issue
26130
+ # #112: `Next action: change-close` followed by `keel gate change-close
26131
+ # --change x` failing on the argument that stage requires — a wrong attempt
26132
+ # Keel had everything in the same result to prevent.
26133
+ def validate_next_action_is_a_command_scenario() -> int:
26134
+ label = "the-next-action-is-a-command"
26135
+
26136
+ def repo_with(root: Path, name: str, *, checked: bool, evidence: bool):
26137
+ repo = root / name
26138
+ task = strategy_probe_task(
26139
+ strategy="evidence-first",
26140
+ reason="fixture; nothing here can fail first",
26141
+ )
26142
+ if checked:
26143
+ task = task.replace("- [ ] 1.1", "- [x] 1.1")
26144
+ if evidence:
26145
+ task = task.replace(" - M1: pending", " - M1: pass. ran it.")
26146
+ write_gate_fixture(repo, tasks=task)
26147
+ started = run_keel(
26148
+ repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
26149
+ "--json", "--no-guard",
26150
+ )
26151
+ value = json.loads(started.stdout)["contract"]["fingerprint"]["value"]
26152
+ tasks_path = repo / "openspec/changes/demo/tasks.md"
26153
+ tasks_path.write_text(
26154
+ tasks_path.read_text(encoding="utf-8").replace(
26155
+ " - Contract: pending",
26156
+ f" - Contract: keel-task-capsule/v1 sha256:{value}",
26157
+ ),
26158
+ encoding="utf-8",
26159
+ )
26160
+ return repo
26161
+
26162
+ def context(repo: Path):
26163
+ text = run_keel(repo, "context", "--change", "demo").stdout
26164
+ payload = json.loads(
26165
+ run_keel(repo, "context", "--change", "demo", "--json").stdout
26166
+ )
26167
+ return text, payload
26168
+
26169
+ with tempfile.TemporaryDirectory(prefix="keel-next-command-") as raw:
26170
+ root = Path(raw)
26171
+
26172
+ pending = repo_with(root, "pending", checked=False, evidence=False)
26173
+ text, payload = context(pending)
26174
+ command = (payload.get("nextAction") or {}).get("command")
26175
+ if not command:
26176
+ report(
26177
+ f"{label}: the next action carries no command; "
26178
+ f"{payload.get('nextAction')!r}."
26179
+ )
26180
+ return 1
26181
+ for needed in ("task-start", "--change demo", "--task 1.1"):
26182
+ if needed not in command:
26183
+ report(
26184
+ f"{label}: the reported command omits {needed!r}; "
26185
+ f"{command!r}."
26186
+ )
26187
+ return 1
26188
+ if command not in text:
26189
+ report(
26190
+ f"{label}: the text surface does not carry the same command as "
26191
+ f"--json; {command!r} not in {text!r}."
26192
+ )
26193
+ return 1
26194
+
26195
+ done = repo_with(root, "done", checked=False, evidence=True)
26196
+ _, payload = context(done)
26197
+ if "task-complete" not in str((payload.get("nextAction") or {}).get("command")):
26198
+ report(
26199
+ f"{label}: a task with completion evidence did not report the "
26200
+ f"task-complete invocation; {payload.get('nextAction')!r}."
26201
+ )
26202
+ return 1
26203
+
26204
+ closing = repo_with(root, "closing", checked=True, evidence=True)
26205
+ _, payload = context(closing)
26206
+ close_command = str((payload.get("nextAction") or {}).get("command") or "")
26207
+ if "change-close" not in close_command:
26208
+ report(
26209
+ f"{label}: a change whose tasks are complete did not report the "
26210
+ f"change-close invocation; {payload.get('nextAction')!r}."
26211
+ )
26212
+ return 1
26213
+ if "--action" not in close_command:
26214
+ report(
26215
+ f"{label}: the change-close command omits the argument that "
26216
+ f"stage requires; {close_command!r}."
26217
+ )
26218
+ return 1
26219
+ # Running exactly what was printed must not hit the input error the
26220
+ # report describes.
26221
+ printed = close_command.split()
26222
+ if printed and printed[0] == "keel":
26223
+ printed = printed[1:]
26224
+ ran = run_keel(closing, *printed)
26225
+ if "requires --action" in (ran.stdout + ran.stderr):
26226
+ report(
26227
+ f"{label}: the reported command still fails on the argument it "
26228
+ f"was supposed to supply; {(ran.stdout + ran.stderr)[:200]!r}."
26229
+ )
26230
+ return 1
26231
+
26232
+ idle = root / "idle"
26233
+ write_text(idle / "README.md", "# fixture\n")
26234
+ payload = json.loads(run_keel(idle, "context", "--json").stdout)
26235
+ if (payload.get("nextAction") or {}).get("command"):
26236
+ report(
26237
+ f"{label}: an action with nothing to run reported a command; "
26238
+ f"{payload.get('nextAction')!r}."
26239
+ )
26240
+ return 1
26241
+
26242
+ report(f"{label} scenario passed.")
26243
+ return 0
26244
+
26245
+
26246
+ # `keel --init` writes OpenSpec's command and skill surfaces, whose text says
26247
+ # `openspec new change "<name>"`. With only the Keel package installed globally
26248
+ # a bare `openspec` is not on PATH, which `keel --doctor` reports and names the
26249
+ # working invocation for. Those files belong to OpenSpec and Keel appends a
26250
+ # block rather than editing their body — so the block is where Keel says it.
26251
+ def validate_overlay_names_the_invocation_scenario() -> int:
26252
+ label = "the-overlay-names-the-invocation"
26253
+ start = "<!-- keel:openspec-surface-overlay"
26254
+ end = "<!-- keel:openspec-surface-overlay:end -->"
26255
+ with tempfile.TemporaryDirectory(prefix="keel-overlay-invocation-") as raw:
26256
+ repo = Path(raw)
26257
+ # The surfaces OpenSpec writes, as OpenSpec writes them: the overlay is
26258
+ # merged into files that already exist, and their body is the text this
26259
+ # scenario asserts Keel does not touch.
26260
+ upstream = 'Run `openspec new change "<name>"` to begin.\n'
26261
+ for action in ("propose", "apply", "sync", "archive"):
26262
+ write_text(repo / f".claude/commands/opsx/{action}.md", upstream)
26263
+ for skill in (
26264
+ "openspec-propose",
26265
+ "openspec-apply-change",
26266
+ "openspec-sync-specs",
26267
+ "openspec-archive-change",
26268
+ ):
26269
+ write_text(repo / f".claude/skills/{skill}/SKILL.md", upstream)
26270
+ installed = run_keel(repo, "--install", "--target", "claude")
26271
+ if installed.returncode != 0:
26272
+ report(
26273
+ f"{label}: the fixture install failed; "
26274
+ f"{(installed.stderr or installed.stdout).strip()[:300]}"
26275
+ )
26276
+ return 1
26277
+
26278
+ carrying = [
26279
+ path
26280
+ for path in sorted(repo.rglob("*.md"))
26281
+ if start in path.read_text(encoding="utf-8", errors="replace")
26282
+ ]
26283
+ if not carrying:
26284
+ report(f"{label}: the install wrote no file carrying an overlay.")
26285
+ return 1
26286
+ for path in carrying:
26287
+ content = path.read_text(encoding="utf-8")
26288
+ block = content[content.index(start):content.index(end) + len(end)]
26289
+ for needed in ("keel openspec", "keel --doctor"):
26290
+ if needed not in block:
26291
+ report(
26292
+ f"{label}: the overlay in {path.name} does not name "
26293
+ f"{needed!r}."
26294
+ )
26295
+ return 1
26296
+ # The boundary: everything outside the block is OpenSpec's.
26297
+ outside = (
26298
+ content[: content.index(start)]
26299
+ + content[content.index(end) + len(end):]
26300
+ )
26301
+ if start in outside or "keel openspec" in outside:
26302
+ report(
26303
+ f"{label}: {path.name} carries Keel invocation text outside "
26304
+ "the overlay block, so the OpenSpec-authored body was "
26305
+ "edited."
26306
+ )
26307
+ return 1
26308
+
26309
+ report(f"{label} scenario passed.")
26310
+ return 0
26311
+
26312
+
25971
26313
  # A scenario name, as the registry spells one. Two registered names carry no
25972
26314
  # hyphen — `cli` and `uninstall` — so requiring one would leave exactly those
25973
26315
  # two unchecked, and allowing single words was measured to add no false
@@ -26207,6 +26549,9 @@ SCENARIOS: tuple = (
26207
26549
  ("the-obligation-is-stated-early", validate_obligation_is_stated_early_scenario),
26208
26550
  ("an-explanation-is-printed-once", validate_explanation_is_printed_once_scenario),
26209
26551
  ("a-paused-change-is-not-the-next-action", validate_paused_change_is_not_the_next_action_scenario),
26552
+ ("evidence-survives-what-did-not-change", validate_evidence_survives_what_did_not_change_scenario),
26553
+ ("the-next-action-is-a-command", validate_next_action_is_a_command_scenario),
26554
+ ("the-overlay-names-the-invocation", validate_overlay_names_the_invocation_scenario),
26210
26555
  (
26211
26556
  "authored-scenario-names-are-registered",
26212
26557
  validate_authored_scenario_names_scenario,
@@ -29,12 +29,34 @@ function taskRecords(tasksPath) {
29
29
  }));
30
30
  }
31
31
 
32
+ // The invocation the action's name stands for. Everything it needs is in the
33
+ // same result, and the failure it prevents is the one issue #112 reports: an
34
+ // action name followed by an attempt that fails on an argument Keel knew was
35
+ // required. Change and task are named explicitly rather than left to
36
+ // inference, because a reader may run the printed line later or elsewhere,
37
+ // where inference would answer about a different repository state.
38
+ function nextActionCommand(kind, selection) {
39
+ if (!selection || !selection.change) return null;
40
+ const scope = `--change ${selection.change}`;
41
+ if (kind === "task-start" || kind === "task-complete") {
42
+ if (!selection.task) return null;
43
+ return `keel gate ${kind} ${scope} --task ${selection.task}`;
44
+ }
45
+ if (kind === "change-close") {
46
+ // `--action` is not optional for this stage, so a command printed without
47
+ // it is a command that fails.
48
+ return `keel gate change-close ${scope} --action archive`;
49
+ }
50
+ return null;
51
+ }
52
+
32
53
  function result(status, selection, nextAction, read, reasons = [], contract = null) {
54
+ const command = nextActionCommand(nextAction, selection);
33
55
  const context = {
34
56
  schemaVersion: 1,
35
57
  status,
36
58
  selection,
37
- nextAction: { kind: nextAction },
59
+ nextAction: command ? { kind: nextAction, command } : { kind: nextAction },
38
60
  read,
39
61
  reasons,
40
62
  warnings: [],
@@ -679,6 +701,9 @@ function renderContext(result) {
679
701
  `Keel context: ${result.status}`,
680
702
  `Next action: ${result.nextAction.kind}`,
681
703
  ];
704
+ if (result.nextAction.command) {
705
+ lines.push(`Run: ${result.nextAction.command}`);
706
+ }
682
707
  if (result.selection) {
683
708
  lines.push(
684
709
  `Selection: ${result.selection.change}`
package/src/core/gates.js CHANGED
@@ -288,6 +288,37 @@ function taskStart(repo, options) {
288
288
  // authors to — needs no manual edit. Refusal is kept only for a task with no
289
289
  // anchor at all, which is a malformed capsule rather than a reauthorization,
290
290
  // and it writes nothing, guard manifest included.
291
+ const declaredKeep = Array.isArray(options.keepEvidence)
292
+ ? options.keepEvidence
293
+ : [];
294
+ if (declaredKeep.length > 0 && !options.record) {
295
+ problems.push(
296
+ problem(
297
+ "keep-evidence-without-record",
298
+ "--keep-evidence declares which evidence a re-record leaves standing, "
299
+ + "and there is no re-record here. Pass --record, or drop the "
300
+ + "declaration so nothing reads it."
301
+ )
302
+ );
303
+ }
304
+ if (declaredKeep.length > 0 && compiled.diagnostics.length === 0) {
305
+ const labels = compiled.capsule.verification.commands.map(
306
+ (item) => item.label
307
+ );
308
+ const unknown = declaredKeep.filter((label) => !labels.includes(label));
309
+ if (unknown.length > 0) {
310
+ // Refused rather than ignored: the likeliest cause is a typo or a check
311
+ // that was renamed, and ignoring it would leave the author believing
312
+ // evidence was kept that was not.
313
+ problems.push(
314
+ problem(
315
+ "keep-evidence-unknown-check",
316
+ `--keep-evidence names ${unknown.join(", ")}, which this contract `
317
+ + `does not declare as a check. It declares ${labels.join(", ")}.`
318
+ )
319
+ );
320
+ }
321
+ }
291
322
  let anchorPlan = null;
292
323
  if (options.record && problems.length === 0) {
293
324
  anchorPlan = contractAnchorPlan(selection, task);
@@ -364,11 +395,30 @@ function taskStart(repo, options) {
364
395
  // call to the current agent's Review.
365
396
  const replaced = anchoredFingerprint(anchorPlan.previous);
366
397
  if (replaced && replaced !== compiled.fingerprint.value) {
398
+ // The gate still cannot judge which evidence survives, and this does not
399
+ // make it able to. What it can stop doing is saying "all of it" to an
400
+ // author who can see that one check's assertion did not change — because
401
+ // an author acting on that sentence in good faith re-runs everything,
402
+ // and issue #112 measured four such re-verifications in one session, one
403
+ // of which meant breaking a testbench and restoring it.
404
+ const kept = declaredKeep;
405
+ const labels = compiled.capsule.verification.commands.map(
406
+ (item) => item.label
407
+ );
408
+ const stale = labels.filter((label) => !kept.includes(label));
367
409
  result.warnings.push(
368
410
  `Re-recorded over a different contract: was sha256:${replaced}, now `
369
- + `sha256:${compiled.fingerprint.value}. Execution evidence produced `
370
- + "under the previous contract is stale; clear or re-verify it "
371
- + "before completing this task."
411
+ + `sha256:${compiled.fingerprint.value}. `
412
+ + (kept.length > 0
413
+ ? `Evidence for ${stale.length > 0 ? stale.join(", ") : "no check"}`
414
+ + " is stale; clear or re-verify it before completing this task. "
415
+ + `${kept.join(", ")} ${kept.length > 1 ? "were" : "was"} `
416
+ + "declared unaffected by this contract change — a declaration "
417
+ + "Keel records and does not verify, since it retains only the "
418
+ + "previous fingerprint and cannot compare a check's former text "
419
+ + "to its current one. State the reason in Reauthorizations."
420
+ : "Execution evidence produced under the previous contract is "
421
+ + "stale; clear or re-verify it before completing this task.")
372
422
  );
373
423
  }
374
424
  }