@christang/keel 5.52.0 → 5.53.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,4 @@
1
- <!-- keel:start version=5.52.0 -->
1
+ <!-- keel:start version=5.53.0 -->
2
2
  ## Keel Bootstrap
3
3
 
4
4
  - Start every session with `keel context`; OpenSpec artifacts and Git are the only durable authority — never native memory, goals, or transcripts.
package/package.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "name": "@christang/keel",
3
3
  "displayName": "Keel",
4
4
  "description": "Keel OpenSpec execution discipline CLI for Claude Code, Codex, and OpenCode.",
5
- "version": "5.52.0",
5
+ "version": "5.53.0",
6
6
  "license": "MIT",
7
7
  "repository": {
8
8
  "type": "git",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "keel",
3
- "version": "5.52.0",
3
+ "version": "5.53.0",
4
4
  "description": "Keel OpenSpec execution discipline: stateless continuity, task capsules, deterministic gates, and expectation alignment for Codex and Claude Code.",
5
5
  "author": {
6
6
  "name": "TanglmChris",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "keel",
3
- "version": "5.52.0",
3
+ "version": "5.53.0",
4
4
  "description": "Keel OpenSpec execution discipline: stateless continuity, task capsules, deterministic gates, and expectation alignment for Codex and Claude Code.",
5
5
  "author": {
6
6
  "name": "TanglmChris",
@@ -37,8 +37,8 @@ REQUIRED_SCRIPTS = [
37
37
  "scripts/validate_plugin.py",
38
38
  ]
39
39
 
40
- PACKAGE_VERSION = "5.52.0"
41
- PROTOCOL_VERSION = "5.52.0"
40
+ PACKAGE_VERSION = "5.53.0"
41
+ PROTOCOL_VERSION = "5.53.0"
42
42
  LEGACY_MANAGED_START = "<!-- keel:start version=2.1 -->"
43
43
  OPENSPEC_SCHEMA_NAME = "keel-spec-driven"
44
44
  # Mirrors KEEL_PACKAGE_NAME in scripts/install_to_repo.py, one of the two
@@ -24945,13 +24945,20 @@ def strategy_probe_task(
24945
24945
  " - README.md",
24946
24946
  " - Touch:",
24947
24947
  " - src/example.js",
24948
- f" - {form}:",
24949
24948
  ]
24950
- if strategy is not None:
24951
- label = "Strategy" if form == "Verify" else "Verification Strategy"
24952
- lines.append(f" - {label}: {strategy}")
24953
- if reason is not None:
24954
- lines.append(f" - Reason: {reason}")
24949
+ # In the expanded v3 form the strategy and its reason are task-level fields
24950
+ # beside `Commands`, not entries inside it.
24951
+ if form != "Verify":
24952
+ if strategy is not None:
24953
+ lines.append(f" - Verification Strategy: {strategy}")
24954
+ if reason is not None:
24955
+ lines.append(f" - Verification Reason: {reason}")
24956
+ lines.append(f" - {form}:")
24957
+ if form == "Verify":
24958
+ if strategy is not None:
24959
+ lines.append(f" - Strategy: {strategy}")
24960
+ if reason is not None:
24961
+ lines.append(f" - Reason: {reason}")
24955
24962
  lines.extend(f" - {entry}" for entry in commands)
24956
24963
  lines.extend(
24957
24964
  [
@@ -25516,6 +25523,295 @@ def validate_drift_names_where_to_look_scenario() -> int:
25516
25523
  return 0
25517
25524
 
25518
25525
 
25526
+ # Issue #112's minimal reproduction. An unfilled slot in an `M<n>` declaration
25527
+ # stopped the contract compiling, the completion path fell back to the expanded
25528
+ # v3 `Commands` field a compact task never declares, and every reference to an
25529
+ # `M<n>` was reported as naming a check the task does not declare — first, and
25530
+ # as somebody else's fault. The reporter calls it the only diagnostic in 149
25531
+ # invocations that made them edit the wrong file.
25532
+ def validate_reference_outlives_its_declaration_scenario() -> int:
25533
+ label = "a-reference-outlives-its-declaration"
25534
+
25535
+ def complete(root: Path, name: str, checks, findings, form="Verify"):
25536
+ repo = root / name
25537
+ task = strategy_probe_task(
25538
+ strategy="evidence-first",
25539
+ reason="fixture; nothing here can fail first",
25540
+ form=form,
25541
+ commands=checks,
25542
+ )
25543
+ task = task.replace("- [ ] 1.1", "- [x] 1.1")
25544
+ task = task.replace(" - Findings: none", f" - Findings: {findings}")
25545
+ for entry in checks:
25546
+ lbl = entry.split(":", 1)[0].split(" ")[0]
25547
+ task = task.replace(f" - {lbl}: pending", f" - {lbl}: pass. ran it.")
25548
+ write_gate_fixture(repo, tasks=task)
25549
+ started = run_keel(
25550
+ repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
25551
+ "--json", "--no-guard",
25552
+ )
25553
+ payload = json.loads(started.stdout)
25554
+ value = ((payload.get("contract") or {}).get("fingerprint") or {}).get(
25555
+ "value"
25556
+ ) or "0" * 64
25557
+ tasks_path = repo / "openspec/changes/demo/tasks.md"
25558
+ tasks_path.write_text(
25559
+ tasks_path.read_text(encoding="utf-8").replace(
25560
+ " - Contract: pending",
25561
+ f" - Contract: keel-task-capsule/v1 sha256:{value}",
25562
+ ),
25563
+ encoding="utf-8",
25564
+ )
25565
+ result = run_keel(
25566
+ repo, "gate", "task-complete", "--change", "demo", "--task", "1.1",
25567
+ "--json",
25568
+ )
25569
+ return json.loads(result.stdout)
25570
+
25571
+ slotted = (
25572
+ "M1: the first check asserts the public behavior",
25573
+ "M2: the second check runs make sta RUN=<experiment_id> and succeeds",
25574
+ )
25575
+ clean = (
25576
+ "M1: the first check asserts the public behavior",
25577
+ "M2: the second check runs the suite and succeeds",
25578
+ )
25579
+
25580
+ with tempfile.TemporaryDirectory(prefix="keel-reference-") as raw:
25581
+ root = Path(raw)
25582
+
25583
+ reported = complete(root, "slotted", slotted, "one. Resolved here: M2")
25584
+ text = problem_text(reported)
25585
+ if "<experiment_id>" not in text:
25586
+ report(
25587
+ f"{label}: the unfilled slot is no longer reported; {text!r}."
25588
+ )
25589
+ return 1
25590
+ if "not a check this task declares" in text:
25591
+ report(
25592
+ f"{label}: a reference to M2 was reported as undeclared because "
25593
+ f"M2's own declaration failed to compile; {text!r}."
25594
+ )
25595
+ return 1
25596
+
25597
+ settled = complete(root, "clean", clean, "one. Resolved here: M2")
25598
+ if settled.get("status") != "pass":
25599
+ report(
25600
+ f"{label}: the same task without the slot no longer passes; "
25601
+ f"{problem_text(settled)!r}."
25602
+ )
25603
+ return 1
25604
+
25605
+ # The set narrowed to the truth, not to everything.
25606
+ absent = complete(root, "absent", clean, "one. Resolved here: M9")
25607
+ if "not a check this task declares" not in problem_text(absent):
25608
+ report(
25609
+ f"{label}: a reference to a check the task never declared was "
25610
+ f"accepted; {problem_text(absent)!r}."
25611
+ )
25612
+ return 1
25613
+
25614
+ legacy = complete(
25615
+ root, "legacy", clean, "one. Resolved here: M2", form="Commands"
25616
+ )
25617
+ if legacy.get("status") != "pass":
25618
+ report(
25619
+ f"{label}: an expanded v3 task lost its declared labels; "
25620
+ f"{problem_text(legacy)!r}."
25621
+ )
25622
+ return 1
25623
+
25624
+ report(f"{label} scenario passed.")
25625
+ return 0
25626
+
25627
+
25628
+ # The red-green obligation is a static function of the strategy and the tags,
25629
+ # and the `Verify` block is complete at task-start while the Evidence is all
25630
+ # pending. Issue #112 hit it three times, this repository twice more: the rule
25631
+ # was first heard after the capsule was written, the task implemented, the
25632
+ # checks run, and the Evidence recorded.
25633
+ def validate_obligation_is_stated_early_scenario() -> int:
25634
+ label = "the-obligation-is-stated-early"
25635
+
25636
+ def start(root: Path, name: str, strategy: str, reason=None, checks=None):
25637
+ repo = root / name
25638
+ write_gate_fixture(
25639
+ repo,
25640
+ tasks=strategy_probe_task(
25641
+ strategy=strategy,
25642
+ reason=reason,
25643
+ commands=checks
25644
+ or (
25645
+ "M1: the first check asserts the public behavior",
25646
+ "M2 (regression): the second asserts something green stays green",
25647
+ ),
25648
+ ),
25649
+ )
25650
+ result = run_keel(
25651
+ repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
25652
+ "--json", "--no-guard",
25653
+ )
25654
+ return json.loads(result.stdout)
25655
+
25656
+ with tempfile.TemporaryDirectory(prefix="keel-obligation-") as raw:
25657
+ root = Path(raw)
25658
+
25659
+ redgreen = start(root, "redgreen", "vertical-tdd")
25660
+ if redgreen.get("status") != "pass":
25661
+ report(
25662
+ f"{label}: stating the obligation changed the verdict; "
25663
+ f"{problem_text(redgreen)!r}."
25664
+ )
25665
+ return 1
25666
+ warnings = " ".join(str(x) for x in (redgreen.get("warnings") or []))
25667
+ if ".red" not in warnings or ".green" not in warnings:
25668
+ report(
25669
+ f"{label}: task-start did not state the red-green obligation; "
25670
+ f"{warnings!r}."
25671
+ )
25672
+ return 1
25673
+ obligation = next(
25674
+ (w for w in redgreen["warnings"] if ".red" in str(w)), ""
25675
+ )
25676
+ if "M1" not in obligation:
25677
+ report(
25678
+ f"{label}: the obligation does not name the check that owes it; "
25679
+ f"{obligation!r}."
25680
+ )
25681
+ return 1
25682
+ if "M2" not in obligation or "regression" not in obligation:
25683
+ report(
25684
+ f"{label}: the obligation does not name the exempt check; "
25685
+ f"{obligation!r}."
25686
+ )
25687
+ return 1
25688
+
25689
+ quiet = start(
25690
+ root,
25691
+ "quiet",
25692
+ "evidence-first",
25693
+ reason="fixture; nothing here can fail first",
25694
+ )
25695
+ quiet_warnings = " ".join(str(x) for x in (quiet.get("warnings") or []))
25696
+ if ".red" in quiet_warnings:
25697
+ report(
25698
+ f"{label}: a strategy without red-green was told about it; "
25699
+ f"{quiet_warnings!r}."
25700
+ )
25701
+ return 1
25702
+
25703
+ # A task that fails task-start for another reason still fails with its
25704
+ # own problem: the warning is not a verdict and cannot mask one.
25705
+ broken = start(root, "broken", "made-up-thing")
25706
+ if broken.get("status") != "fail":
25707
+ report(f"{label}: an unsupported strategy stopped failing.")
25708
+ return 1
25709
+ if "unsupported" not in problem_text(broken):
25710
+ report(
25711
+ f"{label}: an unsupported strategy lost its own diagnostic; "
25712
+ f"{problem_text(broken)!r}."
25713
+ )
25714
+ return 1
25715
+
25716
+ report(f"{label} scenario passed.")
25717
+ return 0
25718
+
25719
+
25720
+ # Measured on issue #112's own shape: a two-check vertical-tdd task with no
25721
+ # red-green Evidence produced four problems in 827 characters, of which the
25722
+ # same 84-character rule sentence was four copies. It scales with the number
25723
+ # of checks, and the reporter estimates a third of failure output is this.
25724
+ def validate_explanation_is_printed_once_scenario() -> int:
25725
+ label = "an-explanation-is-printed-once"
25726
+ shared = "Tag the check `(regression)` if it asserts that something already green stays green."
25727
+ with tempfile.TemporaryDirectory(prefix="keel-explanation-") as raw:
25728
+ root = Path(raw)
25729
+ repo = root / "repo"
25730
+ task = strategy_probe_task(
25731
+ strategy="vertical-tdd",
25732
+ commands=(
25733
+ "M1: the first check asserts the public behavior",
25734
+ "M2: the second check asserts the public behavior",
25735
+ ),
25736
+ )
25737
+ task = task.replace("- [ ] 1.1", "- [x] 1.1")
25738
+ for lbl in ("M1", "M2"):
25739
+ task = task.replace(f" - {lbl}: pending", f" - {lbl}: pass. ran it.")
25740
+ write_gate_fixture(repo, tasks=task)
25741
+ started = run_keel(
25742
+ repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
25743
+ "--json", "--no-guard",
25744
+ )
25745
+ value = json.loads(started.stdout)["contract"]["fingerprint"]["value"]
25746
+ tasks_path = repo / "openspec/changes/demo/tasks.md"
25747
+ tasks_path.write_text(
25748
+ tasks_path.read_text(encoding="utf-8").replace(
25749
+ " - Contract: pending",
25750
+ f" - Contract: keel-task-capsule/v1 sha256:{value}",
25751
+ ),
25752
+ encoding="utf-8",
25753
+ )
25754
+
25755
+ rendered = run_keel(
25756
+ repo, "gate", "task-complete", "--change", "demo", "--task", "1.1"
25757
+ ).stdout
25758
+ copies = rendered.count(shared)
25759
+ if copies != 1:
25760
+ report(
25761
+ f"{label}: the shared rule explanation appears {copies} times "
25762
+ "in the rendered text; it belongs once per run."
25763
+ )
25764
+ return 1
25765
+ problems = [
25766
+ line for line in rendered.splitlines() if line.startswith("Problem:")
25767
+ ]
25768
+ if len(problems) != 4:
25769
+ report(
25770
+ f"{label}: expected the four per-label problems, got "
25771
+ f"{len(problems)}: {problems!r}."
25772
+ )
25773
+ return 1
25774
+ for needed in ("M1.red", "M1.green", "M2.red", "M2.green"):
25775
+ if not any(needed in line for line in problems):
25776
+ report(
25777
+ f"{label}: {needed} lost its own problem line; {problems!r}."
25778
+ )
25779
+ return 1
25780
+ if len(rendered) >= 827:
25781
+ report(
25782
+ f"{label}: the rendered failure is {len(rendered)} characters, "
25783
+ "no shorter than the 827 measured before deduplication."
25784
+ )
25785
+ return 1
25786
+
25787
+ payload = json.loads(
25788
+ run_keel(
25789
+ repo, "gate", "task-complete", "--change", "demo", "--task",
25790
+ "1.1", "--json",
25791
+ ).stdout
25792
+ )
25793
+ carried = [
25794
+ entry for entry in (payload.get("problems") or [])
25795
+ if entry.get("code") == "missing-strategy-evidence"
25796
+ ]
25797
+ if len(carried) != 4:
25798
+ report(
25799
+ f"{label}: the JSON result no longer carries all four problems; "
25800
+ f"{len(carried)}."
25801
+ )
25802
+ return 1
25803
+ if not all(shared in json.dumps(entry, ensure_ascii=False) for entry in carried):
25804
+ report(
25805
+ f"{label}: the JSON result was deduplicated; a consumer reading "
25806
+ "problems individually would get a payload whose content "
25807
+ "depends on position."
25808
+ )
25809
+ return 1
25810
+
25811
+ report(f"{label} scenario passed.")
25812
+ return 0
25813
+
25814
+
25519
25815
  # A scenario name, as the registry spells one. Two registered names carry no
25520
25816
  # hyphen — `cli` and `uninstall` — so requiring one would leave exactly those
25521
25817
  # two unchecked, and allowing single words was measured to add no false
@@ -25751,6 +26047,9 @@ SCENARIOS: tuple = (
25751
26047
  ("the-weakest-strategy-states-its-reason", validate_weakest_strategy_states_its_reason_scenario),
25752
26048
  ("a-quoted-marker-is-not-a-disposition", validate_quoted_marker_is_not_a_disposition_scenario),
25753
26049
  ("drift-names-where-to-look", validate_drift_names_where_to_look_scenario),
26050
+ ("a-reference-outlives-its-declaration", validate_reference_outlives_its_declaration_scenario),
26051
+ ("the-obligation-is-stated-early", validate_obligation_is_stated_early_scenario),
26052
+ ("an-explanation-is-printed-once", validate_explanation_is_printed_once_scenario),
25754
26053
  (
25755
26054
  "authored-scenario-names-are-registered",
25756
26055
  validate_authored_scenario_names_scenario,
package/src/core/gates.js CHANGED
@@ -9,6 +9,7 @@ const {
9
9
  ACCEPTED_REVIEW_STATUSES,
10
10
  RED_GREEN_VERIFICATION_STRATEGIES,
11
11
  compileTaskContract,
12
+ declaredCommandLabels,
12
13
  field,
13
14
  isConcrete,
14
15
  isPassingReviewStatus,
@@ -21,8 +22,15 @@ const GATE_STAGES = new Set(["task-start", "task-complete", "change-close"]);
21
22
 
22
23
  class GateInputError extends Error {}
23
24
 
24
- function problem(code, message) {
25
- return { code, message };
25
+ // `note` is the rule behind the problem rather than the problem itself. Every
26
+ // problem carries its own copy in the JSON result, because a consumer reading
27
+ // problems one at a time must not get a payload whose content depends on
28
+ // position; the text renderer prints each distinct note once, because there
29
+ // the repetition is what crowds out the specific lines. Measured on a
30
+ // two-check task: the same 84-character sentence four times in 827 characters,
31
+ // and it scales with the number of checks.
32
+ function problem(code, message, note = null) {
33
+ return note ? { code, message, note } : { code, message };
26
34
  }
27
35
 
28
36
  function gateResult(
@@ -212,6 +220,34 @@ function anchoredFingerprint(previous) {
212
220
  // way to acknowledge a `needs-review`, so making it one would leave a
213
221
  // legitimate split unstartable. The reader is given the other task's id and
214
222
  // compares two things, rather than being told something is wrong.
223
+ // What completion will require, said while the Evidence is still all pending.
224
+ // The `Verify` block is complete at task-start and the obligation is a static
225
+ // function of the strategy and the tags, so the only reason it was first heard
226
+ // at task-complete is that nobody said it earlier — after the capsule was
227
+ // written, the task implemented, the checks run, and the Evidence recorded.
228
+ // A warning, never a refusal: a task whose author has not yet decided which
229
+ // check is a regression is not malformed, it is unfinished.
230
+ function redGreenObligation(compiled) {
231
+ if (!compiled || compiled.diagnostics.length > 0) return [];
232
+ const strategy = compiled.capsule.verification.strategy.toLowerCase();
233
+ if (!RED_GREEN_VERIFICATION_STRATEGIES.has(strategy)) return [];
234
+ const commands = compiled.capsule.verification.commands;
235
+ const owing = commands.filter((item) => !item.regression).map((item) => item.label);
236
+ const exempt = commands.filter((item) => item.regression).map((item) => item.label);
237
+ if (owing.length === 0) return [];
238
+ return [
239
+ `${strategy} will require concrete .red and .green Evidence at completion `
240
+ + `for ${owing.join(", ")}`
241
+ + (exempt.length > 0
242
+ ? `; ${exempt.join(", ")} ${exempt.length > 1 ? "are" : "is"} exempt as `
243
+ + "(regression)"
244
+ : "")
245
+ + ". Tag a check `(regression)` now if it asserts that something already "
246
+ + "green stays green, rather than discovering the obligation once the "
247
+ + "checks have been run.",
248
+ ];
249
+ }
250
+
215
251
  function taskShapeWarnings(repo, selection, task, compiled) {
216
252
  if (!compiled || compiled.diagnostics.length > 0) return [];
217
253
  const strategy = compiled.capsule.verification.strategy.toLowerCase();
@@ -272,7 +308,10 @@ function taskStart(repo, options) {
272
308
  selection.change,
273
309
  [task.id],
274
310
  problems,
275
- taskShapeWarnings(repo, selection, task, compiled),
311
+ [
312
+ ...redGreenObligation(compiled),
313
+ ...taskShapeWarnings(repo, selection, task, compiled),
314
+ ],
276
315
  problems.length === 0
277
316
  ? compiled
278
317
  : null
@@ -887,9 +926,13 @@ function attributeChanged(repo, task, changedList, contract, change, tasks) {
887
926
 
888
927
  function completionChecks(repo, task, contract = null, changeVerify = null, change = null) {
889
928
  const problems = [];
929
+ // Not the compiled capsule alone: when a check's declaration does not
930
+ // compile there is no capsule, and reconstructing the set from the expanded
931
+ // v3 `Commands` field a compact task never declares makes every reference
932
+ // look undeclared. The task's own declarations answer in both cases.
890
933
  const commands = contract
891
934
  ? contract.capsule.verification.commands.map((item) => item.label)
892
- : commandLabels(task);
935
+ : declaredCommandLabels(task);
893
936
  // With no contract, the labels came from the expanded v3 `Commands` field,
894
937
  // which a compact task never declares — so their absence is a fact about the
895
938
  // fallback, not about the task. The compiler's own diagnostics are already in
@@ -926,8 +969,9 @@ function completionChecks(repo, task, contract = null, changeVerify = null, chan
926
969
  problem(
927
970
  "missing-strategy-evidence",
928
971
  `${strategy} requires concrete ${label}.${phase} Evidence for `
929
- + "the same behavior check. Tag the check `(regression)` if it "
930
- + "asserts that something already green stays green."
972
+ + "the same behavior check.",
973
+ "Tag the check `(regression)` if it asserts that something "
974
+ + "already green stays green."
931
975
  )
932
976
  );
933
977
  }
@@ -1626,6 +1670,11 @@ function renderGate(result) {
1626
1670
  : ""),
1627
1671
  ];
1628
1672
  for (const item of result.problems) lines.push(`Problem: ${item.message}`);
1673
+ for (const note of [
1674
+ ...new Set(result.problems.map((item) => item.note).filter(Boolean)),
1675
+ ]) {
1676
+ lines.push(`Note: ${note}`);
1677
+ }
1629
1678
  for (const warning of result.warnings) lines.push(`Warning: ${warning}`);
1630
1679
  if (result.contract) {
1631
1680
  lines.push(
@@ -229,6 +229,21 @@ function verification(task) {
229
229
  };
230
230
  }
231
231
 
232
+ // The labels a task declares, read from the form the task itself uses. Label
233
+ // parsing does not depend on a check being concrete, so this answers exactly
234
+ // when the compiler cannot: a task whose `M2` declaration carries an unfilled
235
+ // slot still declares `M2`, and a reference to it is not a reference to
236
+ // something that does not exist.
237
+ function declaredCommandLabels(task) {
238
+ const source = fieldValues(task, "Verify").length > 0
239
+ ? fieldValues(task, "Verify")
240
+ : fieldValues(task, "Commands");
241
+ return source
242
+ .map((entry) => entry.match(/^(M[1-9]\d*)(?:\s*\([^)\n]*\))?\s*:/))
243
+ .filter(Boolean)
244
+ .map((match) => match[1]);
245
+ }
246
+
232
247
  function commandLabelProblems(task) {
233
248
  // A task that declared no verification form at all is reported once, by
234
249
  // requiredFieldProblems, as the one field it is missing. Its orphan Evidence
@@ -1195,6 +1210,7 @@ function loadTaskContract(repo, change, taskId) {
1195
1210
 
1196
1211
  module.exports = {
1197
1212
  ACCEPTED_REVIEW_STATUSES,
1213
+ declaredCommandLabels,
1198
1214
  RED_GREEN_VERIFICATION_STRATEGIES,
1199
1215
  SUPPORTED_VERIFICATION_STRATEGIES,
1200
1216
  compileTaskContract,