@christang/keel 5.52.0 → 5.54.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -333,6 +333,29 @@ keel lenses add web # copy the web template into keel/lenses/web.md, the
333
333
  keel lenses add web --force # overwrite an existing lens
334
334
  ```
335
335
 
336
+ ## Pausing a change
337
+
338
+ A change you have deliberately stopped — waiting on something outside the repository, or simply
339
+ not the priority — can say so where it lives, in its own `openspec/changes/<name>/.openspec.yaml`:
340
+
341
+ ```yaml
342
+ schema: keel-spec-driven
343
+ created: 2026-09-05
344
+ keel:
345
+ status: paused
346
+ reason: waiting on the competition brief; the priority is knowledge that needs no tooling
347
+ since: 2026-09-05
348
+ ```
349
+
350
+ `keel context` then passes over it when inferring what to do next, and says which change it passed
351
+ over and why — a skip you cannot see would be worse than the wrong recommendation it replaces. If
352
+ every active change is paused, `context` reports that, with each reason, rather than reporting that
353
+ nothing exists.
354
+
355
+ This changes only what Keel *recommends*. No gate, the write guard, and completion all behave
356
+ exactly as they would without it, and `keel context --change <paused>` still selects it — you asked
357
+ for it by name. Keel never pauses a change on its own.
358
+
336
359
  ## Commands
337
360
 
338
361
  ```bash
@@ -1,4 +1,4 @@
1
- <!-- keel:start version=5.52.0 -->
1
+ <!-- keel:start version=5.54.0 -->
2
2
  ## Keel Bootstrap
3
3
 
4
4
  - Start every session with `keel context`; OpenSpec artifacts and Git are the only durable authority — never native memory, goals, or transcripts.
package/package.json CHANGED
@@ -2,7 +2,7 @@
2
2
  "name": "@christang/keel",
3
3
  "displayName": "Keel",
4
4
  "description": "Keel OpenSpec execution discipline CLI for Claude Code, Codex, and OpenCode.",
5
- "version": "5.52.0",
5
+ "version": "5.54.0",
6
6
  "license": "MIT",
7
7
  "repository": {
8
8
  "type": "git",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "keel",
3
- "version": "5.52.0",
3
+ "version": "5.54.0",
4
4
  "description": "Keel OpenSpec execution discipline: stateless continuity, task capsules, deterministic gates, and expectation alignment for Codex and Claude Code.",
5
5
  "author": {
6
6
  "name": "TanglmChris",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "keel",
3
- "version": "5.52.0",
3
+ "version": "5.54.0",
4
4
  "description": "Keel OpenSpec execution discipline: stateless continuity, task capsules, deterministic gates, and expectation alignment for Codex and Claude Code.",
5
5
  "author": {
6
6
  "name": "TanglmChris",
@@ -37,8 +37,8 @@ REQUIRED_SCRIPTS = [
37
37
  "scripts/validate_plugin.py",
38
38
  ]
39
39
 
40
- PACKAGE_VERSION = "5.52.0"
41
- PROTOCOL_VERSION = "5.52.0"
40
+ PACKAGE_VERSION = "5.54.0"
41
+ PROTOCOL_VERSION = "5.54.0"
42
42
  LEGACY_MANAGED_START = "<!-- keel:start version=2.1 -->"
43
43
  OPENSPEC_SCHEMA_NAME = "keel-spec-driven"
44
44
  # Mirrors KEEL_PACKAGE_NAME in scripts/install_to_repo.py, one of the two
@@ -24945,13 +24945,20 @@ def strategy_probe_task(
24945
24945
  " - README.md",
24946
24946
  " - Touch:",
24947
24947
  " - src/example.js",
24948
- f" - {form}:",
24949
24948
  ]
24950
- if strategy is not None:
24951
- label = "Strategy" if form == "Verify" else "Verification Strategy"
24952
- lines.append(f" - {label}: {strategy}")
24953
- if reason is not None:
24954
- lines.append(f" - Reason: {reason}")
24949
+ # In the expanded v3 form the strategy and its reason are task-level fields
24950
+ # beside `Commands`, not entries inside it.
24951
+ if form != "Verify":
24952
+ if strategy is not None:
24953
+ lines.append(f" - Verification Strategy: {strategy}")
24954
+ if reason is not None:
24955
+ lines.append(f" - Verification Reason: {reason}")
24956
+ lines.append(f" - {form}:")
24957
+ if form == "Verify":
24958
+ if strategy is not None:
24959
+ lines.append(f" - Strategy: {strategy}")
24960
+ if reason is not None:
24961
+ lines.append(f" - Reason: {reason}")
24955
24962
  lines.extend(f" - {entry}" for entry in commands)
24956
24963
  lines.extend(
24957
24964
  [
@@ -25516,6 +25523,451 @@ def validate_drift_names_where_to_look_scenario() -> int:
25516
25523
  return 0
25517
25524
 
25518
25525
 
25526
+ # Issue #112's minimal reproduction. An unfilled slot in an `M<n>` declaration
25527
+ # stopped the contract compiling, the completion path fell back to the expanded
25528
+ # v3 `Commands` field a compact task never declares, and every reference to an
25529
+ # `M<n>` was reported as naming a check the task does not declare — first, and
25530
+ # as somebody else's fault. The reporter calls it the only diagnostic in 149
25531
+ # invocations that made them edit the wrong file.
25532
+ def validate_reference_outlives_its_declaration_scenario() -> int:
25533
+ label = "a-reference-outlives-its-declaration"
25534
+
25535
+ def complete(root: Path, name: str, checks, findings, form="Verify"):
25536
+ repo = root / name
25537
+ task = strategy_probe_task(
25538
+ strategy="evidence-first",
25539
+ reason="fixture; nothing here can fail first",
25540
+ form=form,
25541
+ commands=checks,
25542
+ )
25543
+ task = task.replace("- [ ] 1.1", "- [x] 1.1")
25544
+ task = task.replace(" - Findings: none", f" - Findings: {findings}")
25545
+ for entry in checks:
25546
+ lbl = entry.split(":", 1)[0].split(" ")[0]
25547
+ task = task.replace(f" - {lbl}: pending", f" - {lbl}: pass. ran it.")
25548
+ write_gate_fixture(repo, tasks=task)
25549
+ started = run_keel(
25550
+ repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
25551
+ "--json", "--no-guard",
25552
+ )
25553
+ payload = json.loads(started.stdout)
25554
+ value = ((payload.get("contract") or {}).get("fingerprint") or {}).get(
25555
+ "value"
25556
+ ) or "0" * 64
25557
+ tasks_path = repo / "openspec/changes/demo/tasks.md"
25558
+ tasks_path.write_text(
25559
+ tasks_path.read_text(encoding="utf-8").replace(
25560
+ " - Contract: pending",
25561
+ f" - Contract: keel-task-capsule/v1 sha256:{value}",
25562
+ ),
25563
+ encoding="utf-8",
25564
+ )
25565
+ result = run_keel(
25566
+ repo, "gate", "task-complete", "--change", "demo", "--task", "1.1",
25567
+ "--json",
25568
+ )
25569
+ return json.loads(result.stdout)
25570
+
25571
+ slotted = (
25572
+ "M1: the first check asserts the public behavior",
25573
+ "M2: the second check runs make sta RUN=<experiment_id> and succeeds",
25574
+ )
25575
+ clean = (
25576
+ "M1: the first check asserts the public behavior",
25577
+ "M2: the second check runs the suite and succeeds",
25578
+ )
25579
+
25580
+ with tempfile.TemporaryDirectory(prefix="keel-reference-") as raw:
25581
+ root = Path(raw)
25582
+
25583
+ reported = complete(root, "slotted", slotted, "one. Resolved here: M2")
25584
+ text = problem_text(reported)
25585
+ if "<experiment_id>" not in text:
25586
+ report(
25587
+ f"{label}: the unfilled slot is no longer reported; {text!r}."
25588
+ )
25589
+ return 1
25590
+ if "not a check this task declares" in text:
25591
+ report(
25592
+ f"{label}: a reference to M2 was reported as undeclared because "
25593
+ f"M2's own declaration failed to compile; {text!r}."
25594
+ )
25595
+ return 1
25596
+
25597
+ settled = complete(root, "clean", clean, "one. Resolved here: M2")
25598
+ if settled.get("status") != "pass":
25599
+ report(
25600
+ f"{label}: the same task without the slot no longer passes; "
25601
+ f"{problem_text(settled)!r}."
25602
+ )
25603
+ return 1
25604
+
25605
+ # The set narrowed to the truth, not to everything.
25606
+ absent = complete(root, "absent", clean, "one. Resolved here: M9")
25607
+ if "not a check this task declares" not in problem_text(absent):
25608
+ report(
25609
+ f"{label}: a reference to a check the task never declared was "
25610
+ f"accepted; {problem_text(absent)!r}."
25611
+ )
25612
+ return 1
25613
+
25614
+ legacy = complete(
25615
+ root, "legacy", clean, "one. Resolved here: M2", form="Commands"
25616
+ )
25617
+ if legacy.get("status") != "pass":
25618
+ report(
25619
+ f"{label}: an expanded v3 task lost its declared labels; "
25620
+ f"{problem_text(legacy)!r}."
25621
+ )
25622
+ return 1
25623
+
25624
+ report(f"{label} scenario passed.")
25625
+ return 0
25626
+
25627
+
25628
+ # The red-green obligation is a static function of the strategy and the tags,
25629
+ # and the `Verify` block is complete at task-start while the Evidence is all
25630
+ # pending. Issue #112 hit it three times, this repository twice more: the rule
25631
+ # was first heard after the capsule was written, the task implemented, the
25632
+ # checks run, and the Evidence recorded.
25633
+ def validate_obligation_is_stated_early_scenario() -> int:
25634
+ label = "the-obligation-is-stated-early"
25635
+
25636
+ def start(root: Path, name: str, strategy: str, reason=None, checks=None):
25637
+ repo = root / name
25638
+ write_gate_fixture(
25639
+ repo,
25640
+ tasks=strategy_probe_task(
25641
+ strategy=strategy,
25642
+ reason=reason,
25643
+ commands=checks
25644
+ or (
25645
+ "M1: the first check asserts the public behavior",
25646
+ "M2 (regression): the second asserts something green stays green",
25647
+ ),
25648
+ ),
25649
+ )
25650
+ result = run_keel(
25651
+ repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
25652
+ "--json", "--no-guard",
25653
+ )
25654
+ return json.loads(result.stdout)
25655
+
25656
+ with tempfile.TemporaryDirectory(prefix="keel-obligation-") as raw:
25657
+ root = Path(raw)
25658
+
25659
+ redgreen = start(root, "redgreen", "vertical-tdd")
25660
+ if redgreen.get("status") != "pass":
25661
+ report(
25662
+ f"{label}: stating the obligation changed the verdict; "
25663
+ f"{problem_text(redgreen)!r}."
25664
+ )
25665
+ return 1
25666
+ warnings = " ".join(str(x) for x in (redgreen.get("warnings") or []))
25667
+ if ".red" not in warnings or ".green" not in warnings:
25668
+ report(
25669
+ f"{label}: task-start did not state the red-green obligation; "
25670
+ f"{warnings!r}."
25671
+ )
25672
+ return 1
25673
+ obligation = next(
25674
+ (w for w in redgreen["warnings"] if ".red" in str(w)), ""
25675
+ )
25676
+ if "M1" not in obligation:
25677
+ report(
25678
+ f"{label}: the obligation does not name the check that owes it; "
25679
+ f"{obligation!r}."
25680
+ )
25681
+ return 1
25682
+ if "M2" not in obligation or "regression" not in obligation:
25683
+ report(
25684
+ f"{label}: the obligation does not name the exempt check; "
25685
+ f"{obligation!r}."
25686
+ )
25687
+ return 1
25688
+
25689
+ quiet = start(
25690
+ root,
25691
+ "quiet",
25692
+ "evidence-first",
25693
+ reason="fixture; nothing here can fail first",
25694
+ )
25695
+ quiet_warnings = " ".join(str(x) for x in (quiet.get("warnings") or []))
25696
+ if ".red" in quiet_warnings:
25697
+ report(
25698
+ f"{label}: a strategy without red-green was told about it; "
25699
+ f"{quiet_warnings!r}."
25700
+ )
25701
+ return 1
25702
+
25703
+ # A task that fails task-start for another reason still fails with its
25704
+ # own problem: the warning is not a verdict and cannot mask one.
25705
+ broken = start(root, "broken", "made-up-thing")
25706
+ if broken.get("status") != "fail":
25707
+ report(f"{label}: an unsupported strategy stopped failing.")
25708
+ return 1
25709
+ if "unsupported" not in problem_text(broken):
25710
+ report(
25711
+ f"{label}: an unsupported strategy lost its own diagnostic; "
25712
+ f"{problem_text(broken)!r}."
25713
+ )
25714
+ return 1
25715
+
25716
+ report(f"{label} scenario passed.")
25717
+ return 0
25718
+
25719
+
25720
+ # Measured on issue #112's own shape: a two-check vertical-tdd task with no
25721
+ # red-green Evidence produced four problems in 827 characters, of which the
25722
+ # same 84-character rule sentence was four copies. It scales with the number
25723
+ # of checks, and the reporter estimates a third of failure output is this.
25724
+ def validate_explanation_is_printed_once_scenario() -> int:
25725
+ label = "an-explanation-is-printed-once"
25726
+ shared = "Tag the check `(regression)` if it asserts that something already green stays green."
25727
+ with tempfile.TemporaryDirectory(prefix="keel-explanation-") as raw:
25728
+ root = Path(raw)
25729
+ repo = root / "repo"
25730
+ task = strategy_probe_task(
25731
+ strategy="vertical-tdd",
25732
+ commands=(
25733
+ "M1: the first check asserts the public behavior",
25734
+ "M2: the second check asserts the public behavior",
25735
+ ),
25736
+ )
25737
+ task = task.replace("- [ ] 1.1", "- [x] 1.1")
25738
+ for lbl in ("M1", "M2"):
25739
+ task = task.replace(f" - {lbl}: pending", f" - {lbl}: pass. ran it.")
25740
+ write_gate_fixture(repo, tasks=task)
25741
+ started = run_keel(
25742
+ repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
25743
+ "--json", "--no-guard",
25744
+ )
25745
+ value = json.loads(started.stdout)["contract"]["fingerprint"]["value"]
25746
+ tasks_path = repo / "openspec/changes/demo/tasks.md"
25747
+ tasks_path.write_text(
25748
+ tasks_path.read_text(encoding="utf-8").replace(
25749
+ " - Contract: pending",
25750
+ f" - Contract: keel-task-capsule/v1 sha256:{value}",
25751
+ ),
25752
+ encoding="utf-8",
25753
+ )
25754
+
25755
+ rendered = run_keel(
25756
+ repo, "gate", "task-complete", "--change", "demo", "--task", "1.1"
25757
+ ).stdout
25758
+ copies = rendered.count(shared)
25759
+ if copies != 1:
25760
+ report(
25761
+ f"{label}: the shared rule explanation appears {copies} times "
25762
+ "in the rendered text; it belongs once per run."
25763
+ )
25764
+ return 1
25765
+ problems = [
25766
+ line for line in rendered.splitlines() if line.startswith("Problem:")
25767
+ ]
25768
+ if len(problems) != 4:
25769
+ report(
25770
+ f"{label}: expected the four per-label problems, got "
25771
+ f"{len(problems)}: {problems!r}."
25772
+ )
25773
+ return 1
25774
+ for needed in ("M1.red", "M1.green", "M2.red", "M2.green"):
25775
+ if not any(needed in line for line in problems):
25776
+ report(
25777
+ f"{label}: {needed} lost its own problem line; {problems!r}."
25778
+ )
25779
+ return 1
25780
+ if len(rendered) >= 827:
25781
+ report(
25782
+ f"{label}: the rendered failure is {len(rendered)} characters, "
25783
+ "no shorter than the 827 measured before deduplication."
25784
+ )
25785
+ return 1
25786
+
25787
+ payload = json.loads(
25788
+ run_keel(
25789
+ repo, "gate", "task-complete", "--change", "demo", "--task",
25790
+ "1.1", "--json",
25791
+ ).stdout
25792
+ )
25793
+ carried = [
25794
+ entry for entry in (payload.get("problems") or [])
25795
+ if entry.get("code") == "missing-strategy-evidence"
25796
+ ]
25797
+ if len(carried) != 4:
25798
+ report(
25799
+ f"{label}: the JSON result no longer carries all four problems; "
25800
+ f"{len(carried)}."
25801
+ )
25802
+ return 1
25803
+ if not all(shared in json.dumps(entry, ensure_ascii=False) for entry in carried):
25804
+ report(
25805
+ f"{label}: the JSON result was deduplicated; a consumer reading "
25806
+ "problems individually would get a payload whose content "
25807
+ "depends on position."
25808
+ )
25809
+ return 1
25810
+
25811
+ report(f"{label} scenario passed.")
25812
+ return 0
25813
+
25814
+
25815
+ # Issue #112's most valuable finding: a change its owner had stopped kept being
25816
+ # recommended, and the workaround was a paragraph in the project's CLAUDE.md
25817
+ # telling future sessions to ignore keel's own primary output.
25818
+ def validate_paused_change_is_not_the_next_action_scenario() -> int:
25819
+ label = "a-paused-change-is-not-the-next-action"
25820
+
25821
+ def write_change(repo: Path, name: str, keel_block: str | None) -> None:
25822
+ task = strategy_probe_task(strategy="vertical-tdd")
25823
+ change = repo / "openspec" / "changes" / name
25824
+ write_text(
25825
+ change / "tasks.md",
25826
+ "# Tasks\n\n## Invalidates\n\n- None.\n\n## Expectation Coverage\n\n"
25827
+ "- E1:\n - Covered by: 1.1\n\n## Tasks\n\n" + task,
25828
+ )
25829
+ write_text(change / "proposal.md", "# Proposal\n")
25830
+ write_text(change / "design.md", "## Context\n\nfixture\n")
25831
+ write_text(change / "specs/demo/spec.md", "## ADDED Requirements\n")
25832
+ write_text(
25833
+ change / ".openspec.yaml",
25834
+ "schema: keel-spec-driven\ncreated: 2026-09-08\n"
25835
+ + (keel_block or ""),
25836
+ )
25837
+
25838
+ paused_block = (
25839
+ "keel:\n"
25840
+ " status: paused\n"
25841
+ " reason: 打分寻优要等赛题,当前优先级是不依赖工具的知识沉淀\n"
25842
+ " since: 2026-09-05\n"
25843
+ )
25844
+
25845
+ def context(repo: Path, *args):
25846
+ result = run_keel(repo, "context", "--json", *args)
25847
+ try:
25848
+ return json.loads(result.stdout)
25849
+ except json.JSONDecodeError:
25850
+ return {"status": "unparsed", "reasons": [result.stdout[:300]]}
25851
+
25852
+ def said(payload: dict) -> str:
25853
+ return " ".join(
25854
+ str(x) for x in
25855
+ (payload.get("reasons") or []) + (payload.get("warnings") or [])
25856
+ )
25857
+
25858
+ with tempfile.TemporaryDirectory(prefix="keel-paused-") as raw:
25859
+ root = Path(raw)
25860
+
25861
+ # Without the declaration the pair is ambiguous, so the skip is what
25862
+ # changes the answer rather than the fixture shape.
25863
+ both = root / "both"
25864
+ write_change(both, "alpha", None)
25865
+ write_change(both, "beta", None)
25866
+ if context(both).get("status") != "ambiguous":
25867
+ report(
25868
+ f"{label}: two unpaused changes were not ambiguous, so the "
25869
+ "fixture does not isolate what pausing changes."
25870
+ )
25871
+ return 1
25872
+
25873
+ mixed = root / "mixed"
25874
+ write_change(mixed, "alpha", paused_block)
25875
+ write_change(mixed, "beta", None)
25876
+ picked = context(mixed)
25877
+ if picked.get("status") != "ready":
25878
+ report(
25879
+ f"{label}: inference did not pass over the paused change; "
25880
+ f"{picked.get('status')!r} {said(picked)!r}."
25881
+ )
25882
+ return 1
25883
+ if (picked.get("selection") or {}).get("change") != "beta":
25884
+ report(
25885
+ f"{label}: inference selected {picked.get('selection')!r} "
25886
+ "rather than the change that is not paused."
25887
+ )
25888
+ return 1
25889
+ spoken = said(picked)
25890
+ if "alpha" not in spoken or "打分寻优" not in spoken:
25891
+ report(
25892
+ f"{label}: the paused change was skipped silently or without "
25893
+ f"its reason; {spoken!r}."
25894
+ )
25895
+ return 1
25896
+
25897
+ allpaused = root / "allpaused"
25898
+ write_change(allpaused, "alpha", paused_block)
25899
+ write_change(allpaused, "beta", paused_block)
25900
+ idle = context(allpaused)
25901
+ if idle.get("status") != "idle":
25902
+ report(
25903
+ f"{label}: a repository whose every change is paused reported "
25904
+ f"{idle.get('status')!r}."
25905
+ )
25906
+ return 1
25907
+ spoken = said(idle)
25908
+ for needed in ("alpha", "beta", "打分寻优"):
25909
+ if needed not in spoken:
25910
+ report(
25911
+ f"{label}: the all-paused report omits {needed!r}; "
25912
+ f"{spoken!r}."
25913
+ )
25914
+ return 1
25915
+
25916
+ explicit = context(mixed, "--change", "alpha")
25917
+ if (explicit.get("selection") or {}).get("change") != "alpha":
25918
+ report(
25919
+ f"{label}: explicit selection no longer reaches a paused "
25920
+ f"change; {explicit.get('status')!r} {said(explicit)!r}."
25921
+ )
25922
+ return 1
25923
+ if "paused" not in said(explicit).lower():
25924
+ report(
25925
+ f"{label}: explicit selection did not report that the change "
25926
+ f"is paused; {said(explicit)!r}."
25927
+ )
25928
+ return 1
25929
+
25930
+ # D6: a declaration Keel cannot read pauses nothing.
25931
+ broken = root / "broken"
25932
+ write_change(broken, "alpha", "keel:\n status: perhaps-later\n")
25933
+ write_change(broken, "beta", None)
25934
+ unreadable = context(broken)
25935
+ if unreadable.get("status") != "ambiguous":
25936
+ report(
25937
+ f"{label}: an unreadable declaration removed a change from "
25938
+ f"inference; {unreadable.get('status')!r} {said(unreadable)!r}."
25939
+ )
25940
+ return 1
25941
+ if "perhaps-later" not in said(unreadable):
25942
+ report(
25943
+ f"{label}: an unreadable declaration was not reported; "
25944
+ f"{said(unreadable)!r}."
25945
+ )
25946
+ return 1
25947
+
25948
+ # D4: pausing is not a gate.
25949
+ gated = run_keel(
25950
+ mixed, "gate", "task-start", "--change", "alpha", "--task", "1.1",
25951
+ "--json", "--no-guard",
25952
+ )
25953
+ ungated = run_keel(
25954
+ both, "gate", "task-start", "--change", "alpha", "--task", "1.1",
25955
+ "--json", "--no-guard",
25956
+ )
25957
+ if json.loads(gated.stdout).get("status") != json.loads(
25958
+ ungated.stdout
25959
+ ).get("status"):
25960
+ report(
25961
+ f"{label}: pausing changed a gate verdict — paused "
25962
+ f"{json.loads(gated.stdout).get('status')!r} against unpaused "
25963
+ f"{json.loads(ungated.stdout).get('status')!r}."
25964
+ )
25965
+ return 1
25966
+
25967
+ report(f"{label} scenario passed.")
25968
+ return 0
25969
+
25970
+
25519
25971
  # A scenario name, as the registry spells one. Two registered names carry no
25520
25972
  # hyphen — `cli` and `uninstall` — so requiring one would leave exactly those
25521
25973
  # two unchecked, and allowing single words was measured to add no false
@@ -25751,6 +26203,10 @@ SCENARIOS: tuple = (
25751
26203
  ("the-weakest-strategy-states-its-reason", validate_weakest_strategy_states_its_reason_scenario),
25752
26204
  ("a-quoted-marker-is-not-a-disposition", validate_quoted_marker_is_not_a_disposition_scenario),
25753
26205
  ("drift-names-where-to-look", validate_drift_names_where_to_look_scenario),
26206
+ ("a-reference-outlives-its-declaration", validate_reference_outlives_its_declaration_scenario),
26207
+ ("the-obligation-is-stated-early", validate_obligation_is_stated_early_scenario),
26208
+ ("an-explanation-is-printed-once", validate_explanation_is_printed_once_scenario),
26209
+ ("a-paused-change-is-not-the-next-action", validate_paused_change_is_not_the_next_action_scenario),
25754
26210
  (
25755
26211
  "authored-scenario-names-are-registered",
25756
26212
  validate_authored_scenario_names_scenario,
@@ -230,8 +230,21 @@ function resolveExplicit(repo, change, task) {
230
230
  if (task && !/^\d+(?:\.\d+)+$/.test(task)) {
231
231
  return blocked(`Invalid explicit task: ${task}`);
232
232
  }
233
+ // A pause is skipped by inference and never by explicit selection: the owner
234
+ // has said which change they mean. It is still reported, because a session
235
+ // resuming into a paused change should know that is what it is.
236
+ const declaration = pauseDeclaration(repo, change);
237
+ const paused = declaration && declaration.paused
238
+ ? [
239
+ `Explicitly selected change is paused: ${change} — ${declaration.reason}`
240
+ + (declaration.since ? ` (since ${declaration.since})` : "")
241
+ + ". Inference passes over it; you asked for it by name.",
242
+ ]
243
+ : [];
233
244
  if (!task) {
234
- return selectionForChange(repo, change, "explicit");
245
+ const context = selectionForChange(repo, change, "explicit");
246
+ context.warnings.push(...paused);
247
+ return context;
235
248
  }
236
249
 
237
250
  const tasksPath = path.join(repo, "openspec", "changes", change, "tasks.md");
@@ -259,6 +272,58 @@ function activeChanges(repo) {
259
272
  .sort();
260
273
  }
261
274
 
275
+ // A change its owner deliberately stopped. Declared where the change lives,
276
+ // under a `keel:` key of the OpenSpec change config — namespaced because that
277
+ // file is OpenSpec's, and a bare `status:` would be a claim on a key OpenSpec
278
+ // may define differently. Read by inference and by nothing else: pausing says
279
+ // what to recommend, never what is allowed, so every gate behaves identically
280
+ // on a paused change and explicit selection still reaches it.
281
+ //
282
+ // A declaration that cannot be read leaves the change available and is
283
+ // reported. The alternative failure — a change silently dropped from inference
284
+ // because its config had a typo — is this defect pointed the other way.
285
+ function pauseDeclaration(repo, change) {
286
+ const configPath = path.join(
287
+ repo, "openspec", "changes", change, ".openspec.yaml"
288
+ );
289
+ if (!fs.existsSync(configPath)) return null;
290
+ let content;
291
+ try {
292
+ content = fs.readFileSync(configPath, "utf8");
293
+ } catch {
294
+ return { unreadable: "the file could not be read" };
295
+ }
296
+ // The indented body of a top-level `keel:` key: every following line that
297
+ // starts with whitespace. Simpler and safer than a lookahead for the next
298
+ // top-level key, which has to spell "end of input" as well.
299
+ const block = content.match(/^keel:[ \t]*\r?\n((?:[ \t]+\S[^\n]*\r?\n?)*)/m);
300
+ if (!block) return null;
301
+ const entries = new Map();
302
+ for (const line of block[1].split(/\r?\n/)) {
303
+ const match = line.match(/^\s+([A-Za-z_][A-Za-z0-9_-]*)\s*:\s*(.*)$/);
304
+ if (match) entries.set(match[1].toLowerCase(), parseScalar(match[2]));
305
+ }
306
+ if (!entries.has("status")) return null;
307
+ const status = String(entries.get("status") || "").toLowerCase();
308
+ if (status !== "paused") {
309
+ return { unreadable: `status: ${entries.get("status")}` };
310
+ }
311
+ return {
312
+ paused: true,
313
+ reason: entries.get("reason") || "no reason recorded",
314
+ since: entries.get("since") || null,
315
+ };
316
+ }
317
+
318
+ function pauseNote(change, declaration) {
319
+ return (
320
+ `Paused change not inferred: ${change} — ${declaration.reason}`
321
+ + (declaration.since ? ` (since ${declaration.since})` : "")
322
+ + ". Select it explicitly with `keel context --change "
323
+ + `${change}\` if it is what you mean.`
324
+ );
325
+ }
326
+
262
327
  function inferContext(repo) {
263
328
  const changes = activeChanges(repo);
264
329
  if (changes.length === 0) {
@@ -270,20 +335,57 @@ function inferContext(repo) {
270
335
  ["No active OpenSpec change was found."]
271
336
  );
272
337
  }
273
- const contexts = changes.map((change) => selectionForChange(repo, change, "inferred"));
338
+ const declarations = new Map(
339
+ changes.map((change) => [change, pauseDeclaration(repo, change)])
340
+ );
341
+ const unreadable = [...declarations.entries()]
342
+ .filter(([, declaration]) => declaration && declaration.unreadable)
343
+ .map(([change, declaration]) =>
344
+ `Keel configuration for ${change} is not a pause declaration `
345
+ + `(${declaration.unreadable}); the change stays available to `
346
+ + "inference. A pause is `keel:` with `status: paused` and a `reason:`."
347
+ );
348
+ const paused = changes.filter(
349
+ (change) => declarations.get(change) && declarations.get(change).paused
350
+ );
351
+ const pauseNotes = paused.map(
352
+ (change) => pauseNote(change, declarations.get(change))
353
+ );
354
+ const active = changes.filter((change) => !paused.includes(change));
355
+ if (active.length === 0) {
356
+ // "Nothing to do" and "everything here is deliberately on hold" are
357
+ // different states, and the second is the one that tells a returning
358
+ // session whether to un-pause something or start something new.
359
+ return result(
360
+ "idle",
361
+ null,
362
+ "none",
363
+ [],
364
+ [
365
+ "Every active OpenSpec change is paused.",
366
+ ...pauseNotes,
367
+ ...unreadable,
368
+ ]
369
+ );
370
+ }
371
+ const contexts = active.map((change) => selectionForChange(repo, change, "inferred"));
274
372
  const storage = contexts.filter((context) => context.storageOnly);
275
373
  const candidates = contexts.filter((context) => !context.storageOnly);
276
- const warnings = storage.map(
277
- (context) =>
278
- `Storage-only backlog ignored during inference: ${context.selection.change}.`
279
- );
374
+ const warnings = [
375
+ ...storage.map(
376
+ (context) =>
377
+ `Storage-only backlog ignored during inference: ${context.selection.change}.`
378
+ ),
379
+ ...pauseNotes,
380
+ ...unreadable,
381
+ ];
280
382
  if (candidates.length === 0) {
281
383
  return result(
282
384
  "idle",
283
385
  null,
284
386
  "none",
285
387
  storage.flatMap((context) => context.read),
286
- ["No actionable OpenSpec change was found."]
388
+ ["No actionable OpenSpec change was found.", ...pauseNotes, ...unreadable]
287
389
  );
288
390
  }
289
391
  if (candidates.length > 1) {
@@ -297,6 +399,8 @@ function inferContext(repo) {
297
399
  + candidates
298
400
  .map((context) => context.selection?.change || context.read[0])
299
401
  .join(", "),
402
+ ...pauseNotes,
403
+ ...unreadable,
300
404
  ]
301
405
  );
302
406
  }
package/src/core/gates.js CHANGED
@@ -9,6 +9,7 @@ const {
9
9
  ACCEPTED_REVIEW_STATUSES,
10
10
  RED_GREEN_VERIFICATION_STRATEGIES,
11
11
  compileTaskContract,
12
+ declaredCommandLabels,
12
13
  field,
13
14
  isConcrete,
14
15
  isPassingReviewStatus,
@@ -21,8 +22,15 @@ const GATE_STAGES = new Set(["task-start", "task-complete", "change-close"]);
21
22
 
22
23
  class GateInputError extends Error {}
23
24
 
24
- function problem(code, message) {
25
- return { code, message };
25
+ // `note` is the rule behind the problem rather than the problem itself. Every
26
+ // problem carries its own copy in the JSON result, because a consumer reading
27
+ // problems one at a time must not get a payload whose content depends on
28
+ // position; the text renderer prints each distinct note once, because there
29
+ // the repetition is what crowds out the specific lines. Measured on a
30
+ // two-check task: the same 84-character sentence four times in 827 characters,
31
+ // and it scales with the number of checks.
32
+ function problem(code, message, note = null) {
33
+ return note ? { code, message, note } : { code, message };
26
34
  }
27
35
 
28
36
  function gateResult(
@@ -212,6 +220,34 @@ function anchoredFingerprint(previous) {
212
220
  // way to acknowledge a `needs-review`, so making it one would leave a
213
221
  // legitimate split unstartable. The reader is given the other task's id and
214
222
  // compares two things, rather than being told something is wrong.
223
+ // What completion will require, said while the Evidence is still all pending.
224
+ // The `Verify` block is complete at task-start and the obligation is a static
225
+ // function of the strategy and the tags, so the only reason it was first heard
226
+ // at task-complete is that nobody said it earlier — after the capsule was
227
+ // written, the task implemented, the checks run, and the Evidence recorded.
228
+ // A warning, never a refusal: a task whose author has not yet decided which
229
+ // check is a regression is not malformed, it is unfinished.
230
+ function redGreenObligation(compiled) {
231
+ if (!compiled || compiled.diagnostics.length > 0) return [];
232
+ const strategy = compiled.capsule.verification.strategy.toLowerCase();
233
+ if (!RED_GREEN_VERIFICATION_STRATEGIES.has(strategy)) return [];
234
+ const commands = compiled.capsule.verification.commands;
235
+ const owing = commands.filter((item) => !item.regression).map((item) => item.label);
236
+ const exempt = commands.filter((item) => item.regression).map((item) => item.label);
237
+ if (owing.length === 0) return [];
238
+ return [
239
+ `${strategy} will require concrete .red and .green Evidence at completion `
240
+ + `for ${owing.join(", ")}`
241
+ + (exempt.length > 0
242
+ ? `; ${exempt.join(", ")} ${exempt.length > 1 ? "are" : "is"} exempt as `
243
+ + "(regression)"
244
+ : "")
245
+ + ". Tag a check `(regression)` now if it asserts that something already "
246
+ + "green stays green, rather than discovering the obligation once the "
247
+ + "checks have been run.",
248
+ ];
249
+ }
250
+
215
251
  function taskShapeWarnings(repo, selection, task, compiled) {
216
252
  if (!compiled || compiled.diagnostics.length > 0) return [];
217
253
  const strategy = compiled.capsule.verification.strategy.toLowerCase();
@@ -272,7 +308,10 @@ function taskStart(repo, options) {
272
308
  selection.change,
273
309
  [task.id],
274
310
  problems,
275
- taskShapeWarnings(repo, selection, task, compiled),
311
+ [
312
+ ...redGreenObligation(compiled),
313
+ ...taskShapeWarnings(repo, selection, task, compiled),
314
+ ],
276
315
  problems.length === 0
277
316
  ? compiled
278
317
  : null
@@ -887,9 +926,13 @@ function attributeChanged(repo, task, changedList, contract, change, tasks) {
887
926
 
888
927
  function completionChecks(repo, task, contract = null, changeVerify = null, change = null) {
889
928
  const problems = [];
929
+ // Not the compiled capsule alone: when a check's declaration does not
930
+ // compile there is no capsule, and reconstructing the set from the expanded
931
+ // v3 `Commands` field a compact task never declares makes every reference
932
+ // look undeclared. The task's own declarations answer in both cases.
890
933
  const commands = contract
891
934
  ? contract.capsule.verification.commands.map((item) => item.label)
892
- : commandLabels(task);
935
+ : declaredCommandLabels(task);
893
936
  // With no contract, the labels came from the expanded v3 `Commands` field,
894
937
  // which a compact task never declares — so their absence is a fact about the
895
938
  // fallback, not about the task. The compiler's own diagnostics are already in
@@ -926,8 +969,9 @@ function completionChecks(repo, task, contract = null, changeVerify = null, chan
926
969
  problem(
927
970
  "missing-strategy-evidence",
928
971
  `${strategy} requires concrete ${label}.${phase} Evidence for `
929
- + "the same behavior check. Tag the check `(regression)` if it "
930
- + "asserts that something already green stays green."
972
+ + "the same behavior check.",
973
+ "Tag the check `(regression)` if it asserts that something "
974
+ + "already green stays green."
931
975
  )
932
976
  );
933
977
  }
@@ -1626,6 +1670,11 @@ function renderGate(result) {
1626
1670
  : ""),
1627
1671
  ];
1628
1672
  for (const item of result.problems) lines.push(`Problem: ${item.message}`);
1673
+ for (const note of [
1674
+ ...new Set(result.problems.map((item) => item.note).filter(Boolean)),
1675
+ ]) {
1676
+ lines.push(`Note: ${note}`);
1677
+ }
1629
1678
  for (const warning of result.warnings) lines.push(`Warning: ${warning}`);
1630
1679
  if (result.contract) {
1631
1680
  lines.push(
@@ -229,6 +229,21 @@ function verification(task) {
229
229
  };
230
230
  }
231
231
 
232
+ // The labels a task declares, read from the form the task itself uses. Label
233
+ // parsing does not depend on a check being concrete, so this answers exactly
234
+ // when the compiler cannot: a task whose `M2` declaration carries an unfilled
235
+ // slot still declares `M2`, and a reference to it is not a reference to
236
+ // something that does not exist.
237
+ function declaredCommandLabels(task) {
238
+ const source = fieldValues(task, "Verify").length > 0
239
+ ? fieldValues(task, "Verify")
240
+ : fieldValues(task, "Commands");
241
+ return source
242
+ .map((entry) => entry.match(/^(M[1-9]\d*)(?:\s*\([^)\n]*\))?\s*:/))
243
+ .filter(Boolean)
244
+ .map((match) => match[1]);
245
+ }
246
+
232
247
  function commandLabelProblems(task) {
233
248
  // A task that declared no verification form at all is reported once, by
234
249
  // requiredFieldProblems, as the one field it is missing. Its orphan Evidence
@@ -1195,6 +1210,7 @@ function loadTaskContract(repo, change, taskId) {
1195
1210
 
1196
1211
  module.exports = {
1197
1212
  ACCEPTED_REVIEW_STATUSES,
1213
+ declaredCommandLabels,
1198
1214
  RED_GREEN_VERIFICATION_STRATEGIES,
1199
1215
  SUPPORTED_VERIFICATION_STRATEGIES,
1200
1216
  compileTaskContract,