@christang/keel 5.66.0 → 5.68.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -38,8 +38,8 @@ REQUIRED_SCRIPTS = [
38
38
  "scripts/validate_plugin.py",
39
39
  ]
40
40
 
41
- PACKAGE_VERSION = "5.66.0"
42
- PROTOCOL_VERSION = "5.66.0"
41
+ PACKAGE_VERSION = "5.68.0"
42
+ PROTOCOL_VERSION = "5.68.0"
43
43
  LEGACY_MANAGED_START = "<!-- keel:start version=2.1 -->"
44
44
  OPENSPEC_SCHEMA_NAME = "keel-spec-driven"
45
45
  # Mirrors KEEL_PACKAGE_NAME in scripts/install_to_repo.py, one of the two
@@ -14396,6 +14396,27 @@ def validate_native_plugin_manifests_scenario() -> int:
14396
14396
  f"source: {skill_name}"
14397
14397
  )
14398
14398
  return 1
14399
+ # A referenced `guidance.md` travels with the body that names it. The
14400
+ # host reads the plugin copy directly, so a guidance file the parity
14401
+ # check ignored could drift from the criteria it was split out of, and
14402
+ # the drift would be invisible to everything except a reader.
14403
+ canonical_guidance = canonical_skill.parent / "guidance.md"
14404
+ plugin_guidance = plugin_skill.parent / "guidance.md"
14405
+ if canonical_guidance.is_file() and not plugin_guidance.is_file():
14406
+ report(
14407
+ "native-plugin-manifests guidance file diverges from canonical "
14408
+ f"source: {skill_name} has guidance.md that the plugin does not "
14409
+ "ship, so the body's reference resolves to nothing"
14410
+ )
14411
+ return 1
14412
+ if canonical_guidance.is_file() and plugin_guidance.read_text(
14413
+ encoding="utf-8"
14414
+ ) != canonical_guidance.read_text(encoding="utf-8"):
14415
+ report(
14416
+ "native-plugin-manifests guidance file diverges from canonical "
14417
+ f"source: {skill_name}"
14418
+ )
14419
+ return 1
14399
14420
  lenses_root = ROOT / "assets/lenses"
14400
14421
  for template in ("web.md", "hardware.md", "hardware-dsl.md"):
14401
14422
  if not (lenses_root / template).is_file():
@@ -20692,6 +20713,15 @@ def validate_native_goal_capabilities_scenario() -> int:
20692
20713
  return 0
20693
20714
 
20694
20715
 
20716
+ # The referenced half of a split skill. Returns "" for a skill with one body, so
20717
+ # every assertion about a skill's content reads the same shape whether or not it
20718
+ # was split — a split must not be able to drop a required statement, and a
20719
+ # scenario must not have to know which skills were split to stay correct.
20720
+ def skill_guidance_text(skill_md: Path) -> str:
20721
+ guidance = skill_md.parent / "guidance.md"
20722
+ return guidance.read_text(encoding="utf-8") if guidance.is_file() else ""
20723
+
20724
+
20695
20725
  SINGLE_TASK_GOAL_SKILL = "keel-run-single-task-goal"
20696
20726
  OFFICIAL_GOAL_SOURCES = (
20697
20727
  "https://learn.chatgpt.com/use-cases/follow-goals",
@@ -20711,7 +20741,13 @@ def validate_single_task_goal_skill_scenario() -> int:
20711
20741
  if canonical_bytes != projection.read_bytes():
20712
20742
  report("single-task-goal-skill projection is not byte-equal to the canonical source.")
20713
20743
  return 1
20714
- text = canonical_bytes.decode("utf-8")
20744
+ # A skill split into a body and a referenced `guidance.md` carries its
20745
+ # content across both files. What these assertions require is that the skill
20746
+ # states the thing, not that SKILL.md does, so the split must not be able to
20747
+ # drop a required statement by moving it — and must not be able to keep one
20748
+ # only in a file the plugin does not ship, which `native-plugin-manifests`
20749
+ # checks byte-for-byte alongside the body.
20750
+ text = canonical_bytes.decode("utf-8") + skill_guidance_text(canonical)
20715
20751
 
20716
20752
  # Authoritative official sources are linked, and provenance/license is recorded.
20717
20753
  for source in OFFICIAL_GOAL_SOURCES:
@@ -20916,8 +20952,9 @@ def _goal_target_surface(target: str) -> int:
20916
20952
  return 1
20917
20953
 
20918
20954
  # The skill carries the target-specific fallback guidance.
20919
- skill_text = (ROOT / "plugins/keel/skills" / SINGLE_TASK_GOAL_SKILL / "SKILL.md").read_text(
20920
- encoding="utf-8"
20955
+ skill_body = ROOT / "plugins/keel/skills" / SINGLE_TASK_GOAL_SKILL / "SKILL.md"
20956
+ skill_text = skill_body.read_text(encoding="utf-8") + skill_guidance_text(
20957
+ skill_body
20921
20958
  )
20922
20959
  if target == "claude" and "disabled hooks" not in skill_text.lower():
20923
20960
  report("native-goal-claude skill lacks the disabled-hooks fallback.")
@@ -28721,10 +28758,17 @@ def validate_a_count_is_derived_from_what_it_counts_scenario() -> int:
28721
28758
 
28722
28759
  # A header naming every declaration while miscounting them in prose passes:
28723
28760
  # membership is the checkable property, and a count is a lossy restatement.
28724
- miscounted = flat_header.replace(
28725
- "Six independent declarations", "Five independent declarations"
28761
+ # The numeral is located by shape, not by its current value. Pinning the
28762
+ # word here would reintroduce the literal this scenario exists to remove,
28763
+ # one level down: adding `executor_tier` moved the header to "Seven" and the
28764
+ # mutation stopped finding anything to vary.
28765
+ miscounted, varied = re.subn(
28766
+ r"\b\w+ independent declarations\b",
28767
+ "Zero independent declarations",
28768
+ flat_header,
28769
+ count=1,
28726
28770
  )
28727
- if miscounted == flat_header:
28771
+ if not varied:
28728
28772
  report(f"{label}: the header's prose count could not be located to vary it.")
28729
28773
  return 1
28730
28774
  if config_header_problem(names, miscounted):
@@ -28741,6 +28785,766 @@ def validate_a_count_is_derived_from_what_it_counts_scenario() -> int:
28741
28785
  return 0
28742
28786
 
28743
28787
 
28788
+ def validate_a_tier_declares_what_is_skipped_scenario() -> int:
28789
+ """Issue #135: which guidance an executor skips is a declaration, not a judgement.
28790
+
28791
+ A strong executor gets no value from stepwise how-to prose and pays for it on
28792
+ every activation. The report's own argument against letting the executor
28793
+ decide is that "do I need this?" is the judgement it is worst at, so the skip
28794
+ is written down by the repository. The tier reaches guidance and nothing else:
28795
+ a reader who takes it for a relaxation of the gates is worse off than one who
28796
+ never saw it, which is why every surface that reports it says so.
28797
+ """
28798
+ label = "a-tier-declares-what-is-skipped"
28799
+
28800
+ with tempfile.TemporaryDirectory(prefix="keel-tier-") as raw:
28801
+ root = Path(raw)
28802
+
28803
+ def fixture(name: str, body: str) -> Path:
28804
+ repo = root / name
28805
+ repo.mkdir()
28806
+ (repo / "keel").mkdir()
28807
+ (repo / "keel" / "config.yaml").write_text(body, encoding="utf-8")
28808
+ return repo
28809
+
28810
+ def context(repo: Path) -> str:
28811
+ return run_keel(repo, "context").stdout
28812
+
28813
+ # M1 — the declared tier is reported, an absent one reports the default,
28814
+ # and both say guidance is all the tier reaches.
28815
+ for name, body, expected in (
28816
+ ("high", "fast_check: echo high\nexecutor_tier: high\n", "high"),
28817
+ ("absent", "fast_check: echo absent\n", "standard"),
28818
+ ):
28819
+ repo = fixture(name, body)
28820
+ out = context(repo)
28821
+ line = next(
28822
+ (l for l in out.splitlines() if l.startswith("Executor tier:")), None
28823
+ )
28824
+ if line is None:
28825
+ report(
28826
+ f"{label}: no executor tier reported — the {name} repository "
28827
+ "is told nothing about which guidance its skills load."
28828
+ )
28829
+ report(out)
28830
+ return 1
28831
+ if expected not in line:
28832
+ report(
28833
+ f"{label}: the {name} repository reports {line!r} rather than "
28834
+ f"the {expected!r} tier."
28835
+ )
28836
+ return 1
28837
+ if "guidance" not in line:
28838
+ report(
28839
+ f"{label}: no executor tier reported as affecting guidance "
28840
+ f"only; {line!r} leaves a reader free to take it for a "
28841
+ "relaxation of the gates."
28842
+ )
28843
+ return 1
28844
+
28845
+ # M2 — a value Keel cannot read fails closed to reading the guidance.
28846
+ # The author of a typo believes they declared what they typed, so a
28847
+ # misspelling must not silently buy the skip it asked for, and the
28848
+ # refusal names the value and the alternatives rather than the key.
28849
+ typo = fixture("typo", "fast_check: echo typo\nexecutor_tier: aggressive\n")
28850
+ out = context(typo)
28851
+ line = next(
28852
+ (l for l in out.splitlines() if l.startswith("Executor tier:")), None
28853
+ )
28854
+ if line is None:
28855
+ report(
28856
+ f"{label}: accepted a tier outside the set — the projection "
28857
+ "reported no tier at all for an unreadable value, so the "
28858
+ "fallback is invisible."
28859
+ )
28860
+ report(out)
28861
+ return 1
28862
+ if "standard" not in line:
28863
+ report(
28864
+ f"{label}: accepted a tier outside the set — an unreadable value "
28865
+ f"did not fall back to reading the guidance; got {line!r}."
28866
+ )
28867
+ report(out)
28868
+ return 1
28869
+ if "aggressive" not in out:
28870
+ report(
28871
+ f"{label}: the rejected value is not named, so the author is "
28872
+ "left believing they declared what they typed."
28873
+ )
28874
+ report(out)
28875
+ return 1
28876
+ if "high" not in out:
28877
+ report(
28878
+ f"{label}: the refusal does not name the accepted tiers, so the "
28879
+ "reader learns their value is wrong and not what is right."
28880
+ )
28881
+ report(out)
28882
+ return 1
28883
+ doctor = run_keel(typo, "--doctor").stdout
28884
+ if "executor_tier" not in doctor or "aggressive" not in doctor:
28885
+ report(
28886
+ f"{label}: the doctor does not report the declaration as failed "
28887
+ "and name the entry, so it reports health the projection "
28888
+ "contradicts."
28889
+ )
28890
+ report(doctor)
28891
+ return 1
28892
+ healthy = run_keel(fixture("ok", "executor_tier: high\n"), "--doctor").stdout
28893
+ if "executor_tier: high" not in healthy:
28894
+ report(
28895
+ f"{label}: the doctor does not report a readable declaration, so "
28896
+ "the only tier it ever mentions is a broken one."
28897
+ )
28898
+ report(healthy)
28899
+ return 1
28900
+
28901
+ if label not in {name for name, _ in SCENARIOS}:
28902
+ report(f"{label}: the scenario registry does not include it.")
28903
+ return 1
28904
+ report(f"{label} scenario passed.")
28905
+ return 0
28906
+
28907
+
28908
+ # The words Keel states a criterion in. A guidance file is asserted to contain
28909
+ # none of them, which is what makes "the executor tier removes no criterion" a
28910
+ # property of the repository rather than a promise in a design document: a tier
28911
+ # can only skip a file that decides nothing.
28912
+ CRITERION_VOCABULARY = ("MUST", "SHOULD", "refuses", "rejects", "hard-stops")
28913
+
28914
+
28915
+ def guidance_criterion_problem(name: str, text: str) -> str | None:
28916
+ """Return the problem a guidance file's text has, or None.
28917
+
28918
+ Takes the text rather than reading the path, so the rule can be exercised on
28919
+ a planted copy. A rule that could only ever see files already known to be
28920
+ clean would pass forever without anyone learning whether it fires.
28921
+ """
28922
+ for word in CRITERION_VOCABULARY:
28923
+ if word in text:
28924
+ return (
28925
+ f"{name}'s guidance.md states a criterion — it contains "
28926
+ f"{word!r}, and a criterion in the file the executor tier skips "
28927
+ "would let `executor_tier: high` relax a rule rather than skip "
28928
+ "an explanation. Move the sentence back into SKILL.md."
28929
+ )
28930
+ return None
28931
+
28932
+
28933
+ def validate_guidance_is_referenced_and_carries_no_criterion_scenario() -> int:
28934
+ """Issue #135: stepwise guidance is referenced, and it decides nothing.
28935
+
28936
+ Keel is not on the delivery path for a skill body — the host reads
28937
+ `plugins/keel/skills/*/SKILL.md` directly — so the only way the resident cost
28938
+ falls is for the body to be smaller. What leaves it is how-to prose; every
28939
+ criterion stays, and that is checked here by planting one rather than
28940
+ promised, because a tier that could remove a criterion would be a relaxation
28941
+ of the gates wearing a capability label.
28942
+ """
28943
+ label = "guidance-is-referenced-and-carries-no-criterion"
28944
+ skills_root = ROOT / "src/skills"
28945
+ guidance_files = sorted(skills_root.glob("*/guidance.md"))
28946
+
28947
+ # D4 splits exactly one skill today. An empty glob would satisfy every loop
28948
+ # below without reading anything, so the count is asserted first.
28949
+ if not guidance_files:
28950
+ report(
28951
+ f"{label}: no skill has a guidance.md, so every assertion below is "
28952
+ "vacuous — the split this scenario checks did not happen."
28953
+ )
28954
+ return 1
28955
+
28956
+ for guidance in guidance_files:
28957
+ skill = guidance.parent / "SKILL.md"
28958
+ body = skill.read_text(encoding="utf-8")
28959
+ text = guidance.read_text(encoding="utf-8")
28960
+ if "guidance.md" not in body:
28961
+ report(
28962
+ f"{label}: guidance file is not referenced — "
28963
+ f"{guidance.parent.name}'s SKILL.md never names guidance.md, so "
28964
+ "the prose that left the body is unreachable from it."
28965
+ )
28966
+ return 1
28967
+ if "executor_tier" not in body:
28968
+ report(
28969
+ f"{label}: {guidance.parent.name}'s SKILL.md names guidance.md "
28970
+ "without the condition under which it is read, which leaves the "
28971
+ "skip to the executor's own judgement — the judgement #135 says "
28972
+ "it gets most wrong."
28973
+ )
28974
+ return 1
28975
+ if "high" not in body:
28976
+ report(
28977
+ f"{label}: {guidance.parent.name}'s SKILL.md does not name the "
28978
+ "tier that skips the read, so the condition it states cannot be "
28979
+ "evaluated."
28980
+ )
28981
+ return 1
28982
+ # "Measurably smaller than before the split" is checkable only as a
28983
+ # property of what is on disk now: the guidance carries real content,
28984
+ # and that content is no longer duplicated in the body it left.
28985
+ if len(text) < 500:
28986
+ report(
28987
+ f"{label}: {guidance.parent.name}'s guidance.md holds "
28988
+ f"{len(text)} bytes, too few for the split to have moved "
28989
+ "anything; a pointer to an almost-empty file costs a read and "
28990
+ "saves nothing."
28991
+ )
28992
+ return 1
28993
+ criterion = guidance_criterion_problem(guidance.parent.name, text)
28994
+ if criterion:
28995
+ report(f"{label}: {criterion}")
28996
+ return 1
28997
+ for heading in [
28998
+ line for line in text.splitlines() if line.startswith("## ")
28999
+ ]:
29000
+ if heading in body:
29001
+ report(
29002
+ f"{label}: {guidance.parent.name}'s body still carries "
29003
+ f"{heading!r}, so the section was copied rather than moved "
29004
+ "and the resident cost did not fall."
29005
+ )
29006
+ return 1
29007
+
29008
+ # M2 — the property is held by the suite, not promised by the design. A
29009
+ # guidance file that stated a criterion would let the tier skip one, which
29010
+ # would make `executor_tier: high` a relaxation of the gates wearing a
29011
+ # capability label. Planting one is the only way to know the rule fires.
29012
+ sample = guidance_files[0]
29013
+ planted = (
29014
+ sample.read_text(encoding="utf-8")
29015
+ + "\nAn executor MUST record the fingerprint before implementing.\n"
29016
+ )
29017
+ problem = guidance_criterion_problem(sample.parent.name, planted)
29018
+ if not problem:
29019
+ report(
29020
+ f"{label}: states a criterion — a guidance file carrying `MUST` was "
29021
+ "accepted, so nothing stops a criterion from moving into the file "
29022
+ "the tier skips."
29023
+ )
29024
+ return 1
29025
+ if sample.parent.name not in problem:
29026
+ report(
29027
+ f"{label}: the refusal does not name the file that states the "
29028
+ f"criterion; got {problem!r}."
29029
+ )
29030
+ return 1
29031
+ if "MUST" not in problem:
29032
+ report(
29033
+ f"{label}: the refusal does not name the word it objected to, so the "
29034
+ f"author has to guess which sentence to move; got {problem!r}."
29035
+ )
29036
+ return 1
29037
+
29038
+ if label not in {name for name, _ in SCENARIOS}:
29039
+ report(f"{label}: the scenario registry does not include it.")
29040
+ return 1
29041
+ report(f"{label} scenario passed.")
29042
+ return 0
29043
+
29044
+
29045
+ def equivalence_task(
29046
+ *,
29047
+ base: str | None = "HEAD~1",
29048
+ fields: str | None = "wns, tns, cell_count",
29049
+ covers: tuple[str, ...] = ("E1: the measured result does not move",),
29050
+ label: str = "1.1",
29051
+ title: str = "Move the attribute without moving the numbers",
29052
+ ) -> str:
29053
+ """One `equivalence` task, with either declaration omittable."""
29054
+ lines = [
29055
+ f"- [ ] {label} {title}",
29056
+ " - Covers:",
29057
+ ]
29058
+ lines.extend(f" - {entry}" for entry in covers)
29059
+ lines.extend(
29060
+ [
29061
+ " - Read:",
29062
+ " - README.md",
29063
+ " - Touch:",
29064
+ " - src/example.js",
29065
+ " - Verify:",
29066
+ " - Strategy: equivalence",
29067
+ ]
29068
+ )
29069
+ if base is not None:
29070
+ lines.append(f" - Base: {base}")
29071
+ if fields is not None:
29072
+ lines.append(f" - Fields: {fields}")
29073
+ lines.extend(
29074
+ [
29075
+ " - M1: node compare.js --base --head reports every field equal",
29076
+ " - Autonomy boundary:",
29077
+ " - Default: hard-stop",
29078
+ " - Pre-authorized fallback: none",
29079
+ " - Stop Rules:",
29080
+ " - Stop on any field that differs.",
29081
+ " - Evidence:",
29082
+ " - Contract: pending",
29083
+ " - M1: pending",
29084
+ " - Review:",
29085
+ " - Status: pass",
29086
+ " - Acceptance check: every declared field agreed.",
29087
+ " - Scope check: writes stayed inside Touch.",
29088
+ " - Findings: none",
29089
+ " - Blocker: none",
29090
+ ]
29091
+ )
29092
+ return "\n".join(lines) + "\n"
29093
+
29094
+
29095
+ def validate_an_equivalence_claim_names_its_base_scenario() -> int:
29096
+ """Issue #142: the correct evidence for a refactor is zero difference.
29097
+
29098
+ Red-green has no shape for it. The reporting repository re-recorded one
29099
+ task's contract twice to get past the shape — not because a criterion was
29100
+ wrong, but because the criterion had nowhere to live. An A/B against a base
29101
+ is *stronger* than red-green: it catches the change that also, incidentally,
29102
+ moved a result. What it needs is a place to say what it compares against and
29103
+ on which fields, plus a refusal for every way that shape can be complete and
29104
+ still compare nothing.
29105
+ """
29106
+ label = "an-equivalence-claim-names-its-base"
29107
+
29108
+ def git(repo: Path, *args: str) -> subprocess.CompletedProcess[str]:
29109
+ return subprocess.run(
29110
+ ["git", "-C", str(repo), *args], capture_output=True, text=True
29111
+ )
29112
+
29113
+ with tempfile.TemporaryDirectory(prefix="keel-equivalence-") as raw:
29114
+ root = Path(raw)
29115
+
29116
+ def fixture(name: str, task: str) -> Path:
29117
+ repo = (root / name).resolve()
29118
+ repo.mkdir()
29119
+ git(repo, "init", "-q")
29120
+ git(repo, "config", "user.email", "t@example.com")
29121
+ git(repo, "config", "user.name", "keel-test")
29122
+ write_gate_fixture(repo, tasks=task)
29123
+ write_text(repo / "src/example.js", "// product\n")
29124
+ git(repo, "add", "-A")
29125
+ git(repo, "-c", "commit.gpgsign=false", "commit", "-q", "-m", "first")
29126
+ write_text(repo / "src/example.js", "// product, moved\n")
29127
+ git(repo, "add", "-A")
29128
+ git(repo, "-c", "commit.gpgsign=false", "commit", "-q", "-m", "second")
29129
+ return repo
29130
+
29131
+ def start(name: str, **kwargs) -> dict:
29132
+ repo = fixture(name, equivalence_task(**kwargs))
29133
+ result = run_keel(
29134
+ repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
29135
+ "--no-guard", "--json",
29136
+ )
29137
+ try:
29138
+ return json.loads(result.stdout)
29139
+ except json.JSONDecodeError:
29140
+ return {
29141
+ "status": "unparsed",
29142
+ "problems": [{"message": result.stdout[:400]}],
29143
+ }
29144
+
29145
+ # M1 — the strategy is accepted, and it owes no red.
29146
+ payload = start("accepted")
29147
+ if payload.get("status") != "pass":
29148
+ report(
29149
+ f"{label}: unsupported verification strategy — a complete "
29150
+ "equivalence task was refused, so the one task class with the "
29151
+ "strongest criterion still has nowhere to declare it. "
29152
+ f"{problem_text(payload)}"
29153
+ )
29154
+ return 1
29155
+ capsule = (payload.get("contract") or {}).get("capsule") or {}
29156
+ verification = capsule.get("verification") or {}
29157
+ if verification.get("strategy", "").lower() != "equivalence":
29158
+ report(
29159
+ f"{label}: the compiled capsule does not carry the strategy; got "
29160
+ f"{verification.get('strategy')!r}."
29161
+ )
29162
+ return 1
29163
+ if not verification.get("base") or not verification.get("fields"):
29164
+ report(
29165
+ f"{label}: the capsule drops the declarations — base "
29166
+ f"{verification.get('base')!r}, fields "
29167
+ f"{verification.get('fields')!r}. A declaration absent from the "
29168
+ "capsule is absent from the fingerprint, so it could be edited "
29169
+ "after the run without moving the contract."
29170
+ )
29171
+ return 1
29172
+ warnings = " ".join(str(w) for w in (payload.get("warnings") or []))
29173
+ if ".red" in warnings or ".green" in warnings:
29174
+ report(
29175
+ f"{label}: equivalence was given a red-green obligation; the "
29176
+ f"criterion is agreement with a base, not a failing first run. "
29177
+ f"{warnings}"
29178
+ )
29179
+ return 1
29180
+
29181
+ # M2 — every way the shape can be complete and still compare nothing,
29182
+ # each named for the declaration it is about. A diagnostic that named
29183
+ # the strategy would send the author to the line that is correct.
29184
+ for name, kwargs, code, expected in (
29185
+ ("no-base", {"base": None}, "missing-equivalence-base", "Base:"),
29186
+ ("no-fields", {"fields": None}, "missing-equivalence-fields", "Fields:"),
29187
+ ("empty-fields", {"fields": " , "}, "missing-equivalence-fields", "Fields:"),
29188
+ (
29189
+ "bad-base",
29190
+ {"base": "no-such-ref"},
29191
+ "unresolvable-equivalence-base",
29192
+ "no-such-ref",
29193
+ ),
29194
+ ):
29195
+ payload = start(name, **kwargs)
29196
+ if payload.get("status") == "pass":
29197
+ report(
29198
+ f"{label}: accepted an equivalence task that compares "
29199
+ f"nothing — the {name} fixture passed, so the declaration "
29200
+ "is optional in practice."
29201
+ )
29202
+ return 1
29203
+ if code not in problem_codes(payload):
29204
+ report(
29205
+ f"{label}: accepted an equivalence task that compares "
29206
+ f"nothing — the {name} fixture was refused for another "
29207
+ f"reason; expected {code}, got {problem_codes(payload)!r}."
29208
+ )
29209
+ return 1
29210
+ message = problem_text(payload)
29211
+ if expected not in message:
29212
+ report(
29213
+ f"{label}: the {name} refusal does not name {expected!r}, so "
29214
+ f"the author is sent to find which line is wrong; got "
29215
+ f"{message!r}."
29216
+ )
29217
+ return 1
29218
+ # Absent and empty are the same state to the comparison and different
29219
+ # states to the author, so the two are asserted to read differently.
29220
+ absent = problem_text(start("no-fields-message", fields=None))
29221
+ empty = problem_text(start("empty-fields-message", fields=" , "))
29222
+ if absent == empty:
29223
+ report(
29224
+ f"{label}: a missing `Fields:` and one that resolves to an empty "
29225
+ "set produce the same sentence, so an author who wrote the line "
29226
+ "is told they did not."
29227
+ )
29228
+ return 1
29229
+
29230
+ # M1 of 1.2 — `equivalence` owes no red, which makes it the first thing
29231
+ # reached for by a task that should have one. A task covering a scenario
29232
+ # the change *adds* is claiming new behavior and unchanged behavior at
29233
+ # once, and one of the two claims has no proof anywhere.
29234
+ added_spec = (
29235
+ "## ADDED Requirements\n\n"
29236
+ "### Requirement: The moved attribute keeps its effect\n\n"
29237
+ "The attribute SHALL keep its effect after the move.\n\n"
29238
+ "#### Scenario: The effect survives the move\n\n"
29239
+ "- **WHEN** the attribute moves into the flow\n"
29240
+ "- **THEN** the effect is unchanged\n\n"
29241
+ # A second real scenario, so the mismatched-sibling control fails
29242
+ # because the guard refused it and not because its Covers entry
29243
+ # resolves to nothing. A first attempt pointed the sibling at an
29244
+ # invented scenario and passed for that unrelated reason.
29245
+ "#### Scenario: The flow reports the attribute\n\n"
29246
+ "- **WHEN** the flow runs\n"
29247
+ "- **THEN** it reports the attribute\n"
29248
+ )
29249
+ covered = (
29250
+ "demo / The moved attribute keeps its effect / The effect survives "
29251
+ "the move"
29252
+ )
29253
+
29254
+ def escape_fixture(name: str, *, sibling: str | None) -> dict:
29255
+ repo = (root / name).resolve()
29256
+ repo.mkdir()
29257
+ git(repo, "init", "-q")
29258
+ git(repo, "config", "user.email", "t@example.com")
29259
+ git(repo, "config", "user.name", "keel-test")
29260
+ tasks = equivalence_task(covers=(covered,))
29261
+ if sibling is not None:
29262
+ tasks += (
29263
+ "- [ ] 1.2 Add the behavior\n"
29264
+ " - Covers:\n"
29265
+ f" - {sibling}\n"
29266
+ " - Read:\n - README.md\n"
29267
+ " - Touch:\n - src/other.js\n"
29268
+ " - Verify:\n"
29269
+ " - Strategy: vertical-tdd\n"
29270
+ " - M1: node test.js asserts the new behavior\n"
29271
+ " - Autonomy boundary:\n"
29272
+ " - Default: hard-stop\n"
29273
+ " - Pre-authorized fallback: none\n"
29274
+ " - Stop Rules:\n - Stop on failure.\n"
29275
+ " - Evidence:\n - Contract: pending\n - M1: pending\n"
29276
+ " - Review:\n - Status: pass\n"
29277
+ " - Acceptance check: behavior asserted.\n"
29278
+ " - Scope check: inside Touch.\n"
29279
+ " - Findings: none\n"
29280
+ " - Blocker: none\n"
29281
+ )
29282
+ write_gate_fixture(repo, tasks=tasks)
29283
+ write_text(repo / "openspec/changes/demo/specs/demo/spec.md", added_spec)
29284
+ write_text(repo / "src/example.js", "// product\n")
29285
+ git(repo, "add", "-A")
29286
+ git(repo, "-c", "commit.gpgsign=false", "commit", "-q", "-m", "first")
29287
+ write_text(repo / "src/example.js", "// moved\n")
29288
+ git(repo, "add", "-A")
29289
+ git(repo, "-c", "commit.gpgsign=false", "commit", "-q", "-m", "second")
29290
+ result = run_keel(
29291
+ repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
29292
+ "--no-guard", "--json",
29293
+ )
29294
+ try:
29295
+ return json.loads(result.stdout)
29296
+ except json.JSONDecodeError:
29297
+ return {
29298
+ "status": "unparsed",
29299
+ "problems": [{"message": result.stdout[:400]}],
29300
+ }
29301
+
29302
+ payload = escape_fixture("escape-alone", sibling=None)
29303
+ if payload.get("status") == "pass":
29304
+ report(
29305
+ f"{label}: accepted new behavior with no red anywhere — an "
29306
+ "equivalence task covering a scenario the change adds passed, so "
29307
+ "the strategy is a way to author a feature with no red in the "
29308
+ "whole change."
29309
+ )
29310
+ return 1
29311
+ if "equivalence-covers-added-behavior" not in problem_codes(payload):
29312
+ report(
29313
+ f"{label}: accepted new behavior with no red anywhere — refused "
29314
+ f"for another reason; got {problem_codes(payload)!r}."
29315
+ )
29316
+ return 1
29317
+ message = problem_text(payload)
29318
+ if "The effect survives the move" not in message:
29319
+ report(
29320
+ f"{label}: the refusal does not name the covered scenario, so "
29321
+ f"the author cannot tell which Covers entry is the problem; got "
29322
+ f"{message!r}."
29323
+ )
29324
+ return 1
29325
+ if "unchanged" not in message:
29326
+ report(
29327
+ f"{label}: the refusal does not say why the two claims conflict; "
29328
+ f"got {message!r}."
29329
+ )
29330
+ return 1
29331
+
29332
+ # M2 of 1.2 — the guard is satisfied by coverage of the entry it
29333
+ # objected to, and not by a red-green task merely existing in the change.
29334
+ payload = escape_fixture("escape-sibling", sibling=covered)
29335
+ if payload.get("status") != "pass":
29336
+ report(
29337
+ f"{label}: a sibling task covering the same scenario under "
29338
+ f"vertical-tdd did not satisfy the guard. {problem_text(payload)}"
29339
+ )
29340
+ return 1
29341
+ payload = escape_fixture(
29342
+ "escape-other-sibling",
29343
+ sibling=(
29344
+ "demo / The moved attribute keeps its effect / The flow reports "
29345
+ "the attribute"
29346
+ ),
29347
+ )
29348
+ if payload.get("status") == "pass":
29349
+ report(
29350
+ f"{label}: any sibling satisfied the guard — a red-green task "
29351
+ "covering a different scenario was accepted as proof of this "
29352
+ "one, which makes the guard a check that a change contains at "
29353
+ "least one red-green task."
29354
+ )
29355
+ return 1
29356
+ if "equivalence-covers-added-behavior" not in problem_codes(payload):
29357
+ report(
29358
+ f"{label}: any sibling satisfied the guard — refused for another "
29359
+ f"reason; got {problem_codes(payload)!r}."
29360
+ )
29361
+ return 1
29362
+
29363
+ # M3 of 1.2 — the guard fires on new behavior, not on the strategy. A
29364
+ # task covering only identifiers, or a requirement the change does not
29365
+ # add, is an ordinary equivalence task.
29366
+ plain = start("escape-plain")
29367
+ if plain.get("status") != "pass":
29368
+ report(
29369
+ f"{label}: an equivalence task covering no added scenario was "
29370
+ f"refused. {problem_text(plain)}"
29371
+ )
29372
+ return 1
29373
+
29374
+ # M3 — the shape that is complete, resolvable, and still empty. This is
29375
+ # the refusal worth having: nothing about the task looks wrong, and the
29376
+ # check passes having compared a thing against itself.
29377
+ payload = start("base-is-head", base="HEAD")
29378
+ if payload.get("status") == "pass":
29379
+ report(
29380
+ f"{label}: accepted a base that is head — an A/B against itself "
29381
+ "always agrees, so the check proves nothing and looks complete "
29382
+ "doing it."
29383
+ )
29384
+ return 1
29385
+ if "equivalence-base-is-head" not in problem_codes(payload):
29386
+ report(
29387
+ f"{label}: accepted a base that is head — refused for another "
29388
+ f"reason; got {problem_codes(payload)!r}."
29389
+ )
29390
+ return 1
29391
+ message = problem_text(payload)
29392
+ head = subprocess.run(
29393
+ ["git", "-C", str((root / "base-is-head").resolve()), "rev-parse",
29394
+ "HEAD"],
29395
+ capture_output=True, text=True,
29396
+ ).stdout.strip()
29397
+ if head and head not in message:
29398
+ report(
29399
+ f"{label}: the refusal does not name the resolved commit, so the "
29400
+ "author cannot tell which ref collapsed onto HEAD; got "
29401
+ f"{message!r}."
29402
+ )
29403
+ return 1
29404
+ if "always agrees" not in message:
29405
+ report(
29406
+ f"{label}: the refusal does not say why a base that is HEAD is "
29407
+ f"empty rather than merely redundant; got {message!r}."
29408
+ )
29409
+ return 1
29410
+
29411
+ # 1.3 — Evidence may point at the machine output instead of retelling
29412
+ # it. A 244-line tasks.md that is mostly transcribed test output is a
29413
+ # transcription that can be wrong and that nobody can re-check.
29414
+ def artifact_fixture(
29415
+ name: str,
29416
+ *,
29417
+ artifact_path: str = "openspec/changes/demo/evidence/compare.json",
29418
+ body: str = '{"wns": 0.0, "tns": 0.0}\n',
29419
+ recorded: str | None = None,
29420
+ write_at: str | None = None,
29421
+ ) -> dict:
29422
+ repo = (root / name).resolve()
29423
+ repo.mkdir()
29424
+ git(repo, "init", "-q")
29425
+ git(repo, "config", "user.email", "t@example.com")
29426
+ git(repo, "config", "user.name", "keel-test")
29427
+ digest = hashlib.sha256(body.encode("utf-8")).hexdigest()
29428
+ evidence = recorded or f"artifact {artifact_path} sha256:{digest}"
29429
+ tasks = equivalence_task().replace(
29430
+ " - M1: pending", f" - M1: {evidence}"
29431
+ )
29432
+ write_gate_fixture(repo, tasks=tasks)
29433
+ if write_at is not None:
29434
+ write_text(repo / write_at, body)
29435
+ write_text(repo / "src/example.js", "// product\n")
29436
+ git(repo, "add", "-A")
29437
+ git(repo, "-c", "commit.gpgsign=false", "commit", "-q", "-m", "first")
29438
+ write_text(repo / "src/example.js", "// moved\n")
29439
+ git(repo, "add", "-A")
29440
+ git(repo, "-c", "commit.gpgsign=false", "commit", "-q", "-m", "second")
29441
+ run_keel(
29442
+ repo, "gate", "task-start", "--change", "demo", "--task", "1.1",
29443
+ "--record", "--no-guard",
29444
+ )
29445
+ result = run_keel(
29446
+ repo, "gate", "task-complete", "--change", "demo", "--task",
29447
+ "1.1", "--json",
29448
+ )
29449
+ try:
29450
+ return json.loads(result.stdout)
29451
+ except json.JSONDecodeError:
29452
+ return {
29453
+ "status": "unparsed",
29454
+ "problems": [{"message": result.stdout[:400]}],
29455
+ }
29456
+
29457
+ inside = "openspec/changes/demo/evidence/compare.json"
29458
+ payload = artifact_fixture("artifact-ok", write_at=inside)
29459
+ if payload.get("status") != "pass":
29460
+ report(
29461
+ f"{label}: an artifact reference was not verified — a reference "
29462
+ "to a file that is there, with a digest that matches, was "
29463
+ f"refused. {problem_text(payload)}"
29464
+ )
29465
+ return 1
29466
+ payload = artifact_fixture("artifact-absent", write_at=None)
29467
+ if payload.get("status") == "pass":
29468
+ report(
29469
+ f"{label}: an artifact reference was not verified — a reference "
29470
+ "to a file that does not exist was accepted, so the form is "
29471
+ "tolerated as prose rather than checked. Any sentence would "
29472
+ "have passed the same way."
29473
+ )
29474
+ return 1
29475
+ if "artifact-missing" not in problem_codes(payload):
29476
+ report(
29477
+ f"{label}: an artifact reference was not verified — refused for "
29478
+ f"another reason; got {problem_codes(payload)!r}."
29479
+ )
29480
+ return 1
29481
+
29482
+ # M2 of 1.3 — the digest is what makes the pointer worth more than a
29483
+ # path. A file that moved after the digest was recorded is the case a
29484
+ # bare path cannot see, and it is the common one: the command gets
29485
+ # re-run.
29486
+ stale = hashlib.sha256(b"different\n").hexdigest()
29487
+ payload = artifact_fixture(
29488
+ "artifact-stale",
29489
+ recorded=f"artifact {inside} sha256:{stale}",
29490
+ write_at=inside,
29491
+ )
29492
+ if payload.get("status") == "pass":
29493
+ report(
29494
+ f"{label}: accepted a stale digest — the artifact's content does "
29495
+ "not hash to the recorded digest and the reference was accepted, "
29496
+ "so Review reads whatever the file says now."
29497
+ )
29498
+ return 1
29499
+ if "artifact-digest-mismatch" not in problem_codes(payload):
29500
+ report(
29501
+ f"{label}: accepted a stale digest — refused for another reason; "
29502
+ f"got {problem_codes(payload)!r}."
29503
+ )
29504
+ return 1
29505
+ message = problem_text(payload)
29506
+ for expected in (inside, stale[:12]):
29507
+ if expected not in message:
29508
+ report(
29509
+ f"{label}: the refusal does not name {expected!r}, so a "
29510
+ "reader cannot tell a stale record from a wrong path; got "
29511
+ f"{message!r}."
29512
+ )
29513
+ return 1
29514
+
29515
+ # M3 of 1.3 — the inverse of the `Durable owner:` rule, for the opposite
29516
+ # reason: a follow-up pointer must outlive the change, and an evidence
29517
+ # artifact must travel with it. `openspec archive` moves the change
29518
+ # directory, so a path outside it is one the archive leaves behind.
29519
+ outside = "evidence/compare.json"
29520
+ payload = artifact_fixture("artifact-outside", artifact_path=outside, write_at=outside)
29521
+ if payload.get("status") == "pass":
29522
+ report(
29523
+ f"{label}: accepted a path archiving would leave behind — an "
29524
+ f"artifact at {outside!r} was accepted although the archive "
29525
+ "moves only the change directory."
29526
+ )
29527
+ return 1
29528
+ if "artifact-outside-change" not in problem_codes(payload):
29529
+ report(
29530
+ f"{label}: accepted a path archiving would leave behind — "
29531
+ f"refused for another reason; got {problem_codes(payload)!r}."
29532
+ )
29533
+ return 1
29534
+ if "archiv" not in problem_text(payload):
29535
+ report(
29536
+ f"{label}: the refusal does not say that archiving is what "
29537
+ f"breaks the pointer; got {problem_text(payload)!r}."
29538
+ )
29539
+ return 1
29540
+
29541
+ if label not in {name for name, _ in SCENARIOS}:
29542
+ report(f"{label}: the scenario registry does not include it.")
29543
+ return 1
29544
+ report(f"{label} scenario passed.")
29545
+ return 0
29546
+
29547
+
28744
29548
  SCENARIOS: tuple = (
28745
29549
  ("stateless-continuity", validate_stateless_continuity_scenario),
28746
29550
  ("core-gates", validate_core_gates_scenario),
@@ -29103,6 +29907,18 @@ SCENARIOS: tuple = (
29103
29907
  "a-negation-is-not-a-marker",
29104
29908
  validate_a_negation_is_not_a_marker_scenario,
29105
29909
  ),
29910
+ (
29911
+ "an-equivalence-claim-names-its-base",
29912
+ validate_an_equivalence_claim_names_its_base_scenario,
29913
+ ),
29914
+ (
29915
+ "guidance-is-referenced-and-carries-no-criterion",
29916
+ validate_guidance_is_referenced_and_carries_no_criterion_scenario,
29917
+ ),
29918
+ (
29919
+ "a-tier-declares-what-is-skipped",
29920
+ validate_a_tier_declares_what_is_skipped_scenario,
29921
+ ),
29106
29922
  (
29107
29923
  "a-count-is-derived-from-what-it-counts",
29108
29924
  validate_a_count_is_derived_from_what_it_counts_scenario,