@christang/keel 5.66.0 → 5.67.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -0
- package/assets/bootstrap/AGENTS.md +1 -1
- package/bin/keel.js +20 -0
- package/package.json +1 -1
- package/plugins/keel/.claude-plugin/plugin.json +1 -1
- package/plugins/keel/.codex-plugin/plugin.json +1 -1
- package/plugins/keel/skills/keel-run-single-task-goal/SKILL.md +6 -10
- package/plugins/keel/skills/keel-run-single-task-goal/guidance.md +30 -0
- package/scripts/validate_plugin.py +317 -8
- package/src/core/config.js +54 -0
- package/src/core/context.js +16 -0
package/README.md
CHANGED
|
@@ -295,6 +295,34 @@ decision either — routing decides whether a change exists, so there is nothing
|
|
|
295
295
|
to; what Keel does is make sure the rule and your exceptions are in front of the agent when it
|
|
296
296
|
decides.
|
|
297
297
|
|
|
298
|
+
### How much guidance the agent loads
|
|
299
|
+
|
|
300
|
+
Keel's skills carry two kinds of content, and their value moves in opposite directions. *How to do it*
|
|
301
|
+
— how to split a task, what order to run things in — matters less the stronger the executor is. *Make
|
|
302
|
+
yourself falsifiable* — red then green, the failure literal a check predicts, the fingerprint, whether
|
|
303
|
+
a `Durable owner:` reference actually exists — matters more, because a strong executor produces
|
|
304
|
+
confident work and those are the checks that can contradict it.
|
|
305
|
+
|
|
306
|
+
So the stepwise half of a skill lives in a `guidance.md` beside it, and a repository can say it does
|
|
307
|
+
not need that half:
|
|
308
|
+
|
|
309
|
+
```yaml
|
|
310
|
+
executor_tier: high
|
|
311
|
+
```
|
|
312
|
+
|
|
313
|
+
The default is `standard`, which reads the guidance; an absent or misspelled declaration reads it too,
|
|
314
|
+
so the worst an unconfigured repository does is pay for a read. `keel context` and `keel --doctor`
|
|
315
|
+
report the tier.
|
|
316
|
+
|
|
317
|
+
**The tier reaches guidance and nothing else** — no gate, criterion, evidence requirement, or Review
|
|
318
|
+
changes with it. That is not a promise in this README: a guidance file is checked to contain none of
|
|
319
|
+
the words Keel states criteria in, so a tier can only ever skip a file that decides nothing. Deciding
|
|
320
|
+
what to skip is a declaration rather than the agent's own call on purpose — "do I need this help?" is
|
|
321
|
+
the judgement a weak executor gets most wrong, and it would be answering it about itself.
|
|
322
|
+
|
|
323
|
+
One skill is split today, `keel-run-single-task-goal`, and its body is 11% smaller for it. The other
|
|
324
|
+
five are almost entirely criteria, so splitting them would move the half that has to stay.
|
|
325
|
+
|
|
298
326
|
## How the agent uses these
|
|
299
327
|
|
|
300
328
|
You rarely type the commands below. The point of Keel is that the discipline runs itself:
|
package/bin/keel.js
CHANGED
|
@@ -49,6 +49,7 @@ const {
|
|
|
49
49
|
readPrecedentStore,
|
|
50
50
|
readStandingAuthorization,
|
|
51
51
|
readFullModePaths,
|
|
52
|
+
readExecutorTier,
|
|
52
53
|
fullModePathsUnreadableMessage,
|
|
53
54
|
readTriagePolicy,
|
|
54
55
|
triageIssue,
|
|
@@ -1731,6 +1732,7 @@ function runDoctor(options) {
|
|
|
1731
1732
|
printPrecedentSurface(repo);
|
|
1732
1733
|
printTriageSurface(repo);
|
|
1733
1734
|
printRoutingSurface(repo);
|
|
1735
|
+
printExecutorTierSurface(repo);
|
|
1734
1736
|
printFastPrePushSurface(repo);
|
|
1735
1737
|
printSourceRepoCliResolution(repo);
|
|
1736
1738
|
|
|
@@ -1855,6 +1857,24 @@ function printRoutingSurface(repo) {
|
|
|
1855
1857
|
for (const entry of paths) printDoctorLine(entry.path, "Full", entry.reason);
|
|
1856
1858
|
}
|
|
1857
1859
|
|
|
1860
|
+
// Reported whether or not it is declared, because the default is the state a
|
|
1861
|
+
// reader most needs to see: a repository that declared nothing is loading every
|
|
1862
|
+
// skill's guidance and has no other surface that says so.
|
|
1863
|
+
function printExecutorTierSurface(repo) {
|
|
1864
|
+
process.stdout.write("\nExecutor tier:\n");
|
|
1865
|
+
const { declared, tier, unknown, message } = readExecutorTier(repo);
|
|
1866
|
+
if (unknown.length > 0) {
|
|
1867
|
+
printDoctorLine("executor_tier", "unreadable", message);
|
|
1868
|
+
return;
|
|
1869
|
+
}
|
|
1870
|
+
printDoctorLine(
|
|
1871
|
+
"executor_tier",
|
|
1872
|
+
tier,
|
|
1873
|
+
(declared ? "declared in keel/config.yaml" : "undeclared; the default")
|
|
1874
|
+
+ " - affects which skill guidance is read and nothing else"
|
|
1875
|
+
);
|
|
1876
|
+
}
|
|
1877
|
+
|
|
1858
1878
|
function printTriageSurface(repo) {
|
|
1859
1879
|
process.stdout.write("\nUnattended triage:\n");
|
|
1860
1880
|
const { labels, issues, unreadable } = readTriagePolicy(repo);
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "keel",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.67.0",
|
|
4
4
|
"description": "Keel OpenSpec execution discipline: stateless continuity, task capsules, deterministic gates, and expectation alignment for Codex and Claude Code.",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "TanglmChris",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "keel",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.67.0",
|
|
4
4
|
"description": "Keel OpenSpec execution discipline: stateless continuity, task capsules, deterministic gates, and expectation alignment for Codex and Claude Code.",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "TanglmChris",
|
|
@@ -13,16 +13,13 @@ metadata:
|
|
|
13
13
|
|
|
14
14
|
Activate a native goal or subagent runtime to execute exactly one authorized OpenSpec task end to end, while OpenSpec, Git, the task-capsule fingerprint, and deterministic Keel gates stay the only durable authority. The current agent remains the sole holder of write authority and owns Review, gate invocation, the task checkbox, and completion. Where delegation is declared, an authorized delegate may write inside the `Touch` boundary that authority already defined and acquires none of those decisions; the current agent re-runs each `M<n>` check itself before recording Evidence, because a delegate's reported result is a claim and the byte-identity check that validates a read-only helper cannot apply to a writer. A native evaluator declaring success never marks or reports the task complete.
|
|
15
15
|
|
|
16
|
-
##
|
|
16
|
+
## Guidance
|
|
17
17
|
|
|
18
|
-
|
|
18
|
+
Read `guidance.md` beside this file before proceeding, unless `keel/config.yaml` declares `executor_tier: high` — it holds the runtime references, the manual sequence, and the per-target detail. Every criterion is here, so an absent or unreadable declaration costs a read and nothing else.
|
|
19
19
|
|
|
20
|
-
|
|
21
|
-
- Codex subagents: https://developers.openai.com/codex/subagents
|
|
22
|
-
- Claude goal execution: https://code.claude.com/docs/en/goal
|
|
23
|
-
- Claude subagents: https://code.claude.com/docs/en/sub-agents
|
|
20
|
+
## License
|
|
24
21
|
|
|
25
|
-
License note:
|
|
22
|
+
License note: the Keel package license (UNLICENSED, all rights reserved by the author). Linking the official runtime docs relicenses nothing; their provenance is recorded beside the links and their content is never pasted into Keel artifacts.
|
|
26
23
|
|
|
27
24
|
## When to activate
|
|
28
25
|
|
|
@@ -60,9 +57,8 @@ Helpers are optional, read-only evidence producers and never a second writer. Co
|
|
|
60
57
|
|
|
61
58
|
## Manual fallback
|
|
62
59
|
|
|
63
|
-
When native activation is unavailable
|
|
60
|
+
When native activation is unavailable, do not fake activation: run the identical lifecycle by hand. The manual loop preserves the same single-task boundary and the same stop boundary; its steps are in `guidance.md`.
|
|
64
61
|
|
|
65
62
|
## Target activation
|
|
66
63
|
|
|
67
|
-
|
|
68
|
-
- Claude: activate one `/goal` whose condition stays within the 4,000-character budget; the evaluator is transcript-only, so surface command and gate evidence explicitly. If hooks are disabled, policy blocks activation, or trust is missing, report the manual fallback.
|
|
64
|
+
Claude and Codex only. Activate exactly one bounded goal for the selected task; a native evaluator declaring success never marks or reports the task complete. The per-target detail is in `guidance.md`.
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
# keel-run-single-task-goal — guidance
|
|
2
|
+
|
|
3
|
+
How to carry out the lifecycle. Every criterion is in `SKILL.md`; nothing here decides whether a task
|
|
4
|
+
may start, pass, or complete.
|
|
5
|
+
|
|
6
|
+
## Authoritative runtime references
|
|
7
|
+
|
|
8
|
+
Provenance: linked, not copied. Their text and trademarks belong to their owners, and Keel paraphrases
|
|
9
|
+
only the activation semantics it needs. License note: see `SKILL.md`.
|
|
10
|
+
|
|
11
|
+
- Codex goal-following: https://learn.chatgpt.com/use-cases/follow-goals
|
|
12
|
+
- Codex subagents: https://developers.openai.com/codex/subagents
|
|
13
|
+
- Claude goal execution: https://code.claude.com/docs/en/goal
|
|
14
|
+
- Claude subagents: https://code.claude.com/docs/en/sub-agents
|
|
15
|
+
|
|
16
|
+
## Running the lifecycle by hand
|
|
17
|
+
|
|
18
|
+
Native activation can be unavailable — no plugin, disabled hooks, managed policy, missing trust, or an
|
|
19
|
+
unsupported surface. Type the same numbered steps `SKILL.md` lists, in the same order: `keel gate task-start`, record the
|
|
20
|
+
fingerprint, `keel project goal … --json` for the view, implement inside `Touch`, surface every result
|
|
21
|
+
in the transcript, `keel gate task-complete`, check the box. What changes is who types them.
|
|
22
|
+
|
|
23
|
+
## Per-target activation notes
|
|
24
|
+
|
|
25
|
+
- **Codex**: where a callable goal or subagent surface exists, activate one bounded goal for the
|
|
26
|
+
selected task and use subagents only as bounded read-only helpers. Without a callable surface, paste
|
|
27
|
+
the exact `keel project goal` command and treat the capability as advisory.
|
|
28
|
+
- **Claude**: activate one `/goal` whose condition fits the 4,000-character budget. The evaluator sees
|
|
29
|
+
the transcript only, so command and gate evidence has to appear there explicitly. If hooks are
|
|
30
|
+
disabled, policy blocks activation, or trust is missing, the manual sequence above applies.
|
|
@@ -38,8 +38,8 @@ REQUIRED_SCRIPTS = [
|
|
|
38
38
|
"scripts/validate_plugin.py",
|
|
39
39
|
]
|
|
40
40
|
|
|
41
|
-
PACKAGE_VERSION = "5.
|
|
42
|
-
PROTOCOL_VERSION = "5.
|
|
41
|
+
PACKAGE_VERSION = "5.67.0"
|
|
42
|
+
PROTOCOL_VERSION = "5.67.0"
|
|
43
43
|
LEGACY_MANAGED_START = "<!-- keel:start version=2.1 -->"
|
|
44
44
|
OPENSPEC_SCHEMA_NAME = "keel-spec-driven"
|
|
45
45
|
# Mirrors KEEL_PACKAGE_NAME in scripts/install_to_repo.py, one of the two
|
|
@@ -14396,6 +14396,27 @@ def validate_native_plugin_manifests_scenario() -> int:
|
|
|
14396
14396
|
f"source: {skill_name}"
|
|
14397
14397
|
)
|
|
14398
14398
|
return 1
|
|
14399
|
+
# A referenced `guidance.md` travels with the body that names it. The
|
|
14400
|
+
# host reads the plugin copy directly, so a guidance file the parity
|
|
14401
|
+
# check ignored could drift from the criteria it was split out of, and
|
|
14402
|
+
# the drift would be invisible to everything except a reader.
|
|
14403
|
+
canonical_guidance = canonical_skill.parent / "guidance.md"
|
|
14404
|
+
plugin_guidance = plugin_skill.parent / "guidance.md"
|
|
14405
|
+
if canonical_guidance.is_file() and not plugin_guidance.is_file():
|
|
14406
|
+
report(
|
|
14407
|
+
"native-plugin-manifests guidance file diverges from canonical "
|
|
14408
|
+
f"source: {skill_name} has guidance.md that the plugin does not "
|
|
14409
|
+
"ship, so the body's reference resolves to nothing"
|
|
14410
|
+
)
|
|
14411
|
+
return 1
|
|
14412
|
+
if canonical_guidance.is_file() and plugin_guidance.read_text(
|
|
14413
|
+
encoding="utf-8"
|
|
14414
|
+
) != canonical_guidance.read_text(encoding="utf-8"):
|
|
14415
|
+
report(
|
|
14416
|
+
"native-plugin-manifests guidance file diverges from canonical "
|
|
14417
|
+
f"source: {skill_name}"
|
|
14418
|
+
)
|
|
14419
|
+
return 1
|
|
14399
14420
|
lenses_root = ROOT / "assets/lenses"
|
|
14400
14421
|
for template in ("web.md", "hardware.md", "hardware-dsl.md"):
|
|
14401
14422
|
if not (lenses_root / template).is_file():
|
|
@@ -20692,6 +20713,15 @@ def validate_native_goal_capabilities_scenario() -> int:
|
|
|
20692
20713
|
return 0
|
|
20693
20714
|
|
|
20694
20715
|
|
|
20716
|
+
# The referenced half of a split skill. Returns "" for a skill with one body, so
|
|
20717
|
+
# every assertion about a skill's content reads the same shape whether or not it
|
|
20718
|
+
# was split — a split must not be able to drop a required statement, and a
|
|
20719
|
+
# scenario must not have to know which skills were split to stay correct.
|
|
20720
|
+
def skill_guidance_text(skill_md: Path) -> str:
|
|
20721
|
+
guidance = skill_md.parent / "guidance.md"
|
|
20722
|
+
return guidance.read_text(encoding="utf-8") if guidance.is_file() else ""
|
|
20723
|
+
|
|
20724
|
+
|
|
20695
20725
|
SINGLE_TASK_GOAL_SKILL = "keel-run-single-task-goal"
|
|
20696
20726
|
OFFICIAL_GOAL_SOURCES = (
|
|
20697
20727
|
"https://learn.chatgpt.com/use-cases/follow-goals",
|
|
@@ -20711,7 +20741,13 @@ def validate_single_task_goal_skill_scenario() -> int:
|
|
|
20711
20741
|
if canonical_bytes != projection.read_bytes():
|
|
20712
20742
|
report("single-task-goal-skill projection is not byte-equal to the canonical source.")
|
|
20713
20743
|
return 1
|
|
20714
|
-
|
|
20744
|
+
# A skill split into a body and a referenced `guidance.md` carries its
|
|
20745
|
+
# content across both files. What these assertions require is that the skill
|
|
20746
|
+
# states the thing, not that SKILL.md does, so the split must not be able to
|
|
20747
|
+
# drop a required statement by moving it — and must not be able to keep one
|
|
20748
|
+
# only in a file the plugin does not ship, which `native-plugin-manifests`
|
|
20749
|
+
# checks byte-for-byte alongside the body.
|
|
20750
|
+
text = canonical_bytes.decode("utf-8") + skill_guidance_text(canonical)
|
|
20715
20751
|
|
|
20716
20752
|
# Authoritative official sources are linked, and provenance/license is recorded.
|
|
20717
20753
|
for source in OFFICIAL_GOAL_SOURCES:
|
|
@@ -20916,8 +20952,9 @@ def _goal_target_surface(target: str) -> int:
|
|
|
20916
20952
|
return 1
|
|
20917
20953
|
|
|
20918
20954
|
# The skill carries the target-specific fallback guidance.
|
|
20919
|
-
|
|
20920
|
-
|
|
20955
|
+
skill_body = ROOT / "plugins/keel/skills" / SINGLE_TASK_GOAL_SKILL / "SKILL.md"
|
|
20956
|
+
skill_text = skill_body.read_text(encoding="utf-8") + skill_guidance_text(
|
|
20957
|
+
skill_body
|
|
20921
20958
|
)
|
|
20922
20959
|
if target == "claude" and "disabled hooks" not in skill_text.lower():
|
|
20923
20960
|
report("native-goal-claude skill lacks the disabled-hooks fallback.")
|
|
@@ -28721,10 +28758,17 @@ def validate_a_count_is_derived_from_what_it_counts_scenario() -> int:
|
|
|
28721
28758
|
|
|
28722
28759
|
# A header naming every declaration while miscounting them in prose passes:
|
|
28723
28760
|
# membership is the checkable property, and a count is a lossy restatement.
|
|
28724
|
-
|
|
28725
|
-
|
|
28761
|
+
# The numeral is located by shape, not by its current value. Pinning the
|
|
28762
|
+
# word here would reintroduce the literal this scenario exists to remove,
|
|
28763
|
+
# one level down: adding `executor_tier` moved the header to "Seven" and the
|
|
28764
|
+
# mutation stopped finding anything to vary.
|
|
28765
|
+
miscounted, varied = re.subn(
|
|
28766
|
+
r"\b\w+ independent declarations\b",
|
|
28767
|
+
"Zero independent declarations",
|
|
28768
|
+
flat_header,
|
|
28769
|
+
count=1,
|
|
28726
28770
|
)
|
|
28727
|
-
if
|
|
28771
|
+
if not varied:
|
|
28728
28772
|
report(f"{label}: the header's prose count could not be located to vary it.")
|
|
28729
28773
|
return 1
|
|
28730
28774
|
if config_header_problem(names, miscounted):
|
|
@@ -28741,6 +28785,263 @@ def validate_a_count_is_derived_from_what_it_counts_scenario() -> int:
|
|
|
28741
28785
|
return 0
|
|
28742
28786
|
|
|
28743
28787
|
|
|
28788
|
+
def validate_a_tier_declares_what_is_skipped_scenario() -> int:
|
|
28789
|
+
"""Issue #135: which guidance an executor skips is a declaration, not a judgement.
|
|
28790
|
+
|
|
28791
|
+
A strong executor gets no value from stepwise how-to prose and pays for it on
|
|
28792
|
+
every activation. The report's own argument against letting the executor
|
|
28793
|
+
decide is that "do I need this?" is the judgement it is worst at, so the skip
|
|
28794
|
+
is written down by the repository. The tier reaches guidance and nothing else:
|
|
28795
|
+
a reader who takes it for a relaxation of the gates is worse off than one who
|
|
28796
|
+
never saw it, which is why every surface that reports it says so.
|
|
28797
|
+
"""
|
|
28798
|
+
label = "a-tier-declares-what-is-skipped"
|
|
28799
|
+
|
|
28800
|
+
with tempfile.TemporaryDirectory(prefix="keel-tier-") as raw:
|
|
28801
|
+
root = Path(raw)
|
|
28802
|
+
|
|
28803
|
+
def fixture(name: str, body: str) -> Path:
|
|
28804
|
+
repo = root / name
|
|
28805
|
+
repo.mkdir()
|
|
28806
|
+
(repo / "keel").mkdir()
|
|
28807
|
+
(repo / "keel" / "config.yaml").write_text(body, encoding="utf-8")
|
|
28808
|
+
return repo
|
|
28809
|
+
|
|
28810
|
+
def context(repo: Path) -> str:
|
|
28811
|
+
return run_keel(repo, "context").stdout
|
|
28812
|
+
|
|
28813
|
+
# M1 — the declared tier is reported, an absent one reports the default,
|
|
28814
|
+
# and both say guidance is all the tier reaches.
|
|
28815
|
+
for name, body, expected in (
|
|
28816
|
+
("high", "fast_check: echo high\nexecutor_tier: high\n", "high"),
|
|
28817
|
+
("absent", "fast_check: echo absent\n", "standard"),
|
|
28818
|
+
):
|
|
28819
|
+
repo = fixture(name, body)
|
|
28820
|
+
out = context(repo)
|
|
28821
|
+
line = next(
|
|
28822
|
+
(l for l in out.splitlines() if l.startswith("Executor tier:")), None
|
|
28823
|
+
)
|
|
28824
|
+
if line is None:
|
|
28825
|
+
report(
|
|
28826
|
+
f"{label}: no executor tier reported — the {name} repository "
|
|
28827
|
+
"is told nothing about which guidance its skills load."
|
|
28828
|
+
)
|
|
28829
|
+
report(out)
|
|
28830
|
+
return 1
|
|
28831
|
+
if expected not in line:
|
|
28832
|
+
report(
|
|
28833
|
+
f"{label}: the {name} repository reports {line!r} rather than "
|
|
28834
|
+
f"the {expected!r} tier."
|
|
28835
|
+
)
|
|
28836
|
+
return 1
|
|
28837
|
+
if "guidance" not in line:
|
|
28838
|
+
report(
|
|
28839
|
+
f"{label}: no executor tier reported as affecting guidance "
|
|
28840
|
+
f"only; {line!r} leaves a reader free to take it for a "
|
|
28841
|
+
"relaxation of the gates."
|
|
28842
|
+
)
|
|
28843
|
+
return 1
|
|
28844
|
+
|
|
28845
|
+
# M2 — a value Keel cannot read fails closed to reading the guidance.
|
|
28846
|
+
# The author of a typo believes they declared what they typed, so a
|
|
28847
|
+
# misspelling must not silently buy the skip it asked for, and the
|
|
28848
|
+
# refusal names the value and the alternatives rather than the key.
|
|
28849
|
+
typo = fixture("typo", "fast_check: echo typo\nexecutor_tier: aggressive\n")
|
|
28850
|
+
out = context(typo)
|
|
28851
|
+
line = next(
|
|
28852
|
+
(l for l in out.splitlines() if l.startswith("Executor tier:")), None
|
|
28853
|
+
)
|
|
28854
|
+
if line is None:
|
|
28855
|
+
report(
|
|
28856
|
+
f"{label}: accepted a tier outside the set — the projection "
|
|
28857
|
+
"reported no tier at all for an unreadable value, so the "
|
|
28858
|
+
"fallback is invisible."
|
|
28859
|
+
)
|
|
28860
|
+
report(out)
|
|
28861
|
+
return 1
|
|
28862
|
+
if "standard" not in line:
|
|
28863
|
+
report(
|
|
28864
|
+
f"{label}: accepted a tier outside the set — an unreadable value "
|
|
28865
|
+
f"did not fall back to reading the guidance; got {line!r}."
|
|
28866
|
+
)
|
|
28867
|
+
report(out)
|
|
28868
|
+
return 1
|
|
28869
|
+
if "aggressive" not in out:
|
|
28870
|
+
report(
|
|
28871
|
+
f"{label}: the rejected value is not named, so the author is "
|
|
28872
|
+
"left believing they declared what they typed."
|
|
28873
|
+
)
|
|
28874
|
+
report(out)
|
|
28875
|
+
return 1
|
|
28876
|
+
if "high" not in out:
|
|
28877
|
+
report(
|
|
28878
|
+
f"{label}: the refusal does not name the accepted tiers, so the "
|
|
28879
|
+
"reader learns their value is wrong and not what is right."
|
|
28880
|
+
)
|
|
28881
|
+
report(out)
|
|
28882
|
+
return 1
|
|
28883
|
+
doctor = run_keel(typo, "--doctor").stdout
|
|
28884
|
+
if "executor_tier" not in doctor or "aggressive" not in doctor:
|
|
28885
|
+
report(
|
|
28886
|
+
f"{label}: the doctor does not report the declaration as failed "
|
|
28887
|
+
"and name the entry, so it reports health the projection "
|
|
28888
|
+
"contradicts."
|
|
28889
|
+
)
|
|
28890
|
+
report(doctor)
|
|
28891
|
+
return 1
|
|
28892
|
+
healthy = run_keel(fixture("ok", "executor_tier: high\n"), "--doctor").stdout
|
|
28893
|
+
if "executor_tier: high" not in healthy:
|
|
28894
|
+
report(
|
|
28895
|
+
f"{label}: the doctor does not report a readable declaration, so "
|
|
28896
|
+
"the only tier it ever mentions is a broken one."
|
|
28897
|
+
)
|
|
28898
|
+
report(healthy)
|
|
28899
|
+
return 1
|
|
28900
|
+
|
|
28901
|
+
if label not in {name for name, _ in SCENARIOS}:
|
|
28902
|
+
report(f"{label}: the scenario registry does not include it.")
|
|
28903
|
+
return 1
|
|
28904
|
+
report(f"{label} scenario passed.")
|
|
28905
|
+
return 0
|
|
28906
|
+
|
|
28907
|
+
|
|
28908
|
+
# The words Keel states a criterion in. A guidance file is asserted to contain
|
|
28909
|
+
# none of them, which is what makes "the executor tier removes no criterion" a
|
|
28910
|
+
# property of the repository rather than a promise in a design document: a tier
|
|
28911
|
+
# can only skip a file that decides nothing.
|
|
28912
|
+
CRITERION_VOCABULARY = ("MUST", "SHOULD", "refuses", "rejects", "hard-stops")
|
|
28913
|
+
|
|
28914
|
+
|
|
28915
|
+
def guidance_criterion_problem(name: str, text: str) -> str | None:
|
|
28916
|
+
"""Return the problem a guidance file's text has, or None.
|
|
28917
|
+
|
|
28918
|
+
Takes the text rather than reading the path, so the rule can be exercised on
|
|
28919
|
+
a planted copy. A rule that could only ever see files already known to be
|
|
28920
|
+
clean would pass forever without anyone learning whether it fires.
|
|
28921
|
+
"""
|
|
28922
|
+
for word in CRITERION_VOCABULARY:
|
|
28923
|
+
if word in text:
|
|
28924
|
+
return (
|
|
28925
|
+
f"{name}'s guidance.md states a criterion — it contains "
|
|
28926
|
+
f"{word!r}, and a criterion in the file the executor tier skips "
|
|
28927
|
+
"would let `executor_tier: high` relax a rule rather than skip "
|
|
28928
|
+
"an explanation. Move the sentence back into SKILL.md."
|
|
28929
|
+
)
|
|
28930
|
+
return None
|
|
28931
|
+
|
|
28932
|
+
|
|
28933
|
+
def validate_guidance_is_referenced_and_carries_no_criterion_scenario() -> int:
|
|
28934
|
+
"""Issue #135: stepwise guidance is referenced, and it decides nothing.
|
|
28935
|
+
|
|
28936
|
+
Keel is not on the delivery path for a skill body — the host reads
|
|
28937
|
+
`plugins/keel/skills/*/SKILL.md` directly — so the only way the resident cost
|
|
28938
|
+
falls is for the body to be smaller. What leaves it is how-to prose; every
|
|
28939
|
+
criterion stays, and that is checked here by planting one rather than
|
|
28940
|
+
promised, because a tier that could remove a criterion would be a relaxation
|
|
28941
|
+
of the gates wearing a capability label.
|
|
28942
|
+
"""
|
|
28943
|
+
label = "guidance-is-referenced-and-carries-no-criterion"
|
|
28944
|
+
skills_root = ROOT / "src/skills"
|
|
28945
|
+
guidance_files = sorted(skills_root.glob("*/guidance.md"))
|
|
28946
|
+
|
|
28947
|
+
# D4 splits exactly one skill today. An empty glob would satisfy every loop
|
|
28948
|
+
# below without reading anything, so the count is asserted first.
|
|
28949
|
+
if not guidance_files:
|
|
28950
|
+
report(
|
|
28951
|
+
f"{label}: no skill has a guidance.md, so every assertion below is "
|
|
28952
|
+
"vacuous — the split this scenario checks did not happen."
|
|
28953
|
+
)
|
|
28954
|
+
return 1
|
|
28955
|
+
|
|
28956
|
+
for guidance in guidance_files:
|
|
28957
|
+
skill = guidance.parent / "SKILL.md"
|
|
28958
|
+
body = skill.read_text(encoding="utf-8")
|
|
28959
|
+
text = guidance.read_text(encoding="utf-8")
|
|
28960
|
+
if "guidance.md" not in body:
|
|
28961
|
+
report(
|
|
28962
|
+
f"{label}: guidance file is not referenced — "
|
|
28963
|
+
f"{guidance.parent.name}'s SKILL.md never names guidance.md, so "
|
|
28964
|
+
"the prose that left the body is unreachable from it."
|
|
28965
|
+
)
|
|
28966
|
+
return 1
|
|
28967
|
+
if "executor_tier" not in body:
|
|
28968
|
+
report(
|
|
28969
|
+
f"{label}: {guidance.parent.name}'s SKILL.md names guidance.md "
|
|
28970
|
+
"without the condition under which it is read, which leaves the "
|
|
28971
|
+
"skip to the executor's own judgement — the judgement #135 says "
|
|
28972
|
+
"it gets most wrong."
|
|
28973
|
+
)
|
|
28974
|
+
return 1
|
|
28975
|
+
if "high" not in body:
|
|
28976
|
+
report(
|
|
28977
|
+
f"{label}: {guidance.parent.name}'s SKILL.md does not name the "
|
|
28978
|
+
"tier that skips the read, so the condition it states cannot be "
|
|
28979
|
+
"evaluated."
|
|
28980
|
+
)
|
|
28981
|
+
return 1
|
|
28982
|
+
# "Measurably smaller than before the split" is checkable only as a
|
|
28983
|
+
# property of what is on disk now: the guidance carries real content,
|
|
28984
|
+
# and that content is no longer duplicated in the body it left.
|
|
28985
|
+
if len(text) < 500:
|
|
28986
|
+
report(
|
|
28987
|
+
f"{label}: {guidance.parent.name}'s guidance.md holds "
|
|
28988
|
+
f"{len(text)} bytes, too few for the split to have moved "
|
|
28989
|
+
"anything; a pointer to an almost-empty file costs a read and "
|
|
28990
|
+
"saves nothing."
|
|
28991
|
+
)
|
|
28992
|
+
return 1
|
|
28993
|
+
criterion = guidance_criterion_problem(guidance.parent.name, text)
|
|
28994
|
+
if criterion:
|
|
28995
|
+
report(f"{label}: {criterion}")
|
|
28996
|
+
return 1
|
|
28997
|
+
for heading in [
|
|
28998
|
+
line for line in text.splitlines() if line.startswith("## ")
|
|
28999
|
+
]:
|
|
29000
|
+
if heading in body:
|
|
29001
|
+
report(
|
|
29002
|
+
f"{label}: {guidance.parent.name}'s body still carries "
|
|
29003
|
+
f"{heading!r}, so the section was copied rather than moved "
|
|
29004
|
+
"and the resident cost did not fall."
|
|
29005
|
+
)
|
|
29006
|
+
return 1
|
|
29007
|
+
|
|
29008
|
+
# M2 — the property is held by the suite, not promised by the design. A
|
|
29009
|
+
# guidance file that stated a criterion would let the tier skip one, which
|
|
29010
|
+
# would make `executor_tier: high` a relaxation of the gates wearing a
|
|
29011
|
+
# capability label. Planting one is the only way to know the rule fires.
|
|
29012
|
+
sample = guidance_files[0]
|
|
29013
|
+
planted = (
|
|
29014
|
+
sample.read_text(encoding="utf-8")
|
|
29015
|
+
+ "\nAn executor MUST record the fingerprint before implementing.\n"
|
|
29016
|
+
)
|
|
29017
|
+
problem = guidance_criterion_problem(sample.parent.name, planted)
|
|
29018
|
+
if not problem:
|
|
29019
|
+
report(
|
|
29020
|
+
f"{label}: states a criterion — a guidance file carrying `MUST` was "
|
|
29021
|
+
"accepted, so nothing stops a criterion from moving into the file "
|
|
29022
|
+
"the tier skips."
|
|
29023
|
+
)
|
|
29024
|
+
return 1
|
|
29025
|
+
if sample.parent.name not in problem:
|
|
29026
|
+
report(
|
|
29027
|
+
f"{label}: the refusal does not name the file that states the "
|
|
29028
|
+
f"criterion; got {problem!r}."
|
|
29029
|
+
)
|
|
29030
|
+
return 1
|
|
29031
|
+
if "MUST" not in problem:
|
|
29032
|
+
report(
|
|
29033
|
+
f"{label}: the refusal does not name the word it objected to, so the "
|
|
29034
|
+
f"author has to guess which sentence to move; got {problem!r}."
|
|
29035
|
+
)
|
|
29036
|
+
return 1
|
|
29037
|
+
|
|
29038
|
+
if label not in {name for name, _ in SCENARIOS}:
|
|
29039
|
+
report(f"{label}: the scenario registry does not include it.")
|
|
29040
|
+
return 1
|
|
29041
|
+
report(f"{label} scenario passed.")
|
|
29042
|
+
return 0
|
|
29043
|
+
|
|
29044
|
+
|
|
28744
29045
|
SCENARIOS: tuple = (
|
|
28745
29046
|
("stateless-continuity", validate_stateless_continuity_scenario),
|
|
28746
29047
|
("core-gates", validate_core_gates_scenario),
|
|
@@ -29103,6 +29404,14 @@ SCENARIOS: tuple = (
|
|
|
29103
29404
|
"a-negation-is-not-a-marker",
|
|
29104
29405
|
validate_a_negation_is_not_a_marker_scenario,
|
|
29105
29406
|
),
|
|
29407
|
+
(
|
|
29408
|
+
"guidance-is-referenced-and-carries-no-criterion",
|
|
29409
|
+
validate_guidance_is_referenced_and_carries_no_criterion_scenario,
|
|
29410
|
+
),
|
|
29411
|
+
(
|
|
29412
|
+
"a-tier-declares-what-is-skipped",
|
|
29413
|
+
validate_a_tier_declares_what_is_skipped_scenario,
|
|
29414
|
+
),
|
|
29106
29415
|
(
|
|
29107
29416
|
"a-count-is-derived-from-what-it-counts",
|
|
29108
29417
|
validate_a_count_is_derived_from_what_it_counts_scenario,
|
package/src/core/config.js
CHANGED
|
@@ -81,6 +81,7 @@ const CONFIG_DECLARATIONS = [
|
|
|
81
81
|
"triage",
|
|
82
82
|
"delegation",
|
|
83
83
|
"full_mode_paths",
|
|
84
|
+
"executor_tier",
|
|
84
85
|
];
|
|
85
86
|
|
|
86
87
|
const CONFIG_RELATIVE_PATH = path.join("keel", "config.yaml");
|
|
@@ -402,6 +403,55 @@ function readDelegationPolicy(repo) {
|
|
|
402
403
|
return { declared: true, tier, unknown: [], accepted };
|
|
403
404
|
}
|
|
404
405
|
|
|
406
|
+
// Which guidance an executor loads, declared by the repository rather than
|
|
407
|
+
// judged by the executor. #135's argument is that "do I need this help?" is the
|
|
408
|
+
// judgement a weak executor gets most wrong, so the skip is written down; the
|
|
409
|
+
// absent default therefore reads the guidance, and the worst an unconfigured
|
|
410
|
+
// repository does is pay for a read.
|
|
411
|
+
const EXECUTOR_TIERS = ["standard", "high"];
|
|
412
|
+
const EXECUTOR_TIER_DEFAULT = "standard";
|
|
413
|
+
|
|
414
|
+
// The tier reaches guidance and nothing else. It is a separate key from
|
|
415
|
+
// `delegation:`'s tier on purpose: that one names who runs a *delegated* task,
|
|
416
|
+
// and a repository may hand routine work to a weak delegate while its own
|
|
417
|
+
// session is strong, so one field cannot answer both without lying about one.
|
|
418
|
+
function readExecutorTier(repo) {
|
|
419
|
+
const declared = configScalar(repo, "executor_tier");
|
|
420
|
+
const accepted = [...EXECUTOR_TIERS];
|
|
421
|
+
if (!declared) {
|
|
422
|
+
return {
|
|
423
|
+
declared: false,
|
|
424
|
+
tier: EXECUTOR_TIER_DEFAULT,
|
|
425
|
+
unknown: [],
|
|
426
|
+
accepted,
|
|
427
|
+
};
|
|
428
|
+
}
|
|
429
|
+
// Fail closed, the same rule `delegation:` and `authorize:` follow — except
|
|
430
|
+
// that closed here means *reading* the guidance rather than skipping it. A
|
|
431
|
+
// misspelling must not silently buy the reduction it asked for.
|
|
432
|
+
if (!EXECUTOR_TIERS.includes(declared)) {
|
|
433
|
+
return {
|
|
434
|
+
declared: true,
|
|
435
|
+
tier: EXECUTOR_TIER_DEFAULT,
|
|
436
|
+
unknown: [declared],
|
|
437
|
+
accepted,
|
|
438
|
+
message: executorTierUnreadableMessage(declared, accepted),
|
|
439
|
+
};
|
|
440
|
+
}
|
|
441
|
+
return { declared: true, tier: declared, unknown: [], accepted };
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
// Names the value and the alternatives, never just the key: a reader told their
|
|
445
|
+
// declaration is wrong without being told what is right has to go find the
|
|
446
|
+
// documentation the declaration was supposed to replace.
|
|
447
|
+
function executorTierUnreadableMessage(value, accepted) {
|
|
448
|
+
return (
|
|
449
|
+
`keel/config.yaml declares executor_tier: ${value}, which is not one of `
|
|
450
|
+
+ `${accepted.join(", ")}; the ${EXECUTOR_TIER_DEFAULT} tier applies until `
|
|
451
|
+
+ "it is corrected, so skill guidance is read rather than skipped"
|
|
452
|
+
);
|
|
453
|
+
}
|
|
454
|
+
|
|
405
455
|
function configScalar(repo, key) {
|
|
406
456
|
const configPath = path.join(repo, "keel", "config.yaml");
|
|
407
457
|
if (!fs.existsSync(configPath)) return null;
|
|
@@ -548,12 +598,16 @@ function triageIssue(repo, labels, issue = null) {
|
|
|
548
598
|
module.exports = {
|
|
549
599
|
CONFIG_RELATIVE_PATH,
|
|
550
600
|
DELEGATION_TIERS,
|
|
601
|
+
EXECUTOR_TIERS,
|
|
602
|
+
EXECUTOR_TIER_DEFAULT,
|
|
551
603
|
CONFIG_DECLARATIONS,
|
|
552
604
|
STANDING_AUTHORIZATION_ACTIONS,
|
|
553
605
|
SCOPED_AUTHORIZATION_ACTIONS,
|
|
554
606
|
readFullModePaths,
|
|
555
607
|
fullModePathsUnreadableMessage,
|
|
556
608
|
readDelegationPolicy,
|
|
609
|
+
readExecutorTier,
|
|
610
|
+
executorTierUnreadableMessage,
|
|
557
611
|
readPrecedentStore,
|
|
558
612
|
readStandingAuthorization,
|
|
559
613
|
readTriagePolicy,
|
package/src/core/context.js
CHANGED
|
@@ -15,6 +15,7 @@ const {
|
|
|
15
15
|
readStandingAuthorization,
|
|
16
16
|
readFullModePaths,
|
|
17
17
|
fullModePathsUnreadableMessage,
|
|
18
|
+
readExecutorTier,
|
|
18
19
|
} = require("./config");
|
|
19
20
|
|
|
20
21
|
const NEXT_ACTIONS = new Set([
|
|
@@ -696,6 +697,14 @@ function resolveContext(repo, options) {
|
|
|
696
697
|
} else {
|
|
697
698
|
context.routing = routing.paths;
|
|
698
699
|
}
|
|
700
|
+
// Reported every session, declared or not, unlike `full_mode_paths` above:
|
|
701
|
+
// the default is the one a reader most needs to see, because a repository
|
|
702
|
+
// that declared nothing is loading guidance it may not want and has no other
|
|
703
|
+
// surface that would tell it so.
|
|
704
|
+
const executor = readExecutorTier(repo);
|
|
705
|
+
context.executorTier = executor.tier;
|
|
706
|
+
if (executor.unknown.length > 0) context.warnings.push(executor.message);
|
|
707
|
+
|
|
699
708
|
// Set here rather than by the caller, so every consumer of the projection —
|
|
700
709
|
// text, JSON, and any host reading it — carries the version without having
|
|
701
710
|
// to know to add it.
|
|
@@ -748,6 +757,13 @@ function renderContext(result) {
|
|
|
748
757
|
for (const entry of result.routing || []) {
|
|
749
758
|
lines.push(`Routing: ${entry.path} always routes Full — ${entry.reason}`);
|
|
750
759
|
}
|
|
760
|
+
if (result.executorTier) {
|
|
761
|
+
lines.push(
|
|
762
|
+
`Executor tier: ${result.executorTier} — affects which skill guidance is `
|
|
763
|
+
+ "read and nothing else; no gate, criterion, evidence requirement, or "
|
|
764
|
+
+ "Review changes with it"
|
|
765
|
+
);
|
|
766
|
+
}
|
|
751
767
|
for (const reason of result.reasons) lines.push(`Reason: ${reason}`);
|
|
752
768
|
for (const warning of result.warnings) lines.push(`Warning: ${warning}`);
|
|
753
769
|
return `${lines.join("\n")}\n`;
|