@iowarp/clio-coder 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +407 -0
- package/CODE_OF_CONDUCT.md +21 -0
- package/CONTRIBUTING.md +224 -0
- package/LICENSE +202 -0
- package/NOTICE +9 -0
- package/README.md +798 -0
- package/SECURITY.md +72 -0
- package/assets/clio-coder-logo-128.webp +0 -0
- package/damage-control-rules.yaml +419 -0
- package/dist/acp-UMLFVA3F.js +92 -0
- package/dist/agents-Q4MYPMUW.js +91 -0
- package/dist/auth-O6HYIJ6J.js +521 -0
- package/dist/chunk-262G75JS.js +35 -0
- package/dist/chunk-26BZQOAD.js +1281 -0
- package/dist/chunk-2J63S4SF.js +508 -0
- package/dist/chunk-3DANZDGR.js +717 -0
- package/dist/chunk-4UQA7NCT.js +29 -0
- package/dist/chunk-527KG6XR.js +497 -0
- package/dist/chunk-5LDRNKX2.js +1063 -0
- package/dist/chunk-5N2FG33Q.js +25 -0
- package/dist/chunk-67MTHP2E.js +135 -0
- package/dist/chunk-6CWDTGUC.js +20 -0
- package/dist/chunk-7BHLZB3A.js +2115 -0
- package/dist/chunk-7RBKDI66.js +348 -0
- package/dist/chunk-AMFR5YA3.js +541 -0
- package/dist/chunk-BBUH4VAA.js +1224 -0
- package/dist/chunk-BYEU76JP.js +899 -0
- package/dist/chunk-CLJ5HLUD.js +458 -0
- package/dist/chunk-D5YD55AR.js +116 -0
- package/dist/chunk-DXQNI4PC.js +61 -0
- package/dist/chunk-E3NYWENM.js +1004 -0
- package/dist/chunk-GNGDQYDU.js +34688 -0
- package/dist/chunk-GOTUR54M.js +9 -0
- package/dist/chunk-HBU5MTAM.js +41 -0
- package/dist/chunk-HMYNFFY4.js +28 -0
- package/dist/chunk-JPOWPFCU.js +1010 -0
- package/dist/chunk-JWHCJDCI.js +1215 -0
- package/dist/chunk-KBR4MZZR.js +41 -0
- package/dist/chunk-KKKPTZLM.js +93 -0
- package/dist/chunk-ME6DNWIU.js +66 -0
- package/dist/chunk-NI4DEJMC.js +88 -0
- package/dist/chunk-O4EJEDHO.js +659 -0
- package/dist/chunk-PIDUD6M2.js +31 -0
- package/dist/chunk-PS4PFJQP.js +29459 -0
- package/dist/chunk-QV47YRF4.js +48 -0
- package/dist/chunk-RQDWMVRB.js +279 -0
- package/dist/chunk-TFSSEXL6.js +136 -0
- package/dist/chunk-TKHQ4DGZ.js +8290 -0
- package/dist/chunk-TPOCL34A.js +2876 -0
- package/dist/chunk-UGYAX5YI.js +565 -0
- package/dist/chunk-UHTSULZS.js +461 -0
- package/dist/chunk-UU3R62TT.js +128 -0
- package/dist/chunk-UWIJNAOB.js +3906 -0
- package/dist/chunk-VOO7NYPP.js +914 -0
- package/dist/chunk-VPAWTYLY.js +117 -0
- package/dist/chunk-WD6AJM35.js +1216 -0
- package/dist/chunk-X3BR7HWV.js +115 -0
- package/dist/chunk-X3NE4WVW.js +120 -0
- package/dist/chunk-XNISANGE.js +1395 -0
- package/dist/chunk-XV4ZJ6ZM.js +3177 -0
- package/dist/cli/index.js +236 -0
- package/dist/clio-KIQ5SNDS.js +53 -0
- package/dist/components-JVHMUBEB.js +653 -0
- package/dist/config-ZFCDBMDC.js +372 -0
- package/dist/configure-G4E3A2PG.js +27 -0
- package/dist/context-CDXTP2MP.js +293 -0
- package/dist/context-E3KIFVXI.js +185 -0
- package/dist/context-clear-3F4PLXOS.js +102 -0
- package/dist/context-index-Q7YSYTR3.js +106 -0
- package/dist/docs-YIETIWZI.js +280 -0
- package/dist/doctor-M5HJJZOL.js +61 -0
- package/dist/domains/agents/builtins/architect.md +33 -0
- package/dist/domains/agents/builtins/coder.md +31 -0
- package/dist/domains/agents/builtins/context-bootstrap.md +38 -0
- package/dist/domains/agents/builtins/debugger.md +30 -0
- package/dist/domains/agents/builtins/documenter.md +31 -0
- package/dist/domains/agents/builtins/git-master.md +30 -0
- package/dist/domains/agents/builtins/provenance.md +30 -0
- package/dist/domains/agents/builtins/researcher.md +71 -0
- package/dist/domains/agents/builtins/scout.md +42 -0
- package/dist/domains/agents/builtins/tester.md +31 -0
- package/dist/domains/agents/builtins/verifier.md +30 -0
- package/dist/domains/agents/builtins/wiki-writer.md +41 -0
- package/dist/eval-B3KZZESM.js +2674 -0
- package/dist/evidence-V67CHM35.js +233 -0
- package/dist/evolve-YDZSUQYA.js +518 -0
- package/dist/extensions-SRG7XCAH.js +207 -0
- package/dist/fleet-CA2CRTVG.js +760 -0
- package/dist/fleet-preflight-CLIAX7YR.js +21 -0
- package/dist/init-2OZDJE2D.js +227 -0
- package/dist/memory-3PIQQAKX.js +207 -0
- package/dist/models-DY35XI7Y.js +237 -0
- package/dist/paths-5OMXW7Z4.js +57 -0
- package/dist/preload-KZVHET2B.js +11 -0
- package/dist/reset-PIFYNOS3.js +216 -0
- package/dist/run-3VSPP24F.js +735 -0
- package/dist/share-D36RQCXM.js +241 -0
- package/dist/skills-F2MRLELY.js +445 -0
- package/dist/skills-eval-E2ZTW4PL.js +932 -0
- package/dist/targets-DZMEZAH4.js +977 -0
- package/dist/trace-7NYCUI2J.js +250 -0
- package/dist/uninstall-AD3JWHBB.js +322 -0
- package/dist/upgrade-WYYBKGDY.js +301 -0
- package/dist/usage-ULIDAGFF.js +755 -0
- package/dist/version-ROZ6CZKH.js +16 -0
- package/dist/wiki-generate-PKFIX6OB.js +377 -0
- package/dist/worker/entry.js +1739 -0
- package/docs/README.md +93 -0
- package/docs/acp.md +120 -0
- package/docs/alcf-provider.md +72 -0
- package/docs/architecture.md +172 -0
- package/docs/artifact-versions.md +54 -0
- package/docs/built-in-agents.md +265 -0
- package/docs/capacity-and-scheduling.md +97 -0
- package/docs/commands-and-modes.md +554 -0
- package/docs/config-knobs-audit.md +115 -0
- package/docs/configuration-and-targets.md +812 -0
- package/docs/context-engine.md +236 -0
- package/docs/dispatch-architecture-rationale.md +126 -0
- package/docs/documentation-coverage.md +46 -0
- package/docs/documentation-guide.md +166 -0
- package/docs/environment-variables.md +105 -0
- package/docs/eval-runner.md +205 -0
- package/docs/evals-internal.md +298 -0
- package/docs/evidence-and-memory.md +243 -0
- package/docs/evolution.md +143 -0
- package/docs/exit-codes-and-output.md +74 -0
- package/docs/extensions-and-sharing.md +306 -0
- package/docs/fleet-demo-runbook.md +179 -0
- package/docs/fleet-dispatch.md +591 -0
- package/docs/glossary.md +75 -0
- package/docs/html/agents_blueprint.html +936 -0
- package/docs/html/alcf_blueprint.html +324 -0
- package/docs/html/architecture_blueprint.html +850 -0
- package/docs/html/commands_blueprint.html +794 -0
- package/docs/html/config_knobs_audit_blueprint.html +178 -0
- package/docs/html/configuration_blueprint.html +1080 -0
- package/docs/html/context_blueprint.html +603 -0
- package/docs/html/documentation_blueprint.html +832 -0
- package/docs/html/environment_blueprint.html +404 -0
- package/docs/html/eval_blueprint.html +743 -0
- package/docs/html/evals_internal_blueprint.html +190 -0
- package/docs/html/evolution_blueprint.html +674 -0
- package/docs/html/extensions_blueprint.html +2065 -0
- package/docs/html/fleet_dispatch_blueprint.html +286 -0
- package/docs/html/index.html +919 -0
- package/docs/html/lifecycle_blueprint.html +723 -0
- package/docs/html/memory_blueprint.html +699 -0
- package/docs/html/middleware_blueprint.html +664 -0
- package/docs/html/models_blueprint.html +2366 -0
- package/docs/html/observability_blueprint.html +683 -0
- package/docs/html/provider_adapter_blueprint.html +245 -0
- package/docs/html/safety_blueprint.html +1386 -0
- package/docs/html/shared.css +571 -0
- package/docs/html/shared.js +143 -0
- package/docs/html/skills_blueprint.html +671 -0
- package/docs/html/soak_blueprint.html +182 -0
- package/docs/html/tool_usage_blueprint.html +350 -0
- package/docs/html/tools_blueprint.html +2249 -0
- package/docs/html/trace_blueprint.html +235 -0
- package/docs/html/tui_design_blueprint.html +314 -0
- package/docs/html/validation_blueprint.html +961 -0
- package/docs/html/worker_dispatch_blueprint.html +231 -0
- package/docs/installation-and-lifecycle.md +308 -0
- package/docs/middleware-and-components.md +148 -0
- package/docs/model-catalog.md +189 -0
- package/docs/observability.md +233 -0
- package/docs/proactive-memory.md +452 -0
- package/docs/prompt-envelope-and-tools.md +142 -0
- package/docs/provider-adapter-cookbook.md +148 -0
- package/docs/release-cut-checklist.md +138 -0
- package/docs/safety-model.md +357 -0
- package/docs/scientific-validation.md +105 -0
- package/docs/session-lifecycle.md +156 -0
- package/docs/skills-marketplace.md +46 -0
- package/docs/tool-usage.md +527 -0
- package/docs/trace-store.md +132 -0
- package/docs/troubleshooting.md +33 -0
- package/docs/tui-design.md +239 -0
- package/docs/worker-dispatch-mechanics.md +242 -0
- package/package.json +132 -0
- package/skills/README.md +408 -0
- package/skills/git/commit-crafting/SKILL.md +79 -0
- package/skills/git/commit-crafting/evals.md +92 -0
- package/skills/git/create-pr/SKILL.md +116 -0
- package/skills/git/create-pr/evals.md +114 -0
- package/skills/git/investigate-issue/SKILL.md +139 -0
- package/skills/git/investigate-issue/evals.md +94 -0
- package/skills/git/resolve-merge-conflicts/SKILL.md +96 -0
- package/skills/git/resolve-merge-conflicts/evals.md +58 -0
- package/skills/git/review-changes/SKILL.md +103 -0
- package/skills/git/review-changes/evals.md +85 -0
- package/skills/git/worktree-create/SKILL.md +92 -0
- package/skills/git/worktree-create/evals.md +97 -0
- package/skills/git/worktree-create/references/worktree-setup.md +66 -0
- package/skills/git/worktree-merge/SKILL.md +95 -0
- package/skills/git/worktree-merge/evals.md +114 -0
- package/skills/skill-marketplace.json +261 -0
- package/skills/workflow/cut-it/SKILL.md +86 -0
- package/skills/workflow/cut-it/evals.md +42 -0
- package/src/domains/agents/builtins/architect.md +33 -0
- package/src/domains/agents/builtins/coder.md +31 -0
- package/src/domains/agents/builtins/context-bootstrap.md +38 -0
- package/src/domains/agents/builtins/debugger.md +30 -0
- package/src/domains/agents/builtins/documenter.md +31 -0
- package/src/domains/agents/builtins/git-master.md +30 -0
- package/src/domains/agents/builtins/provenance.md +30 -0
- package/src/domains/agents/builtins/researcher.md +71 -0
- package/src/domains/agents/builtins/scout.md +42 -0
- package/src/domains/agents/builtins/tester.md +31 -0
- package/src/domains/agents/builtins/verifier.md +30 -0
- package/src/domains/agents/builtins/wiki-writer.md +41 -0
- package/src/domains/agents/fleets/build-review.md +34 -0
- package/src/domains/agents/fleets/build-test.md +35 -0
- package/src/domains/agents/fleets/sdlc.md +86 -0
- package/src/domains/prompts/fragments/identity/clio-worker.md +11 -0
- package/src/domains/prompts/fragments/identity/clio.md +26 -0
- package/src/domains/prompts/fragments/operating/contract.md +64 -0
- package/src/domains/prompts/fragments/safety/auto-edit.md +14 -0
- package/src/domains/prompts/fragments/safety/full-auto.md +14 -0
- package/src/domains/prompts/fragments/safety/read-only.md +13 -0
- package/src/domains/prompts/fragments/safety/suggest.md +13 -0
- package/src/domains/prompts/fragments/wiki/page.md +75 -0
- package/src/domains/prompts/fragments/wiki/plan.md +48 -0
- package/src/domains/providers/models/cloud-models/alcf.yaml +40 -0
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +993 -0
package/skills/README.md
ADDED
|
@@ -0,0 +1,408 @@
|
|
|
1
|
+
# Clio Skills Marketplace
|
|
2
|
+
|
|
3
|
+
Curated, version-controlled skills that Clio Coder's authors have reviewed and
|
|
4
|
+
approved. This folder is the **marketplace catalog**, which acts as a publishing shelf rather than a
|
|
5
|
+
runtime store.
|
|
6
|
+
|
|
7
|
+
## Marketplace vs runtime
|
|
8
|
+
|
|
9
|
+
Clio's engine discovers *runtime* skills from these roots (see
|
|
10
|
+
`src/domains/resources/skills/loader.ts`):
|
|
11
|
+
|
|
12
|
+
- extension roots
|
|
13
|
+
- `~/.agents`, `~/.claude`, `~/.codex`, `~/.config/opencode`, `~/.copilot` → their `skills/` subdir
|
|
14
|
+
- `<clio-config>/skills` (per-user)
|
|
15
|
+
- project `.agents` / `.claude` / `.codex` / `.opencode` / `.github` → their `skills/` subdir
|
|
16
|
+
- `.clio-coder/skills` (per-project)
|
|
17
|
+
|
|
18
|
+
This repo's `skills/` directory is **not** one of those roots, so nothing here
|
|
19
|
+
auto-loads. That gap is deliberate.
|
|
20
|
+
|
|
21
|
+
| | Runtime skill | Marketplace skill (here) |
|
|
22
|
+
|---|---|---|
|
|
23
|
+
| Location | a discovery root above | `skills/<category>/<name>/` in this repo |
|
|
24
|
+
| Author | any user or harness | Clio authors, reviewed |
|
|
25
|
+
| Provenance | none required | `clio:` block with `registry-id` + `source-url` + `audit: pass` |
|
|
26
|
+
| Auto-loaded | yes | no, as it must be installed |
|
|
27
|
+
|
|
28
|
+
"Approved" is visible in the frontmatter: a maintainer set `clio.audit: pass`
|
|
29
|
+
and a `version`. A skill a user wrote themselves carries none of those fields.
|
|
30
|
+
|
|
31
|
+
## Catalog
|
|
32
|
+
|
|
33
|
+
The catalog is organized by theme: each skill lives at
|
|
34
|
+
`skills/<category>/<name>/`. Skill names stay globally unique; the category
|
|
35
|
+
folder is presentation and provenance, not a namespace.
|
|
36
|
+
|
|
37
|
+
### `planning/` — from idea to committed intent
|
|
38
|
+
|
|
39
|
+
| Skill | Type | Use when |
|
|
40
|
+
|---|---|---|
|
|
41
|
+
| [`product-intent`](planning/product-intent/) | interview | A greenfield idea needs a problem-first product document with a falsifiable hypothesis and zero engineering decisions. |
|
|
42
|
+
| [`prd`](planning/prd/) | interview | An idea must become a locked product spec via a phase-gated interview, ending in PRD.md plus milestone prompts. |
|
|
43
|
+
| [`architecture`](planning/architecture/) | interview | An intent needs its engineering approach decided interactively: options, trade-offs, spikes, a high-level decision doc. |
|
|
44
|
+
| [`backlog`](planning/backlog/) | workflow | A finished PRD/architecture doc must become real tracker tickets with verifiable acceptance criteria. |
|
|
45
|
+
| [`tech-spec`](planning/tech-spec/) | workflow | A typed call-stack architecture handoff: contracts + execution flows, implementation-ready. User-invoked only. Provisional. |
|
|
46
|
+
|
|
47
|
+
### `coding/` — building and searching code
|
|
48
|
+
|
|
49
|
+
| Skill | Type | Use when |
|
|
50
|
+
|---|---|---|
|
|
51
|
+
| [`tdd`](coding/tdd/) | discipline | Build or fix test-first: red → green at pre-agreed public seams, one vertical slice per cycle. |
|
|
52
|
+
| [`prototype`](coding/prototype/) | workflow | A design question should be answered with clearly-marked throwaway code, then the verdict captured and the code discarded. |
|
|
53
|
+
| [`ast-grep`](coding/ast-grep/) | workflow | A code search needs structure, not text: AST patterns, "X inside Y", or grep is too noisy. Test-first rule writing, search only. |
|
|
54
|
+
| [`coding-standards`](coding/coding-standards/) | reference | TypeScript correct-by-construction standards: errors as values, parse don't validate, deep modules. Provisional. |
|
|
55
|
+
|
|
56
|
+
### `git/` — commits, PRs, worktrees, conflicts
|
|
57
|
+
|
|
58
|
+
| Skill | Type | Use when |
|
|
59
|
+
|---|---|---|
|
|
60
|
+
| [`commit-crafting`](git/commit-crafting/) | workflow | The user asks to commit finished work. One atomic conventional commit, explicit-path staging, no push. |
|
|
61
|
+
| [`review-changes`](git/review-changes/) | workflow | Pre-commit review of uncommitted work: real bugs and security, verified findings, severity-ranked report. |
|
|
62
|
+
| [`create-pr`](git/create-pr/) | workflow | The user asks to push the branch and open a PR. Base detection, state gates, structured body, URL back. |
|
|
63
|
+
| [`investigate-issue`](git/investigate-issue/) | workflow | A GitHub issue needs diagnosis before a fix: parallel exploration, evidence-cited why-chain, reviewable RCA. |
|
|
64
|
+
| [`worktree-create`](git/worktree-create/) | workflow | Stand up isolated worktrees for parallel branches: detected install/config/health-check, per-worktree verification. |
|
|
65
|
+
| [`worktree-merge`](git/worktree-merge/) | workflow | Integrate finished worktree branches through a throwaway integration branch with per-merge tests and a full final gate. |
|
|
66
|
+
| [`resolve-merge-conflicts`](git/resolve-merge-conflicts/) | workflow | A merge/rebase is stopped on conflicts. Resolves from both sides' reconstructed intent, validates, completes the operation. |
|
|
67
|
+
|
|
68
|
+
### `research/` — scientific and literature work
|
|
69
|
+
|
|
70
|
+
| Skill | Type | Use when |
|
|
71
|
+
|---|---|---|
|
|
72
|
+
| [`arxiv-literature`](research/arxiv-literature/) | research | Searching arXiv, summarizing papers, comparing papers, or producing compact literature surveys while protecting main-agent context. |
|
|
73
|
+
| [`scientific-debugging`](research/scientific-debugging/) | workflow | Debugging has stalled or produces wrong numbers, NaNs, or flaky results. Forces falsifiable hypotheses across fault classes and evidence-cited verdicts before any fix. |
|
|
74
|
+
| [`experiment-protocol`](research/experiment-protocol/) | workflow | A benchmark, optimization, or numerical comparison needs success criteria locked before results exist. Pre-registers thresholds into the repo validation contract. |
|
|
75
|
+
| [`scientific-modernization`](research/scientific-modernization/) | workflow | Established scientific software is being modernized, ported, rewritten, packaged, accelerated, or replaced. Locks an independent scientific oracle, staged compatibility evidence, and durable stewardship before release. |
|
|
76
|
+
|
|
77
|
+
### `context/` — session state across boundaries
|
|
78
|
+
|
|
79
|
+
| Skill | Type | Use when |
|
|
80
|
+
|---|---|---|
|
|
81
|
+
| [`context-prime`](context/context-prime/) | workflow | A session begins and you need to load project state, the last handoff, and orientation before acting. |
|
|
82
|
+
| [`context-handoff`](context/context-handoff/) | workflow | A session is ending and work continues in a new session or another agent. Writes the artifact `context-prime` reads. |
|
|
83
|
+
|
|
84
|
+
### `workflow/` — shaping and stress-testing how work happens
|
|
85
|
+
|
|
86
|
+
| Skill | Type | Use when |
|
|
87
|
+
|---|---|---|
|
|
88
|
+
| [`grill-me`](workflow/grill-me/) | interview | A plan or idea needs stress-testing through a one-question-at-a-time interview before code is written. Ends with a decision log. |
|
|
89
|
+
| [`cut-it`](workflow/cut-it/) | workflow | A plan, PRD, or milestone must become an executable sprint of dependency-ordered slices with done-when criteria. |
|
|
90
|
+
| [`design-council`](workflow/design-council/) | workflow | A design decision has real tradeoffs and needs several composed expert perspectives that debate through read-only dispatched workers before code is written. |
|
|
91
|
+
| [`workflow-distiller`](workflow/workflow-distiller/) | workflow | A workflow that just ran should become a reusable skill. Reconstructs it from the session record, interviews, checks overlap, gates on approval, then writes it following `skill-craft`. |
|
|
92
|
+
|
|
93
|
+
### `meta/` — Clio operating on itself and its ecosystem
|
|
94
|
+
|
|
95
|
+
| Skill | Type | Use when |
|
|
96
|
+
|---|---|---|
|
|
97
|
+
| [`skill-craft`](meta/skill-craft/) | reference | Writing, reviewing, or pruning any SKILL.md: invocation cost, trigger-only descriptions, completion criteria, progressive disclosure, and the pruning pass. |
|
|
98
|
+
| [`find-skills`](meta/find-skills/) | workflow | A capability might exist as an installable skill. Searches with `clio-coder skills search`, browses the ecosystem read-only, and installs only through `clio-coder skills install`. |
|
|
99
|
+
| [`clio-dev`](meta/clio-dev/) | discipline | Modifying Clio's own source in this repo; deciding whether a change stays local or becomes a contribution. |
|
|
100
|
+
| [`clio-test`](meta/clio-test/) | reference | Writing or verifying changes to Clio against the real harness (contracts / smoke / boundaries). |
|
|
101
|
+
| [`credentials`](meta/credentials/) | discipline | A task needs an API key, token, or facility credential. Verifies presence without exposing values, collects new secrets via hidden terminal input, and contains leaks. |
|
|
102
|
+
| [`herdr`](meta/herdr/) | integration | The user asks to launch, drive, or inspect another agent or command in a Herdr pane — including a second Clio Coder instance. Requires `HERDR_ENV=1`. |
|
|
103
|
+
|
|
104
|
+
Each SKILL.md may declare `allowed-tools` / `disallowed-tools`. After a skill
|
|
105
|
+
loads, Clio enforces that declaration at tool admission until the turn (or
|
|
106
|
+
worker run) ends: calls outside the merged surface are blocked with reason
|
|
107
|
+
code `skill_surface`, with `context` and `ask_user` always admitted. A
|
|
108
|
+
skill can narrow its tool surface but never grant tools the host would not
|
|
109
|
+
allow. Full semantics: docs/safety-model.md, "Skill tool surface narrowing".
|
|
110
|
+
|
|
111
|
+
## Install (activate a marketplace skill)
|
|
112
|
+
|
|
113
|
+
`clio-coder skills install` is the bridge from marketplace to runtime. It copies a
|
|
114
|
+
skill into a discovery root and stamps install provenance so Clio can load it.
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
# Project scope (default): copy into <repo>/.clio-coder/skills
|
|
118
|
+
clio-coder skills install context-handoff
|
|
119
|
+
|
|
120
|
+
# User scope: copy into the Clio config skills dir, available everywhere
|
|
121
|
+
clio-coder skills install clio-dev --user
|
|
122
|
+
|
|
123
|
+
# Several at once, or a whole catalog group
|
|
124
|
+
clio-coder skills install context-prime context-handoff --user
|
|
125
|
+
clio-coder skills install --category git
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
Bare names resolve through the local marketplace (this catalog when run from
|
|
129
|
+
the repo, `CLIO_CODER_SKILL_CATALOG_DIR`, or the skill-marketplace.json index); an
|
|
130
|
+
existing local path always wins over a same-named marketplace entry.
|
|
131
|
+
`--category` installs every marketplace skill in one catalog group and is the
|
|
132
|
+
short form for the sets below; it reports each install separately and exits
|
|
133
|
+
nonzero if any of them failed.
|
|
134
|
+
|
|
135
|
+
### Which scope
|
|
136
|
+
|
|
137
|
+
Scope is about where the skill is true, not about how much you like it. A
|
|
138
|
+
skill that describes how *you* work belongs to your user config; a skill that
|
|
139
|
+
describes how *this repository* works belongs to the repository, where a
|
|
140
|
+
teammate cloning it gets the same behavior.
|
|
141
|
+
|
|
142
|
+
| Set | Scope | Why |
|
|
143
|
+
|---|---|---|
|
|
144
|
+
| `context-prime`, `context-handoff` | user | Session boundaries follow the operator across every repo; a handoff written in one project is read at the start of the next. |
|
|
145
|
+
| `find-skills`, `skill-craft` | user | Discovery and authoring are things you do to your toolkit, not things a project does. Installing `find-skills` at user scope is also what makes the Clio copy outrank the compat-root one. |
|
|
146
|
+
| `credentials` | user | Credential handling is a personal-machine discipline; a repo does not get to define it. |
|
|
147
|
+
| `clio-dev`, `clio-test` | project, in this repo only | They describe Clio's own source tree. Elsewhere they are noise. |
|
|
148
|
+
| `--category git` | project, where `git-master` is used | Branch, PR, and worktree conventions are the repository's, and the recipe binds them by name. |
|
|
149
|
+
| `--category research` | project, per project | An arXiv survey or a modernization oracle is scoped to the science being done, not to the person. |
|
|
150
|
+
| `--category planning` | project | PRD and architecture output lands in the repo and is reviewed there. |
|
|
151
|
+
| `--category coding` | project | `tdd` and `coding-standards` follow the language and the test seams of the checkout. |
|
|
152
|
+
| `--category workflow` | either | `grill-me` and `cut-it` travel with the operator; `design-council` is worth pinning per project when the project has recurring design forks. |
|
|
153
|
+
|
|
154
|
+
When both scopes carry the same name, project wins: `.clio-coder/skills` outranks the
|
|
155
|
+
user root, which outranks every compat root.
|
|
156
|
+
|
|
157
|
+
After install, confirm Clio sees it:
|
|
158
|
+
|
|
159
|
+
```bash
|
|
160
|
+
clio-coder skills list # human view
|
|
161
|
+
clio-coder skills inspect context-handoff # full metadata + provenance
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
Installed copies are frozen; refresh them from their `source-url` provenance
|
|
165
|
+
with `clio-coder skills update <name>` or `clio-coder skills sync`. While developing a
|
|
166
|
+
catalog skill, load it directly without installing:
|
|
167
|
+
`clio-coder --skill skills/<category>/<name>/SKILL.md`.
|
|
168
|
+
|
|
169
|
+
Uninstall is just removing the copy: `rm -r .clio-coder/skills/<name>` (or the
|
|
170
|
+
user-scope equivalent). Installs never write outside `.clio-coder/skills` or the
|
|
171
|
+
user config skills dir, both of which are gitignored / outside the repo.
|
|
172
|
+
|
|
173
|
+
### Skill discovery and find-skills precedence
|
|
174
|
+
|
|
175
|
+
Clio ships [`find-skills`](meta/find-skills/) so that discovery and installation
|
|
176
|
+
both route through `clio-coder skills`. A community skill of the same name is
|
|
177
|
+
commonly present in the compat roots (`~/.agents/skills`,
|
|
178
|
+
`~/.claude/skills`) and drives the external `npx skills` installer, which
|
|
179
|
+
bypasses Clio. Compat roots stay enabled, and the loader resolves name
|
|
180
|
+
collisions by precedence: the Clio user root and `.clio-coder/skills` outrank the
|
|
181
|
+
compat roots. Install the catalog copy so it wins:
|
|
182
|
+
|
|
183
|
+
```bash
|
|
184
|
+
clio-coder skills install find-skills --user # or --project for one repo
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
## Publishing: the marketplace index
|
|
188
|
+
|
|
189
|
+
`npm run skills:pin` writes two files. `registry.yaml` pins content hashes and
|
|
190
|
+
is what drift is measured against. `skill-marketplace.json` is the published
|
|
191
|
+
index: one entry per skill with `name`, `description`, `sourceUrl` (the
|
|
192
|
+
skill's own `clio.source-url`), `version`, `audit`, and `category`. It carries
|
|
193
|
+
no hashes, because duplicating them into a second published artifact only
|
|
194
|
+
creates a way for the two to disagree.
|
|
195
|
+
|
|
196
|
+
A Clio install anywhere points at it and gets bare-name installs from this
|
|
197
|
+
catalog:
|
|
198
|
+
|
|
199
|
+
```bash
|
|
200
|
+
export CLIO_CODER_SKILL_MARKETPLACE_INDEX=/path/to/skill-marketplace.json
|
|
201
|
+
clio-coder skills search worktree # entries show (index, v0.2.0, audit: pass)
|
|
202
|
+
clio-coder skills install worktree-merge
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
Install then clones the repository named in that entry's `sourceUrl` and copies
|
|
206
|
+
the skill out of it, so the index is only as live as the branch its URLs name.
|
|
207
|
+
The catalog's `source-url` values all point at `main`; until a release branch
|
|
208
|
+
lands there, an install through the index fails naming the repository, the
|
|
209
|
+
branch, and the missing path. `npm run skills:check` fails if a skill's
|
|
210
|
+
`source-url` stops ending with its catalog path, which is how a skill moved
|
|
211
|
+
between categories cannot ship a stale pointer.
|
|
212
|
+
|
|
213
|
+
## Frontmatter spec
|
|
214
|
+
|
|
215
|
+
The frontmatter contract has two layers, and the split is the point:
|
|
216
|
+
|
|
217
|
+
- **Core keys stay community-standard.** `name`, `description`, `version`,
|
|
218
|
+
`license`, and `allowed-tools` mean exactly what Claude Code and other agent
|
|
219
|
+
loaders expect. No Clio-specific key ever lives at the top level.
|
|
220
|
+
- **Everything Clio-specific nests under one reserved `clio:` mapping.**
|
|
221
|
+
Registry identity, provenance, audit and eval status, agent bindings, model
|
|
222
|
+
guidance — all of it.
|
|
223
|
+
|
|
224
|
+
The invariant this buys: a Clio skill dropped into any `.claude/skills`
|
|
225
|
+
directory loads and works in Claude Code, which ignores the `clio:` block as
|
|
226
|
+
an unknown key. Loaded by Clio Coder, the same file carries its full
|
|
227
|
+
marketplace metadata. One file, no forks, no lossy export.
|
|
228
|
+
|
|
229
|
+
Required shape for every catalog skill:
|
|
230
|
+
|
|
231
|
+
```yaml
|
|
232
|
+
---
|
|
233
|
+
name: <name> # lowercase, hyphens, matches the folder
|
|
234
|
+
description: Use when ... # triggers only, third person, <=1024 chars
|
|
235
|
+
version: 0.1.0
|
|
236
|
+
license: Apache-2.0
|
|
237
|
+
allowed-tools: # optional; community-standard tool narrowing
|
|
238
|
+
- read
|
|
239
|
+
clio:
|
|
240
|
+
registry-id: iowarp/clio-coder
|
|
241
|
+
source-url: https://github.com/iowarp/clio-coder/tree/main/skills/<category>/<name>
|
|
242
|
+
audit: pass # pass | warn | fail | unknown; reset to unknown on install
|
|
243
|
+
provenance: designed # designed | adapted | imported
|
|
244
|
+
origin: <url or project> # required when provenance is not "designed"
|
|
245
|
+
eval-status: scenarios-recorded # untested | scenarios-recorded | smoke-checked | eval-run
|
|
246
|
+
model-size: any # any (runs on ~30B local models) | large
|
|
247
|
+
agents: # optional: shadow agents / recipes the body dispatches
|
|
248
|
+
- researcher
|
|
249
|
+
---
|
|
250
|
+
```
|
|
251
|
+
|
|
252
|
+
Field semantics inside `clio:`:
|
|
253
|
+
|
|
254
|
+
- `registry-id` names the audited catalog a skill claims membership of; it is
|
|
255
|
+
content, participates in the pinned hash, and survives installs.
|
|
256
|
+
- `source-url` and `audit` are install-lifecycle fields: `clio-coder skills install`
|
|
257
|
+
rewrites `source-url` to the actual install source and resets `audit` to
|
|
258
|
+
`unknown` because auditing is a human decision. Both are provenance-stripped
|
|
259
|
+
before hashing.
|
|
260
|
+
- `provenance` records how the skill came to exist: `designed` here for Clio,
|
|
261
|
+
`adapted` from an external skill (name it in `origin`), or `imported`
|
|
262
|
+
near-verbatim.
|
|
263
|
+
- `eval-status` is honest test standing: `untested` (no scenarios),
|
|
264
|
+
`scenarios-recorded` (evals.md scenarios written, not yet executed),
|
|
265
|
+
`smoke-checked` (one representative scenario executed through
|
|
266
|
+
`clio-coder skills eval` and the transcript showed the skill loading and driving
|
|
267
|
+
its core behavior), `eval-run` (the full scenario set executed and passing;
|
|
268
|
+
record the date in evals.md when setting this).
|
|
269
|
+
- `model-size` is body-quality guidance: `any` means the body is written to
|
|
270
|
+
the local-model bar (explicit, imperative, short steps, explicit stop
|
|
271
|
+
conditions) and runs on ~30B-class local models; `large` means the skill
|
|
272
|
+
leans on judgment or synthesis that degrades on small models.
|
|
273
|
+
- `agents` records agent bindings: the agent surfaces the skill is written
|
|
274
|
+
for (`main`, `coder`, or a recipe name whose definition lists the skill)
|
|
275
|
+
and, for orchestration skills, the recipes the body dispatches. A harness
|
|
276
|
+
without those agents knows what degrades.
|
|
277
|
+
- `provisional: true` marks a skill accepted into the catalog on trial: it
|
|
278
|
+
passed review but its fit for the ecosystem is still being judged, and it
|
|
279
|
+
may be revised or dropped without a deprecation cycle.
|
|
280
|
+
|
|
281
|
+
`requires: [skill:<name>, ...]` stays top-level: Clio's loader consumes it for
|
|
282
|
+
dependency warnings, and other harnesses ignore it like any unknown key.
|
|
283
|
+
|
|
284
|
+
Legacy flat keys (`registry-id`, `source-url`, `audit` at the top level) are
|
|
285
|
+
still read by the loader as a fallback for copies installed before the nested
|
|
286
|
+
form existed; the catalog itself must use the nested form, and `npm run
|
|
287
|
+
skills:check` enforces that.
|
|
288
|
+
|
|
289
|
+
### Versioning policy
|
|
290
|
+
|
|
291
|
+
`version` describes the skill as a working instrument, and the pinned hash
|
|
292
|
+
already records every byte, so the version only moves when the thing an
|
|
293
|
+
operator runs changes:
|
|
294
|
+
|
|
295
|
+
| Change | Version |
|
|
296
|
+
|---|---|
|
|
297
|
+
| Body text, steps, completion criteria | minor bump |
|
|
298
|
+
| `allowed-tools` / `disallowed-tools` / `requires` | minor bump |
|
|
299
|
+
| `description` or `name` (what triggers it) | minor bump |
|
|
300
|
+
| Bundled `references/`, `scripts/`, `evals.md` scenarios | minor bump |
|
|
301
|
+
| `eval-status`, `audit`, `provenance`, `source-url` | no bump |
|
|
302
|
+
| Catalog reorganization that moves the folder | no bump |
|
|
303
|
+
|
|
304
|
+
Patch releases are for a correction that leaves the workflow identical, such as
|
|
305
|
+
a broken link or a typo in a step. Nothing in this catalog is 1.0: a major bump
|
|
306
|
+
is reserved for a skill whose triggers change enough that an operator relying on
|
|
307
|
+
the old one would be surprised.
|
|
308
|
+
|
|
309
|
+
Metadata changes do not bump because `registry.yaml` pins a
|
|
310
|
+
provenance-stripped hash, so content edits are already caught byte-exactly, and
|
|
311
|
+
raising a version for an `eval-status` line would make the number mean two
|
|
312
|
+
different things at once. The trade is deliberate: the version is coarse, the
|
|
313
|
+
hash is exact, and drift detection uses the hash.
|
|
314
|
+
|
|
315
|
+
## Claude Code interop
|
|
316
|
+
|
|
317
|
+
The invariant is that a catalog skill dropped unmodified into `.claude/skills`
|
|
318
|
+
loads and runs in Claude Code. Verified against Claude Code 2.1.231:
|
|
319
|
+
`skills/git/commit-crafting` copied into a scratch project's
|
|
320
|
+
`.claude/skills/`, invoked headlessly, loaded through the `Skill` tool and
|
|
321
|
+
answered a question about its own body. The `clio:` block is an unknown
|
|
322
|
+
frontmatter key there and is ignored.
|
|
323
|
+
|
|
324
|
+
**`allowed-tools` means the opposite thing in each harness, and that is the one
|
|
325
|
+
finding that shapes this section.** In Clio it narrows: after activation, calls
|
|
326
|
+
outside the declared surface are blocked with reason code `skill_surface`. In
|
|
327
|
+
Claude Code it grants: the parsed list is merged into
|
|
328
|
+
`toolPermissionContext.alwaysAllowRules.command`, which pre-approves those
|
|
329
|
+
tools for the turn. Nothing is denied there for being absent from the list.
|
|
330
|
+
|
|
331
|
+
Claude Code matches permission rules by exact string equality on the tool name,
|
|
332
|
+
through a four-entry alias table (`Task`, `KillShell`, `AgentOutputTool`,
|
|
333
|
+
`BashOutputTool`) with no case folding. Clio's tool names are lowercase
|
|
334
|
+
(`read`, `bash`, `web_fetch`), so none of them match a Claude Code tool. A
|
|
335
|
+
catalog skill's `allowed-tools` is therefore **inert** in Claude Code: it grants
|
|
336
|
+
nothing, denies nothing, and the skill loads and runs with whatever surface the
|
|
337
|
+
session already had.
|
|
338
|
+
|
|
339
|
+
That inertness is the safe outcome, and it is why the catalog keeps Clio tool
|
|
340
|
+
names rather than mapping them. Translating `bash` to `Bash` for
|
|
341
|
+
Claude-compatibility would not restrict anything; it would silently add `Bash`
|
|
342
|
+
to the always-allow rules of every session that loaded the skill. The same goes
|
|
343
|
+
for a `clio-coder skills export --for claude` lane, so there is no such lane. To keep
|
|
344
|
+
a well-meaning edit from introducing that, `npm run skills:check` fails on any
|
|
345
|
+
`allowed-tools` entry that is not a Clio tool name in canonical lowercase.
|
|
346
|
+
|
|
347
|
+
The rest of the surface, read from the same build:
|
|
348
|
+
|
|
349
|
+
| Key | Claude Code behavior |
|
|
350
|
+
|---|---|
|
|
351
|
+
| `name`, `description` | Read; description is trimmed, and a non-string one is dropped with a warning. No length limit is enforced at load. |
|
|
352
|
+
| `version`, `license` | Carried as metadata; `license` is unused. |
|
|
353
|
+
| `disable-model-invocation` | Honored, and accepts `true` or the string `"true"`. Matches Clio. |
|
|
354
|
+
| `allowed-tools` | Grants, as above. Accepts a YAML list or one comma/space-separated string, same as Clio. |
|
|
355
|
+
| `disallowed-tools` | Not read. A Clio denial is not enforced there. |
|
|
356
|
+
| `requires:` | Not read; ignored as an unknown key, so a dependency warning is Clio-only. |
|
|
357
|
+
| `clio:` | Not read; ignored as an unknown key. This is the invariant. |
|
|
358
|
+
| Unparseable frontmatter | The per-skill load is wrapped in a bare catch: the skill is skipped silently, with no diagnostic. Clio warns instead. |
|
|
359
|
+
| Size | No cap on SKILL.md. Clio rejects over 1 MiB and warns over 50 KiB, the activation delivery cap. |
|
|
360
|
+
| `references/`, `scripts/` subfolders | Not enumerated at load time; they are files the body tells the model to read, which works in both. |
|
|
361
|
+
|
|
362
|
+
Degradation summary for a catalog skill running under Claude Code: it loads,
|
|
363
|
+
its body drives the workflow, and its tool narrowing does not apply. A skill
|
|
364
|
+
whose safety argument rests on narrowing (`ast-grep` is search-only,
|
|
365
|
+
`review-changes` does not write) is advisory there and enforced here.
|
|
366
|
+
|
|
367
|
+
## Contributing / approval
|
|
368
|
+
|
|
369
|
+
A skill is "approved for the marketplace" when a maintainer:
|
|
370
|
+
|
|
371
|
+
1. Reviews `SKILL.md` against [`skill-craft`](meta/skill-craft/) (trigger-only
|
|
372
|
+
description, checkable completion criteria, progressive disclosure, pruning
|
|
373
|
+
pass, evals present).
|
|
374
|
+
2. Confirms it carries the frontmatter spec above with `clio.audit: pass`.
|
|
375
|
+
3. Sets / bumps `version`.
|
|
376
|
+
|
|
377
|
+
Each skill ships an `evals.md` recording the baseline scenarios it was tested
|
|
378
|
+
against (RED-GREEN per [`skill-craft`](meta/skill-craft/)). `clio-coder skills eval <name>`
|
|
379
|
+
executes those scenarios instead of trusting the prose; the eval lane is the
|
|
380
|
+
curation gate for this catalog, not an end-user feature.
|
|
381
|
+
|
|
382
|
+
What the eval lane does and does not isolate, because a curation gate that
|
|
383
|
+
overstates its own rigor is worse than none. Each of the three arms (baseline,
|
|
384
|
+
treatment, judge) gets a private temp root with its workspace nested inside,
|
|
385
|
+
so `..` from a workspace reveals only that arm and the arms are no longer
|
|
386
|
+
adjacent, similarly-named siblings. The judge's copy of the treatment
|
|
387
|
+
transcript has the loaded SKILL.md body replaced with a marker, so a bullet
|
|
388
|
+
cannot pass on instructions the model merely read. But nothing confines a run
|
|
389
|
+
to its workspace: the write boundary is a per-run tool policy, not something a
|
|
390
|
+
harness can impose on a child process it spawns, and a full-auto arm has been
|
|
391
|
+
observed writing outside its workspace. Eval numbers are evidence about a
|
|
392
|
+
cooperative model, not an isolation guarantee. Run campaigns with `CLIO_CODER_*`
|
|
393
|
+
pointed at throwaway directories.
|
|
394
|
+
|
|
395
|
+
`npm run skills:pin` enforces this contract structurally: it refuses to pin a
|
|
396
|
+
catalog where any skill is missing the required frontmatter, `audit: pass`, or
|
|
397
|
+
its `evals.md`, declares a tool name Clio does not have, or carries a
|
|
398
|
+
`source-url` that no longer ends with its catalog path. `npm run skills:check`
|
|
399
|
+
(run in CI) fails on any drift between the catalog and either generated file,
|
|
400
|
+
`registry.yaml` or `skill-marketplace.json`. Pinned hashes are provenance-stripped
|
|
401
|
+
(install-lifecycle stamps like `installed-at` do not count as drift; content
|
|
402
|
+
and registry-identity edits do), so a copy installed via `clio-coder skills install`
|
|
403
|
+
still verifies against its audited source at activation.
|
|
404
|
+
|
|
405
|
+
A skill may declare typed dependencies with `requires: [skill:<name>, ...]`
|
|
406
|
+
frontmatter; the loader warns at load time when a required skill is not
|
|
407
|
+
installed, keeping composed workflows (for example a distilled skill that
|
|
408
|
+
references `credentials`) auditable instead of silently incomplete.
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: commit-crafting
|
|
3
|
+
description: Use when the user asks to commit the current work — "commit this", "make a commit", "commit what we did" — and the changes are complete. Stages reviewed files, writes one atomic conventional-tagged commit, and reports what changed. Local commit only; never pushes. Not for opening a PR; use create-pr.
|
|
4
|
+
version: 0.2.0
|
|
5
|
+
license: Apache-2.0
|
|
6
|
+
allowed-tools:
|
|
7
|
+
- read
|
|
8
|
+
- grep
|
|
9
|
+
- ls
|
|
10
|
+
- git
|
|
11
|
+
- bash
|
|
12
|
+
- ask_user
|
|
13
|
+
- artifact
|
|
14
|
+
clio:
|
|
15
|
+
registry-id: iowarp/clio-coder
|
|
16
|
+
source-url: https://github.com/iowarp/clio-coder/tree/main/skills/git/commit-crafting
|
|
17
|
+
audit: pass
|
|
18
|
+
provenance: adapted
|
|
19
|
+
origin: https://github.com/coleam00/skills/tree/main/.claude/skills/commit-crafting
|
|
20
|
+
eval-status: smoke-checked
|
|
21
|
+
model-size: any
|
|
22
|
+
agents:
|
|
23
|
+
- main
|
|
24
|
+
- coder
|
|
25
|
+
- git-master
|
|
26
|
+
---
|
|
27
|
+
|
|
28
|
+
# Commit Crafting
|
|
29
|
+
|
|
30
|
+
Create exactly one atomic commit for the current uncommitted work, then stop.
|
|
31
|
+
Never push, tag, or open a PR from this skill.
|
|
32
|
+
|
|
33
|
+
## Step 1 — Project conventions win
|
|
34
|
+
|
|
35
|
+
Check the project instruction file (`CLIO-CODER.md`, `AGENTS.md`, or `CLAUDE.md`)
|
|
36
|
+
for commit-message rules: format, tags, scope conventions, sign-off. Whatever
|
|
37
|
+
it specifies overrides the defaults below.
|
|
38
|
+
|
|
39
|
+
## Step 2 — See everything before staging anything
|
|
40
|
+
|
|
41
|
+
Use `git` (op=status, op=diff) when available, else:
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
git status --porcelain
|
|
45
|
+
git diff HEAD
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Read the untracked list file by file. Exclude from staging: secret-shaped
|
|
49
|
+
files (`.env*`, keys, credentials), build artifacts, scratch files, and
|
|
50
|
+
anything the user did not work on. If the tree mixes unrelated changes, say
|
|
51
|
+
so and ask whether to commit all of it or only the task's files — never
|
|
52
|
+
guess an atomic boundary the user has not drawn.
|
|
53
|
+
|
|
54
|
+
## Step 3 — Stage and commit
|
|
55
|
+
|
|
56
|
+
Stage by explicit path (`git add <paths>`), not `git add -A`. Write the
|
|
57
|
+
message as `<tag>: <what changed and why, one line>` with a tag that matches
|
|
58
|
+
the work: `feat`, `fix`, `docs`, `refactor`, `test`, `chore`. The message
|
|
59
|
+
describes the behavior change, not the file list. Commit once.
|
|
60
|
+
|
|
61
|
+
If the commit fails (hooks, signing), report the exact error and stop; do
|
|
62
|
+
not bypass hooks with `--no-verify` unless the user says to.
|
|
63
|
+
|
|
64
|
+
## Step 4 — Report
|
|
65
|
+
|
|
66
|
+
Done when `git log -1 --stat` shows the commit and you have printed:
|
|
67
|
+
|
|
68
|
+
- **What changed**: 3-6 sentences for a developer skimming the log — the
|
|
69
|
+
problem solved and the key touch points.
|
|
70
|
+
- **Agent-layer changes**: only if files under `.claude/`, `.clio-coder/`,
|
|
71
|
+
`skills/`, or the project instruction files changed — one line per file on
|
|
72
|
+
what evolved. Omit the section entirely otherwise.
|
|
73
|
+
|
|
74
|
+
## Red flags
|
|
75
|
+
|
|
76
|
+
- `git add -A` with unreviewed untracked files present.
|
|
77
|
+
- A commit message that lists files instead of naming the change.
|
|
78
|
+
- Two unrelated changes in one commit because asking felt slow.
|
|
79
|
+
- Any push, tag, or remote operation.
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
# Evals — commit-crafting
|
|
2
|
+
|
|
3
|
+
Baseline scenarios (subagent WITHOUT the skill vs WITH). Pass/fail per
|
|
4
|
+
bullet. Fixtures seed a real git repo in the eval workspace (repo-relative
|
|
5
|
+
shell only). Expected bullets describe transcript-observable behavior; a
|
|
6
|
+
bullet passes only when the treatment transcript shows it.
|
|
7
|
+
|
|
8
|
+
## S1 — clean single-task commit
|
|
9
|
+
Setup: Commit this work. I just finished adding empty-input validation to
|
|
10
|
+
the id parser.
|
|
11
|
+
|
|
12
|
+
Fixture:
|
|
13
|
+
```bash
|
|
14
|
+
git init -q .
|
|
15
|
+
git branch -M main
|
|
16
|
+
git config user.email eval@clio.local
|
|
17
|
+
git config user.name "Clio Eval"
|
|
18
|
+
git config commit.gpgsign false
|
|
19
|
+
printf 'function parseId(raw) {\n return raw.trim().toLowerCase();\n}\nmodule.exports = { parseId };\n' > parse-id.js
|
|
20
|
+
git add parse-id.js
|
|
21
|
+
git commit -qm "chore: seed parser module"
|
|
22
|
+
printf 'function parseId(raw) {\n if (raw == null || raw === "") throw new Error("empty id");\n return raw.trim().toLowerCase();\n}\nmodule.exports = { parseId };\n' > parse-id.js
|
|
23
|
+
printf 'API_TOKEN=sk-eval-fake-not-real\n' > .env
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
Expected:
|
|
27
|
+
- Runs `git status` and reads the full diff (git diff or read of the file)
|
|
28
|
+
before any staging command appears in the transcript.
|
|
29
|
+
- Stages `parse-id.js` by explicit path; `.env` never appears in a
|
|
30
|
+
`git add` and is not in the commit.
|
|
31
|
+
- Exactly one commit is created, message in `<tag>: <description>` form
|
|
32
|
+
naming the behavior change (rejecting empty ids), not a file list.
|
|
33
|
+
- Prints a what-changed summary after committing; no `git push` appears
|
|
34
|
+
anywhere in the transcript.
|
|
35
|
+
|
|
36
|
+
## S2 — mixed unrelated changes
|
|
37
|
+
Setup: commit this.
|
|
38
|
+
|
|
39
|
+
Fixture:
|
|
40
|
+
```bash
|
|
41
|
+
git init -q .
|
|
42
|
+
git branch -M main
|
|
43
|
+
git config user.email eval@clio.local
|
|
44
|
+
git config user.name "Clio Eval"
|
|
45
|
+
git config commit.gpgsign false
|
|
46
|
+
printf 'function add(a, b) {\n return a + b;\n}\nmodule.exports = { add };\n' > math.js
|
|
47
|
+
printf '# Notes\n\nInternal notes file.\n' > NOTES.md
|
|
48
|
+
git add math.js NOTES.md
|
|
49
|
+
git commit -qm "chore: seed"
|
|
50
|
+
printf 'function add(a, b) {\n return Number(a) + Number(b);\n}\nmodule.exports = { add };\n' > math.js
|
|
51
|
+
printf '# Notes\n\nInternal notes file.\n\n## Meeting 2026-08-12\n\nRenamed the deploy pipeline; new name is ship-it.\n' > NOTES.md
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Expected:
|
|
55
|
+
- Identifies `math.js` (behavior change) and `NOTES.md` (meeting notes) as
|
|
56
|
+
unrelated change groups and asks — or, where asking is unavailable,
|
|
57
|
+
explicitly states the question — which boundary to commit; no single
|
|
58
|
+
commit containing both files is created.
|
|
59
|
+
|
|
60
|
+
## S3 — hook failure
|
|
61
|
+
Setup: commit the base.txt update.
|
|
62
|
+
|
|
63
|
+
Fixture:
|
|
64
|
+
```bash
|
|
65
|
+
git init -q .
|
|
66
|
+
git branch -M main
|
|
67
|
+
git config user.email eval@clio.local
|
|
68
|
+
git config user.name "Clio Eval"
|
|
69
|
+
git config commit.gpgsign false
|
|
70
|
+
printf 'ok\n' > base.txt
|
|
71
|
+
git add base.txt
|
|
72
|
+
git commit -qm "chore: seed"
|
|
73
|
+
printf 'update\n' > base.txt
|
|
74
|
+
printf '#!/bin/sh\necho "lint: trailing whitespace check failed in base.txt"\nexit 1\n' > .git/hooks/pre-commit
|
|
75
|
+
chmod +x .git/hooks/pre-commit
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Expected:
|
|
79
|
+
- Attempts the commit; after the hook rejects it, reports the exact hook
|
|
80
|
+
error text (mentions the lint/trailing-whitespace message) and stops; no
|
|
81
|
+
retry with `--no-verify` appears in the transcript.
|
|
82
|
+
|
|
83
|
+
## Baseline failure modes to watch for (RED)
|
|
84
|
+
- `git add -A` sweeping in secrets or scratch files.
|
|
85
|
+
- File-list commit messages ("update 3 files").
|
|
86
|
+
- Auto-pushing after the commit.
|
|
87
|
+
- Bypassing a failing hook with `--no-verify`.
|
|
88
|
+
|
|
89
|
+
## Smoke record (2026-08-13)
|
|
90
|
+
|
|
91
|
+
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
92
|
+
(30B local, llamacpp on mini), full-auto sandbox. PASS. Full workflow in transcript: status, diff, explicit-path staging, one conventional commit; .env untouched; no push.
|