aiblueprint-cli 1.4.104 → 1.4.105

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/README.md +0 -7
  2. package/package.json +1 -1
  3. package/agents-config/skills/audit/SKILL.md +0 -126
  4. package/agents-config/skills/audit/agents/openai.yaml +0 -10
  5. package/agents-config/skills/audit/assets/codex-icon.svg +0 -20
  6. package/agents-config/skills/commit/SKILL.md +0 -44
  7. package/agents-config/skills/commit/agents/openai.yaml +0 -10
  8. package/agents-config/skills/commit/assets/codex-icon.svg +0 -17
  9. package/agents-config/skills/create-pr/SKILL.md +0 -55
  10. package/agents-config/skills/create-pr/agents/openai.yaml +0 -10
  11. package/agents-config/skills/create-pr/assets/codex-icon.svg +0 -17
  12. package/agents-config/skills/oneshot/SKILL.md +0 -44
  13. package/agents-config/skills/oneshot/agents/openai.yaml +0 -10
  14. package/agents-config/skills/oneshot/assets/codex-icon.svg +0 -18
  15. package/agents-config/skills/tools/SKILL.md +0 -149
  16. package/agents-config/skills/use-artifacts/SKILL.md +0 -211
  17. package/agents-config/skills/use-artifacts/agents/openai.yaml +0 -7
  18. package/agents-config/skills/use-artifacts/assets/codex-icon.svg +0 -18
  19. package/agents-config/skills/use-artifacts/assets/local-runtime.js +0 -299
  20. package/agents-config/skills/use-artifacts/scripts/create_artifact.py +0 -317
  21. package/agents-config/skills/use-delegate/SKILL.md +0 -97
  22. package/agents-config/skills/use-delegate/agents/openai.yaml +0 -10
  23. package/agents-config/skills/use-delegate/assets/codex-icon.svg +0 -20
  24. package/agents-config/skills/use-delegate/references/models.md +0 -32
  25. package/agents-config/skills/use-goal/SKILL.md +0 -121
  26. package/agents-config/skills/use-goal/agents/openai.yaml +0 -7
  27. package/agents-config/skills/use-goal/assets/codex-icon.svg +0 -18
  28. package/agents-config/skills/use-goal/references/claude-code-goal.md +0 -65
  29. package/agents-config/skills/use-goal/references/codex-goal.md +0 -70
  30. package/agents-config/skills/use-goal/references/verification-harnesses.md +0 -108
@@ -1,317 +0,0 @@
1
- #!/usr/bin/env python3
2
- """Scaffold a Claude-style local HTML artifact workspace."""
3
-
4
- from __future__ import annotations
5
-
6
- import argparse
7
- import json
8
- import re
9
- from datetime import datetime, timezone
10
- from pathlib import Path
11
-
12
- GLOBAL_ARTIFACTS_DIR = Path.home() / ".agents" / "artifacts"
13
-
14
-
15
- def slugify(value: str) -> str:
16
- slug = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-")
17
- return slug[:48].strip("-") or "artifact"
18
-
19
-
20
- def unique_dir(base: Path, slug: str) -> Path:
21
- candidate = base / slug
22
- if not candidate.exists():
23
- return candidate
24
-
25
- stamp = datetime.now(timezone.utc).strftime("%Y%m%d-%H%M%S")
26
- return base / f"{slug}-{stamp}"
27
-
28
-
29
- def html_template(title: str, style: str, kind: str) -> str:
30
- escaped_title = (
31
- title.replace("&", "&")
32
- .replace("<", "&lt;")
33
- .replace(">", "&gt;")
34
- .replace('"', "&quot;")
35
- )
36
- return f"""<!doctype html>
37
- <html lang="en">
38
- <head>
39
- <meta charset="utf-8">
40
- <meta name="viewport" content="width=device-width, initial-scale=1">
41
- <title>{escaped_title}</title>
42
- <style>
43
- :root {{
44
- color-scheme: dark;
45
- --bg: #000000;
46
- --panel: #111111;
47
- --ink: #ffffff;
48
- --muted: #888888;
49
- --border: #333333;
50
- --accent: #0070f3;
51
- }}
52
- * {{ box-sizing: border-box; }}
53
- body {{
54
- margin: 0;
55
- min-height: 100vh;
56
- background: var(--bg);
57
- color: var(--ink);
58
- font-family: Geist, Inter, ui-sans-serif, system-ui, -apple-system, BlinkMacSystemFont, "Segoe UI", sans-serif;
59
- }}
60
- main {{
61
- width: min(1120px, calc(100vw - 32px));
62
- margin: 0 auto;
63
- padding: 48px 0;
64
- }}
65
- .shell {{
66
- display: grid;
67
- gap: 24px;
68
- }}
69
- .header {{
70
- display: flex;
71
- flex-wrap: wrap;
72
- align-items: end;
73
- justify-content: space-between;
74
- gap: 16px;
75
- border-bottom: 1px solid var(--border);
76
- padding-bottom: 20px;
77
- }}
78
- .eyebrow {{
79
- margin: 0 0 8px;
80
- color: var(--accent);
81
- font-size: 12px;
82
- font-weight: 700;
83
- letter-spacing: .08em;
84
- text-transform: uppercase;
85
- }}
86
- h1 {{
87
- margin: 0;
88
- max-width: 760px;
89
- font-size: clamp(40px, 7vw, 88px);
90
- line-height: .95;
91
- letter-spacing: -.03em;
92
- }}
93
- .meta {{
94
- color: var(--muted);
95
- font-size: 14px;
96
- line-height: 1.5;
97
- }}
98
- .panel {{
99
- min-height: 420px;
100
- border: 1px solid var(--border);
101
- border-radius: 8px;
102
- background: var(--panel);
103
- padding: 28px;
104
- }}
105
- .panel h2 {{
106
- margin: 0 0 12px;
107
- font-size: 24px;
108
- letter-spacing: 0;
109
- }}
110
- .panel p {{
111
- max-width: 680px;
112
- color: var(--muted);
113
- font-size: 17px;
114
- line-height: 1.65;
115
- }}
116
- button {{
117
- border: 0;
118
- border-radius: 6px;
119
- background: var(--ink);
120
- color: var(--bg);
121
- cursor: pointer;
122
- font: inherit;
123
- font-weight: 650;
124
- padding: 10px 16px;
125
- }}
126
- .callout {{
127
- border: 1px solid var(--border);
128
- border-radius: 0;
129
- background: var(--panel);
130
- padding: 20px;
131
- }}
132
- .callout h2 {{ color: var(--ink); }}
133
- .grid {{
134
- display: grid;
135
- gap: 16px;
136
- grid-template-columns: repeat(auto-fit, minmax(220px, 1fr));
137
- }}
138
- .croquis {{
139
- min-height: 180px;
140
- border: 1px solid var(--border);
141
- border-radius: 0;
142
- background: var(--bg);
143
- padding: 14px;
144
- }}
145
- .bar {{
146
- height: 14px;
147
- border-radius: 0;
148
- background: #666666;
149
- margin-bottom: 10px;
150
- }}
151
- .box {{
152
- height: 52px;
153
- border: 1px solid var(--border);
154
- border-radius: 0;
155
- background: var(--panel);
156
- margin-bottom: 10px;
157
- }}
158
- .note {{
159
- color: var(--muted);
160
- font-size: 14px;
161
- line-height: 1.45;
162
- }}
163
- pre {{
164
- overflow-x: auto;
165
- border: 1px solid var(--border);
166
- border-radius: 0;
167
- background: var(--bg);
168
- padding: 16px;
169
- }}
170
- code {{
171
- border-radius: 0;
172
- background: var(--panel);
173
- padding: 2px 5px;
174
- font-family: ui-monospace, SFMono-Regular, Menlo, Consolas, monospace;
175
- font-size: .92em;
176
- }}
177
- </style>
178
- </head>
179
- <body>
180
- <main>
181
- <section class="shell" aria-label="{escaped_title}">
182
- <div class="header">
183
- <div>
184
- <p class="eyebrow">{kind} artifact</p>
185
- <h1>{escaped_title}</h1>
186
- </div>
187
- <p class="meta">Style: {style}<br>Entrypoint: index.html</p>
188
- </div>
189
- <div class="callout">
190
- <h2>Key finding or recommendation</h2>
191
- <p>
192
- Put the highest-signal conclusion near the top. Use this area for the
193
- main security finding, product decision, implementation warning, or
194
- core recommendation that frames the rest of the artifact.
195
- </p>
196
- </div>
197
- <div class="panel">
198
- <h2>Public reasoning surface</h2>
199
- <p>
200
- Turn the answer into a readable page: context, model, tradeoffs,
201
- examples, edge cases, rollout steps, and verification notes. Show
202
- conclusions and evidence, not private chain-of-thought.
203
- </p>
204
- <pre><code>// Add focused snippets when they clarify the plan.
205
- function example() {{
206
- return "replace this with the requested artifact content";
207
- }}</code></pre>
208
- </div>
209
- <div class="panel">
210
- <h2>Page draft</h2>
211
- <p>
212
- For plan artifacts, draft the actual page or content structure here:
213
- title, lede, sections, key copy, calls to action, states, and the
214
- narrative blocks the user should see.
215
- </p>
216
- </div>
217
- <div class="grid">
218
- <article class="panel">
219
- <h2>Page croquis A</h2>
220
- <div class="croquis">
221
- <div class="bar" style="width: 72%;"></div>
222
- <div class="box"></div>
223
- <div class="box" style="width: 64%;"></div>
224
- </div>
225
- <p class="note">Sketch the page hierarchy quickly. This is for seeing and understanding, not final UI.</p>
226
- </article>
227
- <article class="panel">
228
- <h2>Page croquis B</h2>
229
- <div class="croquis">
230
- <div class="box" style="height: 84px;"></div>
231
- <div class="bar" style="width: 46%;"></div>
232
- <div class="bar" style="width: 78%; opacity: .35;"></div>
233
- </div>
234
- <p class="note">Annotate what changes: layout, emphasis, rhythm, audience, or tradeoff.</p>
235
- </article>
236
- </div>
237
- </section>
238
- </main>
239
- </body>
240
- </html>
241
- """
242
-
243
-
244
- def highlogic_template(title: str, style: str, kind: str, created_at: str) -> str:
245
- return f"""# {title}
246
-
247
- ## Intent
248
-
249
- - Kind: {kind}
250
- - Style: {style}
251
- - Created: {created_at}
252
-
253
- ## User Request
254
-
255
- TODO: Summarize the user's artifact request.
256
-
257
- ## Core Logic
258
-
259
- TODO: Describe the public reasoning structure, sections, data model, interactions, and key design decisions. Capture conclusions, evidence, assumptions, and tradeoffs without private chain-of-thought.
260
-
261
- ## Verification
262
-
263
- TODO: Record how the artifact was opened or tested, including browser/runtime checks when relevant.
264
-
265
- ## Iteration Notes
266
-
267
- - Initial scaffold created.
268
- """
269
-
270
-
271
- def main() -> int:
272
- parser = argparse.ArgumentParser(description="Create a local HTML artifact workspace.")
273
- parser.add_argument("title", help="Short artifact title or slug")
274
- parser.add_argument("--style", default="custom", help="style label recorded in the manifest (requested, project:<app-name>, or subject-specific)")
275
- parser.add_argument("--kind", default="thinking", help="artifact kind")
276
- args = parser.parse_args()
277
-
278
- base = GLOBAL_ARTIFACTS_DIR.resolve()
279
- artifact_id = slugify(args.title)
280
- artifact_dir = unique_dir(base, artifact_id)
281
- created_at = datetime.now(timezone.utc).isoformat()
282
-
283
- artifact_dir.mkdir(parents=True, exist_ok=False)
284
- (artifact_dir / "versions").mkdir()
285
-
286
- title = args.title.strip()
287
- (artifact_dir / "index.html").write_text(
288
- html_template(title, args.style, args.kind),
289
- encoding="utf-8",
290
- )
291
- (artifact_dir / "HIGHLOGIC.md").write_text(
292
- highlogic_template(title, args.style, args.kind, created_at),
293
- encoding="utf-8",
294
- )
295
- manifest = {
296
- "id": artifact_dir.name,
297
- "title": title,
298
- "kind": args.kind,
299
- "style": args.style,
300
- "created_at": created_at,
301
- "entrypoint": "index.html",
302
- "files": ["index.html", "HIGHLOGIC.md", "manifest.json"],
303
- }
304
- (artifact_dir / "manifest.json").write_text(
305
- json.dumps(manifest, indent=2) + "\n",
306
- encoding="utf-8",
307
- )
308
-
309
- index_path = artifact_dir / "index.html"
310
- print(f"created={artifact_dir}")
311
- print(f"index={index_path}")
312
- print(f"url={index_path.as_uri()}")
313
- return 0
314
-
315
-
316
- if __name__ == "__main__":
317
- raise SystemExit(main())
@@ -1,97 +0,0 @@
1
- ---
2
- name: use-delegate
3
- description: "Delegation mode: the host agent (Claude or Codex) plans and reviews while heavy work runs on cheap executors: OpenCode Kimi K3, Codex GPT-5.6 terra/sol. Use when the user invokes /use-delegate, says 'use delegate', 'delegate mode', 'orchestrator mode', or wants to save tokens/rate limits."
4
- disable-model-invocation: true
5
- metadata:
6
- opencode/autoinvoke: "false"
7
- opencode/slash: "true"
8
- ---
9
-
10
- # Use Delegate
11
-
12
- The host agent (Claude Code or Codex, whichever is running this skill) is the thinker, never the typist. Its tokens are scarce; the executors below are cheap and steerable. The host decides **what** to do and judges **whether it was done well**: everything token-hungry runs elsewhere and reports back.
13
-
14
- ## Core rule
15
-
16
- The host does not execute. It may: read a few targeted files, search, inspect git state, think, plan, decompose, write specs and delegation prompts, review diffs and reports, judge outputs, and talk to the user.
17
-
18
- The host must NOT directly do:
19
-
20
- - Implementation, refactors, migrations, test writing (any multi-file or >~15-line change)
21
- - Codebase-wide exploration or analysis (reading many files to "understand")
22
- - Computer use, browser automation, UI/UX verification
23
- - Log triage, data analysis, bulk mechanical edits
24
- - Running long build/test loops and reading their full output
25
-
26
- The only direct edits allowed: trivial single-file tweaks (a config value, a typo, a one-liner) where writing a delegation prompt would cost more than the edit itself.
27
-
28
- ## Executors
29
-
30
- | Executor | Command | Use for |
31
- |---|---|---|
32
- | OpenCode · Kimi K3 | `opencode run "<prompt>" -m kimi-for-coding/k3` | Default general executor: implementation, refactors, tests, bulk edits |
33
- | Codex · GPT-5.6 terra (high) | `codex exec -m gpt-5.6-terra "<prompt>"` | Low-stakes tasks: mechanical edits, scripts, quick investigations, log triage |
34
- | Codex · GPT-5.6 sol (high) | `codex exec "<prompt>"` (config default = sol + high) | Compute-heavy tasks: hard bugs, migrations, architecture-sensitive changes, computer use / UI verification |
35
- | Host-native subagents | Claude `Agent` tool / Codex collab threads | Exploration summaries the host plans from; taste-sensitive work (UI, copy, API design) |
36
-
37
- Read-only investigation: `codex exec -s read-only`, `opencode run --agent plan` (built-in read-only agent).
38
-
39
- Model rankings move fast: current DeepSWE scores, API pricing, and the refresh protocol live in `references/models.md`. Check its `Last verified` date before planning a big batch; if older than 14 days, delegate a refresh first (DeepSWE leaderboard + pricing pages), never guess rankings from memory.
40
-
41
- ## Invocation mechanics
42
-
43
- Both CLIs: always end the command with `< /dev/null` and run in background (codex hangs forever on open stdin: full codex mechanics in `~/.claude/rules/launch-codex.md`).
44
-
45
- **Codex** (background):
46
-
47
- ```bash
48
- codex exec -C <repo-root> -m gpt-5.6-terra \
49
- --output-last-message <scratchpad>/codex-<task>.md \
50
- "<self-contained prompt>" < /dev/null
51
- ```
52
-
53
- Effort override: `-c model_reasoning_effort=high`. Non-git dir: `--skip-git-repo-check`.
54
-
55
- **OpenCode** (background):
56
-
57
- ```bash
58
- opencode run "<self-contained prompt>" \
59
- -m kimi-for-coding/k3 --title "<task>" \
60
- > <scratchpad>/oc-<task>.log 2>&1 < /dev/null
61
- ```
62
-
63
- - Final answer = tail of the log; `--format json` for machine-readable events.
64
- - Steer or continue a session: `opencode run -s <sessionID> "<follow-up>"`.
65
- - Standalone specialized agent: `--agent <name>`: the `~/.config/opencode/agent/` roster (worker, explore-fast, verifier, snipper, code-reviewer…) runs standalone with any `-m` model.
66
- - Permissions are pre-allowed in the user config (build agent allows all); no interactive prompt will block a non-interactive run.
67
-
68
- ## Batch / multi-process
69
-
70
- - **Parallel processes** (verified): launch N independent `opencode run` / `codex exec` in background; each opencode run spins up its own server + session, results stay isolated.
71
- - **Shared server** for large batches: `opencode serve --port <p>` once, then N × `opencode run --attach http://localhost:<p> --dir <workdir> ...`: one server, many sessions, less startup overhead. Kill the serve process when done.
72
- - **In-executor subagents**: Kimi in opencode spawns its own task-tool subagents; codex spawns collab threads (config caps 6). Prefer one executor process per independent task over one giant prompt.
73
-
74
- ## The loop
75
-
76
- 1. **Think.** Understand the request. Missing context → delegate the exploration, think on the summaries.
77
- 2. **Spec.** Write a self-contained delegation prompt: exact files, goal, constraints, and what "done" looks like (tests pass, lint clean, behavior X). The executor can't see this conversation: spell everything out.
78
- 3. **Delegate.** Fire independent tasks in parallel in background. Stay available to steer.
79
- 4. **Verify.** Delegate verification too: read-only pass, test run, or `verifier` agent. The host reads the report and the diff, not the whole tree.
80
- 5. **Judge.** Output misses the bar → refine the spec and re-delegate (better prompts beat manual fixes). Escalate terra → Kimi K3 → sol → host-native only when the cheaper tier keeps failing.
81
-
82
- ## Delegation prompt checklist
83
-
84
- - Names exact files/paths and the repo root
85
- - States the goal in one sentence, then constraints (style, deletion safety: `trash` not `rm -rf`, no scope creep)
86
- - Defines done: commands to run, expected results
87
- - Asks for a report: changed files, what was done, tests run, risks
88
-
89
- ## Anti-patterns
90
-
91
- - "It's faster if I just do it": beyond a trivial tweak it isn't, and it burns the budget the whole session depends on.
92
- - Reading 10 files to plan: delegate exploration, plan from the summary.
93
- - Fixing an executor's output by hand: refine the prompt and rerun.
94
- - Serializing independent delegations: parallelize.
95
- - Escalating everything to sol or the host: terra and Kimi K3 handle most well-spec'd work.
96
-
97
- If no delegation path works (CLIs unavailable, Bash denied), say so explicitly and ask the user before falling back to direct execution: never silently drop out of the mode.
@@ -1,10 +0,0 @@
1
- interface:
2
- display_name: "Use Delegate"
3
- short_description: "Delegate heavy work to OpenCode Kimi K3 and Codex"
4
- icon_small: "./assets/codex-icon.svg"
5
- icon_large: "./assets/codex-icon.svg"
6
- brand_color: "#802F83"
7
- default_prompt: "Use $use-delegate to orchestrate this task through cheap executors."
8
-
9
- policy:
10
- allow_implicit_invocation: false
@@ -1,20 +0,0 @@
1
- <!-- @license lucide-static v1.24.0 - ISC -->
2
- <svg role="img" aria-label="use-delegate skill icon"
3
- class="lucide lucide-bot"
4
- xmlns="http://www.w3.org/2000/svg"
5
- width="128"
6
- height="128"
7
- viewBox="0 0 24 24"
8
- fill="none"
9
- stroke="#F5F5F5"
10
- stroke-width="2"
11
- stroke-linecap="round"
12
- stroke-linejoin="round"
13
- >
14
- <path d="M12 8V4H8" />
15
- <rect width="16" height="12" x="4" y="8" rx="2" />
16
- <path d="M2 14h2" />
17
- <path d="M20 14h2" />
18
- <path d="M15 13v2" />
19
- <path d="M9 13v2" />
20
- </svg>
@@ -1,32 +0,0 @@
1
- # Delegation models: current rankings and pricing
2
-
3
- Last verified: 2026-07-20
4
-
5
- **Refresh protocol**: if the date above is older than 14 days, refresh BEFORE planning a big delegation batch. Delegate the research (exa-search skill or a read-only executor): pull the DeepSWE leaderboard (https://deepswe.datacurve.ai/) and the provider pricing pages, then update both tables and the date. DeepSWE is the reference signal: 113 original long-horizon engineering tasks, contamination-free, cost-per-task published per model.
6
-
7
- ## DeepSWE leaderboard (best config per model, snapshot 2026-07-17)
8
-
9
- | Model | Pass@1 | Avg cost/task | Read |
10
- |---|---|---|---|
11
- | gpt-5.6-sol [max] | 73% | $8.39 | Top score, best cost among frontier |
12
- | claude-fable-5 [max] | 70% | $21.63 | Host tier: 2.6x sol cost, never a delegation target |
13
- | gpt-5.6-terra [max] | 70% | $4.95 | Sol-level score at 59% of the cost |
14
- | kimi-k3 [max] | 69% | $4.65 | Within noise of terra/sol, cheapest of the top pack |
15
- | gpt-5.6-luna [max] | 67% | $3.03 | Acceptable floor for trivial bulk work |
16
- | gpt-5.5 [xhigh] | 67% | $7.23 | Superseded by the 5.6 family |
17
- | claude-opus-4.8 [max] | 59% | $13.22 | Dominated: lower score, higher cost |
18
-
19
- ## API list pricing (per 1M tokens)
20
-
21
- | Model | Input | Cached input | Output | Context |
22
- |---|---|---|---|---|
23
- | Kimi K3 (Moonshot) | $3.00 | $0.30 | $15.00 | 1M |
24
- | GPT-5.6 Sol | $5.00 | $0.50 | $30.00 | 1.05M |
25
- | GPT-5.6 Terra | $2.50 | $0.25 | $15.00 | 1.05M |
26
- | GPT-5.6 Luna | $1.00 | ~$0.10 | $6.00 | 1.05M |
27
-
28
- ## Access notes (this machine)
29
-
30
- - Kimi K3: flat-rate through the kimi-for-coding subscription in opencode (`-m kimi-for-coding/k3`), so marginal cost per delegation is ~zero. The opencode-go gateway is unfunded (insufficient balance): do not route through it. Open weights due 2026-07-27; 1M context; native vision; Terminal-Bench 88.3.
31
- - GPT-5.6 sol/terra/luna: covered by the Codex subscription (`codex exec -m gpt-5.6-<tier>`); config default is sol + effort high.
32
- - Current call (2026-07-20): Kimi K3 is the default executor (top-pack score, subscription-covered). Terra for low-stakes tasks. Sol for compute-heavy work. Luna only for trivial bulk edits.
@@ -1,121 +0,0 @@
1
- ---
2
- name: use-goal
3
- description: Use when the user asks to create, draft, set, start, or refine a Codex or Claude Code `/goal` objective for persistent multi-turn work.
4
- ---
5
-
6
- # Use Goal
7
-
8
- Create or draft a Codex or Claude Code Goal that follows the official `/goal` contract: one persistent objective with evidence-based completion criteria.
9
-
10
- ## When To Use
11
-
12
- Use this skill when the user explicitly asks to:
13
-
14
- - create, set, start, or use a Goal
15
- - turn a task into a strong `/goal`
16
- - make Codex or Claude Code continue until an outcome is actually done
17
- - define success criteria for longer debugging, optimization, migration, refactor, benchmark, flaky-test, or research work
18
-
19
- Do not introduce a Goal for a one-off edit, short explanation, simple code review, or single answer unless the user explicitly asks for Goal mode.
20
- Do not use a Goal for a loose backlog or unrelated task list. A good Goal is bigger than one prompt but smaller than an open-ended project.
21
-
22
- ## Pick The Platform
23
-
24
- Before drafting or creating a Goal, identify the active platform from the runtime and available tools:
25
-
26
- - **Codex**: use `references/codex-goal.md`.
27
- - **Claude Code**: use `references/claude-code-goal.md`.
28
- - **Unknown platform**: draft a plain `/goal ...` command and state that the user should run it in the target agent.
29
-
30
- If Goal tools are available in Codex, use them rather than only printing a slash command. If the runtime only exposes slash commands, return the exact `/goal ...` command unless the harness can dispatch it directly.
31
-
32
- ## Goal Shape
33
-
34
- Before writing or creating the Goal, think through the verification strategy. Do a short discovery pass when the evidence surface is not already obvious:
35
-
36
- - Inspect repository docs, package scripts, test commands, CI config, benchmark scripts, failing logs, linked issue text, plans, or referenced files.
37
- - Identify which command, artifact, report, screenshot, benchmark, source document, or manual check can prove completion.
38
- - For external libraries, APIs, or current product behavior, use the appropriate docs/research skill before relying on memory.
39
- - Prefer existing project commands and documented workflows over inventing new validation.
40
- - If no reliable verification surface exists, ask one concise question or make the Goal explicitly require creating one.
41
-
42
- For broad refactors, deletions, migrations, moving files, or "remove all X" goals, read `references/verification-harnesses.md` before creating the Goal. Default to a measurable harness: establish a baseline count/list first, then make the Goal continue until the validation command exits successfully at the target condition, such as count `0`.
43
-
44
- Write the Goal as a compact, well-formatted contract with these fields embedded in natural language:
45
-
46
- 1. Outcome: what must be true when the work is done.
47
- 2. Verification surface: the tests, commands, benchmarks, artifacts, reports, logs, source material, or other concrete evidence that proves it.
48
- 3. Constraints: what must not regress or be violated.
49
- 4. Boundaries: allowed files, tools, repositories, data, and resources when relevant.
50
- 5. Iteration policy: how to choose the next best action after each attempt.
51
- 6. Blocked stop condition: when to stop and what to report if no defensible path remains.
52
-
53
- For long-running implementation work, also include:
54
-
55
- - one objective and one stopping condition
56
- - the files, docs, issue, logs, or plan the agent should inspect first
57
- - the commands or artifacts that prove progress
58
- - checkpoint behavior and a short progress log requirement
59
-
60
- Prefer this pattern:
61
-
62
- ```text
63
- <desired end state>, verified by <specific evidence>, while preserving <constraints>. Use <allowed inputs, tools, or boundaries>. Between iterations, <how to choose and record the next best action>. If blocked or no valid paths remain, stop with <attempted paths, evidence gathered, blocker, and next input needed>.
64
- ```
65
-
66
- For implementation Goals, include exact command names when known:
67
-
68
- ```text
69
- <desired end state>, verified by `<test or build command>` and <artifact/manual check>, while preserving <constraints>. First inspect <files/docs/logs>. Work in checkpoints: after each change, run the narrowest relevant verification, record the result, and choose the next smallest defensible step. Stop only when the verification passes, or stop blocked with the failed command output, attempted paths, and the missing input needed.
70
- ```
71
-
72
- Keep the objective non-empty and at most 4,000 characters. If the needed instructions are longer, create or point to a file and make the Goal refer to that file.
73
-
74
- ## Create Or Draft
75
-
76
- When goal tools are available, use this order:
77
-
78
- 1. Call the status tool first to check whether a Goal already exists.
79
- 2. If the user explicitly asked to create, set, start, or use a new Goal and no Goal exists, call the create tool with the refined objective.
80
- 3. Set a token budget only when the user explicitly provided one.
81
- 4. If a Goal already exists, do not overwrite, clear, pause, or resume it unless the user explicitly asks for that lifecycle action.
82
-
83
- For Codex, this means calling `get_goal` first, then `create_goal` with the refined objective when creation is requested and no active Goal blocks it. Read `references/codex-goal.md` before doing so.
84
-
85
- For Claude Code, `/goal <condition>` is a real slash command, but the model cannot launch it unless the harness exposes a slash-command dispatch tool. If no dispatch tool is available, return the exact manual `/goal ...` command, ask the user to paste/run it, and wait for confirmation before continuing Goal-driven work. Do not say Claude Code lacks `/goal`, do not call it Codex-only, and do not substitute a task list or goal file as equivalent unless the user explicitly asks for that fallback. Read `references/claude-code-goal.md` before drafting or instructing a Claude Code Goal.
86
-
87
- If the user asks only to draft, rewrite, explain, or refine a Goal, return the final `/goal ...` text instead of activating it.
88
-
89
- Ask a clarifying question only when a missing detail would make the Goal unverifiable or unsafe. Prefer one concise question. Otherwise infer conservative defaults from the repository, task, and available evidence.
90
-
91
- ## Evidence Rules
92
-
93
- Completion must be evidence-based. Do not mark a Goal complete because the work seems likely done. First compare the objective to concrete evidence such as changed files, command output, tests, benchmarks, generated artifacts, logs, or source-backed research findings.
94
-
95
- If a budget limit is reached, stop substantive work, summarize progress and blockers, and identify the next useful step. Do not treat budget exhaustion as completion.
96
-
97
- If blocked, report the attempted paths, evidence gathered, blocker, and exact input or external change that would unlock progress.
98
-
99
- If status reports become vague, tighten the Goal instead of adding more one-off instructions. Name the current checkpoint, what was verified, what remains, and what should cause a pause.
100
-
101
- Only mark a Goal complete after verifying the stated stopping condition. Only mark it blocked when the same blocking condition has repeated enough that no meaningful progress is possible without user input or an external change.
102
-
103
- ## References
104
-
105
- - `references/codex-goal.md`: Use for Codex Goal mode, `get_goal` / `create_goal`, CLI/app `/goal`, feature setup, and completion/blocking rules.
106
- - `references/claude-code-goal.md`: Use for Claude Code manual `/goal`, evaluator behavior, requirements, status, clear/resume behavior, and non-interactive usage.
107
- - `references/verification-harnesses.md`: Use for measurable refactor, deletion, migration, move, rename, dependency-removal, and "remove all X" Goals.
108
-
109
- ## Good Examples
110
-
111
- ```text
112
- /goal Reduce p95 checkout latency below 120 ms, verified by the checkout benchmark, while keeping the correctness suite green. Use only the checkout service, benchmark fixtures, and related tests. Between iterations, record what changed, what the benchmark showed, and the next best experiment to try. If the benchmark cannot run or no valid paths remain, stop with the attempted paths, the evidence gathered, the blocker, and the next input needed.
113
- ```
114
-
115
- ```text
116
- /goal Make the checkout test suite pass on the current branch, verified by the repository's documented test command, while preserving public API behavior. Use the failing tests, adjacent implementation files, and existing test helpers. Between iterations, inspect the latest failure, make the smallest defensible change, and rerun the relevant test surface. If no valid path remains, stop with the failures, changes tried, and the missing decision or dependency.
117
- ```
118
-
119
- ```text
120
- /goal Produce the strongest evidence-backed reproduction report for the provided paper using available materials and local resources. Attempt the headline claims where feasible, verify outputs where possible, and end with a report that separates confirmed findings, approximate reconstructions, blocked claims, and remaining uncertainty.
121
- ```
@@ -1,7 +0,0 @@
1
- interface:
2
- display_name: "Use Goal"
3
- short_description: "Use when the user asks to create, draft, set, start, or..."
4
- icon_small: "./assets/codex-icon.svg"
5
- icon_large: "./assets/codex-icon.svg"
6
- brand_color: "#C70A64"
7
- default_prompt: "Use $use-goal to help with this task."
@@ -1,18 +0,0 @@
1
- <!-- @license lucide-static v1.24.0 - ISC -->
2
- <svg role="img" aria-label="use-goal skill icon"
3
- class="lucide lucide-sparkles"
4
- xmlns="http://www.w3.org/2000/svg"
5
- width="128"
6
- height="128"
7
- viewBox="0 0 24 24"
8
- fill="none"
9
- stroke="#F5F5F5"
10
- stroke-width="2"
11
- stroke-linecap="round"
12
- stroke-linejoin="round"
13
- >
14
- <path d="M11.017 2.814a1 1 0 0 1 1.966 0l1.051 5.558a2 2 0 0 0 1.594 1.594l5.558 1.051a1 1 0 0 1 0 1.966l-5.558 1.051a2 2 0 0 0-1.594 1.594l-1.051 5.558a1 1 0 0 1-1.966 0l-1.051-5.558a2 2 0 0 0-1.594-1.594l-5.558-1.051a1 1 0 0 1 0-1.966l5.558-1.051a2 2 0 0 0 1.594-1.594z" />
15
- <path d="M20 2v4" />
16
- <path d="M22 4h-4" />
17
- <circle cx="4" cy="20" r="2" />
18
- </svg>