@hybridlabor-api/aos 4.12.1 → 4.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. package/.agents/vendor-manifest.json +6 -0
  2. package/.claude/agents/reviewer.md +1 -1
  3. package/.claude/agents/techlead.md +1 -1
  4. package/.opencode/agents/architect.md +1 -1
  5. package/.opencode/agents/reviewer.md +1 -1
  6. package/.opencode/agents/techlead.md +1 -1
  7. package/README.de.md +10 -9
  8. package/README.md +10 -9
  9. package/README.pt.md +10 -9
  10. package/THIRD_PARTY_NOTICES.md +94 -0
  11. package/bin/aos-store.mjs +48 -45
  12. package/installer.js +22 -20
  13. package/lib/scenario-store-index.json +1 -0
  14. package/lib/store-shared.mjs +81 -4
  15. package/lib/store-ui/index.html +14 -5
  16. package/lib/store-ui/server.mjs +44 -21
  17. package/package.json +2 -2
  18. package/scripts/build-plugin-manifest.mjs +33 -10
  19. package/scripts/build-scenario-store-index.mjs +154 -0
  20. package/skills/bdbrainstorm/SKILL.md +1 -0
  21. package/skills/global_config/agent-pipeline/SKILL.md +4 -0
  22. package/skills/global_config/aos-store/SKILL.md +3 -3
  23. package/skills/global_config/design-control-loop/SKILL.md +184 -0
  24. package/skills/global_config/design-control-loop/references/agent-iteration.ts +174 -0
  25. package/skills/global_config/design-control-loop/references/agent-runner-templates.md +157 -0
  26. package/skills/global_config/design-control-loop/references/control-loop-taxonomy.md +75 -0
  27. package/skills/global_config/design-control-loop/references/example-control-loop.md +57 -0
  28. package/skills/global_config/design-control-loop/references/example-skill.md +171 -0
  29. package/skills/global_config/design-control-loop/references/memory-template.md +7 -0
  30. package/skills/global_config/design-control-loop/references/prompt-template.md +58 -0
  31. package/skills/global_config/design-control-loop/references/response-template.md +103 -0
  32. package/skills/global_config/design-control-loop/references/skill-template.md +57 -0
  33. package/skills/global_config/design-control-loop/references/workflow-template.yml +273 -0
@@ -0,0 +1,154 @@
1
+ #!/usr/bin/env node
2
+ // Usage: node scripts/build-scenario-store-index.mjs <path-to-local-clone> <commit>
3
+ // Offline: reads blobs of <commit> from the local clone with git; never touches the network and never runs anything from it.
4
+ import { execFileSync } from 'node:child_process';
5
+ import { createHash } from 'node:crypto';
6
+ import { writeFileSync } from 'node:fs';
7
+ import { dirname, join, resolve } from 'node:path';
8
+ import { fileURLToPath } from 'node:url';
9
+
10
+ const ROOT = resolve(dirname(fileURLToPath(import.meta.url)), '..');
11
+ const OUTPUT_PATH = join(ROOT, 'lib', 'scenario-store-index.json');
12
+ const CAP = 300 * 1024;
13
+ const SOURCE = {
14
+ id: 'scenario', label: 'Scenario', upstream: 'https://github.com/scenario-labs/skills',
15
+ raw_base: 'https://raw.githubusercontent.com/scenario-labs/skills',
16
+ license: 'MIT', copyright: 'Copyright (c) 2026 Scenario', third_party: true,
17
+ };
18
+
19
+ // Starter set (production_artifacts/09 section 4, minus skills whose dependencies are outside the set).
20
+ // Excluded: scenario-blender-rigging (bx_rig.mannequin imports scenario-blender-sculpting), scenario-unity-2d (SKILL.md
21
+ // names scenario-sprite-pipeline, which does not exist upstream), scenario-text-overlay (Scenario MCP workflow, chevron + network font fetch).
22
+ const BLENDER = 'scenario-blender-expert';
23
+ const UNITY = 'scenario-unity-expert';
24
+ const SET = {
25
+ [BLENDER]: { dir: 'skills/dcc/blender', category: 'media-eventtech' },
26
+ 'scenario-blender-retopology': { dir: 'skills/dcc/blender', category: 'media-eventtech', requires: [BLENDER] },
27
+ 'scenario-blender-uv-baking': { dir: 'skills/dcc/blender', category: 'media-eventtech', requires: [BLENDER] },
28
+ 'scenario-blender-texturing-shading': { dir: 'skills/dcc/blender', category: 'media-eventtech', requires: [BLENDER] },
29
+ 'scenario-blender-hard-surface': { dir: 'skills/dcc/blender', category: 'media-eventtech', requires: [BLENDER] },
30
+ 'scenario-blender-lighting-rendering': { dir: 'skills/dcc/blender', category: 'media-eventtech', requires: [BLENDER] },
31
+ 'scenario-blender-animation': { dir: 'skills/dcc/blender', category: 'media-eventtech', requires: [BLENDER] },
32
+ 'scenario-blender-geometry-nodes': { dir: 'skills/dcc/blender', category: 'media-eventtech', requires: [BLENDER] },
33
+ [UNITY]: { dir: 'skills/game-engines/unity', category: 'media-eventtech' },
34
+ 'scenario-unity-performance': { dir: 'skills/game-engines/unity', category: 'engineering-method', requires: [UNITY] },
35
+ 'scenario-unity-pipeline-automation': { dir: 'skills/game-engines/unity', category: 'engineering-method', requires: [UNITY] },
36
+ 'scenario-unity-shaders': { dir: 'skills/game-engines/unity', category: 'media-eventtech', requires: [UNITY] },
37
+ 'scenario-unity-ui': { dir: 'skills/game-engines/unity', category: 'design-ui-ux', requires: [UNITY] },
38
+ };
39
+
40
+ const [cloneArg, commit] = process.argv.slice(2);
41
+ if (!cloneArg || !/^[a-f0-9]{40}$/.test(commit || '')) {
42
+ console.error('Usage: build-scenario-store-index.mjs <clone-path> <40-hex-commit>');
43
+ process.exit(2);
44
+ }
45
+ const clone = resolve(cloneArg);
46
+ const git = (args, opts = {}) => execFileSync('git', ['-C', clone, ...args], { maxBuffer: 64 << 20, ...opts });
47
+ if (git(['rev-parse', `${commit}^{commit}`], { encoding: 'utf8' }).trim() !== commit) throw new Error('commit not found in clone');
48
+ const hash = (data) => createHash('sha256').update(data).digest('hex');
49
+
50
+ function frontmatterValue(text, key) {
51
+ const match = text.match(/^---\s*\n([\s\S]*?)\n---/);
52
+ if (!match) return '';
53
+ const lines = match[1].split(/\r?\n/);
54
+ const start = lines.findIndex((line) => new RegExp(`^${key}:`, 'i').test(line));
55
+ if (start < 0) return '';
56
+ const first = lines[start].slice(lines[start].indexOf(':') + 1).trim();
57
+ if (first && !/^[>|][-+]?$/.test(first)) return first.replace(/^['"]|['"]$/g, '');
58
+ const values = [];
59
+ for (let i = start + 1; i < lines.length; i += 1) {
60
+ if (lines[i] && !/^\s/.test(lines[i])) break;
61
+ values.push(lines[i].trim());
62
+ }
63
+ return values.join(' ').trim();
64
+ }
65
+
66
+ function skillFiles(dir) {
67
+ return git(['ls-tree', '-r', '-z', commit, `${dir}/`]).toString('latin1').split('\0').filter(Boolean).map((line) => {
68
+ const [meta, path] = line.split('\t');
69
+ const [mode, kind, oid] = meta.split(' ');
70
+ if (kind !== 'blob' || mode === '120000') throw new Error(`Unsupported entry in ${dir}: ${line}`);
71
+ const data = git(['cat-file', 'blob', oid]);
72
+ return { path: path.slice(dir.length + 1), sha256: hash(data), size: data.length, exec: mode === '100755', data };
73
+ }).sort((a, b) => (a.path < b.path ? -1 : 1));
74
+ }
75
+
76
+ // Audit detectors run on every script (files under scripts/). execBridge = a file that both executes code it
77
+ // receives (exec/eval/runtime compile/reflective invoke) and has a receive channel (socket, server, inbox).
78
+ const EXEC = [
79
+ [/(^|[^\w.])(exec|eval)\s*\(/m, 'calls exec()/eval()'],
80
+ [/\b(CSharpScript|CSharpCodeProvider|CompileAssemblyFromSource)\b/, 'compiles C# at runtime'],
81
+ [/\.Invoke\(null, null\)/, 'invokes a method named in a request'],
82
+ [/\bAssembly\.Load\b/, 'loads assemblies at runtime'],
83
+ ];
84
+ const CHANNEL = [
85
+ [/\b(HttpListener|TcpListener|UdpClient)\b/, 'network listener'],
86
+ [/\.listen\s*\(|\.bind\s*\(|socketserver|http\.server|socket\.socket\(/, 'socket/server'],
87
+ [/\binbox\b/i, 'file inbox polled for requests'],
88
+ ];
89
+ const firstHit = (rules, text) => rules.find(([re]) => re.test(text))?.[1];
90
+
91
+ function audit(files) {
92
+ const scripts = files.filter((f) => f.path.startsWith('scripts/'));
93
+ const bridgeFiles = [];
94
+ for (const f of scripts) {
95
+ const text = f.data.toString('utf8');
96
+ const exec = firstHit(EXEC, text);
97
+ const channel = firstHit(CHANNEL, text);
98
+ if (exec && channel) bridgeFiles.push({ path: f.path, reason: `${exec} + ${channel}` });
99
+ }
100
+ return { hasScripts: scripts.length > 0, scriptCount: scripts.length, execBridge: bridgeFiles.length > 0, execBridgeFiles: bridgeFiles };
101
+ }
102
+
103
+ const KNOWN_SOFTWARE = ['Blender', 'Unity'];
104
+ function software(description) {
105
+ return KNOWN_SOFTWARE.flatMap((name) => {
106
+ const m = description.match(new RegExp(`\\b${name}\\b(?:\\s(\\d+(?:\\.\\d+)+))?`));
107
+ return m ? [m[1] ? `${name} ${m[1]}` : name] : [];
108
+ });
109
+ }
110
+
111
+ const skills = {};
112
+ for (const [name, cfg] of Object.entries(SET)) {
113
+ const dir = `${cfg.dir}/${name}`;
114
+ const files = skillFiles(dir);
115
+ const skillMd = files.find((f) => f.path === 'SKILL.md');
116
+ if (!skillMd) throw new Error(`${name}: no SKILL.md in ${dir}`);
117
+ const text = skillMd.data.toString('utf8');
118
+ if ((frontmatterValue(text, 'name') || '').trim() !== name) throw new Error(`${name}: SKILL.md name differs from directory`);
119
+ if (!/^[A-Za-z0-9_-]+$/.test(name)) throw new Error(`Invalid store name: ${name}`);
120
+ const description = frontmatterValue(text, 'description').replace(/\s+/g, ' ').trim();
121
+ skills[name] = {
122
+ description: description.length > 100 ? `${description.slice(0, 99)}…` : description,
123
+ category: cfg.category,
124
+ tier: 'free',
125
+ origin: 'scenario',
126
+ requires_auth: false,
127
+ sha256: skillMd.sha256,
128
+ upstream_path: `${dir}/SKILL.md`,
129
+ ...(cfg.requires ? { requires: cfg.requires } : {}),
130
+ software: software(description),
131
+ ...audit(files),
132
+ files: files.map(({ data, ...rest }) => rest),
133
+ };
134
+ }
135
+ for (const [name, item] of Object.entries(skills)) {
136
+ for (const dep of item.requires || []) if (!skills[dep]) throw new Error(`${name} requires ${dep}, which is not in the set`);
137
+ }
138
+
139
+ const index = {
140
+ version: '1.1.0',
141
+ generated_at: git(['show', '-s', '--format=%cI', commit], { encoding: 'utf8' }).trim(),
142
+ pinned_commit: commit,
143
+ source: SOURCE,
144
+ harness_support: ['claude', 'antigravity', 'codex', 'cursor', 'roo'],
145
+ skills: Object.fromEntries(Object.entries(skills).sort(([a], [b]) => a.localeCompare(b))),
146
+ subagents: {},
147
+ };
148
+ const output = JSON.stringify(index) + '\n';
149
+ if (Buffer.byteLength(output) >= CAP) throw new Error(`Store index exceeds 300 KB: ${Buffer.byteLength(output)} bytes`);
150
+ writeFileSync(OUTPUT_PATH, output);
151
+ for (const [name, item] of Object.entries(skills)) {
152
+ console.log(`${name}\t${item.category}\tfiles=${item.files.length}\tscripts=${item.scriptCount}\texecBridge=${item.execBridge}\t${item.execBridgeFiles.map((f) => `${f.path} (${f.reason})`).join('; ')}`);
153
+ }
154
+ console.log(`Wrote ${OUTPUT_PATH} with ${Object.keys(skills).length} skills (${Buffer.byteLength(output)} bytes).`);
@@ -39,6 +39,7 @@ You are strictly required to enforce the following 6 pillars in your process:
39
39
  - **Mandatory — Plan Canvas review.** Write the aligned plan to a file, then run `aos-plan-canvas open <file>` followed by `aos-plan-canvas await <file>` and leave it running. The user reviews and annotates in the browser (Mermaid diagrams render live, click-to-annotate, chat rail); do not write `state.goal` or hand off to `/startcycle-graph` before an `approve` verdict comes back. A `request_changes` verdict means revise the plan file and reopen the session — it live-reloads. This runs identically regardless of which agent harness is executing this skill; it is a plain CLI, not a Claude-Code-specific mechanism. See the `plan-canvas` skill.
40
40
  - Write the plan in the agenttrail component convention (`## Name {#id}` components with `needs:` / `files:` lines, tasks as `- [ ] ... {#id}`) and render the architecture with `aos-archify` so the canvas review includes the diagram; after the `approve` verdict, Trigger A starts the live map. See the `agenttrail` and `archify` skills.
41
41
  - Present the aligned plan and hand off to `/startcycle-graph` for execution — write `state.goal` from this session's output and let `/startcycle-graph`'s dispatcher take it from there (see `.agents/graph.md`). This skill does not invoke `/startcycle-graph`'s agents itself; it produces the goal they read.
42
+ - For a recurring quality goal, run `/design-control-loop` after shipping (manual, opt-in; not part of the graph).
42
43
 
43
44
  ## Execution Rules
44
45
  1. **Never skip the debate:** Ideas must be contested by subagents and the user before finalization.
@@ -48,6 +48,10 @@ This file exists so anyone (human or agent) asking "what's the BDB software
48
48
  lifecycle?" gets the real shape of it, without re-deriving it from `graph.md`
49
49
  line by line. It is descriptive, not an alternate entry point.
50
50
 
51
+ Optional post-ship step, outside the graph: `/design-control-loop` sets up a
52
+ recurring improvement loop in CI for a quality goal. It is started manually and
53
+ the dispatcher never calls it.
54
+
51
55
  ## 2. When to Use
52
56
  - Use when you need the lifecycle framing (define → plan → build → verify →
53
57
  ship) to structure a conversation or a manual walkthrough.
@@ -1,6 +1,6 @@
1
1
  ---
2
2
  name: aos-store
3
- description: Browse, preview, and install AOS Core + ECC skills and agents from a local web UI. Use when exploring or adding new skills to a project.
3
+ description: Browse, preview, and install AOS Core, ECC and Scenario skills and agents from a local web UI. Use when exploring or adding new skills to a project.
4
4
  category: bdb-core
5
5
  metadata:
6
6
  version: "1.0.0"
@@ -13,7 +13,7 @@ A web UI for discovering and installing AOS skills and agents. Lists what is alr
13
13
 
14
14
  ## When to Use
15
15
 
16
- - You want to explore available AOS Core + ECC skills and agents in a local web UI.
16
+ - You want to explore available AOS Core, ECC and Scenario skills and agents in a local web UI.
17
17
  - You need to preview a skill or agent before installing it.
18
18
  - You want to install skills or agents globally (all harnesses) or into the current project, with explicit confirmation before each install.
19
19
 
@@ -43,7 +43,7 @@ Some harnesses (e.g. OpenCode) cannot open a browser automatically. **Always pri
43
43
 
44
44
  ## Store Features
45
45
 
46
- - **Browse:** Lists all available AOS Core and ECC skills and agents.
46
+ - **Browse:** Lists all available AOS Core, ECC and Scenario skills and agents. Scenario items are MIT-licensed third-party skills; some contain scripts (the preview warns, and flags bridges that execute received code).
47
47
  - **Show Installed:** Marks what is already installed and which items are AOS Core (always included).
48
48
  - **Preview:** "+ Add" first shows the exact target paths for the chosen scope (global or project); nothing is written yet.
49
49
  - **Confirm:** Every install requires explicit user confirmation — nothing is installed silently.
@@ -0,0 +1,184 @@
1
+ ---
2
+ name: design-control-loop
3
+ description: interview the user to design an agentic control loop (sensor, controller, actuator under disturbances) tailored to their codebase, then build it as locally-runnable components plus a scheduled coding-agent workflow
4
+ category: bdb-core
5
+ source: humanlayer
6
+ license: MIT (see THIRD_PARTY_NOTICES.md, humanlayer/skills)
7
+ ---
8
+
9
+ # Design Control Loop
10
+
11
+ <!-- AOS notes: start (local addition, not upstream) -->
12
+ ## AOS notes
13
+
14
+ - **Manual, opt-in.** Start it yourself (`/design-control-loop`) for a recurring improvement loop in CI. It is not a pipeline node and not part of `/startcycle-graph`; the dispatcher never calls it. After shipping, it is an optional follow-up.
15
+ - **Runs outside the GO gate.** The loops it generates run unattended in CI, often with `--dangerously-skip-permissions` and API secrets, so the AOS GO gate does not cover them. That is a deliberate design choice; you decide per loop. When generating workflows, pin CLI versions instead of `@latest` and keep secrets in the CI secret store.
16
+ - **Output lives in the target repo.** The generated actuator skill, scripts and workflow are written to the TARGET repo, not to AOS.
17
+ - **Source.** Vendored from https://github.com/humanlayer/skills at commit `ca7c8088db69e315a8b2deea43820270457f8f3c` (MIT, see `THIRD_PARTY_NOTICES.md`). Everything below this section is upstream text.
18
+ <!-- AOS notes: end -->
19
+
20
+ Use this skill when a user wants to drive some property of their codebase toward a target with small, low-risk, reviewable changes on a schedule — an **agentic control loop**.
21
+
22
+ Your job is to **interview the user, design the loop _with_ them, and then build it for them**. The design must be tailored to *their* codebase and the tooling they already use. There is no fixed toolset and no template to reproduce: propose options grounded in what you find in the repo, discuss trade-offs, agree on a design, then implement it.
23
+
24
+ ## The mental model
25
+
26
+ Borrow from control theory. The codebase is a dynamic system being changed continuously (by teammates, dependencies, and generated code — the **disturbances**). A control loop drives it toward a desired state instead of all at once:
27
+
28
+ - **Set point** — the desired end state for some property of the codebase.
29
+ - **Sensor** — measures the current state, producing the gap to the set point.
30
+ - **Controller** — decides the next small, low-risk change from that measurement.
31
+ - **Actuator** — a coding agent that applies the change and opens a PR.
32
+ - The result feeds back into the next run. A human stays *on* the loop to steer it.
33
+
34
+ Read `references/control-loop-taxonomy.md` and walk the user through these concepts before designing anything. For one fully worked example, see `references/example-control-loop.md` — treat it as an illustration, not a blueprint.
35
+
36
+ ## How to run this skill
37
+
38
+ - **Read the repo before you ask** (Phase A). Come to the interview with proposals, not a blank form.
39
+ - **Tailor every component.** The right sensor, controller, and actuator depend entirely on the user's problem and stack. The lists in the references are examples to spark discussion, never a checklist to push.
40
+ - **Make each component runnable locally and standalone before wiring it into CI** (Phase D). The workflow should only orchestrate pieces the user can already run by hand.
41
+ - **Capture the agreed design in writing** before building, so the user can correct it cheaply.
42
+
43
+ ## Outputs
44
+
45
+ Create or update these in the target repo, tailored to the agreed design:
46
+
47
+ - The **sensor** and **controller** as version-controlled commands/scripts the user can run locally.
48
+ - `.claude/skills/<skill-name>/SKILL.md` — the **actuator** skill capturing the agent's judgement (path may be `.agents/skills/...` per repo convention).
49
+ - The recurring **workflow** that runs the loop and opens a PR (GitHub Actions by default; whatever CI the repo uses).
50
+ - A **memory/feedback file** that carries standing feedback between runs.
51
+ - Optionally, a **dampener** (regression gate) that keeps the problem from getting worse while the loop improves it.
52
+
53
+ ## Workflow
54
+
55
+ ### Phase A — Understand the system
56
+
57
+ **Read the following references:** `references/example-control-loop.md`.
58
+
59
+ Read before asking setup questions:
60
+
61
+ - Existing CI: `.github/workflows/*.yml`, `.github/actions/**`, or the repo's non-GitHub CI config — runner, checkout, dependency install, cache, and PR conventions.
62
+ - Package manager files (`package.json`, `bun.lock`, `pnpm-lock.yaml`, `yarn.lock`, `package-lock.json`, `pyproject.toml`, `go.mod`, `Cargo.toml`, …).
63
+ - Existing validation scripts: typecheck, lint, test, quality, format, and package-scoped commands.
64
+ - Existing `.claude/skills` / `.agents/skills` and any existing agent loops (workflows, `agent-memory`, where glue scripts live) to mirror conventions instead of inventing new ones.
65
+ - The static-analysis, linting, codegen, and test tooling already in the repo — these are the most likely raw material for a sensor.
66
+ - Discover packages, services, and repo purpose at a high level.
67
+
68
+ Completion criterion: you can name the repo's package manager, install command, likely validation commands, CI platform, and any existing loop conventions. You understand the packages/services/applications it contains at a high level.
69
+
70
+ ### Phase B — Design the loop with the user
71
+
72
+ **Read the following references:** `references/control-loop-taxonomy.md`, `references/example-control-loop.md`, `references/agent-runner-templates.md`.
73
+
74
+ This is an interview. Work through each component below. Start by asking the user questions about the set point. Proposing options grounded in Phase A and surfacing trade-offs rather than mandating any choice. Record the decisions as you go.
75
+
76
+ 1. **Set point.** What property are we driving, and to what target? Examples: an invariant ("no procedures use the old pattern"), a threshold ("test coverage ≥ X in these packages"), or a direction ("reduce occurrences each run"). Also pin the **scope**: which directories/packages the loop may change, and which it may only read.
77
+
78
+ 2. **Sensor.** How will the loop measure the gap to the set point? Inspect the codebase and the user's existing tooling and propose the options that fit *their* stack — a static-analysis or lint tool, a structural/AST search, a test suite, a type checker, a telemetry or error query, a custom script, or even an agent-based check. Discuss the trade-offs that matter to them (stability, cost, repeatability, and whether the measurement can be silently disabled) instead of mandating any property. Aim for a measurement the controller can act on repeatably.
79
+
80
+ 3. **Controller.** How will the loop choose the next increment from the measurement, sized to stay low-risk and reviewable? Design this *with* the user: how to prioritize targets, how big one increment is, and what "one reviewable unit of work" means here. A controller can be anything from fully deterministic (a script that selects the next target) to fully agentic (an agent that decides from natural-language criteria), and it may be **fused** with the sensor or the actuator. The controller is the part you will **tune over time** from loop output — start simple and expect to revise it.
81
+
82
+ 4. **Actuator.** A coding agent plus a repo-local skill applies the change.
83
+ - **Agent + credentials.** Pick the CLI coding agent (Claude Code, Codex, OpenCode, CodeLayer, …), its secret, and its headless command from `references/agent-runner-templates.md`.
84
+ - **Golden patterns first.** Before automating, establish what a good change looks like: ask the user whether existing patterns in the codebase should be followed, and inspect the code to find them. Capture these in the actuator skill (Phase C).
85
+ - **Validation.** Decide which commands must pass before the agent commits (propose these from Phase A and confirm).
86
+
87
+ 5. **Disturbances + dampener (offer).** Name what changes the system outside the loop (teammates shipping concurrently, dependency bumps, generated code). Then **offer** a dampener: a check that keeps the measured problem from getting worse while the scheduled loop chips away at it — for example a PR check that compares the sensor's output against a baseline and surfaces (or eventually blocks) newly introduced deviations. This is optional; some loops do not need one.
88
+
89
+ Completion criterion: a short written design naming the set point, sensor, controller, actuator (agent + skill + validation), and disturbances/dampener — with each component something the user can run locally.
90
+
91
+ ### Phase C — Build the actuator skill
92
+
93
+ **Read the following references:** `references/skill-template.md`, `references/example-skill.md`, `references/response-template.md`.
94
+
95
+ Write a repo-local skill that captures the actuator's judgement for this task. It can use repo-specific paths, package names, and conventions since it lives in the repository.
96
+
97
+ - Put ordered behavior in `SKILL.md` as steps with checkable completion criteria; move long templates and examples into sibling reference files.
98
+ - Encode the golden patterns from Phase B4 so the agent follows established conventions.
99
+ - Keep one source of truth for each rule; do not repeat the same guidance in the skill, the prompt, and the memory file.
100
+ - Include a response template (e.g. `references/response-template.md`) defining how the agent formats its final output, which becomes the PR body. Instruct the skill to read and follow it.
101
+ - Use `references/skill-template.md` as the skeleton and `references/example-skill.md` as a concrete example. See https://agentskills.io/specification for the skill spec.
102
+
103
+ **IMPORTANT:** the `name` in the skill's frontmatter must match its directory slug — a skill named `migrate-foo` lives at `.claude/skills/migrate-foo/SKILL.md` (or `.agents/skills/migrate-foo/SKILL.md`).
104
+
105
+ Completion criterion: the skill explains the job clearly enough that the agent can do it unattended, including how to format its final response.
106
+
107
+ ### Phase D — Make each component runnable locally
108
+
109
+ **Read the following references:** `references/agent-runner-templates.md`.
110
+
111
+ Before any CI exists, land the sensor and controller as version-controlled commands or scripts (follow the repo's convention for where such scripts live), and verify the whole loop works by hand:
112
+
113
+ - Run the **sensor** standalone and confirm it produces a stable, usable measurement.
114
+ - Run the **controller** on real sensor output and confirm it selects a sensible next increment.
115
+ - Run the **actuator** locally via its headless CLI command on a controller-selected target, and confirm it makes the change and passes validation.
116
+
117
+ Only proceed to CI once each piece runs locally on its own. This keeps the loop debuggable and makes the workflow a thin orchestrator of things the user can already run.
118
+
119
+ Completion criterion: the user can run sensor, controller, and actuator locally and independently.
120
+
121
+ ### Phase E — Wire the loop into CI
122
+
123
+ **Read the following references:** `references/workflow-template.yml`, `references/prompt-template.md`, `references/agent-runner-templates.md`.
124
+
125
+ Assemble the components into a recurring job. GitHub Actions is the default because it already has the code, the secrets, version control, and scheduling/dispatch — but use whatever CI the repo uses.
126
+
127
+ - Run the loop as **discrete steps: sensor → controller → actuator**, then commit and open a PR using the agent's final message as the body. (When components are fused — e.g. the sensor already prioritizes, or one agent both selects and changes — collapse them into a single step; do not invent separation the design does not have.)
128
+ - Reusable logic can live in a custom composite action.
129
+ - Decide the **cadence** (daily, weekdays, weekly, monthly, manual-only, or custom cron) based on task risk and review burden.
130
+ - Interpolate the memory file (Phase F) into the actuator's context.
131
+ - Use `references/workflow-template.yml` as the base and `references/prompt-template.md` for the embedded prompt. Pull the agent run + response-extraction steps from `references/agent-runner-templates.md` (each agent outputs differently; get the final response into `/tmp/pr-body.md`).
132
+
133
+ Completion criterion: the workflow can run from `workflow_dispatch` without relying on files that do not exist.
134
+
135
+ ### Phase F — Put a human on the loop
136
+
137
+ **Read the following references:** `references/memory-template.md`, `references/agent-iteration.ts`.
138
+
139
+ A scheduled loop drifts without steering. Give the human two channels, both of which should change future behavior, not just the current PR:
140
+
141
+ - **Memory/feedback file.** A version-controlled markdown file (e.g. `.github/agent-memory/<task-slug>.md`) loaded deterministically into the actuator's context **after the controller** on every run. Use `references/memory-template.md`. Good entries: permanent scope exclusions, known false-positive areas, and reviewer feedback that should change future selections — not one-off instructions or single-run logs.
142
+ - **`/iterate` on the PR.** Label each loop's PRs and embed a hidden marker so each workflow only handles comments on PRs it created. When a maintainer comments `/iterate`, the matching workflow loads the PR context (diff, comments) and the feedback, and the agent updates its memory and the PR. Install `references/agent-iteration.ts` (modes: `footer` and `prompt`) where the repo keeps CI scripts.
143
+
144
+ Frame this for the user as **how you tune the controller and skill over time** — the loop gets better because a human keeps correcting it.
145
+
146
+ Completion criterion: standing feedback survives between runs, and `/iterate` (if enabled) updates the existing PR.
147
+
148
+ ### Phase G — Flow control
149
+
150
+ **Read the following references:** `references/workflow-template.yml`.
151
+
152
+ Bound work-in-progress so the loop never produces PRs faster than they can be reviewed. **Recommended default: one open PR per loop.**
153
+
154
+ - The workflow checks for open PRs with this loop's label and no-ops on scheduled runs when the bound is met; manual `workflow_dispatch` runs bypass the check.
155
+ - Decide PR metadata: label name, PR title prefix, branch prefix.
156
+
157
+ Without this, a daily loop can stack up duplicate or conflicting PRs while no one is reviewing. Completion criterion: scheduled runs no-op when the open-PR bound for this loop is already met.
158
+
159
+ ### Phase H — Validate, dry-run, and iterate faster
160
+
161
+ **Read the following references:** `references/workflow-template.yml`.
162
+
163
+ **Validate** the workflow YAML (`bunx js-yaml file.yml`, `python -c "import yaml,sys; yaml.safe_load(open(sys.argv[1]))" file.yml`, or `yq`) and confirm every path named by the skill, workflow, and memory file exists or is created by this task.
164
+
165
+ **Dry-run.** A workflow cannot be `workflow_dispatch`-ed until it has run once. Temporarily add a `push` trigger for the current branch, push, watch it run, then remove the trigger. Review any PR it opens to confirm the loop's behavior.
166
+
167
+ **Ready to iterate faster** (once the loop is tuned and producing consistent, high-quality output): increase the schedule frequency; widen the controller's batch (e.g. select N targets per run); run the sense→control→actuate cycle N times per workflow run; or run the workflow multiple times and assign one PR to each teammate.
168
+
169
+ Completion criterion: the workflow YAML parses, all referenced files exist, and the loop has produced at least one reviewed PR.
170
+
171
+ ## Reference Files
172
+
173
+ Each phase above names the references relevant to it — read each one when you reach that phase. Full index:
174
+
175
+ - `references/control-loop-taxonomy.md` — the control-loop components and the design questions to ask; read this first and use it to teach the user.
176
+ - `references/example-control-loop.md` — one fully worked loop, annotated component-by-component. An illustration, not a template.
177
+ - `references/agent-runner-templates.md` — local + CI headless commands and secrets for Claude Code, Codex, OpenCode, and CodeLayer, with response extraction.
178
+ - `references/workflow-template.yml` — recurring loop workflow skeleton with discrete sensor/controller/actuator steps.
179
+ - `references/prompt-template.md` — embedded prompt structure for the actuator step.
180
+ - `references/memory-template.md` — memory/feedback file skeleton.
181
+ - `references/skill-template.md` — skeleton for the generated actuator skill.
182
+ - `references/response-template.md` — examples for how the agent should format its final response (the PR body).
183
+ - `references/example-skill.md` — a concrete example of a well-formed task skill.
184
+ - `references/agent-iteration.ts` — helper for `/iterate` support (PR footer marker + iteration prompt building).
@@ -0,0 +1,174 @@
1
+ #!/usr/bin/env bun
2
+
3
+ // Shared helper for recurring coding agent workflows.
4
+ //
5
+ // `footer` mode appends visible /iterate instructions plus a hidden workflow marker
6
+ // to agent-created PR bodies. The marker is what lets many workflows listen for the
7
+ // same /iterate command while only the workflow that opened the PR continues.
8
+ //
9
+ // `prompt` mode builds the iteration prompt after the workflow has passed its cheap
10
+ // pre-check and checked out the PR branch. It intentionally includes raw PR context
11
+ // plus the workflow memory file, then asks the agent to update the PR and distill
12
+ // durable guidance back into that memory file when appropriate.
13
+ //
14
+ // Installation options:
15
+ // 1. Place in `.github/scripts/agent-iteration.ts` (recommended)
16
+ // 2. Place in `ci-scripts/agent-iteration.ts`
17
+ // 3. Place in `scripts/agent-iteration.ts`
18
+ //
19
+ // Usage:
20
+ // bun .github/scripts/agent-iteration.ts --command footer --workflow <workflow-id> --memory <memory-path>
21
+ // bun .github/scripts/agent-iteration.ts --command prompt --workflow <workflow-id> --memory <memory-path> --repo <owner/repo> --pr-number <number> --comment-body <body>
22
+
23
+ interface Args {
24
+ command: 'footer' | 'prompt'
25
+ workflow: string
26
+ memory: string
27
+ repo?: string
28
+ prNumber?: string
29
+ commentBody?: string
30
+ }
31
+
32
+ const args = parseArgs(Bun.argv.slice(2))
33
+
34
+ if (args.command === 'footer') {
35
+ process.stdout.write(renderFooter(args.workflow, args.memory))
36
+ } else {
37
+ if (!args.repo || !args.prNumber) throw new Error('--repo and --pr-number are required for prompt')
38
+ process.stdout.write(await renderIterationPrompt(args))
39
+ }
40
+
41
+ function renderFooter(workflow: string, memory: string): string {
42
+ return `
43
+ ---
44
+
45
+ ### Iterating on this agent run
46
+
47
+ This PR was opened by a coding agent workflow. Maintainers can comment:
48
+
49
+ - \`/iterate <feedback>\` to ask the same workflow to update this PR and learn durable guidance for future runs.
50
+
51
+ The workflow stores durable feedback in its agent memory file and injects that memory into future runs.
52
+
53
+ <!-- codelayer-agent:workflow=${workflow};memory=${memory};version=1 -->
54
+ `.trim()
55
+ }
56
+
57
+ async function renderIterationPrompt(args: Args): Promise<string> {
58
+ const memory = await readTextIfExists(args.memory)
59
+ const pr = await ghJson(`repos/${args.repo}/pulls/${args.prNumber}`)
60
+ const [issueComments, reviewComments] = await Promise.all([
61
+ ghJson(`repos/${args.repo}/issues/${args.prNumber}/comments --paginate`),
62
+ ghJson(`repos/${args.repo}/pulls/${args.prNumber}/comments --paginate`),
63
+ ])
64
+
65
+ return `# Iteration Request
66
+
67
+ ${stripIterateCommand(args.commentBody ?? '')}
68
+
69
+ # Workflow Identity
70
+
71
+ - Workflow: ${args.workflow}
72
+ - Agent memory file: ${args.memory}
73
+
74
+ # Instructions
75
+
76
+ You are iterating on an open PR that was created by this coding agent workflow.
77
+
78
+ 1. Treat the Iteration Request as the user's current instruction.
79
+ 2. Update this PR branch if the feedback asks for a code, workflow, prompt, or documentation change.
80
+ 3. Distill durable feedback into ${args.memory} when it should influence future scheduled/manual runs of this workflow.
81
+ 4. Keep ${args.memory} concise and human-readable. Do not append raw transcripts or one-off PR details.
82
+ 5. If the feedback is only PR-specific, update the PR but do not add it to memory.
83
+ 6. Commit and push any changes you make.
84
+ 7. Finish with a concise GitHub-flavored markdown summary of what changed and whether memory was updated.
85
+
86
+ # Current Agent Memory
87
+
88
+ ${memory || '(memory file is empty or missing)'}
89
+
90
+ # Pull Request
91
+
92
+ - Number: #${pr.number}
93
+ - Title: ${pr.title}
94
+ - State: ${pr.state}
95
+ - Base: ${pr.base?.ref ?? '(unknown)'}
96
+ - Head: ${pr.head?.ref ?? '(unknown)'}
97
+
98
+ ## PR Body
99
+
100
+ ${pr.body ?? '(empty)'}
101
+
102
+ # PR Issue Comments
103
+
104
+ ${formatIssueComments(issueComments)}
105
+
106
+ # PR Review Comments
107
+
108
+ ${formatReviewComments(reviewComments)}
109
+ `
110
+ }
111
+
112
+ async function readTextIfExists(path: string): Promise<string> {
113
+ try {
114
+ const file = Bun.file(path)
115
+ if (!(await file.exists())) return ''
116
+ return await file.text()
117
+ } catch {
118
+ return ''
119
+ }
120
+ }
121
+
122
+ async function ghJson(pathAndArgs: string): Promise<any> {
123
+ const result = await Bun.$`bash -lc ${`gh api ${pathAndArgs}`}`.text()
124
+ return JSON.parse(result)
125
+ }
126
+
127
+ function formatIssueComments(comments: any): string {
128
+ if (!Array.isArray(comments) || comments.length === 0) return '(no issue comments)'
129
+ return comments
130
+ .map((comment) => {
131
+ return `Comment ${comment.id} by ${comment.user?.login ?? 'unknown'} at ${comment.created_at ?? 'unknown time'}:\n\n${comment.body ?? ''}`
132
+ })
133
+ .join('\n\n---\n\n')
134
+ }
135
+
136
+ function formatReviewComments(comments: any): string {
137
+ if (!Array.isArray(comments) || comments.length === 0) return '(no review comments)'
138
+ return comments
139
+ .map((comment) => {
140
+ return `Review comment ${comment.id} by ${comment.user?.login ?? 'unknown'} on ${comment.path ?? 'unknown path'}:${comment.line ?? comment.original_line ?? 'unknown line'} at ${comment.created_at ?? 'unknown time'}:\n\n${comment.body ?? ''}`
141
+ })
142
+ .join('\n\n---\n\n')
143
+ }
144
+
145
+ function stripIterateCommand(body: string): string {
146
+ return body.trim().replace(/^\/iterate\b\s*/i, '').trim() || 'Iterate on this PR using the available feedback.'
147
+ }
148
+
149
+ function parseArgs(argv: string[]): Args {
150
+ const parsed: Record<string, string> = {}
151
+ for (let i = 0; i < argv.length; i += 1) {
152
+ const arg = argv[i]
153
+ if (!arg?.startsWith('--')) continue
154
+ const key = arg.slice(2)
155
+ const value = argv[i + 1]
156
+ if (!value || value.startsWith('--')) throw new Error(`Missing value for ${arg}`)
157
+ parsed[key] = value
158
+ i += 1
159
+ }
160
+
161
+ const command = parsed.command
162
+ if (command !== 'footer' && command !== 'prompt') throw new Error('--command must be footer or prompt')
163
+ if (!parsed.workflow) throw new Error('--workflow is required')
164
+ if (!parsed.memory) throw new Error('--memory is required')
165
+
166
+ return {
167
+ command,
168
+ workflow: parsed.workflow,
169
+ memory: parsed.memory,
170
+ repo: parsed.repo,
171
+ prNumber: parsed['pr-number'],
172
+ commentBody: parsed['comment-body'],
173
+ }
174
+ }