@enderfga/claw-orchestrator 3.3.1 → 3.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -87,9 +87,23 @@ await manager.councilStart("Design and implement an auth system", {
87
87
  });
88
88
  ```
89
89
 
90
+ ### Autoloop (autonomous workspace iteration)
91
+
92
+ Given a git workspace, a `plan.md` (intent + scope), and a `goal.json` (success criteria — scalar metric and/or structural gates), the loop runs `BOOTSTRAP → propose → execute → measure → ratchet → maybe compress` until the goal is met or caps fire. Asymmetric reviewer (separate process, sandboxed cwd) defaults to reset; non-blocking pushes via `openclaw message send` on new-best / plateau / aspirational gate / termination.
93
+
94
+ ```ts
95
+ await manager.autoloopStart({
96
+ workspace: "/path/to/repo",
97
+ plan_path: "/path/to/repo/plan.md",
98
+ goal_path: "/path/to/repo/goal.json",
99
+ });
100
+ ```
101
+
102
+ Resume after process death with `autoloopResume(workspace, taskId)`. SSE event stream at `GET /autoloop/<id>/events`. See [`skills/references/autoloop.md`](./skills/references/autoloop.md) for two worked scenarios (Karpathy-style scalar improvement + paper-review gates).
103
+
90
104
  ### Tool Orchestration
91
105
 
92
- Expose coding sessions as tools so other agents and systems can control them. The runtime registers 35 tools, including:
106
+ Expose coding sessions as tools so other agents and systems can control them. The runtime registers 40 tools, including:
93
107
 
94
108
  ```txt
95
109
  session_start session_send coding_session_status
@@ -97,6 +111,7 @@ session_grep session_compact session_inbox
97
111
  team_send team_list coding_agents_list
98
112
  council_start council_review council_accept
99
113
  ultraplan_start ultrareview_start
114
+ autoloop_start autoloop_resume autoloop_inject
100
115
  ```
101
116
 
102
117
  ---
@@ -0,0 +1,30 @@
1
+ # Autoloop — BOOTSTRAP Phase
2
+
3
+ You are the BOOTSTRAP agent for autoloop task `{{task_id}}`. This phase runs ONCE before the iteration loop begins. If you fail, the loop does not start.
4
+
5
+ ## Your Job
6
+
7
+ 1. Read `tasks/{{task_id}}/plan.md` to understand the user's intent.
8
+ 2. Read `tasks/{{task_id}}/goal.json` to understand what success looks like.
9
+ 3. Verify the workspace is in a runnable state. If `goal.scalar.extract_cmd` exists, run it once and capture the baseline scalar. Run every locked gate `cmd` and record pass/fail.
10
+ 4. Write the **first** `tasks/{{task_id}}/current.md`: a short summary of the workspace's current state and your initial proposal for what to try first.
11
+ 5. **If the task involves deep research** (no scalar, gate-driven goal): propose up to {{max_aspirational}} `aspirational_gates` derived from the user's plan. Append them to `goal.json`'s `aspirational_gates` array. The runner will push these to the user for approval.
12
+ 6. Commit the resulting state on the autoloop branch with message `autoloop(bootstrap): baseline established`.
13
+
14
+ ## Hard Rules
15
+
16
+ - **No code/policy changes in BOOTSTRAP.** You may add files under `tasks/{{task_id}}/` only. Do NOT modify the user's source code in this phase.
17
+ - **If the workspace cannot run** (missing deps, broken scripts), do NOT try to fix it silently. Write a clear failure note to `tasks/{{task_id}}/bootstrap-failure.md` describing what's broken and stop. The loop will abort.
18
+ - **Do not invent gates.** Aspirational gates must trace to specific items in the user's plan. Each one needs a verifiable `cmd`.
19
+ - **No interactive prompts.** Stay non-blocking.
20
+
21
+ ## Output
22
+
23
+ When done, your final message must include:
24
+ - Workspace status: clean / runnable / failed (with reason)
25
+ - Baseline scalar value (if any)
26
+ - Initial gate pass/fail counts
27
+ - Aspirational gates count proposed (if any)
28
+ - One-line summary of `current.md` contents
29
+
30
+ Use tools to do real work. Truthful reporting only — every claim must trace to a tool call.
@@ -0,0 +1,50 @@
1
+ # Autoloop — COMPRESS Phase
2
+
3
+ You are the COMPRESS agent for autoloop task `{{task_id}}`. Run every {{compress_every_k}} iterations. Your job is to fold recent iter logs into `history.md` so PROPOSE has a parseable summary instead of an ever-growing pile of artifacts.
4
+
5
+ ## Read
6
+
7
+ - All `tasks/{{task_id}}/iter/<n>/` directories from iter `{{compress_from}}` to `{{compress_to}}` inclusive
8
+ - Existing `tasks/{{task_id}}/history.md` (may not exist yet)
9
+ - `tasks/{{task_id}}/state.json` — `best` field
10
+
11
+ ## Write
12
+
13
+ Replace `tasks/{{task_id}}/history.md` with new content following this **fixed schema** (PROPOSE relies on these section names — do not rename, do not omit):
14
+
15
+ ```markdown
16
+ # History — autoloop {{task_id}}
17
+
18
+ ## Iters {{compress_from}}–{{compress_to}} (compressed at iter {{compress_to_plus_1}})
19
+
20
+ **Best so far**: <metric> at iter <n> (sha <git_sha>).
21
+
22
+ **Tried and worked**:
23
+ - iter <n>: <one-line description from current.md> → <metric_pre> → <metric_post>
24
+ - ...
25
+
26
+ **Tried and rolled back** (reasons):
27
+ - iter <n>: <one-line description> → <reset reason from ratchet.json>
28
+ - ...
29
+
30
+ **Open hypotheses** (carry forward — things to try next):
31
+ - <hypothesis 1, 1 line>
32
+ - ...
33
+
34
+ **Aspirational gates approved this segment**: <count>
35
+ ```
36
+
37
+ After writing `history.md`, **delete** the per-iter directories `iter/<compress_from>/` through `iter/<compress_to>/` to reclaim disk. Keep the most recent 5 iter dirs intact (don't delete those even if in range).
38
+
39
+ ## Hard Rules
40
+
41
+ - The schema is fixed. If you cannot fill a section truthfully, write `(none this segment)` — do not omit the section header.
42
+ - Do not commit changes outside `history.md` and the deleted `iter/` dirs.
43
+ - Commit message: `autoloop(compress): iters {{compress_from}}-{{compress_to}}`
44
+
45
+ ## Output
46
+
47
+ Report (≤80 words):
48
+ - Iters compressed
49
+ - Lines in new `history.md`
50
+ - Iter dirs deleted
@@ -0,0 +1,57 @@
1
+ # Autoloop — PROPOSE Phase (iter {{iter}})
2
+
3
+ You are the PROPOSE agent for autoloop task `{{task_id}}`. This is iteration `{{iter}}`. Your job is to make ONE incremental change that you believe will improve the metric or pass more gates, then hand off to EXECUTE.
4
+
5
+ ## Read Order (don't skip)
6
+
7
+ 1. `tasks/{{task_id}}/plan.md` — user's stated intent and constraints
8
+ 2. `tasks/{{task_id}}/goal.json` — what counts as success
9
+ 3. `tasks/{{task_id}}/current.md` — current best summary + last suggestion
10
+ 4. `tasks/{{task_id}}/history.md` — what's already been tried, what to avoid (may be empty in early iters)
11
+ 5. `tasks/{{task_id}}/state.json` — current iter, best so far, plateau count
12
+ 6. The **last 2** `tasks/{{task_id}}/iter/*/ratchet.json` files — what RATCHET said about recent attempts. Heed reset reasons.
13
+
14
+ ## Your Change Must
15
+
16
+ - **Be focused.** One hypothesis per iteration. Do not bundle a refactor + a metric tweak. RATCHET will reset bundled changes.
17
+ - **Be neutral on existing gates.** Every locked gate that passed before this iteration must still pass after. Test data and `cmd` scripts are out of bounds — do not modify them.
18
+ - **Be aware of plateau.** If `state.json.plateau_count >= 3`, prefer a more exploratory change (try a different region of the design space rather than incremental tuning).
19
+ - **Respect `plan.md` scope.** If `plan.md` has a section like `## Scope`, `## Constraints`, `## Read-only files`, `## Forbidden paths`, or `## Allowed paths`, those statements are HARD constraints — equivalent to a locked gate failing if you violate them. Specifically:
20
+ - "do not modify X" → treat X as if it were a frozen test file
21
+ - "only change Y/" → all changes must be inside Y/; touching anything else is grounds for RATCHET reset
22
+ - "tunable hyperparameters: A, B, C" → only A, B, C may move; do not touch architecture, data loading, eval code
23
+ - When the constraint is ambiguous, default to the narrower interpretation. RATCHET will reset on plausible scope violations.
24
+
25
+ ## What You May Modify
26
+
27
+ - The user's source code in `{{workspace}}` (anything outside `tasks/{{task_id}}/`)
28
+ - `tasks/{{task_id}}/current.md` (must update with: the change you made + your prediction of the metric direction + your reasoning in ≤200 words)
29
+
30
+ ## What You May NOT Modify
31
+
32
+ - `tasks/{{task_id}}/goal.json` (locked gates and scalar definition are user-controlled)
33
+ - `tasks/{{task_id}}/state.json` (only RATCHET writes the decision; runner writes other fields)
34
+ - `tasks/{{task_id}}/metric.json` / `iter/*/eval.json` (MEASURE writes)
35
+ - Any test fixture, eval data, or gate-check script that the user listed as out-of-bounds in `plan.md`
36
+ - `tasks/{{task_id}}/regression*.md` if present (frozen reference data)
37
+
38
+ ## Aspirational Gates
39
+
40
+ If you believe the goal needs an additional gate to be considered "done" (a coverage gap you discovered), append a candidate to `goal.json.aspirational_gates`. Cap: `state.json.pending_aspirational_count` must not exceed `goal.termination.max_pending_aspirational` after your addition. The runner will push it to the user; do not block waiting for approval.
41
+
42
+ ## Commit
43
+
44
+ Stage your changes (code + `current.md`) and commit on the autoloop branch with:
45
+ ```
46
+ autoloop(iter-{{iter}}): <one-line description of the hypothesis>
47
+ ```
48
+
49
+ EXECUTE will then run the workspace and gates against your change. RATCHET will decide commit-or-reset.
50
+
51
+ ## Output
52
+
53
+ Report (≤150 words):
54
+ - The single hypothesis you tested this iter
55
+ - Files changed (paths only)
56
+ - Predicted metric direction + your confidence (low/med/high)
57
+ - Whether you added an aspirational gate
@@ -0,0 +1,56 @@
1
+ # Autoloop — RATCHET Phase (iter {{iter}})
2
+
3
+ You are the RATCHET reviewer for autoloop task `{{task_id}}`. **Your default verdict is `reset`.** A `commit` requires positive evidence of improvement that you have personally verified. You do NOT have access to the source code or workspace — by design. You see only the artifacts piped into this prompt, and that is all you should base your decision on.
4
+
5
+ ## What You See
6
+
7
+ - `goal.json` — locked gates and scalar definition
8
+ - `eval.json` (this iter) — gate results + scalar (if any)
9
+ - `metric.json` — full history of metric points
10
+ - `current.md` (this iter) — PROPOSE's claimed change + prediction
11
+ - `state.json.best` — incumbent best to beat
12
+ - `last_ratchet.json` (previous iter) — your prior decision, for continuity
13
+
14
+ ## What You Must NOT Trust
15
+
16
+ - **The PROPOSE agent's prediction.** It is biased toward optimism.
17
+ - **The scalar value alone.** Reward hacking is real (see Anthropic AAR — agents have flipped test labels in the past). If the scalar moved more than the noise_floor in one iter, suspect manipulation **unless** the change in `current.md` plausibly explains it.
18
+ - **A passing gate is not the same as a working feature.** Check that the gate `cmd` in `goal.json` is actually probing what it claims to probe, given what `current.md` says was changed.
19
+
20
+ ## Decision Rules (apply in order)
21
+
22
+ 1. **Gate regression**: any locked gate that previously passed but now fails → `reset`. No exceptions.
23
+ 2. **Scope violation**: if `current.md` describes changes to files / modules / hyperparameters that `plan.md`'s Scope / Constraints / Read-only / Forbidden / Allowed paths sections would forbid → `reset`. Default to the narrower interpretation when ambiguous.
24
+ 3. **Aspirational-only progress**: if all locked gates are unchanged and only aspirational gates moved → `reset` (locked gates are the source of truth; aspirational ones don't ratchet).
25
+ 4. **No improvement beyond noise**: if `isImprovement(eval.scalar_or_gate_completion, state.best.metric, goal)` is false → `reset`.
26
+ 5. **Plausibility check**: if the change in `current.md` could not, by your reading, plausibly cause the metric move → `reset` and flag possible reward hacking in `reason`.
27
+ 6. **Otherwise**: `commit`.
28
+
29
+ ## When To Push the User (`push_user`)
30
+
31
+ - `kind: "new_best"` — when committing AND this is a new best (strictly better than `state.best.metric`).
32
+ - `kind: "plateau"` — when resetting AND `state.plateau_count + 1 >= goal.termination.plateau_iters`. Ask: "continue / redirect / stop?"
33
+ - `kind: "unsure_no_metric"` — when goal has no scalar and your gate-based judgment is genuinely ambiguous (rare; default to `reset`).
34
+ - `kind: "aspirational_proposed"` — never set this yourself; the runner sets it when PROPOSE adds an aspirational gate.
35
+
36
+ ## Output Format (strict JSON, no other text)
37
+
38
+ You MUST output exactly one valid JSON object. No prose before or after. No code fence. The runner parses your stdout/last-text-block as JSON.
39
+
40
+ ```json
41
+ {
42
+ "decision": "commit" | "reset",
43
+ "reason": "<one or two sentences explaining the decision, citing specific eval.json fields>",
44
+ "push_user": null | {
45
+ "kind": "new_best" | "plateau" | "unsure_no_metric",
46
+ "text": "<message to push to user>"
47
+ }
48
+ }
49
+ ```
50
+
51
+ If you cannot decide due to malformed inputs, output:
52
+ ```json
53
+ { "decision": "reset", "reason": "malformed inputs: <what was wrong>", "push_user": null }
54
+ ```
55
+
56
+ Be terse. The point of RATCHET is to not waffle.
@@ -0,0 +1,218 @@
1
+ /**
2
+ * Types for the autoloop feature.
3
+ *
4
+ * Contracts are documented in tasks/autoloop.md. Schemas here are intentionally
5
+ * narrow — additions belong in the design doc first.
6
+ */
7
+ import type { EngineType } from './types.js';
8
+ export type AutoloopPhase = 'BOOTSTRAP' | 'PROPOSE' | 'EXECUTE' | 'MEASURE' | 'RATCHET' | 'COMPRESS' | 'IDLE' | 'TERMINATED';
9
+ export type AutoloopStatus = 'starting' | 'running' | 'paused' | 'completed' | 'error' | 'stopped';
10
+ export interface ScalarSpec {
11
+ /** Name of the metric for display / logs */
12
+ name: string;
13
+ /** Optimisation direction */
14
+ direction: 'min' | 'max';
15
+ /**
16
+ * Shell command that prints the scalar value to stdout (one number).
17
+ * Run by EXECUTE in the workspace. Must be deterministic.
18
+ */
19
+ extract_cmd: string;
20
+ /**
21
+ * Hard wall-clock cap on extract_cmd in seconds. Default 600 (10 min).
22
+ * For long ML evals (training + measure), set this to your real upper
23
+ * bound — e.g. 14400 for a 4-hour training run.
24
+ */
25
+ extract_timeout_sec?: number;
26
+ /** Stop when scalar reaches this. For 'min' direction, stop when ≤ target. */
27
+ target?: number;
28
+ /** Changes within ±noise_floor are not considered improvements. Default 0. */
29
+ noise_floor?: number;
30
+ }
31
+ export interface GateSpec {
32
+ /** Stable identifier (used in state.json gate accounting) */
33
+ name: string;
34
+ /**
35
+ * Shell command. Exit code 0 = pass, non-zero = fail.
36
+ * Time-limited (see GateSpec.timeout_sec).
37
+ */
38
+ cmd: string;
39
+ /**
40
+ * Pass condition. Currently only 'exit-0' is supported.
41
+ * Future: 'stdout-matches:<regex>', 'json-field:<path>=<value>'.
42
+ */
43
+ must: 'exit-0';
44
+ /** Per-gate timeout in seconds. Default 300. */
45
+ timeout_sec?: number;
46
+ }
47
+ export interface TerminationSpec {
48
+ scalar_target_hit?: boolean;
49
+ max_iters: number;
50
+ /** Plateau triggers async push but loop continues unless user stops. */
51
+ plateau_iters: number;
52
+ max_cost_usd: number;
53
+ /** Cap on un-locked aspirational gates (agent must rotate before adding more). */
54
+ max_pending_aspirational: number;
55
+ }
56
+ export interface GoalSpec {
57
+ /**
58
+ * Optional: when null/absent, the loop has no faithful scalar; goal_completion
59
+ * becomes the de-facto scalar (= locked_gates_passed / locked_gates_total).
60
+ */
61
+ scalar?: ScalarSpec;
62
+ /** Locked gates (user-approved). Hard requirements; all must pass for ratchet. */
63
+ gates: GateSpec[];
64
+ /**
65
+ * Aspirational gates proposed by agent during BOOTSTRAP or mid-loop.
66
+ * Don't count toward goal_completion until user moves them to `gates`.
67
+ * Capped by termination.max_pending_aspirational.
68
+ */
69
+ aspirational_gates?: GateSpec[];
70
+ termination: TerminationSpec;
71
+ }
72
+ export interface BestPoint {
73
+ iter: number;
74
+ metric: number;
75
+ git_sha: string;
76
+ gate_completion: number;
77
+ }
78
+ export interface MetricPoint {
79
+ iter: number;
80
+ metric: number;
81
+ gate_completion: number;
82
+ }
83
+ export interface StateTree {
84
+ /** For serial v1: always points to last committed iter. */
85
+ parent_iter: number | null;
86
+ /** For population (v2): list of forks. v1 always [iter] or []. */
87
+ children_iters: number[];
88
+ }
89
+ export type RatchetDecision = 'commit' | 'reset' | 'pending';
90
+ export interface AutoloopState {
91
+ task_id: string;
92
+ branch: string;
93
+ phase: AutoloopPhase;
94
+ status: AutoloopStatus;
95
+ iter: number;
96
+ started_at: string;
97
+ best: BestPoint | null;
98
+ last_metric: MetricPoint | null;
99
+ plateau_count: number;
100
+ /** Set by RATCHET only. PROPOSE / MEASURE may not write this. */
101
+ decision: RatchetDecision | null;
102
+ decision_reason: string | null;
103
+ /**
104
+ * Git sha of the BOOTSTRAP commit. Used as the floor for git reset when no
105
+ * `best` exists yet (otherwise we'd reset to HEAD~1, which after a failed
106
+ * propose-then-resume can be the wrong target).
107
+ */
108
+ bootstrap_sha: string | null;
109
+ tree: StateTree;
110
+ termination: {
111
+ fired: boolean;
112
+ reason: string | null;
113
+ };
114
+ cost_usd_so_far: number;
115
+ /** Pushed-but-not-yet-locked aspirational gates the agent has proposed. */
116
+ pending_aspirational_count: number;
117
+ }
118
+ export interface MetricHistoryEntry {
119
+ iter: number;
120
+ ts: string;
121
+ metric: number;
122
+ gate_completion: number;
123
+ phase_at_record: AutoloopPhase;
124
+ git_sha_pre?: string;
125
+ git_sha_post?: string;
126
+ }
127
+ export interface GateResult {
128
+ name: string;
129
+ passed: boolean;
130
+ exit_code: number;
131
+ duration_ms: number;
132
+ /** Truncated stdout/stderr tail for diagnostic context. */
133
+ output_tail: string;
134
+ }
135
+ export interface EvalOutput {
136
+ iter: number;
137
+ ts: string;
138
+ /** All locked-gate results, in goal.json order. */
139
+ gates: GateResult[];
140
+ /** null when no scalar in goal.json or extract_cmd failed. */
141
+ scalar: number | null;
142
+ /** Computed = (gates that passed) / (locked gates total). */
143
+ gate_completion: number;
144
+ /** True if every locked gate passed. */
145
+ all_gates_passed: boolean;
146
+ }
147
+ export interface RatchetOutput {
148
+ iter: number;
149
+ decision: RatchetDecision;
150
+ reason: string;
151
+ /** True if RATCHET wants the runner to push the user (e.g. unsure / new-best / plateau). */
152
+ push_user?: {
153
+ kind: 'new_best' | 'plateau' | 'unsure_no_metric' | 'aspirational_proposed';
154
+ text: string;
155
+ };
156
+ }
157
+ export type PushKind = 'bootstrap_aspirational' | 'new_best' | 'plateau' | 'unsure_no_metric' | 'aspirational_proposed' | 'termination' | 'hard_error';
158
+ export interface PushEvent {
159
+ kind: PushKind;
160
+ text: string;
161
+ task_id: string;
162
+ iter: number;
163
+ ts: string;
164
+ }
165
+ export interface AutoloopConfig {
166
+ /** Workspace path — must be a git repo. tasks/<id>/ will be created here. */
167
+ workspace: string;
168
+ /** Path to plan.md (will be copied into tasks/<id>/plan.md). */
169
+ plan_path: string;
170
+ /** Path to goal.json (will be copied + validated into tasks/<id>/goal.json). */
171
+ goal_path: string;
172
+ /** Override for the task id; defaults to a timestamped slug. */
173
+ task_id?: string;
174
+ /** Engine for PROPOSE/BOOTSTRAP/COMPRESS. Default 'claude'. */
175
+ propose_engine?: EngineType;
176
+ /** Model for PROPOSE/BOOTSTRAP/COMPRESS. Default 'opus'. */
177
+ propose_model?: string;
178
+ /** Engine for RATCHET. Default 'claude'. */
179
+ ratchet_engine?: EngineType;
180
+ /** Model for RATCHET. Default 'opus'. */
181
+ ratchet_model?: string;
182
+ /** How often to run COMPRESS. Default 10. */
183
+ compress_every_k?: number;
184
+ /** Per-iter wall clock for PROPOSE/EXECUTE phases (ms). Default 600_000 (10 min). */
185
+ per_iter_timeout_ms?: number;
186
+ /** Push hook command (defaults to `openclaw message send`). Set to null to disable pushes. */
187
+ push_cmd?: string | null;
188
+ }
189
+ export interface AutoloopHandle {
190
+ id: string;
191
+ status: AutoloopStatus;
192
+ task_dir: string;
193
+ started_at: string;
194
+ ended_at?: string;
195
+ /** Last-known phase from state.json (cheap to read). */
196
+ current_phase?: AutoloopPhase;
197
+ current_iter?: number;
198
+ best_metric?: number;
199
+ error?: string;
200
+ }
201
+ export declare class GoalSpecError extends Error {
202
+ constructor(message: string);
203
+ }
204
+ export declare function validateGoalSpec(raw: unknown): GoalSpec;
205
+ /**
206
+ * Derive the metric value used for ratcheting from an EvalOutput.
207
+ * Rule:
208
+ * - If goal has a scalar AND eval produced one, use the scalar.
209
+ * - Otherwise, use gate_completion (∈ [0, 1]).
210
+ */
211
+ export declare function deriveMetric(eval_out: EvalOutput, goal: GoalSpec): number;
212
+ /**
213
+ * Did `candidate` improve over `incumbent` for this goal? Honours noise_floor.
214
+ * If no scalar, "improve" means strictly higher gate_completion.
215
+ */
216
+ export declare function isImprovement(candidate: number, incumbent: number | null, goal: GoalSpec): boolean;
217
+ /** Has the scalar/gate target been hit? */
218
+ export declare function isTargetReached(metric: number, gate_completion: number, goal: GoalSpec): boolean;
@@ -0,0 +1,130 @@
1
+ /**
2
+ * Types for the autoloop feature.
3
+ *
4
+ * Contracts are documented in tasks/autoloop.md. Schemas here are intentionally
5
+ * narrow — additions belong in the design doc first.
6
+ */
7
+ // ─── Validation Helpers ─────────────────────────────────────────────────────
8
+ export class GoalSpecError extends Error {
9
+ constructor(message) {
10
+ super(`goal.json: ${message}`);
11
+ this.name = 'GoalSpecError';
12
+ }
13
+ }
14
+ export function validateGoalSpec(raw) {
15
+ if (!raw || typeof raw !== 'object')
16
+ throw new GoalSpecError('must be an object');
17
+ const o = raw;
18
+ if (!Array.isArray(o.gates))
19
+ throw new GoalSpecError('`gates` must be an array');
20
+ const gates = o.gates.map((g, i) => validateGate(g, `gates[${i}]`));
21
+ let scalar;
22
+ if (o.scalar != null)
23
+ scalar = validateScalar(o.scalar);
24
+ let aspirational_gates;
25
+ if (Array.isArray(o.aspirational_gates)) {
26
+ aspirational_gates = o.aspirational_gates.map((g, i) => validateGate(g, `aspirational_gates[${i}]`));
27
+ }
28
+ if (!o.termination || typeof o.termination !== 'object') {
29
+ throw new GoalSpecError('`termination` is required');
30
+ }
31
+ const termination = validateTermination(o.termination);
32
+ // Cross-check: if no scalar, gates must be non-empty (otherwise the loop has nothing to ratchet against)
33
+ if (!scalar && gates.length === 0) {
34
+ throw new GoalSpecError('must have at least one of `scalar` or `gates`');
35
+ }
36
+ return { scalar, gates, aspirational_gates, termination };
37
+ }
38
+ function validateScalar(raw) {
39
+ const o = raw;
40
+ if (typeof o.name !== 'string')
41
+ throw new GoalSpecError('scalar.name must be string');
42
+ if (o.direction !== 'min' && o.direction !== 'max') {
43
+ throw new GoalSpecError('scalar.direction must be "min" or "max"');
44
+ }
45
+ if (typeof o.extract_cmd !== 'string' || !o.extract_cmd.trim()) {
46
+ throw new GoalSpecError('scalar.extract_cmd must be a non-empty string');
47
+ }
48
+ return {
49
+ name: o.name,
50
+ direction: o.direction,
51
+ extract_cmd: o.extract_cmd,
52
+ extract_timeout_sec: typeof o.extract_timeout_sec === 'number' && o.extract_timeout_sec > 0 ? o.extract_timeout_sec : 600,
53
+ target: typeof o.target === 'number' ? o.target : undefined,
54
+ noise_floor: typeof o.noise_floor === 'number' ? o.noise_floor : 0,
55
+ };
56
+ }
57
+ function validateGate(raw, ctx) {
58
+ const o = raw;
59
+ if (typeof o.name !== 'string')
60
+ throw new GoalSpecError(`${ctx}.name must be string`);
61
+ if (typeof o.cmd !== 'string' || !o.cmd.trim()) {
62
+ throw new GoalSpecError(`${ctx}.cmd must be a non-empty string`);
63
+ }
64
+ if (o.must !== 'exit-0') {
65
+ throw new GoalSpecError(`${ctx}.must must be "exit-0" (other modes not yet supported)`);
66
+ }
67
+ return {
68
+ name: o.name,
69
+ cmd: o.cmd,
70
+ must: 'exit-0',
71
+ timeout_sec: typeof o.timeout_sec === 'number' && o.timeout_sec > 0 ? o.timeout_sec : 300,
72
+ };
73
+ }
74
+ function validateTermination(raw) {
75
+ const o = raw;
76
+ const num = (k, def) => {
77
+ if (typeof o[k] === 'number' && Number.isFinite(o[k]))
78
+ return o[k];
79
+ if (def !== undefined)
80
+ return def;
81
+ throw new GoalSpecError(`termination.${k} must be a number`);
82
+ };
83
+ return {
84
+ scalar_target_hit: o.scalar_target_hit !== false,
85
+ max_iters: num('max_iters', 200),
86
+ plateau_iters: num('plateau_iters', 10),
87
+ max_cost_usd: num('max_cost_usd', 200),
88
+ max_pending_aspirational: num('max_pending_aspirational', 5),
89
+ };
90
+ }
91
+ /**
92
+ * Derive the metric value used for ratcheting from an EvalOutput.
93
+ * Rule:
94
+ * - If goal has a scalar AND eval produced one, use the scalar.
95
+ * - Otherwise, use gate_completion (∈ [0, 1]).
96
+ */
97
+ export function deriveMetric(eval_out, goal) {
98
+ if (goal.scalar && eval_out.scalar != null)
99
+ return eval_out.scalar;
100
+ return eval_out.gate_completion;
101
+ }
102
+ /**
103
+ * Did `candidate` improve over `incumbent` for this goal? Honours noise_floor.
104
+ * If no scalar, "improve" means strictly higher gate_completion.
105
+ */
106
+ export function isImprovement(candidate, incumbent, goal) {
107
+ if (incumbent === null)
108
+ return true;
109
+ if (goal.scalar) {
110
+ const noise = goal.scalar.noise_floor ?? 0;
111
+ if (goal.scalar.direction === 'min')
112
+ return candidate < incumbent - noise;
113
+ return candidate > incumbent + noise;
114
+ }
115
+ // No scalar: gate_completion, higher is better, no noise floor.
116
+ return candidate > incumbent;
117
+ }
118
+ /** Has the scalar/gate target been hit? */
119
+ export function isTargetReached(metric, gate_completion, goal) {
120
+ if (!goal.termination.scalar_target_hit)
121
+ return false;
122
+ if (goal.scalar?.target != null) {
123
+ if (goal.scalar.direction === 'min')
124
+ return metric <= goal.scalar.target;
125
+ return metric >= goal.scalar.target;
126
+ }
127
+ // No scalar target set; "done" means all gates passing.
128
+ return gate_completion >= 1.0;
129
+ }
130
+ //# sourceMappingURL=autoloop-types.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"autoloop-types.js","sourceRoot":"","sources":["../../src/autoloop-types.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AA2PH,+EAA+E;AAE/E,MAAM,OAAO,aAAc,SAAQ,KAAK;IACtC,YAAY,OAAe;QACzB,KAAK,CAAC,cAAc,OAAO,EAAE,CAAC,CAAC;QAC/B,IAAI,CAAC,IAAI,GAAG,eAAe,CAAC;IAC9B,CAAC;CACF;AAED,MAAM,UAAU,gBAAgB,CAAC,GAAY;IAC3C,IAAI,CAAC,GAAG,IAAI,OAAO,GAAG,KAAK,QAAQ;QAAE,MAAM,IAAI,aAAa,CAAC,mBAAmB,CAAC,CAAC;IAClF,MAAM,CAAC,GAAG,GAA8B,CAAC;IAEzC,IAAI,CAAC,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,KAAK,CAAC;QAAE,MAAM,IAAI,aAAa,CAAC,0BAA0B,CAAC,CAAC;IACjF,MAAM,KAAK,GAAG,CAAC,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,YAAY,CAAC,CAAC,EAAE,SAAS,CAAC,GAAG,CAAC,CAAC,CAAC;IAEpE,IAAI,MAA8B,CAAC;IACnC,IAAI,CAAC,CAAC,MAAM,IAAI,IAAI;QAAE,MAAM,GAAG,cAAc,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC;IAExD,IAAI,kBAA0C,CAAC;IAC/C,IAAI,KAAK,CAAC,OAAO,CAAC,CAAC,CAAC,kBAAkB,CAAC,EAAE,CAAC;QACxC,kBAAkB,GAAG,CAAC,CAAC,kBAAkB,CAAC,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,EAAE,EAAE,CAAC,YAAY,CAAC,CAAC,EAAE,sBAAsB,CAAC,GAAG,CAAC,CAAC,CAAC;IACvG,CAAC;IAED,IAAI,CAAC,CAAC,CAAC,WAAW,IAAI,OAAO,CAAC,CAAC,WAAW,KAAK,QAAQ,EAAE,CAAC;QACxD,MAAM,IAAI,aAAa,CAAC,2BAA2B,CAAC,CAAC;IACvD,CAAC;IACD,MAAM,WAAW,GAAG,mBAAmB,CAAC,CAAC,CAAC,WAAW,CAAC,CAAC;IAEvD,yGAAyG;IACzG,IAAI,CAAC,MAAM,IAAI,KAAK,CAAC,MAAM,KAAK,CAAC,EAAE,CAAC;QAClC,MAAM,IAAI,aAAa,CAAC,+CAA+C,CAAC,CAAC;IAC3E,CAAC;IAED,OAAO,EAAE,MAAM,EAAE,KAAK,EAAE,kBAAkB,EAAE,WAAW,EAAE,CAAC;AAC5D,CAAC;AAED,SAAS,cAAc,CAAC,GAAY;IAClC,MAAM,CAAC,GAAG,GAA8B,CAAC;IACzC,IAAI,OAAO,CAAC,CAAC,IAAI,KAAK,QAAQ;QAAE,MAAM,IAAI,aAAa,CAAC,4BAA4B,CAAC,CAAC;IACtF,IAAI,CAAC,CAAC,SAAS,KAAK,KAAK,IAAI,CAAC,CAAC,SAAS,KAAK,KAAK,EAAE,CAAC;QACnD,MAAM,IAAI,aAAa,CAAC,yCAAyC,CAAC,CAAC;IACrE,CAAC;IACD,IAAI,OAAO,CAAC,CAAC,WAAW,KAAK,QAAQ,IAAI,CAAC,CAAC,CAAC,WAAW,CAAC,IAAI,EAAE,EAAE,CAAC;QAC/D,MAAM,IAAI,aAAa,CAAC,+CAA+C,CAAC,CAAC;IAC3E,CAAC;IACD,OAAO;QACL,IAAI,EAAE,CAAC,CAAC,IAAI;QACZ,SAAS,EAAE,CAAC,CAAC,SAAS;QACtB,WAAW,EAAE,CAAC,CAAC,WAAW;QAC1B,mBAAmB,EACjB,OAAO,CAAC,CAAC,mBAAmB,KAAK,QAAQ,IAAI,CAAC,CAAC,mBAAmB,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,mBAAmB,CAAC,CAAC,CAAC,GAAG;QACtG,MAAM,EAAE,OAAO,CAAC,CAAC,MAAM,KAAK,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAC,MAAM,CAAC,CAAC,CAAC,SAAS;QAC3D,WAAW,EAAE,OAAO,CAAC,CAAC,WAAW,KAAK,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAC,WAAW,CAAC,CAAC,CAAC,CAAC;KACnE,CAAC;AACJ,CAAC;AAED,SAAS,YAAY,CAAC,GAAY,EAAE,GAAW;IAC7C,MAAM,CAAC,GAAG,GAA8B,CAAC;IACzC,IAAI,OAAO,CAAC,CAAC,IAAI,KAAK,QAAQ;QAAE,MAAM,IAAI,aAAa,CAAC,GAAG,GAAG,sBAAsB,CAAC,CAAC;IACtF,IAAI,OAAO,CAAC,CAAC,GAAG,KAAK,QAAQ,IAAI,CAAC,CAAC,CAAC,GAAG,CAAC,IAAI,EAAE,EAAE,CAAC;QAC/C,MAAM,IAAI,aAAa,CAAC,GAAG,GAAG,iCAAiC,CAAC,CAAC;IACnE,CAAC;IACD,IAAI,CAAC,CAAC,IAAI,KAAK,QAAQ,EAAE,CAAC;QACxB,MAAM,IAAI,aAAa,CAAC,GAAG,GAAG,wDAAwD,CAAC,CAAC;IAC1F,CAAC;IACD,OAAO;QACL,IAAI,EAAE,CAAC,CAAC,IAAI;QACZ,GAAG,EAAE,CAAC,CAAC,GAAG;QACV,IAAI,EAAE,QAAQ;QACd,WAAW,EAAE,OAAO,CAAC,CAAC,WAAW,KAAK,QAAQ,IAAI,CAAC,CAAC,WAAW,GAAG,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,WAAW,CAAC,CAAC,CAAC,GAAG;KAC1F,CAAC;AACJ,CAAC;AAED,SAAS,mBAAmB,CAAC,GAAY;IACvC,MAAM,CAAC,GAAG,GAA8B,CAAC;IACzC,MAAM,GAAG,GAAG,CAAC,CAAS,EAAE,GAAY,EAAU,EAAE;QAC9C,IAAI,OAAO,CAAC,CAAC,CAAC,CAAC,KAAK,QAAQ,IAAI,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,CAAC,CAAW,CAAC;YAAE,OAAO,CAAC,CAAC,CAAC,CAAW,CAAC;QACvF,IAAI,GAAG,KAAK,SAAS;YAAE,OAAO,GAAG,CAAC;QAClC,MAAM,IAAI,aAAa,CAAC,eAAe,CAAC,mBAAmB,CAAC,CAAC;IAC/D,CAAC,CAAC;IACF,OAAO;QACL,iBAAiB,EAAE,CAAC,CAAC,iBAAiB,KAAK,KAAK;QAChD,SAAS,EAAE,GAAG,CAAC,WAAW,EAAE,GAAG,CAAC;QAChC,aAAa,EAAE,GAAG,CAAC,eAAe,EAAE,EAAE,CAAC;QACvC,YAAY,EAAE,GAAG,CAAC,cAAc,EAAE,GAAG,CAAC;QACtC,wBAAwB,EAAE,GAAG,CAAC,0BAA0B,EAAE,CAAC,CAAC;KAC7D,CAAC;AACJ,CAAC;AAED;;;;;GAKG;AACH,MAAM,UAAU,YAAY,CAAC,QAAoB,EAAE,IAAc;IAC/D,IAAI,IAAI,CAAC,MAAM,IAAI,QAAQ,CAAC,MAAM,IAAI,IAAI;QAAE,OAAO,QAAQ,CAAC,MAAM,CAAC;IACnE,OAAO,QAAQ,CAAC,eAAe,CAAC;AAClC,CAAC;AAED;;;GAGG;AACH,MAAM,UAAU,aAAa,CAAC,SAAiB,EAAE,SAAwB,EAAE,IAAc;IACvF,IAAI,SAAS,KAAK,IAAI;QAAE,OAAO,IAAI,CAAC;IACpC,IAAI,IAAI,CAAC,MAAM,EAAE,CAAC;QAChB,MAAM,KAAK,GAAG,IAAI,CAAC,MAAM,CAAC,WAAW,IAAI,CAAC,CAAC;QAC3C,IAAI,IAAI,CAAC,MAAM,CAAC,SAAS,KAAK,KAAK;YAAE,OAAO,SAAS,GAAG,SAAS,GAAG,KAAK,CAAC;QAC1E,OAAO,SAAS,GAAG,SAAS,GAAG,KAAK,CAAC;IACvC,CAAC;IACD,gEAAgE;IAChE,OAAO,SAAS,GAAG,SAAS,CAAC;AAC/B,CAAC;AAED,2CAA2C;AAC3C,MAAM,UAAU,eAAe,CAAC,MAAc,EAAE,eAAuB,EAAE,IAAc;IACrF,IAAI,CAAC,IAAI,CAAC,WAAW,CAAC,iBAAiB;QAAE,OAAO,KAAK,CAAC;IACtD,IAAI,IAAI,CAAC,MAAM,EAAE,MAAM,IAAI,IAAI,EAAE,CAAC;QAChC,IAAI,IAAI,CAAC,MAAM,CAAC,SAAS,KAAK,KAAK;YAAE,OAAO,MAAM,IAAI,IAAI,CAAC,MAAM,CAAC,MAAM,CAAC;QACzE,OAAO,MAAM,IAAI,IAAI,CAAC,MAAM,CAAC,MAAM,CAAC;IACtC,CAAC;IACD,wDAAwD;IACxD,OAAO,eAAe,IAAI,GAAG,CAAC;AAChC,CAAC"}
@@ -0,0 +1,70 @@
1
+ /**
2
+ * Autoloop driver — phase machine for autonomous workspace iteration.
3
+ *
4
+ * Contract: tasks/autoloop.md. This file implements the runner; phase prompts
5
+ * live in configs/autoloop-*-prompt.md.
6
+ *
7
+ * Lifecycle (one runner per task):
8
+ * start() → BOOTSTRAP → loop { PROPOSE → EXECUTE → MEASURE → RATCHET → maybe COMPRESS } → TERMINATED
9
+ *
10
+ * Termination triggers: scalar target hit, max_iters, max_cost_usd, hard error,
11
+ * explicit stop(). Plateau pushes the user but does NOT halt the loop (per
12
+ * design constraint C9 — proactivity).
13
+ */
14
+ import { EventEmitter } from 'node:events';
15
+ import type { SessionManager } from './session-manager.js';
16
+ import type { Logger } from './logger.js';
17
+ import type { EngineType } from './types.js';
18
+ import type { AutoloopConfig, AutoloopHandle } from './autoloop-types.js';
19
+ export declare class AutoloopRunner extends EventEmitter {
20
+ readonly id: string;
21
+ readonly config: AutoloopConfig;
22
+ readonly taskDir: string;
23
+ readonly branch: string;
24
+ private state;
25
+ private goal;
26
+ private stopRequested;
27
+ private startedAt;
28
+ private endedAt?;
29
+ private status;
30
+ private errorMsg?;
31
+ private readonly manager;
32
+ private readonly logger;
33
+ constructor(manager: SessionManager, config: AutoloopConfig, logger?: Logger);
34
+ /** Start the loop. Resolves after BOOTSTRAP; iteration runs in the background. */
35
+ start(): Promise<void>;
36
+ /**
37
+ * Resume an interrupted autoloop from on-disk state. Skips BOOTSTRAP and
38
+ * jumps straight to the iteration loop. Resets the workspace to the last
39
+ * known best (or the BOOTSTRAP baseline if no best yet) so any half-finished
40
+ * PROPOSE that crashed mid-flight is discarded.
41
+ */
42
+ resume(): Promise<void>;
43
+ /** Ask the loop to halt at the next phase boundary. Resolves when actually stopped. */
44
+ stop(): Promise<void>;
45
+ /** Inject a hint that the next PROPOSE will read. */
46
+ inject(text: string): void;
47
+ /** Cheap status snapshot from in-memory state. */
48
+ handle(): AutoloopHandle;
49
+ private prepareTaskDir;
50
+ private loadAndValidateGoal;
51
+ private initialState;
52
+ private ensureBranch;
53
+ private runBootstrap;
54
+ private runLoop;
55
+ private runPropose;
56
+ private runExecute;
57
+ private runMeasure;
58
+ private runRatchet;
59
+ private runCompress;
60
+ private persistState;
61
+ private gitSha;
62
+ private gitReset;
63
+ private readMetricHistory;
64
+ private readLastRatchet;
65
+ private tryRead;
66
+ private push;
67
+ private terminate;
68
+ private fail;
69
+ }
70
+ export type { EngineType };