jonah-fleet 1.6.0 → 1.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +130 -0
- package/README.md +25 -2
- package/dist/commands/daemon.d.ts.map +1 -1
- package/dist/commands/init.d.ts +3 -0
- package/dist/commands/init.d.ts.map +1 -1
- package/dist/commands/labels.d.ts +11 -0
- package/dist/commands/labels.d.ts.map +1 -0
- package/dist/commands/status.d.ts.map +1 -1
- package/dist/commands/sync.d.ts.map +1 -1
- package/dist/commands/telemetry.d.ts.map +1 -1
- package/dist/index.js +2527 -519
- package/dist/lib/daemon-keys.d.ts +103 -0
- package/dist/lib/daemon-keys.d.ts.map +1 -0
- package/dist/lib/daemon.d.ts +8 -2
- package/dist/lib/daemon.d.ts.map +1 -1
- package/dist/lib/fleet-query.d.ts.map +1 -1
- package/dist/lib/labels.d.ts +68 -0
- package/dist/lib/labels.d.ts.map +1 -0
- package/dist/lib/manifest.d.ts +6 -0
- package/dist/lib/manifest.d.ts.map +1 -1
- package/dist/lib/presets.d.ts +52 -1
- package/dist/lib/presets.d.ts.map +1 -1
- package/dist/lib/runner.d.ts +114 -0
- package/dist/lib/runner.d.ts.map +1 -1
- package/dist/lib/telemetry.d.ts.map +1 -1
- package/dist/lib/terminal-card.d.ts +24 -1
- package/dist/lib/terminal-card.d.ts.map +1 -1
- package/package.json +1 -1
- package/schema.json +74 -1
- package/templates/prompts/ORCHESTRATION.md +84 -41
- package/templates/prompts/_prompt-template.md +10 -13
- package/templates/prompts/analytics-review.md +4 -2
- package/templates/prompts/autowork.md +76 -35
- package/templates/prompts/dependency-update-security-check.md +4 -2
- package/templates/prompts/design-review.md +125 -0
- package/templates/prompts/issues-housekeeping.md +10 -8
- package/templates/prompts/optimizer.md +31 -12
- package/templates/prompts/peer-review.md +42 -7
- package/templates/prompts/product-planning.md +14 -10
- package/templates/workflows/autowork-cron.yml +137 -38
- package/templates/workflows/dependency-check-cron.yml +127 -2
- package/templates/workflows/design-review-cron.yml +221 -0
- package/templates/workflows/issues-housekeeping-cron.yml +127 -2
- package/templates/workflows/prompt-optimizer-cron.yml +134 -9
- package/templates/workflows/sync-fleet.yml +4 -0
- package/templates/workflows/trigger-autowork-manual.yml +154 -11
- package/templates/workflows/trigger-autowork-on-bug.yml +177 -9
- package/templates/workflows/trigger-autowork-on-merge.yml +178 -9
- package/templates/workflows/trigger-review-routine.yml +162 -22
package/schema.json
CHANGED
|
@@ -26,7 +26,8 @@
|
|
|
26
26
|
"issues-housekeeping": { "type": "boolean" },
|
|
27
27
|
"dependency-update-security-check": { "type": "boolean" },
|
|
28
28
|
"product-planning": { "type": "boolean" },
|
|
29
|
-
"analytics-review": { "type": "boolean" }
|
|
29
|
+
"analytics-review": { "type": "boolean" },
|
|
30
|
+
"design-review": { "type": "boolean" }
|
|
30
31
|
},
|
|
31
32
|
"additionalProperties": false,
|
|
32
33
|
"description": "Granular routine toggles"
|
|
@@ -36,6 +37,65 @@
|
|
|
36
37
|
"items": { "type": "string" },
|
|
37
38
|
"description": "List of core engineering skills to install/sync"
|
|
38
39
|
},
|
|
40
|
+
"models": {
|
|
41
|
+
"type": "object",
|
|
42
|
+
"properties": {
|
|
43
|
+
"default": { "type": "string", "description": "Default model used across routines" },
|
|
44
|
+
"autowork": { "type": "string", "description": "Target model for autowork routine" },
|
|
45
|
+
"peer-review": { "type": "string", "description": "Target model for peer-review routine" },
|
|
46
|
+
"optimizer": { "type": "string", "description": "Target model for optimizer routine" },
|
|
47
|
+
"issues-housekeeping": { "type": "string", "description": "Target model for issues-housekeeping routine" },
|
|
48
|
+
"dependency-update-security-check": { "type": "string", "description": "Target model for dependency-update-security-check routine" },
|
|
49
|
+
"product-planning": { "type": "string", "description": "Target model for product-planning routine" },
|
|
50
|
+
"analytics-review": { "type": "string", "description": "Target model for analytics-review routine" },
|
|
51
|
+
"design-review": { "type": "string", "description": "Target model for design-review routine" }
|
|
52
|
+
},
|
|
53
|
+
"additionalProperties": { "type": "string" },
|
|
54
|
+
"description": "Per-routine model overrides (e.g. gemini-3.8-flash-high, gemini-3.8-flash-medium)"
|
|
55
|
+
},
|
|
56
|
+
"budgets": {
|
|
57
|
+
"type": "object",
|
|
58
|
+
"properties": {
|
|
59
|
+
"weeklyTokens": {
|
|
60
|
+
"type": "number",
|
|
61
|
+
"description": "Weekly token budget limit across routines"
|
|
62
|
+
},
|
|
63
|
+
"maxIterations": {
|
|
64
|
+
"type": "object",
|
|
65
|
+
"properties": {
|
|
66
|
+
"default": { "type": "number" },
|
|
67
|
+
"autowork": { "type": "number" },
|
|
68
|
+
"peer-review": { "type": "number" },
|
|
69
|
+
"optimizer": { "type": "number" },
|
|
70
|
+
"issues-housekeeping": { "type": "number" },
|
|
71
|
+
"dependency-update-security-check": { "type": "number" },
|
|
72
|
+
"product-planning": { "type": "number" },
|
|
73
|
+
"analytics-review": { "type": "number" },
|
|
74
|
+
"design-review": { "type": "number" }
|
|
75
|
+
},
|
|
76
|
+
"additionalProperties": { "type": "number" },
|
|
77
|
+
"description": "Maximum tool call iterations per routine"
|
|
78
|
+
},
|
|
79
|
+
"timeoutMinutes": {
|
|
80
|
+
"type": "object",
|
|
81
|
+
"properties": {
|
|
82
|
+
"default": { "type": "number" },
|
|
83
|
+
"autowork": { "type": "number" },
|
|
84
|
+
"peer-review": { "type": "number" },
|
|
85
|
+
"optimizer": { "type": "number" },
|
|
86
|
+
"issues-housekeeping": { "type": "number" },
|
|
87
|
+
"dependency-update-security-check": { "type": "number" },
|
|
88
|
+
"product-planning": { "type": "number" },
|
|
89
|
+
"analytics-review": { "type": "number" },
|
|
90
|
+
"design-review": { "type": "number" }
|
|
91
|
+
},
|
|
92
|
+
"additionalProperties": { "type": "number" },
|
|
93
|
+
"description": "Workflow runner timeout in minutes per routine"
|
|
94
|
+
}
|
|
95
|
+
},
|
|
96
|
+
"additionalProperties": false,
|
|
97
|
+
"description": "Per-routine token, iteration, and timeout budgets"
|
|
98
|
+
},
|
|
39
99
|
"schedules": {
|
|
40
100
|
"type": "object",
|
|
41
101
|
"properties": {
|
|
@@ -45,6 +105,7 @@
|
|
|
45
105
|
"issues-housekeeping": { "type": "string", "description": "Custom cron schedule for issues housekeeping" },
|
|
46
106
|
"dependency-update-security-check": { "type": "string", "description": "Custom cron schedule for dependency security checks" },
|
|
47
107
|
"analytics-review": { "type": "string", "description": "Custom cron schedule for analytics review" },
|
|
108
|
+
"design-review": { "type": "string", "description": "Custom cron schedule for design review" },
|
|
48
109
|
"sync-fleet": { "type": "string", "description": "Custom cron schedule for fleet sync" }
|
|
49
110
|
},
|
|
50
111
|
"additionalProperties": { "type": "string" },
|
|
@@ -90,6 +151,18 @@
|
|
|
90
151
|
},
|
|
91
152
|
"additionalProperties": false,
|
|
92
153
|
"description": "Configuration for priority-driven dual local/cloud agent execution"
|
|
154
|
+
},
|
|
155
|
+
"labels": {
|
|
156
|
+
"type": "object",
|
|
157
|
+
"properties": {
|
|
158
|
+
"protected": {
|
|
159
|
+
"type": "array",
|
|
160
|
+
"items": { "type": "string" },
|
|
161
|
+
"description": "User-defined list or wildcard patterns of protected labels retained during label pruning"
|
|
162
|
+
}
|
|
163
|
+
},
|
|
164
|
+
"additionalProperties": false,
|
|
165
|
+
"description": "Configuration for label management and protection shields"
|
|
93
166
|
}
|
|
94
167
|
},
|
|
95
168
|
"required": ["version", "preset", "routines", "skills"],
|
|
@@ -6,17 +6,17 @@ How agent routines in this repository are dispatched, claimed, and reconciled
|
|
|
6
6
|
|
|
7
7
|
This project's automation is a GitHub-native implementation of the orchestration pattern formalized by OpenAI's [Symphony specification](https://github.com/openai/symphony/blob/main/SPEC.md) for orchestrating autonomous coding agents against an issue tracker. There is **no long-running orchestrator daemon**; the roles map onto GitHub primitives:
|
|
8
8
|
|
|
9
|
-
| Symphony Concept
|
|
10
|
-
|
|
11
|
-
| `WORKFLOW.md` (repo-owned config + prompt templates)
|
|
12
|
-
| Orchestrator (poll, dispatch, reconcile)
|
|
13
|
-
| Issue tracker (Linear in Symphony)
|
|
14
|
-
| Agent runner (Codex app-server in per-issue workspace)
|
|
15
|
-
| Tracker is reader/scheduler; mutations happen via agent tools | Routines only schedule; the agent session makes every GitHub write
|
|
16
|
-
|
|
9
|
+
| Symphony Concept | Implementation in this repo |
|
|
10
|
+
| ------------------------------------------------------------- | ----------------------------------------------------------------------------- |
|
|
11
|
+
| `WORKFLOW.md` (repo-owned config + prompt templates) | `AGENTS.md` (aliased as `GEMINI.md`/`CLAUDE.md`) + `.github/prompts/*.md` |
|
|
12
|
+
| Orchestrator (poll, dispatch, reconcile) | GitHub Actions triggers + scheduled routine sessions |
|
|
13
|
+
| Issue tracker (Linear in Symphony) | GitHub Issues |
|
|
14
|
+
| Agent runner (Codex app-server in per-issue workspace) | An ephemeral agent session (Antigravity CLI `agy`) in an isolated fresh clone |
|
|
15
|
+
| Tracker is reader/scheduler; mutations happen via agent tools | Routines only schedule; the agent session makes every GitHub write |
|
|
17
16
|
|
|
18
17
|
Dispatch is both **scheduled** and **event-driven**. All routines run as ephemeral agent sessions via **Antigravity CLI (`agy`)** powered by **Gemini 3.7 Flash (High reasoning)**. The routine suite is calibrated to operate within a **strict 70% weekly token ceiling across all routines combined**, supervised by `optimizer.md`:
|
|
19
|
-
|
|
18
|
+
|
|
19
|
+
- **Scheduled cron sweeps**: Autowork runs periodically (`autowork-cron.yml`), complemented by prompt optimization (`prompt-optimizer-cron.yml`), issues housekeeping (`issues-housekeeping-cron.yml`), dependency security checks (`dependency-check-cron.yml`), analytics review (`analytics-review-cron.yml`), product planning (`product-planning-cron.yml`), and design review (`design-review-cron.yml`).
|
|
20
20
|
- **Event-driven & manual triggers**: GitHub Actions workflows fire routines on events and interactive commands so work starts within seconds instead of waiting for scheduled ticks:
|
|
21
21
|
- `trigger-review-routine.yml` fires Peer Review automatically when a PR is marked ready for review, updated, or review is requested (`ready_for_review`, `opened`, `reopened`, `synchronize`, `review_requested`). It can also be manually (re)triggered via `workflow_dispatch` (with optional `pr_number` for Targeted mode or blank for Scan mode) or by commenting `/review`, `/peer-review`, `/retrigger`, or `/re-review` on any open pull request.
|
|
22
22
|
- `trigger-autowork-on-merge.yml` fires Autowork in **Targeted mode** when a PR merges to `main` and unblocks the next unit of chained work.
|
|
@@ -27,22 +27,23 @@ Autowork triggers pass the target issue via environment variables (`TARGET_ISSUE
|
|
|
27
27
|
|
|
28
28
|
Invariants deliberately upheld from this spec:
|
|
29
29
|
|
|
30
|
-
- **Single-flight per issue and PR convergence** — at most one run works an issue or pull request at a time, enforced by the autowork claim protocol (assign → read-back → earliest-timestamp tiebreak). For **umbrella** issues, single-flight is maintained at the
|
|
30
|
+
- **Single-flight per issue and PR convergence** — at most one run works an issue or pull request at a time, enforced by the autowork claim protocol (assign → read-back → earliest-timestamp tiebreak). For **umbrella** issues, single-flight is maintained at the _child-issue_ level so slices progress cleanly.
|
|
31
31
|
- **Recover dead-run claims** — a crashed run's orphaned claim is released back to the pool rather than starving the issue or PR, both opportunistically during candidate selection and periodically via issues housekeeping.
|
|
32
|
-
- **Reader/writer separation** — the routine that authors a PR never merges it; the Peer Review routine is the sole merge authority for
|
|
32
|
+
- **Reader/writer separation** — the routine that authors a PR never merges it; the Peer Review routine is the sole merge authority for pull requests.
|
|
33
33
|
- **Warm-Context Review Synchronization** — Autowork maintains an active warm session during implementation, polling for Peer Review's verdict. When Peer Review bounces a PR to draft with findings, Autowork immediately detects the draft state in-session, applies fixes directly to its warm working tree, and re-marks the PR ready—re-firing Peer Review for Round N+1 without cold-start overhead.
|
|
34
34
|
|
|
35
35
|
---
|
|
36
36
|
|
|
37
37
|
## Stale-Claim Definition
|
|
38
38
|
|
|
39
|
-
Single source of truth for both autowork candidate reclamation and housekeeping sweeps. An assigned issue is a
|
|
39
|
+
Single source of truth for both autowork candidate reclamation and housekeeping sweeps. An assigned issue is a _stale claim_ (a dead autowork run's orphaned reservation, safe to release) only when **all** of these hold:
|
|
40
40
|
|
|
41
41
|
1. **It is an autowork claim, not a manual one.** The issue carries a `🔒 Claimed by autowork run …` comment. An assigned issue with **no** such comment is never stale; leave it alone (it may be a person working manually).
|
|
42
42
|
2. **No live work exists.** There is **no open PR** referencing the issue (`Closes #N`). An open PR is live, recoverable work that autowork Phase 1 owns — never reclaim it, at any age.
|
|
43
43
|
3. **The claim is old.** The most recent `🔒 Claimed by autowork run …` comment's GitHub creation time (`created_at`) is **more than 6 hours** ago. Measure age from that `created_at` only — never the issue's `updated_at`.
|
|
44
44
|
|
|
45
45
|
**Releasing a stale claim is a destructive write and MUST be guarded:**
|
|
46
|
+
|
|
46
47
|
- **Re-read immediately before writing.** Re-read the issue (`issue_read`) right before the unassign and re-confirm conditions 1–3 still hold. If any no longer holds, abort the release and move on.
|
|
47
48
|
- **Remove only the named dead owner.** Unassign that specific login; never blindly clear all assignees.
|
|
48
49
|
|
|
@@ -50,7 +51,7 @@ Single source of truth for both autowork candidate reclamation and housekeeping
|
|
|
50
51
|
|
|
51
52
|
## PR Stale-Claim Definition (Phase 1 Convergence)
|
|
52
53
|
|
|
53
|
-
Single source of truth for Autowork Phase 1 pull request convergence. An assigned pull request or draft PR with unaddressed review comments is a
|
|
54
|
+
Single source of truth for Autowork Phase 1 pull request convergence. An assigned pull request or draft PR with unaddressed review comments is a _stale claim_ (safe to reclaim and reassign by another runner) only when **all** of these hold:
|
|
54
55
|
|
|
55
56
|
1. **It carries an autowork claim comment**: The PR thread contains `🔒 Addressing review findings by autowork run …` or `🔒 Addressing review findings by local autowork session …`.
|
|
56
57
|
2. **The claim is old**: The most recent claim comment's `created_at` is **more than 2 hours** ago. (2 hours instead of 6 hours because PR review convergence is a rapid turnaround loop).
|
|
@@ -66,15 +67,16 @@ Single source of truth for Autowork Phase 1 pull request convergence. An assigne
|
|
|
66
67
|
2. **Content match** — compare the task against the routine table below. When matching an interactive request from a human, name the matched routine and confirm before proceeding.
|
|
67
68
|
3. **No match** — follow the general Working Practices, PR Workflow, and documentation rules with no routine-specific constraints.
|
|
68
69
|
|
|
69
|
-
| Routine
|
|
70
|
-
|
|
71
|
-
| Autowork
|
|
72
|
-
| Peer Review
|
|
73
|
-
| Prompt Optimizer
|
|
74
|
-
| Issues Housekeeping
|
|
75
|
-
| Dependency Update & Security Check | `.github/prompts/dependency-update-security-check.md` | Checking dependencies for updates and known vulnerabilities, opening actionable PRs
|
|
76
|
-
| Product Planning
|
|
77
|
-
| Analytics Review
|
|
70
|
+
| Routine | File | Applies when the conversation is about... |
|
|
71
|
+
| ---------------------------------- | ----------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
|
72
|
+
| Autowork | `.github/prompts/autowork.md` | Converging on open work: addressing PR review comments, closing issues whose PRs merged, then claiming and implementing the highest-priority unclaimed issue |
|
|
73
|
+
| Peer Review | `.github/prompts/peer-review.md` | Reviewing a pull request (a named PR or scan mode) and merging it or leaving findings and bouncing to draft |
|
|
74
|
+
| Prompt Optimizer | `.github/prompts/optimizer.md` | Diagnosing failures, inefficiency, token anomalies, and analyzing resolved bugs to propose prompt/test/workflow fixes and upstream contributions |
|
|
75
|
+
| Issues Housekeeping | `.github/prompts/issues-housekeeping.md` | Sweeping open issues for staleness, duplicates, label drift, priority accuracy, and orphaned claims |
|
|
76
|
+
| Dependency Update & Security Check | `.github/prompts/dependency-update-security-check.md` | Checking dependencies for updates and known vulnerabilities, opening actionable PRs |
|
|
77
|
+
| Product Planning | `.github/prompts/product-planning.md` | Turning roadmap priorities into staged issues (`/to-tickets`) and formal PRDs (`/to-spec`) |
|
|
78
|
+
| Analytics Review | `.github/prompts/analytics-review.md` | Evaluating telemetry & measurement trackers against success metrics, emitting action directives (PIVOT/DEPRECATE/ITERATE), and bridging to product planning |
|
|
79
|
+
| Design Review | `.github/prompts/design-review.md` | Auditing player-facing surfaces for design system token deviations, visual clutter, feature pruning, and UX improvements |
|
|
78
80
|
|
|
79
81
|
---
|
|
80
82
|
|
|
@@ -102,17 +104,54 @@ The routines invoke specialized engineering skills at key workflow checkpoints:
|
|
|
102
104
|
- `/to-tickets`: Decomposes approved epics/proposals into dependency-linked issues.
|
|
103
105
|
- **Prompt Optimizer (`optimizer.md`)**:
|
|
104
106
|
- `/writing-for-agents`: Drafts crisp, token-efficient prompt and rule updates.
|
|
107
|
+
- **Design Review (`design-review.md`)**:
|
|
108
|
+
- `/design-system`: Analyzes styling diffs and static components for token purity and pattern alignment.
|
|
109
|
+
- `/run`: Boots dev environment and automates mobile/desktop viewport captures.
|
|
110
|
+
- `/design-critique`: Analyzes captured screenshots for visual hierarchy, clutter, spacing, and contrast.
|
|
105
111
|
|
|
106
112
|
---
|
|
107
113
|
|
|
108
|
-
##
|
|
109
|
-
|
|
110
|
-
Single source of truth for every routine's
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
114
|
+
## Routine Issue Logging & Telemetry Protocol
|
|
115
|
+
|
|
116
|
+
Single source of truth for every routine's logging and telemetry lifecycle:
|
|
117
|
+
|
|
118
|
+
1. **GitHub Issues as Operational Ledger**: Routine run logs are recorded as GitHub Issues instead of git commits, keeping `main` and Git history 100% clean. No operational markdown files are committed to Git, eliminating push conflicts, merge races, and draft-PR fallbacks.
|
|
119
|
+
2. **Two-Phase Issue Lifecycle**:
|
|
120
|
+
- **Start**: The workflow harness or local runner pre-step creates a tracking issue via `gh issue create` titled `[routine-name] run {timestamp}` with labels `routine-log`, `routine:{name}`, `status:running`, and `runner:{github-actions|local}`. The issue number is exported as `ROUTINE_ISSUE_NUMBER`.
|
|
121
|
+
- **Execution**: The agent performs the routine and records its structured execution report (Definition of Done table, telemetry metrics, execution trace) to `.jonah-fleet/run-report.md`.
|
|
122
|
+
- **Finish**: The harness post-step reconciles the tracking issue body:
|
|
123
|
+
- On **SUCCESS**: Applies label `status:success`, removes `status:running`, and closes the issue immediately via `gh issue close $ROUTINE_ISSUE_NUMBER --reason completed`.
|
|
124
|
+
- On **FAILURE / CRASH / TIMEOUT**: Applies labels `status:failure`, `needs-attention`, removes `status:running`, and leaves the issue **OPEN** in the issue tracker for human maintainer triage and optimizer diagnosis.
|
|
125
|
+
3. **Local Daemon Parity & Offline Fallback**: `jonah-fleet daemon` and `jonah-fleet run` always record runs locally in `.jonah-fleet/runs/{timestamp}.json` (ignored by git). When online with valid `gh` auth, local daemons create and close tracking issues labeled `runner:local`. If offline or unauthenticated, runs gracefully fall back to local-only logging without interrupting agent execution.
|
|
126
|
+
4. **Issue Backlog Isolation**: Routine run issues carry `label:routine-log`. Human maintainers and product queries filter `-label:routine-log` in issue searches to keep product backlogs pristine.
|
|
127
|
+
|
|
128
|
+
### Routine Issue Progress Reporting & Milestone Protocol
|
|
129
|
+
|
|
130
|
+
How live execution progress is reported during autonomous routine runs:
|
|
131
|
+
|
|
132
|
+
1. **Comment Stream for Live Telemetry**: Rather than leaving tracking issues static until completion, routines emit structured milestone comments into the thread of `$ROUTINE_ISSUE_NUMBER`. This gives maintainers real-time visibility into agent decisions and progress with GitHub timestamps without needing to inspect raw Actions logs.
|
|
133
|
+
2. **Issue Isolation Invariant (Negative Rule)**: Milestone comments are posted exclusively to the routine tracking issue (`gh issue comment "$ROUTINE_ISSUE_NUMBER"`). Agents MUST NEVER post progress telemetry comments to target product issues or pull requests (except for required PR linkage and review comments).
|
|
134
|
+
3. **Compact Milestone Cards (5-Point Schema)**: Every milestone comment follows this standard structure:
|
|
135
|
+
```markdown
|
|
136
|
+
### <Emoji> Milestone: <Milestone Name>
|
|
137
|
+
- **Phase**: `<Phase Identifier>`
|
|
138
|
+
- **Status**: <Status Emoji + Summary>
|
|
139
|
+
- **Target / Context**: `<Target Issue/PR or Context>`
|
|
140
|
+
- **Key Decision / Finding**: <Summary of key decision, root cause, or verification outcome>
|
|
141
|
+
- **Next**: <Next planned milestone>
|
|
142
|
+
```
|
|
143
|
+
4. **Bounded 4-Stage Milestone Cadence**: Routines enforce strictly 3–4 bounded milestones per flight:
|
|
144
|
+
- **Milestone 1 (Intake & Strategy)**: Target claimed, scope clarified, initial strategy/reproduction plan established.
|
|
145
|
+
- **Milestone 2 (Verification & Tests)**: Implementation complete, unit/integration tests passing, lint clean.
|
|
146
|
+
- **Milestone 3 (Autonomous Handoff)**: Branch pushed, PR opened with link, or review decision submitted.
|
|
147
|
+
- **Milestone 4 (Run Completed)**: Final compact status closing out the comment stream.
|
|
148
|
+
5. **Soft Failure & Offline Tolerance**: All milestone commands use `|| true`:
|
|
149
|
+
```bash
|
|
150
|
+
gh issue comment "$ROUTINE_ISSUE_NUMBER" --body "..." || true
|
|
151
|
+
```
|
|
152
|
+
If offline, unauthenticated, or rate-limited, agent execution continues uninterrupted.
|
|
153
|
+
6. **Harness Interruption Card**: If a routine run fails, times out, or crashes abruptly before Milestone 4, the workflow harness or local runner post-step automatically appends an Interruption Card (`### ❌ Milestone: Run Interrupted / Failed`) before marking `status:failure`.
|
|
154
|
+
7. **Final State Reconciliation**: The final comment is Milestone 4 (Compact completion card). The workflow harness replaces the top-level issue body with the comprehensive telemetry & audit report (`.jonah-fleet/run-report.md`) and closes the issue on success.
|
|
116
155
|
|
|
117
156
|
---
|
|
118
157
|
|
|
@@ -141,14 +180,15 @@ How external and human contributor pull requests are reconciled into the issue t
|
|
|
141
180
|
2. **Autonomous Synthesis on Merge**: When `peer-review.md` approves a pull request lacking a `Closes #N` link, the review routine automatically synthesizes a tracking issue before merging:
|
|
142
181
|
- Creates a tracked issue via `gh issue create` capturing the PR title, body, and deliverables.
|
|
143
182
|
- Appends `Closes #<synthesized_issue_id>` to the PR description via `gh pr edit`.
|
|
144
|
-
3. **Audit & Single-Flight Lineage**: When the PR is squash-merged,
|
|
183
|
+
3. **Audit & Single-Flight Lineage**: When the PR is squash-merged, `peer-review` explicitly closes the tracking issue (`gh issue close <ISSUE_NUMBER>`) to guarantee tracking closure, even when bot credentials or draft PR states bypass GitHub's native issue auto-close. This maintains 100% issue auditability, project board tracking, telemetry metrics, and release changelogs without leaving stray open issues.
|
|
145
184
|
|
|
146
185
|
---
|
|
147
186
|
|
|
148
187
|
## Fleet Telemetry & Weekly Token Economics
|
|
149
188
|
|
|
150
189
|
Cross-repository telemetry aggregation and token tracking protocol:
|
|
151
|
-
|
|
190
|
+
|
|
191
|
+
1. **Lightweight Routine Telemetry**: Every autonomous routine execution emits a structured `RoutineTelemetrySummary` (recorded in the GitHub Issue body and local `.jonah-fleet/runs/*.json` cache) capturing routine identity, duration, iterations, result, failure category, cost, and tokens.
|
|
152
192
|
2. **Opt-in Emission Step**: GitHub Actions workflows (`autowork-cron.yml`, `trigger-review-routine.yml`, `prompt-optimizer-cron.yml`) run `jonah-fleet telemetry --emit` using optional `JONAH_FLEET_TELEMETRY_ENDPOINT` secrets.
|
|
153
193
|
3. **Global 70% Budget Ceiling**: Tracks rolling 7-day spend across all fleet repositories against the global ceiling (~8.75M tokens/week).
|
|
154
194
|
4. **Health Thresholds**:
|
|
@@ -168,23 +208,27 @@ How the fleet guarantees continuous review throughput, recovers from transient A
|
|
|
168
208
|
3. **Interactive Re-triggering**: Any team member or author can immediately re-dispatch review by commenting `/review`, `/peer-review`, `/retrigger`, or `/re-review` on any open pull request, or manually triggering `trigger-review-routine.yml` via `workflow_dispatch`.
|
|
169
209
|
4. **Autowork Phase 1 Watchdog**: During Phase 1 convergence, Autowork actively identifies open ready PRs that have received no review feedback for $>2$ hours, re-triggering review via draft toggle or `/review` comment before picking up new work. Review re-triggering is strictly conditioned on all CI checks having passed (`conclusion: SUCCESS`, `mergeStateStatus: CLEAN`) and is strictly prohibited if checks are in-progress or awaiting approval (`ACTION_REQUIRED`).
|
|
170
210
|
|
|
171
|
-
|
|
172
211
|
---
|
|
173
212
|
|
|
174
|
-
## Upstream Symphony Intel & Architectural Evaluation Framework
|
|
213
|
+
## Upstream Symphony & Funes Intel & Architectural Evaluation Framework
|
|
175
214
|
|
|
176
|
-
How changes and innovations from [openai/symphony](https://github.com/openai/symphony) are systematically audited and evaluated for incorporation into Jonah Fleet:
|
|
215
|
+
How changes and innovations from upstream ecosystems—[openai/symphony](https://github.com/openai/symphony) for issue-tracker orchestration and [huggingface/funes](https://github.com/huggingface/funes) for agent memory & session indexing—are systematically audited and evaluated for incorporation into Jonah Fleet:
|
|
177
216
|
|
|
178
|
-
1. **Automated Radar (`symphony-radar.yml`)**: A weekly scheduled workflow runs `.github/scripts/fetch-symphony-radar.js` to inspect upstream commits, specification updates (`SPEC.md`), and releases, generating an actionable digest issue in Jonah Fleet.
|
|
217
|
+
1. **Automated Ecosystem Radar (`symphony-radar.yml`)**: A weekly scheduled workflow runs `.github/scripts/fetch-symphony-radar.js` to inspect upstream commits, specification updates (`SPEC.md`), and releases across `openai/symphony` (orchestration) and `huggingface/funes` (memory tooling), generating an actionable digest issue in Jonah Fleet.
|
|
179
218
|
2. **The 4 Evaluation Layers**:
|
|
180
219
|
- **Layer 1 (Zero-Daemon Invariant)**: Can the enhancement execute in ephemeral GitHub Actions and `agy` CLI sessions without requiring a 24/7 background server or persistent WebSocket?
|
|
181
220
|
- **Layer 2 (Issue Tracker Abstraction)**: Does the pattern map cleanly to native GitHub Issues, labels, and PR checks without proprietary tracker dependencies?
|
|
182
221
|
- **Layer 3 (Token & Cost Economy)**: Does the change optimize LLM spend within Jonah Fleet's 70% weekly token ceiling (~8.75M tokens)?
|
|
183
222
|
- **Layer 4 (Multi-Repo Portability)**: Can the routine or skill be distributed via `agents-manifest.json` and `jonah-fleet sync` across any consumer repository?
|
|
184
|
-
3. **
|
|
185
|
-
-
|
|
186
|
-
-
|
|
187
|
-
-
|
|
223
|
+
3. **Agent Memory & Session Indexing Evaluation Dimensions (Funes Watch)**:
|
|
224
|
+
- **Zero-LLM Ingestion**: Deterministic parsing of agent session traces (`.jsonl`/Parquet) into LanceDB without spending LLM tokens from the weekly budget.
|
|
225
|
+
- **Pull-Based Memory Delivery**: Memory served strictly on demand via MCP (`recall`, `get`) to prevent prompt context bloat.
|
|
226
|
+
- **Cross-Session Provenance**: Verbatim turns and provenance retention instead of lossy summary drift.
|
|
227
|
+
- **Multi-Agent Portability**: Standardized trace ingestion across Antigravity CLI (`agy`), Claude Code, and Codex.
|
|
228
|
+
4. **Classification & Action Protocol**:
|
|
229
|
+
- **🟢 Category A (Adopt Directly)**: Security guardrails, claim lock invariants, reader/writer rules, prompt engineering optimizations, deterministic zero-LLM indexing.
|
|
230
|
+
- **🟡 Category B (Adapt to Actions/CLI)**: Dynamic orchestrator pacing, backpressure controls, multi-stage review checks, pull-based memory MCP integrations.
|
|
231
|
+
- **🔴 Category C (Skip)**: Elixir/OTP supervision trees, BEAM memory tuning, proprietary runtime internals, always-loaded memory context dumps.
|
|
188
232
|
|
|
189
233
|
---
|
|
190
234
|
|
|
@@ -213,4 +257,3 @@ How agent routines are partitioned between cloud GitHub Actions (24/7 cloud runn
|
|
|
213
257
|
- Local agent PR convergence claims post: `🔒 Addressing review findings by local autowork session (host: <hostname>) <timestamp>`.
|
|
214
258
|
- Local processes trap `SIGINT`/`SIGTERM` to unassign claims and remove worktrees cleanly on exit.
|
|
215
259
|
- Standard stale-claim rules (6h for issues, 2h for PRs) safely reclaim orphaned local claims if a machine powers down unexpectedly.
|
|
216
|
-
|
|
@@ -34,22 +34,19 @@ If any criterion cannot be met, stop immediately and log FAILURE with the reason
|
|
|
34
34
|
|
|
35
35
|
## Logging
|
|
36
36
|
|
|
37
|
-
After completing (SUCCESS or FAILURE),
|
|
37
|
+
After completing (SUCCESS or FAILURE), record run execution details to `.jonah-fleet/run-report.md`. Include:
|
|
38
38
|
- The prompt SHA (run `git rev-parse --short HEAD:.github/prompts/{routine-name}.md`)
|
|
39
39
|
- Every Definition of Done criterion with YES/NO and evidence
|
|
40
40
|
- Full execution trace with tool calls
|
|
41
41
|
- If FAILURE: root cause, category, and suggested fix
|
|
42
42
|
|
|
43
|
-
**
|
|
44
|
-
- **Negative Rule**: NEVER commit or push run logs to
|
|
45
|
-
- **
|
|
46
|
-
- **
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
git push origin main
|
|
53
|
-
```
|
|
54
|
-
Follow the Log delivery fallback in `ORCHESTRATION.md` if direct push fails.
|
|
43
|
+
**Issue Logging Protocol & Invariants**:
|
|
44
|
+
- **Negative Rule**: NEVER commit or push run logs to any git branch (`main` or feature branches). Run logs are recorded exclusively as GitHub Issues and local `.jonah-fleet/runs/*.json` artifacts.
|
|
45
|
+
- **Routine Issue Progress Milestones**: Emit bounded Compact Milestone Cards (3–5 bullet points) to `$ROUTINE_ISSUE_NUMBER` at major phase transitions using `gh issue comment "$ROUTINE_ISSUE_NUMBER" --body "..." || true`.
|
|
46
|
+
- **Milestone 4 (Run Completed)**: Append the final compact milestone card to close out the comment stream.
|
|
47
|
+
- **Harness Reconciliation**: When running under GitHub Actions or `jonah-fleet daemon`, the harness initializes the run issue (`status:running`), replaces the top-level issue body with `.jonah-fleet/run-report.md`, and reconciles the final status upon completion:
|
|
48
|
+
- On **SUCCESS**: Labels `status:success`, removes `status:running`, and closes the issue.
|
|
49
|
+
- On **FAILURE**: Posts the Interruption Card, edits the body, and marks `status:failure,needs-attention`.
|
|
50
|
+
- On clean local runs outside a harness, write `.jonah-fleet/runs/{timestamp}.json`.
|
|
51
|
+
- Follow the Routine Issue Logging & Telemetry Protocol in `ORCHESTRATION.md`.
|
|
55
52
|
|
|
@@ -82,10 +82,12 @@ When a measurement tracker is conclusive or reaches sub-threshold adoption (<2%
|
|
|
82
82
|
|
|
83
83
|
## Logging
|
|
84
84
|
|
|
85
|
-
After completing (SUCCESS or FAILURE),
|
|
85
|
+
After completing (SUCCESS or FAILURE), record run execution details to `.jonah-fleet/run-report.md`. Include:
|
|
86
86
|
- The prompt SHA (run `git rev-parse --short HEAD:.github/prompts/analytics-review.md`)
|
|
87
87
|
- Every Definition of Done criterion with YES/NO and evidence
|
|
88
88
|
- Full execution trace with evaluated trackers and emitted recommendations
|
|
89
89
|
- If FAILURE: root cause, category, and suggested fix
|
|
90
90
|
|
|
91
|
-
**
|
|
91
|
+
**Issue Logging Protocol**:
|
|
92
|
+
- Record run execution details to `.jonah-fleet/run-report.md` (or update `$ROUTINE_ISSUE_NUMBER`).
|
|
93
|
+
- Follow the Routine Issue Logging & Telemetry Protocol in `ORCHESTRATION.md`. Never commit run logs to git branches.
|