maf 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (136) hide show
  1. checksums.yaml +7 -0
  2. data/CHANGELOG.md +11 -0
  3. data/LICENSE.txt +21 -0
  4. data/README.md +411 -0
  5. data/assets/agents-contract.md +80 -0
  6. data/assets/analyst +240 -0
  7. data/assets/coord +2936 -0
  8. data/assets/dashboard +553 -0
  9. data/assets/dashboard.html +341 -0
  10. data/assets/dispatcher +1687 -0
  11. data/assets/doc-graph-refresh +286 -0
  12. data/assets/env.sh +6 -0
  13. data/assets/git-hooks/post-commit +7 -0
  14. data/assets/git-hooks/post-merge +7 -0
  15. data/assets/git-hooks/pre-commit +32 -0
  16. data/assets/harness-hooks/board-watch-opencode.js +87 -0
  17. data/assets/harness-hooks/board-watch.rb +286 -0
  18. data/assets/harness-hooks/context-watch.rb +268 -0
  19. data/assets/harness-hooks/next-task-hermes.sh +48 -0
  20. data/assets/harness-hooks/next-task.rb +97 -0
  21. data/assets/harness-hooks/session-guard.rb +128 -0
  22. data/assets/taskrc.append +11 -0
  23. data/assets/vault +224 -0
  24. data/assets/worktree-env.example.rb +26 -0
  25. data/exe/maf +14 -0
  26. data/install.md +326 -0
  27. data/lib/maf/bootstrap/claude_settings.rb +55 -0
  28. data/lib/maf/bootstrap/dependencies.rb +37 -0
  29. data/lib/maf/bootstrap/git_hook_planner.rb +68 -0
  30. data/lib/maf/bootstrap/global_taskrc_warning.rb +33 -0
  31. data/lib/maf/bootstrap/graph_home.rb +62 -0
  32. data/lib/maf/bootstrap/hook_merger.rb +53 -0
  33. data/lib/maf/bootstrap/installer.rb +66 -0
  34. data/lib/maf/bootstrap/layout_planner.rb +18 -0
  35. data/lib/maf/bootstrap/marked_block.rb +44 -0
  36. data/lib/maf/bootstrap/memory_branch.rb +77 -0
  37. data/lib/maf/bootstrap/options.rb +34 -0
  38. data/lib/maf/bootstrap/project.rb +77 -0
  39. data/lib/maf/bootstrap/script_planner.rb +81 -0
  40. data/lib/maf/bootstrap/text_planner.rb +42 -0
  41. data/lib/maf/bootstrap/vault_starter.rb +41 -0
  42. data/lib/maf/bootstrap/writer.rb +69 -0
  43. data/lib/maf/bootstrap.rb +162 -0
  44. data/lib/maf/budget.rb +59 -0
  45. data/lib/maf/cli.rb +135 -0
  46. data/lib/maf/env_exclude.rb +23 -0
  47. data/lib/maf/flow/agent_links.rb +79 -0
  48. data/lib/maf/flow/bootstrapper.rb +36 -0
  49. data/lib/maf/flow/codex_hooks.rb +50 -0
  50. data/lib/maf/flow/generator.rb +63 -0
  51. data/lib/maf/flow/harness_linker.rb +37 -0
  52. data/lib/maf/flow/hermes_hook.rb +48 -0
  53. data/lib/maf/flow/hermes_hook_setup.rb +69 -0
  54. data/lib/maf/flow/hook_files.rb +16 -0
  55. data/lib/maf/flow/hook_installer.rb +33 -0
  56. data/lib/maf/flow/legacy_codex_hook.rb +71 -0
  57. data/lib/maf/flow/manifest.rb +51 -0
  58. data/lib/maf/flow/mcp_config.rb +72 -0
  59. data/lib/maf/flow/mcp_installer.rb +45 -0
  60. data/lib/maf/flow/models.rb +61 -0
  61. data/lib/maf/flow/options.rb +65 -0
  62. data/lib/maf/flow/prompt_builder.rb +85 -0
  63. data/lib/maf/flow/prompt_text.rb +263 -0
  64. data/lib/maf/flow/report.rb +89 -0
  65. data/lib/maf/flow/role_catalog.rb +40 -0
  66. data/lib/maf/flow/role_files.rb +72 -0
  67. data/lib/maf/flow/role_stub.rb +38 -0
  68. data/lib/maf/flow/roster.rb +28 -0
  69. data/lib/maf/flow/validator.rb +38 -0
  70. data/lib/maf/flow/workflow.rb +28 -0
  71. data/lib/maf/flow.rb +84 -0
  72. data/lib/maf/local_exclude.rb +53 -0
  73. data/lib/maf/menu.rb +101 -0
  74. data/lib/maf/migrate/moves.rb +44 -0
  75. data/lib/maf/migrate/rewrites.rb +53 -0
  76. data/lib/maf/migrate/role_files.rb +35 -0
  77. data/lib/maf/migrate/runner.rb +66 -0
  78. data/lib/maf/migrate/worktrees.rb +65 -0
  79. data/lib/maf/migrate.rb +62 -0
  80. data/lib/maf/prompt.rb +40 -0
  81. data/lib/maf/retire.rb +116 -0
  82. data/lib/maf/role_limits.rb +49 -0
  83. data/lib/maf/setup_agent/args.rb +57 -0
  84. data/lib/maf/setup_agent/dispatch.rb +44 -0
  85. data/lib/maf/setup_agent/hermes_launcher.rb +34 -0
  86. data/lib/maf/setup_agent/hermes_skill.rb +26 -0
  87. data/lib/maf/setup_agent/launcher.rb +85 -0
  88. data/lib/maf/setup_agent/manifest.rb +35 -0
  89. data/lib/maf/setup_agent/project.rb +9 -0
  90. data/lib/maf/setup_agent/role_file.rb +30 -0
  91. data/lib/maf/setup_agent/runtime_hooks.rb +37 -0
  92. data/lib/maf/setup_agent/worktree.rb +50 -0
  93. data/lib/maf/setup_agent.rb +111 -0
  94. data/lib/maf/shared/git_exclude.rb +33 -0
  95. data/lib/maf/shared/git_identity.rb +41 -0
  96. data/lib/maf/shared/peak_rate.rb +20 -0
  97. data/lib/maf/shared/processes.rb +31 -0
  98. data/lib/maf/shared/project.rb +34 -0
  99. data/lib/maf/shared/roles.rb +19 -0
  100. data/lib/maf/team.rb +114 -0
  101. data/lib/maf/team_command.rb +73 -0
  102. data/lib/maf/uninstall/claude_settings.rb +40 -0
  103. data/lib/maf/uninstall/codex_hooks.rb +18 -0
  104. data/lib/maf/uninstall/commit_guard.rb +16 -0
  105. data/lib/maf/uninstall/coordination.rb +15 -0
  106. data/lib/maf/uninstall/doc_graph_hooks.rb +38 -0
  107. data/lib/maf/uninstall/git.rb +13 -0
  108. data/lib/maf/uninstall/local_files.rb +33 -0
  109. data/lib/maf/uninstall/manifest.rb +29 -0
  110. data/lib/maf/uninstall/marked_files.rb +37 -0
  111. data/lib/maf/uninstall/mcp_entries.rb +43 -0
  112. data/lib/maf/uninstall/notes.rb +31 -0
  113. data/lib/maf/uninstall/owned.rb +12 -0
  114. data/lib/maf/uninstall/role_files.rb +51 -0
  115. data/lib/maf/uninstall/runner.rb +67 -0
  116. data/lib/maf/uninstall/scripts.rb +35 -0
  117. data/lib/maf/uninstall/vault_watcher.rb +21 -0
  118. data/lib/maf/uninstall/worktrees.rb +30 -0
  119. data/lib/maf/uninstall.rb +59 -0
  120. data/lib/maf/untrack.rb +90 -0
  121. data/lib/maf/version.rb +5 -0
  122. data/lib/maf/worker_archive.rb +63 -0
  123. data/lib/maf/worker_control.rb +137 -0
  124. data/lib/maf/workers.rb +37 -0
  125. data/lib/maf.rb +5 -0
  126. data/templates/claude.md.erb +16 -0
  127. data/templates/codex.md.erb +7 -0
  128. data/templates/hermes.md.erb +12 -0
  129. data/templates/opencode.md.erb +24 -0
  130. data/templates/role-stub.yml.erb +15 -0
  131. data/templates/roles.yml +289 -0
  132. data/templates/workflows/panel.md +20 -0
  133. data/templates/workflows/plan-review.md +9 -0
  134. data/templates/workflows/simple.md +4 -0
  135. data/templates/workflows/tdd.md +8 -0
  136. metadata +193 -0
data/assets/dispatcher ADDED
@@ -0,0 +1,1687 @@
1
+ #!/usr/bin/env ruby
2
+ # frozen_string_literal: true
3
+
4
+ # dispatcher - task board and inbox monitor that starts one-shot agents.
5
+ #
6
+ # The dispatcher solves two problems:
7
+ # 1. A long-lived interactive agent session burns tokens while idle. The
8
+ # dispatcher starts an agent only when there is work, and the agent exits
9
+ # when the work is done (one-shot mode).
10
+ # 2. An agent that stops has no external watcher to restart it. The
11
+ # dispatcher polls the inbox and the task board on a fixed interval.
12
+ #
13
+ # Usage:
14
+ # dispatcher ROLE [--harness NAME | --command TEMPLATE] [--model M]
15
+ # [--skill NAME | --no-skill] [--interval S] [--max-turns N]
16
+ # [--timeout S] [--idle-timeout S] [--grace S]
17
+ # [--completion-signal TEXT] [--abort-signal TEXT]
18
+ # [--cache-window S] [--max-session-runs N] [--max-context N]
19
+ # [--no-poll-tasks]
20
+ # [--full-harness] [--once] [--verbose]
21
+ #
22
+ # dispatcher tester # hermes (default)
23
+ # dispatcher reviewer --harness claude --model sonnet
24
+ # dispatcher backend-developer --harness opencode --model openrouter/qwen3-coder
25
+ # dispatcher reviewer --harness codex
26
+ # dispatcher reviewer --once # one poll cycle, then exit
27
+ #
28
+ # Flags:
29
+ # --harness NAME Built-in adapter: hermes (default), claude, codex, or
30
+ # opencode. Each one resumes sessions.
31
+ # --command TEMPLATE Any other harness. A shell command with %{prompt},
32
+ # %{role}, %{skill}, and %{model} placeholders, each
33
+ # shell-escaped. The command must exit when done.
34
+ # Sessions are not resumed for custom commands.
35
+ # --model M Model for the agent, in the harness CLI's own format.
36
+ # --skill NAME Hermes skill to preload. Default: <project>-<role> when
37
+ # flow.rb generated it.
38
+ # --no-skill Do not preload a skill.
39
+ # --interval S Poll interval in seconds (default: 60).
40
+ # --max-turns N Hermes max tool-calling turns per run (default: 50).
41
+ # --timeout S Max wall-clock seconds per run. Default: team.timeouts.<role>
42
+ # in .maf/config.json, else 1500. The
43
+ # dispatcher kills the agent's process group after it.
44
+ # --idle-timeout S Fail a run that prints no output for S seconds
45
+ # (default: 0, off). Claude Code prints its JSON only at
46
+ # the end, so keep this off for --harness claude.
47
+ # --grace S Seconds to wait for the agent to exit after the
48
+ # completion signal, and for a child that holds stdout
49
+ # (default: 5).
50
+ # --completion-signal TEXT The agent prints TEXT when the work is complete.
51
+ # The run is a success, and the dispatcher stops the
52
+ # agent after the grace window. Default: off.
53
+ # --abort-signal TEXT The agent prints TEXT when it gives up. The run is a
54
+ # failure. Default: off.
55
+ # --cache-window S Resume a session only if its last run ended less than
56
+ # S seconds ago. Default: 300 for codex, else 3300
57
+ # (55 minutes). 0 means never resume.
58
+ # --max-session-runs N Start a fresh session after N runs in one session
59
+ # (default: 5). Each resume sends the whole old context
60
+ # again, so a long session costs more on every call.
61
+ # 0 means no limit.
62
+ # --max-context N Start a fresh session when the last model call of the
63
+ # session had N or more context tokens (default:
64
+ # 150000). Each task in a resumed session adds to the
65
+ # context. Only opencode reports the size. 0 means no limit.
66
+ # --no-poll-tasks Dispatch on inbox messages only.
67
+ # --full-harness Codex and Claude Code: keep the user's full setup. By
68
+ # default a dispatched Codex run starts without plugins,
69
+ # apps, browser and computer tools, subagents, and MCP
70
+ # servers. A dispatched Claude Code run starts without
71
+ # skills and without MCP servers. Both keep only the
72
+ # graphify server of this project: no person reads the
73
+ # run, and each extra tool or skill adds context to
74
+ # every model call of the run.
75
+ # --once Run one poll cycle, then exit (tests, cron).
76
+ # --verbose Log idle cycles too.
77
+ #
78
+ # Messages: the dispatcher owns the role's inbox. It moves each unread
79
+ # message into inbox/<role>/processing/ (an atomic rename, so two dispatchers
80
+ # never take the same message), bundles all of them into one prompt, and runs
81
+ # the agent once. On success the messages move to read/. On failure they go
82
+ # back to the inbox, and after MAX_ATTEMPTS failures to failed/. An FYI
83
+ # message (`coord msg --fyi`) starts no run. The next run takes it along.
84
+ #
85
+ # Usage limits: a run that fails on a provider usage or rate limit never
86
+ # reached the model. The dispatcher returns its messages without an attempt
87
+ # and pauses new runs: 15 minutes, then a doubling pause up to MAX_BACKOFF.
88
+ #
89
+ # Unknown model: a run that fails because the harness does not know the
90
+ # model never reached the model, and a retry fails the same way. The
91
+ # dispatcher returns its messages without an attempt, starts no new run until
92
+ # it restarts, and escalates once to the project manager. The claims stay.
93
+ #
94
+ # Tasks: when `coord next ROLE` lists unclaimed tasks, the dispatcher claims
95
+ # one task for its worker and runs the agent on that task only. The claim is
96
+ # atomic, so a run never starts for a task that another worker took. If the
97
+ # same task set is still unclaimed afterwards, it waits with a doubling
98
+ # backoff (capped at MAX_BACKOFF) before it tries again. A changed task set
99
+ # dispatches at once.
100
+ #
101
+ # Sessions and the prompt cache: the session ID is saved to
102
+ # .maf/coordination/sessions/<worker>.session and resumed on the next run, so the
103
+ # agent keeps its context without staying alive. LLM providers cache the
104
+ # session prefix for a limited time after the last request (1 hour for
105
+ # Claude Code, a few minutes for Codex). A resume after that time resends the whole context at full
106
+ # price. So the dispatcher resumes only inside --cache-window, and for at
107
+ # most --max-session-runs runs, below --max-context tokens. Otherwise the dispatcher starts a fresh
108
+ # session. A run that changed the state of the work puts a short handoff note
109
+ # in its final reply. The dispatcher writes the note to
110
+ # .maf/coordination/sessions/<worker>.handoff.md, and a fresh session starts
111
+ # with that note. If the harness reports an unknown session,
112
+ # the dispatcher also starts fresh. Other failures keep the session.
113
+ #
114
+ # The agent runs with COORD_DIR, COORD_ROLE, COORD_WORKER (default:
115
+ # <role>-bot), TASKRC, and COORD_DISPATCHED=1 set, and with stdin closed. Every built-in
116
+ # adapter skips permission prompts (hermes --yolo, claude bypassPermissions,
117
+ # codex --dangerously-bypass-approvals-and-sandbox): the agent runs every
118
+ # tool call without approval. Run it only in a worktree or sandbox you trust
119
+ # it with. Two dispatchers for one role need different COORD_WORKER values.
120
+ #
121
+ # Environment:
122
+ # COORD_DIR Path to .maf/coordination/ (default: .maf/coordination).
123
+ # COORD_WORKER Worker ID for the agent (default: <role>-bot).
124
+ # DISPATCHER_LOG Append log lines to this file (default: stderr).
125
+ #
126
+ # Ruby: 3.0+ (endless method defs, same as coord).
127
+
128
+ require "json"
129
+ require "fileutils"
130
+ require "optparse"
131
+ require "rbconfig"
132
+ require "shellwords"
133
+ begin
134
+ require_relative "../lib/maf/shared/processes"
135
+ require_relative "../lib/maf/shared/project"
136
+ require_relative "../lib/maf/shared/roles"
137
+ require_relative "../lib/maf/shared/peak_rate"
138
+ require_relative "../lib/maf/shared/git_identity"
139
+ rescue LoadError
140
+ abort "dispatcher: the maf shared library is missing. Run maf update."
141
+ end
142
+
143
+ module Dispatcher
144
+ COORD_BIN = ".maf/bin/coord"
145
+ MAX_ATTEMPTS = 3
146
+ MAX_BACKOFF = 3600
147
+ REPORT_RETRIES = 1
148
+
149
+ Config = Struct.new(:role, :harness, :command, :model, :skill, :no_skill, :interval, :max_turns,
150
+ :timeout, :idle_timeout, :grace, :completion_signal, :abort_signal,
151
+ :cache_window, :max_session_runs, :max_context, :once, :poll_tasks, :worker, :coord_dir,
152
+ :verbose, :lean,
153
+ keyword_init: true)
154
+
155
+ module Log
156
+ class << self
157
+ # The last message. The worker status shows it as the result of a run.
158
+ attr_reader :last
159
+ end
160
+
161
+ def self.line(message)
162
+ @last = message
163
+ text = "[#{Time.now.utc.strftime("%Y-%m-%dT%H:%M:%SZ")}] dispatcher: #{message}"
164
+ path = ENV["DISPATCHER_LOG"]
165
+ path ? File.open(path, "a") { |file| file.puts(text) } : warn(text)
166
+ end
167
+ end
168
+
169
+ # Project finds the main checkout from any worktree. Lead roles own no
170
+ # tasks: they get an inbox prompt and no task polling.
171
+ Project = Maf::Shared::Project
172
+ LEADS = Maf::Shared::Roles::LEADS
173
+
174
+ # FlowConfig finds the MCP config of the flow in .maf/mcp/ (next to the
175
+ # coordination folder). The project's .mcp.json and opencode.json stay as
176
+ # they are. NOTE: lib/maf/setup_agent/launcher.rb passes the same files.
177
+ module FlowConfig
178
+ def self.path(config, name) = File.join(File.dirname(config.coord_dir), "mcp", name)
179
+ def self.claude_mcp(config) = File.exist?(file = path(config, "claude.json")) ? ["--mcp-config", file] : []
180
+
181
+ def self.opencode_env(config)
182
+ file = path(config, "opencode.json")
183
+ File.exist?(file) ? { "OPENCODE_CONFIG" => file } : {}
184
+ end
185
+ end
186
+
187
+ # EditGrant reads each agent's can_edit from .maf/config.json. On Hermes,
188
+ # a role with can_edit false starts without the file-writing toolsets.
189
+ module EditGrant
190
+ READ_ONLY_TOOLSETS = Maf::Shared::Roles::READ_ONLY_TOOLSETS
191
+
192
+ def self.read_only?(role)
193
+ Project.manifest.fetch("agents", []).any? { |agent| agent["role"] == role && agent["can_edit"] == false }
194
+ end
195
+ end
196
+
197
+ # Presence writes .maf/coordination/presence/<worker>.json once per cycle, so
198
+ # `coord who` shows the dispatcher as live while its pid runs.
199
+ module Presence
200
+ def self.write(config)
201
+ dir = File.join(config.coord_dir, "presence")
202
+ FileUtils.mkdir_p(dir)
203
+ File.write(File.join(dir, "#{config.worker}.json"), JSON.generate(record(config)))
204
+ end
205
+
206
+ def self.record(config)
207
+ { "role" => config.role, "worker" => config.worker, "mode" => "dispatch", "pid" => Process.pid,
208
+ "started" => Maf::Shared::Processes.started_at(Process.pid),
209
+ "seen_at" => Time.now.utc.strftime("%Y-%m-%dT%H:%M:%SZ") }
210
+ end
211
+ end
212
+
213
+ # HermesSkill resolves the skill flow.rb generated for a role: the name is
214
+ # <project>-<role>, and the dir is the manifest's hermes_dir (set by
215
+ # `flow.rb --hermes-dir`) or flow.rb's default ~/.hermes/skills.
216
+ module HermesSkill
217
+ DEFAULT_DIR = File.join(Dir.home, ".hermes", "skills")
218
+
219
+ def self.name(role) = "#{File.basename(Project.root)}-#{role}"
220
+ def self.dir = Project.manifest.fetch("hermes_dir", DEFAULT_DIR)
221
+ def self.installed?(role) = File.exist?(File.join(dir, name(role), "SKILL.md"))
222
+ end
223
+
224
+ # Session persists the agent's session ID across runs, per worker: two
225
+ # dispatchers for one role (different COORD_WORKER values) never share a
226
+ # session or a handoff note. The session file's mtime is the time of the
227
+ # last successful run in that session, so it tells whether the LLM prompt
228
+ # cache is still warm. A failed run does not update it, so the next run
229
+ # treats the cache as colder than it may be. That errs on the cheap side.
230
+ class Session
231
+ attr_reader :path, :handoff_path
232
+
233
+ def initialize(coord_dir, worker)
234
+ dir = File.join(coord_dir, "sessions")
235
+ @path = File.join(dir, "#{worker}.session")
236
+ @handoff_path = File.join(dir, "#{worker}.handoff.md")
237
+ end
238
+
239
+ def id = lines.first.to_s.empty? ? nil : lines.first
240
+
241
+ # The second line counts the runs of the session. Each resume adds the old
242
+ # context again, so --max-session-runs caps the count.
243
+ def runs = lines[1].to_i
244
+
245
+ # The third line holds the context tokens of the last model call, if the
246
+ # harness reports them. --max-context caps the size.
247
+ def context_tokens = lines[2].to_i
248
+
249
+ def save(session_id, context = nil)
250
+ count = session_id == id ? runs + 1 : 1
251
+ FileUtils.mkdir_p(File.dirname(@path))
252
+ File.write(@path, "#{session_id}\n#{count}\n#{context}\n")
253
+ end
254
+
255
+ def lines = File.exist?(@path) ? File.read(@path).split("\n").map(&:strip) : []
256
+ def clear = FileUtils.rm_f(@path)
257
+ def idle_seconds = File.exist?(@path) ? Time.now - File.mtime(@path) : Float::INFINITY
258
+ def handoff = File.exist?(@handoff_path) ? File.read(@handoff_path).strip : ""
259
+ def write_handoff(text) = FileUtils.mkdir_p(File.dirname(@handoff_path)) && File.write(@handoff_path, text)
260
+ end
261
+
262
+ Message = Struct.new(:path, :from, :text, :worker)
263
+
264
+ # Mailbox owns a role's inbox: take moves unread messages into processing/,
265
+ # ack moves them to read/, release moves them back (or to failed/).
266
+ class Mailbox
267
+ def initialize(coord_dir, role)
268
+ @dir = File.join(coord_dir, "inbox", role)
269
+ @worker_dir = /\A#{Regexp.escape(role)}-\d+\z/
270
+ end
271
+
272
+ # A crash mid-run leaves messages in processing/. Return them to the inbox.
273
+ def recover = Dir.glob(File.join(@dir, "processing", "*.md")).each { |path| move(path, @dir) }
274
+
275
+ # FYI messages alone start no run. A waking message takes them along.
276
+ def take
277
+ fold_worker_inboxes
278
+ paths = Dir.glob(File.join(@dir, "*.md")).sort
279
+ paths.all? { |path| Mailbox.fyi?(path) } ? [] : paths.filter_map { |path| claim(path) }
280
+ end
281
+
282
+ # NOTE: assets/coord writes the same name mark (coord msg --fyi).
283
+ def self.fyi?(path) = File.basename(path).include?(".fyi.")
284
+
285
+ def ack(messages) = messages.each { |message| move(message.path, File.join(@dir, "read")) }
286
+
287
+ def release(messages) = messages.each { |message| move(message.path, *retry_target(message.path)) }
288
+
289
+ # A run that hit a usage limit never reached the model, so it uses no attempt.
290
+ def restore(messages) = messages.each { |message| move(message.path, @dir) }
291
+
292
+ private
293
+
294
+ def claim(path)
295
+ target = move(path, File.join(@dir, "processing"))
296
+ target && parse(target)
297
+ end
298
+
299
+ def parse(path)
300
+ header, _blank, body = File.read(path).partition(/\n\n/)
301
+ Message.new(path, header[/^# from: (.+)$/, 1].to_s.strip, body.strip, header[/^# for: (.+)$/, 1]&.strip)
302
+ end
303
+
304
+ # Older versions of coord wrote a message for a worker to inbox/<worker>/.
305
+ # No session reads that folder. Move its unread messages to the role inbox.
306
+ # NOTE: assets/coord folds the same folders.
307
+ def fold_worker_inboxes
308
+ dirs = Dir.glob("#{@dir}-*").select { |dir| File.basename(dir).match?(@worker_dir) }
309
+ dirs.flat_map { |dir| Dir.glob(File.join(dir, "*.md")) }.each { |path| move(path, @dir) }
310
+ end
311
+
312
+ # The attempt count lives in the file name (<stamp>.retry2.md), so it
313
+ # survives a dispatcher restart. The stamp prefix keeps the sort order.
314
+ def retry_target(path)
315
+ attempts = path[/\.retry(\d+)\.md\z/, 1].to_i + 1
316
+ name = File.basename(path).sub(/(\.retry\d+)?\.md\z/, ".retry#{attempts}.md")
317
+ [attempts >= MAX_ATTEMPTS ? File.join(@dir, "failed") : @dir, name]
318
+ end
319
+
320
+ # Returns the new path, or nil when another process moved the file first.
321
+ def move(path, dir, name = File.basename(path))
322
+ FileUtils.mkdir_p(dir)
323
+ target = File.join(dir, name)
324
+ File.rename(path, target) && target
325
+ rescue Errno::ENOENT
326
+ nil
327
+ end
328
+ end
329
+
330
+ # TaskWatch decides when unclaimed tasks justify another agent run. A new
331
+ # task set dispatches at once; an unchanged one waits a doubling backoff.
332
+ # A set that vanishes and comes back unchanged keeps its old backoff.
333
+ class TaskWatch
334
+ def initialize(interval)
335
+ @interval = interval
336
+ @ids = []
337
+ @attempts = 0
338
+ @next_at = Time.at(0)
339
+ end
340
+
341
+ def due?(ids, now = Time.now) = !ids.empty? && (ids.sort != @ids || now >= @next_at)
342
+
343
+ def dispatched(ids, now = Time.now)
344
+ @attempts = ids.sort == @ids ? @attempts + 1 : 0
345
+ @ids = ids.sort
346
+ @next_at = now + [@interval * (2**(@attempts + 1)), MAX_BACKOFF].min
347
+ end
348
+ end
349
+
350
+ # Lessons reads the dead ends and corrections of graphify-out/reflections/
351
+ # LESSONS.md for one task, newest first. coord done and coord lesson save the
352
+ # notes that LESSONS.md sums up. A lesson stays when its note cites a node of
353
+ # a file that the graph query of the task found, or cites no known node.
354
+ # Without query output all lessons stay. LESSONS.md shows no nodes for a
355
+ # correction, so the nodes come from the notes in graphify-out/memory/.
356
+ module Lessons
357
+ HEADS = { "**Known dead ends**" => "dead end", "**Corrections**" => "correction" }.freeze
358
+ QUESTION = /\A[^:]+: "(.*?)"(?: — `| → |$)/
359
+ SOURCE = /\[src=(.+?)(?: loc=| community=|\])/
360
+
361
+ def self.for_task(out_dir, query)
362
+ file = File.join(out_dir, "reflections", "LESSONS.md")
363
+ lines = File.exist?(file) ? lines(File.read(file)) : []
364
+ files = query.to_s.scan(SOURCE).flatten.map { |src| relative(src, out_dir) }
365
+ (files.empty? ? lines : on_topic(lines, out_dir, files)).reverse
366
+ end
367
+
368
+ # Only the overall section counts. "## By topic" repeats the same lessons.
369
+ def self.lines(text)
370
+ kind = nil
371
+ text.split(/^## By topic/).first.lines.filter_map do |line|
372
+ kind = HEADS.find { |head, _| line.start_with?(head) }&.last if line.start_with?("**")
373
+ "#{kind}: #{line.delete_prefix("- ")}" if kind && line.start_with?("- ")
374
+ end
375
+ end
376
+
377
+ def self.on_topic(lines, out_dir, files)
378
+ nodes = cited_nodes(File.join(out_dir, "memory"))
379
+ where = node_files(out_dir, nodes.values.flatten.uniq)
380
+ lines.select { |line| topic?(nodes[line[QUESTION, 1]].to_a.filter_map { |id| where[id] }, files) }
381
+ end
382
+
383
+ def self.topic?(cited, files) = cited.empty? || !(cited & files).empty?
384
+
385
+ # The question of a note maps to the nodes that its notes cite.
386
+ def self.cited_nodes(dir)
387
+ Dir.glob(File.join(dir, "*.md")).each_with_object({}) do |path, map|
388
+ text = File.read(path)
389
+ question = quoted(text[/^question: (".*")$/, 1])
390
+ (map[question] ||= []).concat(Array(quoted(text[/^source_nodes: (\[.*\])$/, 1]))) if question
391
+ end
392
+ end
393
+
394
+ def self.quoted(text)
395
+ text && JSON.parse(text)
396
+ rescue JSON::ParserError
397
+ nil
398
+ end
399
+
400
+ # graphify may store the source path relative to the project or absolute.
401
+ def self.node_files(out_dir, ids)
402
+ return {} if ids.empty?
403
+
404
+ nodes = JSON.parse(File.read(File.join(out_dir, "graph.json")))["nodes"] || []
405
+ nodes.select { |node| ids.include?(node["id"]) }
406
+ .to_h { |node| [node["id"], relative(node["source_file"], out_dir)] }
407
+ rescue JSON::ParserError, SystemCallError
408
+ {}
409
+ end
410
+
411
+ def self.relative(path, out_dir) = path.to_s.delete_prefix("#{File.dirname(out_dir)}/")
412
+ end
413
+
414
+ # Prefetch adds live context to a dispatch prompt: the spec of the task the
415
+ # run works on (`coord show ID`) and `git log --oneline -10`. Each part is
416
+ # cut to LIMIT characters. A failed command is logged and left out. A run
417
+ # without a task (a message run) gets only the git log. The agent queries
418
+ # the knowledge graph itself, with a question that fits the task.
419
+ # A lead run also gets the artifact list of the open goals, with sizes, so
420
+ # it reads single files by name instead of printing a whole folder.
421
+ module Prefetch
422
+ LIMIT = 2000
423
+ GRAPH_BUDGET = 500
424
+ ARTIFACTS = "artifacts of the open goals"
425
+
426
+ def self.context(poller, task_id = nil, lead: false)
427
+ sections = commands(poller, task_id, lead).filter_map { |label, run| section(label, run) }
428
+ sections.empty? ? "" : "\nLive context at dispatch time:\n\n#{sections.join("\n")}"
429
+ end
430
+
431
+ def self.commands(poller, task_id, lead)
432
+ log = { "git log --oneline -10" => -> { git_log } }
433
+ log = log.merge(ARTIFACTS => -> { artifacts(poller) }) if lead
434
+ task_id ? task_sections(poller, task_id).merge(log) : log
435
+ end
436
+
437
+ # A graph query on the task description names the code of the task. Some
438
+ # models skip the graph rule and grep instead, and each grep output stays in
439
+ # the context of the run. The query runs only when the graph exists.
440
+ def self.task_sections(poller, task_id)
441
+ spec = poller.show(task_id)
442
+ { "coord show #{task_id}" => -> { spec } }.merge(graph(poller.coord_dir, spec.to_s[/^description: (.+)$/, 1]))
443
+ end
444
+
445
+ # The lessons section runs after the graph query and keeps the lessons of
446
+ # the files that the query found.
447
+ def self.graph(coord_dir, words)
448
+ path = File.expand_path(File.join(coord_dir, "..", "..", "graphify-out", "graph.json"))
449
+ return {} unless words && File.exist?(path)
450
+
451
+ query = nil
452
+ { "graphify query <task description> --budget #{GRAPH_BUDGET}" => -> { query = graph_query(words, path) } }
453
+ .merge(lessons(File.dirname(path), -> { query }))
454
+ end
455
+
456
+ LESSONS = "dead ends and corrections from graphify-out/reflections/LESSONS.md"
457
+
458
+ # No lesson for the task adds no section.
459
+ def self.lessons(out_dir, query = -> {})
460
+ { LESSONS => -> { Lessons.for_task(out_dir, query.call).join } }
461
+ end
462
+
463
+ def self.graph_query(words, path)
464
+ output = IO.popen(["graphify", "query", words, "--budget", GRAPH_BUDGET.to_s, "--graph", path], err: File::NULL,
465
+ &:read)
466
+ $?.success? ? output : nil
467
+ rescue SystemCallError
468
+ nil
469
+ end
470
+
471
+ def self.artifacts(poller)
472
+ files = poller.goal_ids&.flat_map { |id| Dir.glob(File.join(poller.coord_dir, "artifacts", id, "*")) }
473
+ files && (files.empty? ? "(none)" : files.sort.map { |path| artifact_line(path) }.join("\n"))
474
+ end
475
+
476
+ def self.artifact_line(path)
477
+ "#{File.basename(File.dirname(path))}/#{File.basename(path)} #{(File.size(path) / 1024.0).round(1)}k " \
478
+ "#{File.mtime(path).utc.strftime("%H:%M")}"
479
+ end
480
+
481
+ def self.section(label, run)
482
+ text = run.call
483
+ Log.line("prefetch failed: #{label}; the prompt goes out without it") unless text
484
+ text && !text.strip.empty? && "$ #{label}\n#{bound(text.strip)}\n"
485
+ end
486
+
487
+ def self.bound(text) = text.length > LIMIT ? "#{text[0, LIMIT]}\n[cut at #{LIMIT} characters]" : text
488
+
489
+ def self.git_log
490
+ output = IO.popen(%w[git log --oneline -10], err: File::NULL, &:read)
491
+ $?.success? ? output : nil
492
+ rescue SystemCallError
493
+ nil
494
+ end
495
+ end
496
+
497
+ # Poller reads the role's tasks through coord.
498
+ class Poller
499
+ TASK_ID = /^([0-9a-f-]{36})\b/
500
+
501
+ def initialize(env, role)
502
+ @env = env
503
+ @role = role
504
+ end
505
+
506
+ def unclaimed_task_ids = next_output.to_s.scan(TASK_ID).flatten
507
+
508
+ # The tasks that this worker claimed and did not finish.
509
+ def claimed_task_ids = read("next", @role, "--mine").to_s.scan(TASK_ID).flatten
510
+
511
+ # The first task that this worker can claim, or nil. `coord claim` is
512
+ # atomic, so two workers never get one task, and a run always has work.
513
+ def claim_first(ids) = ids.find { |id| coord("claim", id) }
514
+
515
+ def coord(*args) = CoordCall.run(@env, *args)
516
+ def next_output = read("next", @role)
517
+ def show(id) = read("show", id)
518
+ def goal_ids = read("goal", "list")&.scan(TASK_ID)&.flatten
519
+ def coord_dir = @env.fetch("COORD_DIR")
520
+
521
+ private
522
+
523
+ # The output of a coord command, or nil when the command fails.
524
+ def read(*args)
525
+ output = IO.popen(@env, CoordCall.command + args, err: File::NULL, &:read)
526
+ $?.success? ? output : nil
527
+ rescue Errno::ENOENT
528
+ nil
529
+ end
530
+ end
531
+
532
+ # CoordCall runs coord with this Ruby, so an old system Ruby on the
533
+ # shebang path never parses it.
534
+ module CoordCall
535
+ def self.command
536
+ local = File.expand_path(COORD_BIN)
537
+ [RbConfig.ruby, File.executable?(local) ? local : File.join(Project.root, COORD_BIN)]
538
+ end
539
+
540
+ # Returns true when the command succeeds.
541
+ def self.run(env, *args)
542
+ system(env, *command, *args, out: File::NULL, err: File::NULL)
543
+ rescue Errno::ENOENT
544
+ false
545
+ end
546
+ end
547
+
548
+ # LeadChores runs the periodic coord checks in the architect dispatcher, at
549
+ # most one time in EVERY seconds. `coord reap` releases the claims of
550
+ # stalled workers. `coord review-watch --once` reads new pull request
551
+ # reviews, only with a github section in .maf/config.json. Both commands
552
+ # write to the architect inbox, so the next cycle starts an architect run.
553
+ class LeadChores
554
+ EVERY = 300
555
+
556
+ def initialize(poller, role)
557
+ @poller, @role = poller, role
558
+ end
559
+
560
+ def run
561
+ return unless due?
562
+
563
+ @last = Time.now
564
+ commands.each { |args| @poller.coord(*args) || Log.line("coord #{args.first} failed") }
565
+ end
566
+
567
+ private
568
+
569
+ def commands = [["reap"], (%w[review-watch --once] if github?)].compact
570
+ def due? = @role == "architect" && (@last.nil? || Time.now - @last >= EVERY)
571
+ def github? = !Project.manifest.dig("github", "bot_user").to_s.empty?
572
+ end
573
+
574
+ Result = Struct.new(:output, :success, :timed_out, :reason)
575
+
576
+ # Limits are the run limits besides the hard --timeout. idle: seconds
577
+ # without output before the run fails (0 means off). grace: seconds to
578
+ # wait after a completion signal, and for a child that holds stdout.
579
+ # complete and abort: signal strings in the output (nil means off).
580
+ Limits = Struct.new(:idle, :grace, :complete, :abort) do
581
+ def self.from(config)
582
+ new(config.idle_timeout || 0, config.grace || Spawn::GRACE, config.completion_signal, config.abort_signal)
583
+ end
584
+ end
585
+
586
+ # Stream reads the output of a child in a thread and records the time of
587
+ # the last output.
588
+ class Stream
589
+ def initialize(reader)
590
+ @reader = reader
591
+ @lock, @text, @last_at = Mutex.new, String.new, Time.now
592
+ @thread = Thread.new { pump }
593
+ end
594
+
595
+ def idle_seconds = Time.now - @last_at
596
+ def include?(signal) = @lock.synchronize { @text.include?(signal.b) }
597
+
598
+ # A background grandchild can hold the pipe open after the agent exits.
599
+ # Then the reader thread is still blocked: kill it before closing the pipe,
600
+ # so it never dies with an IOError and never lingers.
601
+ def close(grace)
602
+ @thread.join(grace)
603
+ @thread.kill
604
+ @reader.close
605
+ @lock.synchronize { @text.dup.force_encoding(Encoding::UTF_8) }
606
+ end
607
+
608
+ private
609
+
610
+ def pump
611
+ loop { append(@reader.readpartial(4096)) }
612
+ rescue EOFError, IOError
613
+ nil
614
+ end
615
+
616
+ def append(chunk) = @lock.synchronize { (@text << chunk.b) && (@last_at = Time.now) }
617
+ end
618
+
619
+ # Watch decides why a run ends. :exit means the process exited. :complete
620
+ # and :abort mean a signal string appeared in the output. :idle means no
621
+ # output for limits.idle seconds. :timeout means the hard --timeout passed.
622
+ # After a completion signal, the process gets limits.grace seconds to exit.
623
+ class Watch
624
+ TICK = 0.2
625
+
626
+ def initialize(timeout, limits)
627
+ @deadline = Time.now + timeout
628
+ @limits = limits
629
+ end
630
+
631
+ def grace = @limits.grace
632
+
633
+ def wait(waiter, stream)
634
+ loop do
635
+ reason = signal(stream) || (waiter.join(TICK) ? :exit : limit(stream))
636
+ return settle(reason, waiter) if reason
637
+ end
638
+ end
639
+
640
+ # A signal in the last output chunk can arrive after the exit.
641
+ def final(reason, output)
642
+ return reason unless reason == :exit
643
+
644
+ signal(output) || :exit
645
+ end
646
+
647
+ private
648
+
649
+ def settle(reason, waiter)
650
+ waiter.join(grace) if reason == :complete
651
+ reason
652
+ end
653
+
654
+ def signal(text)
655
+ return :abort if seen?(text, @limits.abort)
656
+
657
+ :complete if seen?(text, @limits.complete)
658
+ end
659
+
660
+ def seen?(text, signal) = !signal.to_s.empty? && text.include?(signal)
661
+
662
+ def limit(stream)
663
+ return :timeout if Time.now >= @deadline
664
+
665
+ :idle if @limits.idle.positive? && stream.idle_seconds >= @limits.idle
666
+ end
667
+ end
668
+
669
+ # Spawn runs a command in its own process group. Any end reason except a
670
+ # normal exit kills the whole group (TERM, then KILL), which also stops the
671
+ # children of an `sh -c` template. A completion signal is success even when
672
+ # the process must be killed. An abort signal is failure.
673
+ module Spawn
674
+ GRACE = 5
675
+ TIMEOUTS = %i[timeout idle].freeze
676
+
677
+ def self.run(cmd, env:, timeout:, limits: Limits.new(0, GRACE))
678
+ reader, writer = IO.pipe
679
+ pid = Process.spawn(env, *cmd, in: File::NULL, out: writer, err: writer, pgroup: true)
680
+ collect(pid, Stream.new(reader.tap { writer.close }), Watch.new(timeout, limits))
681
+ rescue Errno::ENOENT => e
682
+ Result.new(e.message, false, false, :missing)
683
+ end
684
+
685
+ def self.collect(pid, stream, watch)
686
+ waiter = Process.detach(pid)
687
+ reason = watch.wait(waiter, stream)
688
+ stop(pid, waiter) unless reason == :exit
689
+ result(stream.close(watch.grace), waiter, watch, reason)
690
+ end
691
+
692
+ def self.result(output, waiter, watch, reason)
693
+ reason = watch.final(reason, output)
694
+ success = reason == :complete || (reason == :exit && waiter.value.success?)
695
+ Result.new(output, success, TIMEOUTS.include?(reason), reason)
696
+ end
697
+
698
+ def self.stop(pid, waiter)
699
+ Process.kill("TERM", -pid)
700
+ Process.kill("KILL", -pid) unless waiter.join(GRACE)
701
+ true
702
+ rescue Errno::ESRCH
703
+ true
704
+ end
705
+ end
706
+
707
+ # Events reads the JSON event lines of a harness run. Each adapter takes
708
+ # its session ID from one specific event, never from free text, so a
709
+ # prompt or a handoff note that the agent echoes cannot supply a false ID.
710
+ module Events
711
+ def self.all(output) = output.lines.filter_map { |line| parse(line.strip) }
712
+
713
+ def self.last(output, &match) = all(output).reverse.find(&match) || {}
714
+
715
+ def self.parse(line)
716
+ event = line.start_with?("{") ? JSON.parse(line) : nil
717
+ event.is_a?(Hash) ? event : nil
718
+ rescue JSON::ParserError
719
+ nil
720
+ end
721
+ end
722
+
723
+ # TokenUsage reads the token usage of one run and adds it to the totals of
724
+ # the worker in .maf/coordination/usage/<worker>.json. Claude Code, Hermes,
725
+ # Codex, and opencode report usage. A run without usage data adds nothing. A usage error
726
+ # never fails a run.
727
+ #
728
+ # input_tokens counts every input token, cache reads included.
729
+ # cached_input_tokens counts the cache reads, which cost a fraction of the
730
+ # input price. cache_write_input_tokens counts the cache writes, which cost
731
+ # more than the input price. A resume with a large cache write found a cold
732
+ # cache: the --cache-window is too long for the harness. Claude Code reports
733
+ # cache reads and cache writes outside its input_tokens, so they are added.
734
+ # Codex reports them inside.
735
+ class TokenUsage
736
+ def self.normalize(data)
737
+ return nil unless data.is_a?(Hash)
738
+
739
+ { "input_tokens" => input(data),
740
+ "cached_input_tokens" => int(data, "cached_input_tokens", "cache_read_input_tokens"),
741
+ "cache_write_input_tokens" => int(data, "cache_write_input_tokens", "cache_creation_input_tokens"),
742
+ "output_tokens" => int(data, "output_tokens", "completion_tokens") }
743
+ end
744
+
745
+ def self.input(data)
746
+ int(data, "input_tokens", "prompt_tokens") + int(data, "cache_read_input_tokens") +
747
+ int(data, "cache_creation_input_tokens")
748
+ end
749
+
750
+ def self.int(data, *names) = data.values_at(*names).compact.first.to_i
751
+
752
+ def self.label(usage) = usage.map { |key, value| "#{key.delete_suffix("_tokens")}=#{value}" }.join(" ")
753
+
754
+ def self.sum(list) = list.empty? ? nil : list.reduce { |total, usage| total.merge(usage) { |_, a, b| a + b } }
755
+
756
+ def initialize(coord_dir, worker)
757
+ @path = File.join(coord_dir, "usage", "#{worker}.json")
758
+ end
759
+
760
+ def add(usage)
761
+ return unless usage
762
+
763
+ totals = read
764
+ write(usage.to_h { |key, value| [key, totals[key].to_i + value] }.merge("runs" => totals["runs"].to_i + 1))
765
+ rescue SystemCallError, JSON::ParserError => e
766
+ Log.line("token usage not saved: #{e.message}")
767
+ end
768
+
769
+ private
770
+
771
+ def read = File.exist?(@path) ? JSON.parse(File.read(@path)) : {}
772
+ def write(data) = FileUtils.mkdir_p(File.dirname(@path)) && File.write(@path, JSON.generate(data))
773
+ end
774
+
775
+ # RunHistory appends one line per run to .maf/coordination/usage/<worker>.runs.jsonl:
776
+ # the usage, the session run (1 is a fresh session), the context of the last
777
+ # model call, the peak rate, and the session limits of the run. The dashboard
778
+ # turns the history into hints. A run without usage data adds no line.
779
+ class RunHistory
780
+ def initialize(config)
781
+ @config = config
782
+ @path = File.join(config.coord_dir, "usage", "#{config.worker}.runs.jsonl")
783
+ end
784
+
785
+ def add(usage, session_run, context)
786
+ return unless usage
787
+
788
+ FileUtils.mkdir_p(File.dirname(@path))
789
+ File.write(@path, "#{JSON.generate(line(usage, session_run, context))}\n", mode: "a")
790
+ rescue SystemCallError => e
791
+ Log.line("run history not saved: #{e.message}")
792
+ end
793
+
794
+ private
795
+
796
+ def line(usage, session_run, context)
797
+ { "at" => Time.now.utc.strftime("%Y-%m-%dT%H:%M:%SZ"), "session_run" => session_run, "context" => context,
798
+ "peak" => PeakRate.notice(@config) ? true : nil, "limits" => limits }.compact.merge(usage)
799
+ end
800
+
801
+ def limits = { "max_context" => @config.max_context, "max_session_runs" => @config.max_session_runs,
802
+ "cache_window" => @config.cache_window }
803
+ end
804
+
805
+ # PeakRate tells when a dispatched opencode run uses DeepSeek at the peak
806
+ # rate (Maf::Shared::PeakRate). The notice can show on a Chinese holiday.
807
+ module PeakRate
808
+ def self.deepseek?(config) = config.harness == "opencode" && OpencodeModel.for(config).to_s.include?("deepseek")
809
+
810
+ def self.notice(config, time = Time.now.utc)
811
+ return unless Maf::Shared::PeakRate.peak?(time) && deepseek?(config)
812
+
813
+ "info: DeepSeek peak hours (#{time.strftime("%H:%M")} UTC): this run costs 2x the off-peak rate"
814
+ end
815
+ end
816
+
817
+ # OpencodeModel finds the model of an opencode run in the opencode order:
818
+ # --model, the agent file, the project config, the user config.
819
+ module OpencodeModel
820
+ CONFIG_MODEL = /^\s*"model"\s*:\s*"([^"]+)"/
821
+ AGENT_MODEL = /\A---\n(?:(?!^---$).)*?^model:\s*(\S+)/m
822
+
823
+ def self.for(config) = config.model || agent(config.role) || configured
824
+ def self.agent(role) = find(File.join(".opencode", "agents", "#{role}.md"), AGENT_MODEL)
825
+ def self.configured = config_paths.lazy.filter_map { |path| find(path, CONFIG_MODEL) }.first
826
+ def self.config_paths = %w[opencode.json opencode.jsonc] + user_paths
827
+ def self.user_paths = %w[opencode.jsonc opencode.json].map { |name| File.join(home, name) }
828
+ def self.home = File.join(ENV.fetch("XDG_CONFIG_HOME", File.join(Dir.home, ".config")), "opencode")
829
+ def self.find(path, pattern) = File.exist?(path) ? File.read(path)[pattern, 1] : nil
830
+ end
831
+
832
+ module Harness
833
+ # Adapter is the part every built-in adapter shares. Each adapter sets
834
+ # STALE_SESSION to the error its CLI prints for an unknown session ID.
835
+ module Adapter
836
+ def stale_session?(output) = output.include?(self::STALE_SESSION)
837
+ def resume(session_id, *flag) = session_id ? [*flag, session_id] : []
838
+ def present(value) = value.to_s.empty? ? nil : value
839
+ def usage(_output) = nil
840
+ def context(_output) = nil
841
+ def model(_output) = nil
842
+ end
843
+
844
+ # Hermes: `hermes chat --oneshot`, JSONL output, skill preload.
845
+ module Hermes
846
+ extend Adapter
847
+ STALE_SESSION = "Session not found"
848
+
849
+ def self.build_command(config, prompt, session_id)
850
+ cmd = %w[hermes chat --oneshot --yolo --format stream-json -Q] + options(config)
851
+ cmd += ["--max-turns", config.max_turns.to_s, "--run-budget", run_budget(config).to_s]
852
+ cmd + resume(session_id, "--resume") + ["-q", prompt]
853
+ end
854
+
855
+ def self.options(config)
856
+ tools = EditGrant.read_only?(config.role) ? ["-t", EditGrant::READ_ONLY_TOOLSETS] : []
857
+ [*(["--skills", config.skill] if config.skill), *(["--model", config.model] if config.model), *tools]
858
+ end
859
+
860
+ # Stop Hermes before the dispatcher's kill, so it can still write its result event.
861
+ def self.run_budget(config) = [config.timeout - Spawn::GRACE, 1].max
862
+ def self.result_event(output) = Events.last(output) { |event| event["type"] == "result" }
863
+ def self.result_text(output) = result_event(output)["text"]
864
+ def self.session_id(output) = present(result_event(output)["session_id"])
865
+ def self.usage(output) = TokenUsage.normalize(result_event(output)["usage"])
866
+ end
867
+
868
+ # Claude Code: `claude -p` prints one JSON object with session_id and
869
+ # result. `--agent` loads .claude/agents/<role>.md as the session's role.
870
+ # `--mcp-config` takes many values. Put `--` before the prompt, or claude
871
+ # reads the prompt as one more config file and exits.
872
+ module Claude
873
+ extend Adapter
874
+ STALE_SESSION = "No conversation found with session ID"
875
+
876
+ # LEAN starts no MCP server outside --mcp-config and lists no skill.
877
+ # The user's CLAUDE.md and hooks stay.
878
+ LEAN = %w[--strict-mcp-config --disable-slash-commands].freeze
879
+
880
+ def self.build_command(config, prompt, session_id)
881
+ cmd = %w[claude -p --output-format json --permission-mode bypassPermissions --agent] + [config.role]
882
+ cmd += ["--model", config.model] if config.model
883
+ cmd += LEAN if config.lean
884
+ cmd + FlowConfig.claude_mcp(config) + resume(session_id, "--resume") + ["--", prompt]
885
+ end
886
+
887
+ def self.result_event(output) = Events.last(output) { |event| event.key?("result") }
888
+ def self.result_text(output) = result_event(output)["result"]
889
+ def self.session_id(output) = present(result_event(output)["session_id"])
890
+ def self.usage(output) = TokenUsage.normalize(result_event(output)["usage"])
891
+ # The result names each model that the run used. The main model comes first.
892
+ def self.model(output) = result_event(output)["modelUsage"]&.keys&.first
893
+ end
894
+
895
+ # Codex: `codex exec` (or `codex exec resume ID`) with JSONL events. Codex
896
+ # has no agent flag, so a fresh session gets .codex/prompts/<role>.md
897
+ # in front of the prompt. A resumed session already has it.
898
+ module Codex
899
+ extend Adapter
900
+ STALE_SESSION = "no rollout found for thread id"
901
+
902
+ def self.build_command(config, prompt, session_id)
903
+ cmd = %w[codex exec] + (session_id ? ["resume"] : [])
904
+ cmd += %w[--json --skip-git-repo-check --dangerously-bypass-approvals-and-sandbox]
905
+ cmd += ["-m", config.model] if config.model
906
+ cmd += Lean.flags if config.lean
907
+ cmd + resume(session_id) + [session_id ? prompt : role_prompt(config.role) + prompt]
908
+ end
909
+
910
+ # Lean drops the parts of the user's Codex setup that a one-shot run
911
+ # does not use. Each one adds tool schemas, skill lists, or MCP calls
912
+ # to every model call. Only the graphify server of this project stays.
913
+ module Lean
914
+ FEATURES = %w[plugins apps browser_use computer_use multi_agent].freeze
915
+ SERVER = /^\[mcp_servers\.([\w-]+)\]\s*$/
916
+
917
+ def self.flags = FEATURES.flat_map { |name| ["--disable", name] } + servers.flat_map { |name| off(name) }
918
+ def self.off(name) = ["-c", "mcp_servers.#{name}.enabled=false"]
919
+ def self.servers = File.exist?(config_path) ? File.read(config_path).scan(SERVER).flatten.uniq - [keep] : []
920
+ def self.keep = "graphify-#{File.basename(Project.root)}"
921
+ def self.config_path = File.join(ENV.fetch("CODEX_HOME", File.join(Dir.home, ".codex")), "config.toml")
922
+ end
923
+
924
+ def self.role_prompt(role)
925
+ path = File.join(".codex", "prompts", "#{role}.md")
926
+ File.exist?(path) ? "#{File.read(path)}\n\n" : ""
927
+ end
928
+
929
+ def self.result_text(output)
930
+ Events.last(output) { |event| event.dig("item", "type") == "agent_message" }.dig("item", "text")
931
+ end
932
+
933
+ def self.session_id(output)
934
+ present(Events.last(output) { |event| event["type"] == "thread.started" }["thread_id"])
935
+ end
936
+
937
+ # Codex reports the usage of each turn in a turn.completed event.
938
+ def self.usage(output)
939
+ turns = Events.all(output).select { |event| event["type"] == "turn.completed" }
940
+ TokenUsage.sum(turns.filter_map { |event| TokenUsage.normalize(event["usage"]) })
941
+ end
942
+ end
943
+
944
+ # opencode: `opencode run --format json`. `--agent` loads
945
+ # .opencode/agents/<role>.md. The result event shape is not verified, so
946
+ # the log shows no result text. opencode prints terminal escape codes and
947
+ # can glue two JSON objects onto one line, so the session ID is matched on
948
+ # the start of an event object (captured from a real run), not by parsing
949
+ # whole lines.
950
+ module Opencode
951
+ extend Adapter
952
+ STALE_SESSION = "Session not found"
953
+ EVENT_SESSION = /\{"type":"[\w.-]+","timestamp":\d+,"sessionID":"([^"]+)"/
954
+
955
+ def self.session_id(output) = output.scan(EVENT_SESSION).flatten.last
956
+
957
+ def self.build_command(config, prompt, session_id)
958
+ cmd = %w[opencode run --format json --agent] + [config.role]
959
+ cmd += ["-m", config.model] if config.model
960
+ cmd + resume(session_id, "--session") + [prompt]
961
+ end
962
+
963
+ def self.result_text(_output) = nil
964
+
965
+ # Each model call ends with a step_finish event. Its tokens.input
966
+ # excludes the cache reads and writes; reasoning is billed as output.
967
+ STEP_FINISH = /\{"type":"step_finish".*/
968
+
969
+ def self.steps(output) = output.scan(STEP_FINISH).filter_map { |json| Events.parse(json)&.dig("part", "tokens") }
970
+ def self.usage(output) = TokenUsage.sum(steps(output).map { |tokens| TokenUsage.normalize(flat(tokens)) })
971
+ def self.context(output) = steps(output).last&.then { |tokens| TokenUsage.input(flat(tokens)) }
972
+
973
+ def self.flat(tokens)
974
+ { "input_tokens" => tokens["input"].to_i, "cache_read_input_tokens" => tokens.dig("cache", "read").to_i,
975
+ "cache_creation_input_tokens" => tokens.dig("cache", "write").to_i,
976
+ "output_tokens" => tokens["output"].to_i + tokens["reasoning"].to_i }
977
+ end
978
+ end
979
+
980
+ # Generic: a user-provided shell template. No session resume, no parsing.
981
+ module Generic
982
+ PLACEHOLDERS = %i[prompt role skill model].freeze
983
+
984
+ def self.build_command(config, prompt, _session_id)
985
+ values = { prompt: prompt, role: config.role, skill: config.skill, model: config.model }
986
+ line = PLACEHOLDERS.reduce(config.command) { |cmd, key| cmd.gsub("%{#{key}}") { values[key].to_s.shellescape } }
987
+ ["sh", "-c", line]
988
+ end
989
+
990
+ def self.stale_session?(_output) = false
991
+ def self.session_id(_output) = nil
992
+ def self.result_text(_output) = nil
993
+ def self.usage(_output) = nil
994
+ def self.context(_output) = nil
995
+ def self.model(_output) = nil
996
+ end
997
+
998
+ REGISTRY = { "hermes" => Hermes, "claude" => Claude, "codex" => Codex, "opencode" => Opencode }.freeze
999
+
1000
+ def self.for(config)
1001
+ return Generic if config.command
1002
+
1003
+ REGISTRY.fetch(config.harness) { abort "dispatcher: unknown harness '#{config.harness}'. Use --command." }
1004
+ end
1005
+ end
1006
+
1007
+ # ReportBlock reads the typed result that an agent puts at the end of its
1008
+ # final reply: <report>{"status":"done","tests":"pass","next":"..."}</report>.
1009
+ # parse returns nil when no block is present. Otherwise it returns a Parsed
1010
+ # with the data or with the validation error.
1011
+ module ReportBlock
1012
+ REPORT_FORMAT = '<report>{"status":"<done|blocked|needs_review>","tests":"<pass|fail>","next":"<next>"}</report>'
1013
+ STATUSES = %w[done blocked needs_review].freeze
1014
+ TESTS = %w[pass fail].freeze
1015
+ PATTERN = %r{<report>(.*?)</report>}m
1016
+ Parsed = Struct.new(:data, :error)
1017
+
1018
+ def self.parse(text) = text.to_s.scan(PATTERN).flatten.last&.then { |body| check(decode(body.strip)) }
1019
+
1020
+ # Raw harness output holds the block inside a JSON string, with escaped quotes.
1021
+ def self.decode(body) = load(body) || load(load(%("#{body}")).to_s)
1022
+
1023
+ def self.load(text)
1024
+ JSON.parse(text)
1025
+ rescue JSON::ParserError
1026
+ nil
1027
+ end
1028
+
1029
+ def self.check(value)
1030
+ return Parsed.new(nil, "the report block is not a JSON object") unless value.is_a?(Hash)
1031
+
1032
+ error = errors(value).first
1033
+ Parsed.new(error ? nil : value, error)
1034
+ end
1035
+
1036
+ def self.errors(value)
1037
+ [("status must be one of #{STATUSES.join(", ")}" unless STATUSES.include?(value["status"])),
1038
+ ("tests must be pass or fail" unless value["tests"].nil? || TESTS.include?(value["tests"])),
1039
+ ("status done needs tests pass, not fail" if value["status"] == "done" && value["tests"] == "fail")].compact
1040
+ end
1041
+ end
1042
+
1043
+ # HandoffBlock reads the handoff note that an agent puts in its final reply:
1044
+ # <handoff>...</handoff>. The dispatcher writes the note to the handoff file,
1045
+ # so the agent spends no tool call and no extra model call on the note.
1046
+ module HandoffBlock
1047
+ PATTERN = %r{<handoff>(.*?)</handoff>}m
1048
+
1049
+ def self.parse(text) = text.to_s.scan(PATTERN).flatten.last&.then { |body| decode(body).strip }
1050
+
1051
+ # Raw harness output holds the block inside a JSON string, with escapes.
1052
+ def self.decode(body) = ReportBlock.load(%("#{body}")).then { |text| text.is_a?(String) ? text : body }
1053
+ end
1054
+
1055
+ # Verify runs the verify command of .maf/config.json after a successful
1056
+ # run, in the worktree. A non-zero exit makes the run no success. No key
1057
+ # means no check. Lead roles own no task, so they get no check.
1058
+ module Verify
1059
+ def self.pass?(config, env)
1060
+ command = Project.manifest["verify"].to_s
1061
+ return true if command.empty? || LEADS.include?(config.role)
1062
+
1063
+ check(Spawn.run(["sh", "-c", command], env: env, timeout: config.timeout), command)
1064
+ end
1065
+
1066
+ def self.check(result, command)
1067
+ Log.line("verify command failed (#{command}): #{result.output.lines.last(5).join.strip}") unless result.success
1068
+ result.success
1069
+ end
1070
+ end
1071
+
1072
+ # Outcome decides what a finished run means: :done, :waiting, or :failed.
1073
+ # No report block keeps the old rule: exit 0 is done. Status done is done.
1074
+ # Status blocked or needs_review is :waiting: the agent handled its input
1075
+ # and now waits for someone else. An invalid block is :failed. The session
1076
+ # stays, so the next run resumes it.
1077
+ class Outcome
1078
+ def initialize(session)
1079
+ @session = session
1080
+ end
1081
+
1082
+ def conclude(block, text)
1083
+ keep_handoff(text)
1084
+ return finished(text, block) if block.nil? || block.data&.fetch("status") == "done"
1085
+
1086
+ unfinished(block)
1087
+ end
1088
+
1089
+ private
1090
+
1091
+ def finished(text, block)
1092
+ Log.line("agent finished#{" (no report block)" unless block}: #{text.slice(0, 80)}")
1093
+ :done
1094
+ end
1095
+
1096
+ def unfinished(block)
1097
+ reason = block.error ? "an invalid report block: #{block.error}" : "status #{block.data["status"]}"
1098
+ Log.line("agent ended with #{reason}; next: #{block.data&.fetch("next", nil) || "-"}")
1099
+ block.error ? :failed : :waiting
1100
+ end
1101
+
1102
+ # Some models skip the handoff instruction. If no note exists, the agent's
1103
+ # final reply becomes the note, so a later fresh session still gets some
1104
+ # context. An existing note stays: a run that changed nothing keeps it.
1105
+ def keep_handoff(text)
1106
+ note = text.gsub(ReportBlock::PATTERN, "").gsub(HandoffBlock::PATTERN, "").strip
1107
+ return if note.empty? || !@session.handoff.empty?
1108
+
1109
+ @session.write_handoff(note)
1110
+ end
1111
+ end
1112
+
1113
+ # WorkerStatus writes .maf/coordination/status/<worker>.json. The dashboard
1114
+ # shows it. NOTE: assets/harness-hooks/context-watch.rb writes the same file
1115
+ # for an interactive session; both run standalone.
1116
+ class WorkerStatus
1117
+ def initialize(config)
1118
+ @config = config
1119
+ @path = File.join(config.coord_dir, "status", "#{config.worker}.json")
1120
+ end
1121
+
1122
+ def started(session_id, runs)
1123
+ save("mode" => "dispatch", "role" => @config.role, "harness" => @config.command ? "command" : @config.harness,
1124
+ "pid" => Process.pid, "running" => true, "run_started_at" => now, "session_id" => session_id,
1125
+ "session_runs" => runs, "model" => @config.model || read["model"])
1126
+ end
1127
+
1128
+ def finished(outcome, detail, model)
1129
+ run = { "finished_at" => now, "success" => outcome == :done, "outcome" => outcome.to_s,
1130
+ "detail" => detail.to_s[0, 300] }
1131
+ save({ "running" => false, "model" => model, "last_run" => run }.compact)
1132
+ end
1133
+
1134
+ # The board shows a paused worker, not a live one that fails each cycle.
1135
+ def paused(time) = save("paused_until" => time&.utc&.strftime("%Y-%m-%dT%H:%M:%SZ"))
1136
+
1137
+ # A halted worker starts no run until its dispatcher restarts.
1138
+ def halted(reason) = save("halted" => reason.to_s[0, 300])
1139
+
1140
+ private
1141
+
1142
+ def now = Time.now.utc.strftime("%Y-%m-%dT%H:%M:%SZ")
1143
+
1144
+ def read
1145
+ File.exist?(@path) ? JSON.parse(File.read(@path)) : {}
1146
+ rescue JSON::ParserError
1147
+ {}
1148
+ end
1149
+
1150
+ def save(data)
1151
+ FileUtils.mkdir_p(File.dirname(@path))
1152
+ File.write("#{@path}.tmp", JSON.generate(read.merge(data)))
1153
+ File.rename("#{@path}.tmp", @path)
1154
+ rescue SystemCallError => e
1155
+ Log.line("worker status not saved: #{e.message}")
1156
+ end
1157
+ end
1158
+
1159
+ # UsageLimit spots a provider usage or rate limit in the end of the output
1160
+ # of a run that exited with an error. Such a run never reached the model.
1161
+ # A timeout is never a limit: the output can be the agent's own text.
1162
+ module UsageLimit
1163
+ PATTERN = /usage limit|rate limit|hit your limit|limit reached|quota exceeded|too many requests/i
1164
+
1165
+ def self.hit?(result) = result.reason == :exit && result.output.lines.last(10).join.match?(PATTERN)
1166
+ end
1167
+
1168
+ # ModelError spots a model name that the harness does not know, such as a
1169
+ # typo in the roster. Such a run never reached the model, and each retry
1170
+ # fails the same way. Only a fix of the model name helps.
1171
+ module ModelError
1172
+ PATTERN = /unrecognized_model|unknown\ model|model[_\ ]?not[_\ ]?found|invalid\ model|
1173
+ model\b.{0,80}\b(?:does\ not\ exist|is\ not\ supported)/ix
1174
+
1175
+ def self.hit?(result) = result.reason == :exit && result.output.lines.last(10).join.match?(PATTERN)
1176
+ end
1177
+
1178
+ # Pause holds new runs after a usage limit. Each limit in a row doubles the
1179
+ # pause, up to MAX_BACKOFF. Any other outcome ends the series.
1180
+ class Pause
1181
+ FIRST = 900
1182
+
1183
+ def initialize
1184
+ @seconds, @until = 0, Time.at(0)
1185
+ end
1186
+
1187
+ attr_reader :until
1188
+
1189
+ def active?(now = Time.now) = now < @until
1190
+
1191
+ def start(now = Time.now)
1192
+ @seconds = @seconds.zero? ? FIRST : [@seconds * 2, MAX_BACKOFF].min
1193
+ @until = now + @seconds
1194
+ end
1195
+
1196
+ # Returns true when a series of limits ended.
1197
+ def clear = @seconds.positive?.tap { @seconds = 0 }
1198
+ end
1199
+
1200
+ # Runner runs one agent dispatch. It resumes the saved session only while
1201
+ # the LLM prompt cache is still warm (see --cache-window). A cold session
1202
+ # would resend the whole context at full price, so the dispatcher starts a
1203
+ # fresh session and hands it the handoff note instead. It also starts
1204
+ # fresh when the harness reports that the saved session no longer exists.
1205
+ class Runner
1206
+ def initialize(config, harness, env)
1207
+ @config = config
1208
+ @harness = harness
1209
+ @env = env
1210
+ @session = Session.new(config.coord_dir, config.worker)
1211
+ @status = WorkerStatus.new(config)
1212
+ end
1213
+
1214
+ # Returns the outcome of the run: :done, :waiting, :failed, :limited, or :broken.
1215
+ def dispatch(prompt) = run(prompt).tap { |outcome| @status.finished(outcome, Log.last, @model) }
1216
+
1217
+ private
1218
+
1219
+ def run(prompt)
1220
+ session_id = warm_session_id
1221
+ @session_run = session_id ? @session.runs + 1 : 1
1222
+ @status.started(session_id, @session.runs)
1223
+ result = attempt(prompt, session_id)
1224
+ result = retry_fresh(prompt) if session_id && @harness.stale_session?(result.output)
1225
+ report(result)
1226
+ end
1227
+
1228
+ def warm_session_id
1229
+ id = @session.id
1230
+ return id unless id && (cold? || full? || large?)
1231
+
1232
+ expire(id)
1233
+ end
1234
+
1235
+ def cold? = @session.idle_seconds > @config.cache_window
1236
+ def full? = @config.max_session_runs.to_i.positive? && @session.runs >= @config.max_session_runs
1237
+ def large? = @config.max_context.to_i.positive? && @session.context_tokens >= @config.max_context
1238
+
1239
+ def expire(id)
1240
+ Log.line("session #{id} #{expire_reason}; starting fresh with the handoff note")
1241
+ @session.clear
1242
+ nil
1243
+ end
1244
+
1245
+ def expire_reason
1246
+ return "idle #{(@session.idle_seconds / 60).round}m, past the cache window" if cold?
1247
+ return "has #{@session.runs} runs, the --max-session-runs limit" if full?
1248
+
1249
+ "has #{@session.context_tokens} context tokens, the --max-context limit"
1250
+ end
1251
+
1252
+ def attempt(prompt, session_id)
1253
+ Log.line("dispatching #{@config.role} (session: #{session_label(session_id)})")
1254
+ PeakRate.notice(@config)&.then { |text| Log.line(text) }
1255
+ text = Prompt.with_handoff(prompt, @session, session_id) + Prompt.signals(@config)
1256
+ spawn(@harness.build_command(@config, text, session_id))
1257
+ end
1258
+
1259
+ def session_label(id) = id ? "#{id}, idle #{(@session.idle_seconds / 60).round}m" : "new"
1260
+
1261
+ def spawn(cmd)
1262
+ result = Spawn.run(cmd, env: @env, timeout: @config.timeout, limits: Limits.from(@config))
1263
+ record(@harness.usage(result.output), result.output)
1264
+ @model = @harness.model(result.output) || @model
1265
+ result
1266
+ end
1267
+
1268
+ def record(usage, output)
1269
+ usage && Log.line("tokens: #{TokenUsage.label(usage)}")
1270
+ TokenUsage.new(@config.coord_dir, @config.worker).add(usage)
1271
+ RunHistory.new(@config).add(usage, @session_run, @harness.context(output))
1272
+ end
1273
+
1274
+ def retry_fresh(prompt)
1275
+ Log.line("session #{@session.id} not found; starting a fresh session")
1276
+ @session.clear
1277
+ attempt(prompt, nil)
1278
+ end
1279
+
1280
+ # A missing or invalid report block resumes the session with the error,
1281
+ # at most REPORT_RETRIES times. A harness without a session gets no retry.
1282
+ # The retry is part of the same run, so it does not count as a session run.
1283
+ def report(result, retries = REPORT_RETRIES)
1284
+ return failed(result) unless result.success
1285
+
1286
+ first_reply(result) if retries == REPORT_RETRIES
1287
+ block = ReportBlock.parse(@harness.result_text(result.output) || result.output)
1288
+ retry_report?(block, retries) ? report(ask_for_report(block), retries - 1) : conclude(block, result)
1289
+ end
1290
+
1291
+ def first_reply(result)
1292
+ @harness.session_id(result.output)&.then { |id| @session.save(id, @harness.context(result.output)) }
1293
+ note = HandoffBlock.parse(@harness.result_text(result.output) || result.output)
1294
+ @session.write_handoff(note) unless note.to_s.empty?
1295
+ end
1296
+
1297
+ def retry_report?(block, retries) = retries.positive? && @session.id && (block.nil? || block.error)
1298
+
1299
+ def ask_for_report(block)
1300
+ problem = block ? "Your report block is invalid: #{block.error}." : "Your final reply has no report block."
1301
+ Log.line("#{problem} Resuming session #{@session.id} once for the report block")
1302
+ spawn(@harness.build_command(@config, "#{problem}\n#{Prompt.report}", @session.id))
1303
+ end
1304
+
1305
+ def conclude(block, result)
1306
+ outcome = Outcome.new(@session).conclude(block, @harness.result_text(result.output).to_s)
1307
+ outcome != :done || Verify.pass?(@config, @env) ? outcome : :failed
1308
+ end
1309
+
1310
+ # Returns :broken for an unknown model, :limited for a usage limit, else :failed.
1311
+ def failed(result)
1312
+ Log.line("agent #{failure_label(result.reason)}: #{result.output.lines.last(5).join.strip}")
1313
+ report_timeout if result.reason == :timeout
1314
+ return :broken if ModelError.hit?(result)
1315
+
1316
+ UsageLimit.hit?(result) ? :limited : :failed
1317
+ end
1318
+
1319
+ TIMEOUT_NOTICE = "Dispatcher run of %<worker>s timed out after %<seconds>ss. The claims stay. " \
1320
+ "If the task needs more time, raise team.timeouts.%<role>s in .maf/config.json."
1321
+
1322
+ # The architect cannot see a dispatcher log. Without this notice, a task waits without a sign.
1323
+ def report_timeout
1324
+ return if @config.role == "architect"
1325
+
1326
+ text = format(TIMEOUT_NOTICE, worker: @config.worker, seconds: @config.timeout, role: @config.role)
1327
+ CoordCall.run(@env, "msg", "--from", @config.role, "architect", text)
1328
+ end
1329
+
1330
+ def failure_label(reason)
1331
+ { timeout: "timed out after #{@config.timeout}s", idle: "idle for #{@config.idle_timeout}s (no output)",
1332
+ abort: "sent the abort signal" }.fetch(reason, "failed")
1333
+ end
1334
+ end
1335
+
1336
+ # Prompt builds the one-shot work prompt. The coordination rules live in
1337
+ # the role prompt, so the prompt only adds what is specific to a dispatched run.
1338
+ module Prompt
1339
+ # Each tool output stays in the context for every later model call of
1340
+ # the run. A skill or a file that the work does not need costs on each call.
1341
+ ONE_SHOT = <<~TEXT
1342
+ You run as role %<role>s in one-shot mode. Follow your role instructions. They hold the coordination rules.
1343
+ Do not use --wait: nobody is waiting on this session.
1344
+ No person reads this run. Skip session-start skills such as caveman. Read a skill file only if the work needs it.
1345
+ TEXT
1346
+
1347
+ # The dispatcher claims each task. A worker that claims a task itself can
1348
+ # take the task of a run that starts next.
1349
+ MESSAGE_END = "Handle the messages above. Do not claim a task: the dispatcher claims tasks for you.\n" \
1350
+ "When every message is handled, stop.\n"
1351
+ LEAD_END = <<~TEXT
1352
+ You are a lead role. Never claim a task. Handle the messages above.
1353
+ Start from your handoff note. Read only the artifacts that the messages or the note name.
1354
+ Never print a whole artifact folder. For a revised artifact, read the diff against the reviewed copy.
1355
+ Do not send a message only to acknowledge. Send a message only with a result, a question, or a decision.
1356
+ When every message is handled, stop.
1357
+ TEXT
1358
+
1359
+ def self.messages(messages, role)
1360
+ bodies = messages.map { |message| "--- #{heading(message)} ---\n#{message.text}\n" }
1361
+ ending = LEADS.include?(role) ? LEAD_END : MESSAGE_END
1362
+ "The dispatcher took these messages from your inbox:\n\n#{bodies.join("\n")}\n#{rules(role)}#{ending}"
1363
+ end
1364
+
1365
+ # A worker has no inbox of its own, so a message for one worker reaches
1366
+ # any worker of the role.
1367
+ def self.heading(message)
1368
+ fyi = " (FYI: no reply needed)" if Mailbox.fyi?(message.path)
1369
+ "message from #{message.from}#{" for worker #{message.worker}" if message.worker}#{fyi}"
1370
+ end
1371
+
1372
+ TASK = <<~TEXT
1373
+ The dispatcher claimed task %<id>s for you. Skip the claim step of your work loop.
1374
+ Do this task only. Finish it with coord done. Then stop. Do not claim another task.
1375
+ TEXT
1376
+
1377
+ def self.task(role, id) = "#{format(TASK, id: id)}#{rules(role)}"
1378
+
1379
+ RESUME = <<~TEXT
1380
+ You hold claimed tasks that a previous run did not finish: %<ids>s.
1381
+ For each task, run coord show <id> first. If the status is not pending, or the worker is not you, skip the task.
1382
+ Otherwise continue the task on its task branch. Finish it with coord done.
1383
+ Do not claim another task. When these tasks are done, stop.
1384
+ TEXT
1385
+
1386
+ def self.resume(role, ids) = "#{format(RESUME, ids: ids.join(", "))}#{rules(role)}"
1387
+ def self.rules(role) = format(ONE_SHOT, role: role)
1388
+
1389
+ HANDOFF = <<~TEXT
1390
+ If this run changed the state of your work, put a handoff note for your next session in your final reply:
1391
+ <handoff>At most 300 words: current state, decisions, open questions, and the next step.</handoff>
1392
+ Do not write the note to a file. If this run changed nothing, leave out the handoff block.
1393
+ TEXT
1394
+
1395
+ # A fresh session starts with the previous session's handoff note. A
1396
+ # resumed session already has that context. Every run writes a new note
1397
+ # while its cache is still warm, which is cheap.
1398
+ REPORT = <<~TEXT
1399
+ End your final reply with one report block on its own line:
1400
+ %<format>s
1401
+ Use status done only when the work is complete. Use blocked or needs_review otherwise.
1402
+ TEXT
1403
+
1404
+ SIGNALS = <<~TEXT
1405
+ When the work is complete, print %<complete>s on its own line as the last output.
1406
+ If you give up, print %<abort>s on its own line instead.
1407
+ TEXT
1408
+
1409
+ def self.signals(config)
1410
+ return "" unless config.completion_signal || config.abort_signal
1411
+
1412
+ format(SIGNALS, complete: config.completion_signal || "-", abort: config.abort_signal || "-")
1413
+ end
1414
+
1415
+ def self.report = format(REPORT, format: ReportBlock::REPORT_FORMAT)
1416
+
1417
+ def self.with_handoff(prompt, session, session_id)
1418
+ note = session.handoff
1419
+ intro = session_id || note.empty? ? "" : "Handoff note from your previous session:\n#{note}\n\n"
1420
+ "#{intro}#{prompt}\n#{HANDOFF}#{report}"
1421
+ end
1422
+ end
1423
+
1424
+ # StopSignal ends the poll loop after the current agent run. `maf retire`
1425
+ # sends TERM. The agent runs in its own process group, so TERM does not
1426
+ # reach the agent, and the agent finishes its work first. The trap writes
1427
+ # to a pipe, so a TERM also ends the wait between two cycles.
1428
+ class StopSignal
1429
+ def initialize
1430
+ @reader, @writer = IO.pipe
1431
+ trap("TERM") { signal }
1432
+ end
1433
+
1434
+ # Returns true after the full interval, false when a stop was requested.
1435
+ def wait(seconds) = IO.select([@reader], nil, nil, seconds).nil?
1436
+
1437
+ private
1438
+
1439
+ def signal
1440
+ @writer.write_nonblock(".")
1441
+ rescue IOError, IO::WaitWritable
1442
+ nil
1443
+ end
1444
+ end
1445
+
1446
+ # Main runs the poll + dispatch loop.
1447
+ class Main
1448
+ def initialize(config)
1449
+ @config = config
1450
+ @runner = Runner.new(config, Harness.for(config), Main.run_env(config))
1451
+ @mailbox = Mailbox.new(config.coord_dir, config.role)
1452
+ @poller = Poller.new(Main.run_env(config), config.role)
1453
+ @watch, @resume = Array.new(2) { TaskWatch.new(config.interval) }
1454
+ @pause, @status = Pause.new, WorkerStatus.new(config)
1455
+ end
1456
+
1457
+ def self.run_env(config)
1458
+ env = { "COORD_DIR" => config.coord_dir, "COORD_ROLE" => config.role, "COORD_WORKER" => config.worker,
1459
+ "COORD_DISPATCHED" => "1", "PWD" => Dir.pwd }
1460
+ taskrc = File.join(config.coord_dir, "taskrc")
1461
+ env.merge(File.exist?(taskrc) ? { "TASKRC" => taskrc } : {}, FlowConfig.opencode_env(config))
1462
+ end
1463
+
1464
+ def run
1465
+ start
1466
+ cycle_forever
1467
+ Log.line("stopped (TERM)") unless @config.once
1468
+ rescue Interrupt
1469
+ Log.line("stopped (interrupted)")
1470
+ end
1471
+
1472
+ # At most one agent run per cycle. A message run comes first. The next
1473
+ # cycle claims a task.
1474
+ def cycle
1475
+ Presence.write(@config)
1476
+ chores.run
1477
+ return held_cycle if held?
1478
+
1479
+ worked = dispatch_messages || (@config.poll_tasks && (dispatch_claimed || dispatch_tasks))
1480
+ Log.line("no work this cycle") if @config.verbose && !worked
1481
+ end
1482
+
1483
+ private
1484
+
1485
+ def context(task_id = nil) = Prefetch.context(@poller, task_id, lead: LEADS.include?(@config.role))
1486
+
1487
+ def dispatch(prompt) = @runner.dispatch(prompt).tap { |outcome| pace(outcome) }
1488
+
1489
+ def pace(outcome)
1490
+ return halt if outcome == :broken
1491
+
1492
+ outcome == :limited ? pause : unpause
1493
+ end
1494
+
1495
+ def unpause = @pause.clear && @status.paused(nil)
1496
+ def held? = @halted || @pause.active?
1497
+ def held_cycle = @config.verbose && Log.line(held_label)
1498
+
1499
+ def held_label
1500
+ @halted ? "halted: the harness does not know the model" : "paused for a usage limit until #{@pause.until.utc}"
1501
+ end
1502
+
1503
+ HALT_NOTICE = "Dispatcher of %<worker>s halted: the %<harness>s harness does not know model %<model>s. " \
1504
+ "Each retry fails the same way. Fix the model in the role file and in " \
1505
+ ".maf/coordination/workers.json, then run `maf worker restart %<worker>s`. The claims stay."
1506
+
1507
+ # An unknown model never fixes itself. The dispatcher stops new runs and
1508
+ # escalates once, so the work does not wait without a sign.
1509
+ def halt
1510
+ @halted = true
1511
+ @status.halted(Log.last)
1512
+ Log.line("unknown model: no new run until the dispatcher restarts")
1513
+ text = format(HALT_NOTICE, worker: @config.worker, harness: @config.harness,
1514
+ model: @config.model || "(default)")
1515
+ CoordCall.run(Main.run_env(@config), "escalate", text)
1516
+ end
1517
+
1518
+ def pause
1519
+ @pause.start
1520
+ @status.paused(@pause.until)
1521
+ Log.line("usage limit: no new run until #{@pause.until.utc}")
1522
+ end
1523
+ def chores = @chores ||= LeadChores.new(@poller, @config.role)
1524
+
1525
+ def start
1526
+ Log.line("started: role=#{@config.role} #{harness_label} #{timing_label}")
1527
+ @mailbox.recover
1528
+ @stop = StopSignal.new
1529
+ end
1530
+ def timing_label = "interval=#{@config.interval}s cache_window=#{@config.cache_window}s " \
1531
+ "max_session_runs=#{@config.max_session_runs} max_context=#{@config.max_context}"
1532
+ def harness_label = @config.command ? "command=#{@config.command.inspect}" : "harness=#{@config.harness}"
1533
+
1534
+ def cycle_forever
1535
+ cycle
1536
+ cycle while !@config.once && @stop.wait(@config.interval)
1537
+ end
1538
+
1539
+ def dispatch_messages
1540
+ messages = @mailbox.take
1541
+ return false if messages.empty?
1542
+
1543
+ settle(messages, dispatch(Prompt.messages(messages, @config.role) + context))
1544
+ end
1545
+
1546
+ # A run that waits for someone else handled its messages, so they are read.
1547
+ # Only a failed run returns them for another try. A usage limit or an
1548
+ # unknown model returns them without an attempt.
1549
+ SETTLE = { failed: :release, limited: :restore, broken: :restore }.freeze
1550
+
1551
+ # Tasks still unclaimed after a message run count as dispatched, so the
1552
+ # next cycle backs off instead of starting another run for them.
1553
+ def settle(messages, outcome)
1554
+ @mailbox.public_send(SETTLE.fetch(outcome, :ack), messages)
1555
+ @watch.dispatched(@poller.unclaimed_task_ids) if @config.poll_tasks
1556
+ true
1557
+ end
1558
+
1559
+ # A run that timed out or crashed leaves its claims. Finish them before new work.
1560
+ def dispatch_claimed
1561
+ ids = @poller.claimed_task_ids
1562
+ return false unless @resume.due?(ids)
1563
+
1564
+ dispatch(Prompt.resume(@config.role, ids) + context(ids.first))
1565
+ @resume.dispatched(ids)
1566
+ true
1567
+ end
1568
+
1569
+ # The claim comes first, so the run never starts without work.
1570
+ def dispatch_tasks
1571
+ ids = @poller.unclaimed_task_ids
1572
+ return false unless @watch.due?(ids)
1573
+
1574
+ id = @poller.claim_first(ids)
1575
+ @watch.dispatched(ids)
1576
+ id && run_task(id)
1577
+ end
1578
+
1579
+ def run_task(id)
1580
+ dispatch(Prompt.task(@config.role, id) + context(id))
1581
+ true
1582
+ end
1583
+ end
1584
+
1585
+ # Options parses the command line into a Config.
1586
+ module Options
1587
+ VALUES = { "--harness NAME" => [:harness, String], "--command TEMPLATE" => [:command, String],
1588
+ "--model M" => [:model, String], "--skill NAME" => [:skill, String],
1589
+ "--interval S" => [:interval, Integer], "--max-turns N" => [:max_turns, Integer],
1590
+ "--timeout S" => [:timeout, Integer], "--cache-window S" => [:cache_window, Integer],
1591
+ "--max-session-runs N" => [:max_session_runs, Integer], "--max-context N" => [:max_context, Integer],
1592
+ "--idle-timeout S" => [:idle_timeout, Integer], "--grace S" => [:grace, Integer],
1593
+ "--completion-signal TEXT" => [:completion_signal, String],
1594
+ "--abort-signal TEXT" => [:abort_signal, String] }.freeze
1595
+ SWITCHES = { "--no-skill" => [:no_skill, true], "--no-poll-tasks" => [:poll_tasks, false],
1596
+ "--full-harness" => [:lean, false],
1597
+ "--once" => [:once, true], "--verbose" => [:verbose, true] }.freeze
1598
+ DEFAULTS = { harness: "hermes", interval: 60, max_turns: 50, idle_timeout: 0, grace: Spawn::GRACE,
1599
+ poll_tasks: true, lean: true,
1600
+ no_skill: false, once: false, verbose: false }.freeze
1601
+
1602
+ DEFAULT_TIMEOUT = 1500
1603
+ DEFAULT_SESSION_RUNS = 5
1604
+ DEFAULT_CONTEXT = 150_000
1605
+
1606
+ # Claude Code caches for 1 hour. The OpenAI cache of Codex is best effort:
1607
+ # a probe missed it after 5 idle minutes. A cold resume costs more than a
1608
+ # fresh session with the handoff note.
1609
+ CACHE_WINDOWS = { "codex" => 300 }.freeze
1610
+ DEFAULT_CACHE_WINDOW = 3300
1611
+
1612
+ def self.parse(argv, env = ENV)
1613
+ config = Config.new(**DEFAULTS)
1614
+ parser(config).parse!(argv)
1615
+ config.role = argv.shift || abort("dispatcher: role is required (usage: dispatcher ROLE)")
1616
+ validate(finish(config, env))
1617
+ end
1618
+
1619
+ # Reviews and data tasks need more time than code tasks. The team section
1620
+ # of .maf/config.json sets a limit for each role: "team": { "timeouts": { "reviewer": 2400 } }.
1621
+ def self.role_timeout(role)
1622
+ value = Project.manifest.dig("team", "timeouts", role)
1623
+ value && Integer(value, exception: false)
1624
+ end
1625
+
1626
+ def self.validate(config)
1627
+ abort "dispatcher: role must not be empty" if config.role.strip.empty?
1628
+ positive = config.interval.positive? && config.timeout.positive?
1629
+ abort "dispatcher: --interval and --timeout must be positive" unless positive
1630
+ abort "dispatcher: --cache-window must be 0 or more" if config.cache_window.negative?
1631
+ validate_limits(config)
1632
+ end
1633
+
1634
+ def self.validate_limits(config)
1635
+ limits = [config.idle_timeout, config.grace, config.max_session_runs, config.max_context]
1636
+ abort "dispatcher: --idle-timeout, --grace, --max-session-runs, and --max-context must be 0 or more" if
1637
+ limits.any?(&:negative?)
1638
+ config
1639
+ end
1640
+
1641
+ def self.parser(config)
1642
+ OptionParser.new("Usage: dispatcher ROLE [options]") do |o|
1643
+ VALUES.each { |flag, (key, type)| o.on(flag, type) { |v| config[key] = v } }
1644
+ SWITCHES.each { |flag, (key, value)| o.on(flag) { config[key] = value } }
1645
+ o.on("-h", "--help") { puts(o) || exit }
1646
+ end
1647
+ end
1648
+
1649
+ def self.finish(config, env)
1650
+ config.coord_dir = File.expand_path(env.fetch("COORD_DIR", ".maf/coordination"))
1651
+ config.worker = env.fetch("COORD_WORKER", "#{config.role}-bot")
1652
+ config.poll_tasks = false if LEADS.include?(config.role)
1653
+ skill(limits(config))
1654
+ end
1655
+
1656
+ def self.limits(config)
1657
+ config.timeout ||= role_timeout(config.role) || DEFAULT_TIMEOUT
1658
+ session_limits(config, Project.manifest.dig("team", "limits", config.role) || {})
1659
+ end
1660
+
1661
+ # The team section of .maf/config.json sets the session limits of a role:
1662
+ # "team": { "limits": { "frontend-developer": { "max_context": 80000 } } }.
1663
+ # `maf worker restart ID --max-context N` writes them. A command-line flag wins.
1664
+ # NOTE: assets/dashboard reads the limits of each run from the run history.
1665
+ def self.session_limits(config, saved)
1666
+ config.cache_window ||= saved["cache_window"] || CACHE_WINDOWS.fetch(config.harness, DEFAULT_CACHE_WINDOW)
1667
+ config.max_session_runs ||= saved["max_session_runs"] || DEFAULT_SESSION_RUNS
1668
+ config.max_context ||= saved["max_context"] || DEFAULT_CONTEXT
1669
+ config
1670
+ end
1671
+
1672
+ # --no-skill drops a skill. Without --skill, the generated Hermes skill of the role loads.
1673
+ def self.skill(config)
1674
+ config.skill = nil if config.no_skill
1675
+ config.skill ||= HermesSkill.name(config.role) if !config.no_skill && HermesSkill.installed?(config.role)
1676
+ config
1677
+ end
1678
+ end
1679
+ end
1680
+
1681
+ if __FILE__ == $PROGRAM_NAME
1682
+ # The agents run `coord`. Put the folder of this script (.maf/bin) on PATH.
1683
+ ENV["PATH"] = [File.dirname(File.expand_path(__FILE__)), ENV["PATH"]].compact.join(File::PATH_SEPARATOR)
1684
+ # Each run inherits the git persona of the agents.
1685
+ Maf::Shared::GitIdentity.apply(Dispatcher::Project.manifest)
1686
+ Dispatcher::Main.new(Dispatcher::Options.parse(ARGV)).run
1687
+ end