flowviant 0.39.0 → 0.40.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/bin/lib/live.mjs CHANGED
@@ -34,7 +34,7 @@ import {
34
34
  ALLOW_PATCHES,
35
35
  } from './config.mjs';
36
36
  import { c, info, ok, warn } from './ui.mjs';
37
- import { sleep } from './claude.mjs';
37
+ import { sleep, runTurn, mcpFor, sawSentinel, blockedId } from './claude.mjs';
38
38
  import {
39
39
  git,
40
40
  resetWorktree,
@@ -44,7 +44,7 @@ import {
44
44
  clearWip,
45
45
  } from './git.mjs';
46
46
  import { applyPatch, fileDiffs, ownerCurrentBranch, withPatchLock } from './patch.mjs';
47
- import { RUNTIMES } from './runtimes.mjs';
47
+ import { RUNTIMES, runtimeById, drivableHere, mediated } from './runtimes.mjs';
48
48
  import { loadPreviewConfig, startPreview } from './preview.mjs';
49
49
  import { materializeInto, scrub as envScrub } from './env.mjs';
50
50
 
@@ -53,16 +53,19 @@ import { materializeInto, scrub as envScrub } from './env.mjs';
53
53
  const LIVE_TARGET_URL = FLEET_URL.replace(/\/agents\/?$/, '/live-target');
54
54
 
55
55
  /**
56
- * What a live session can build, sent on every claim.
56
+ * What THIS WORKER can build, sent on every claim.
57
57
  *
58
- * Read off the registry rather than written as `['claude']` so that the day a
59
- * second runtime gains an SDK session, marking `live: true` there is the whole
60
- * change — a hand-written list here would keep withholding its work and the
61
- * reason would be three files away from the flag that looks like it decides.
58
+ * Deliberately the same predicate the roster report uses (`drivableHere`), not a
59
+ * hand-written list. The claim and the report answer the same question to two
60
+ * different consumers, and if they ever disagree the daemon either claims work it
61
+ * cannot build — the exact bug this argument was added to close — or refuses work
62
+ * it can. One source, so they cannot drift.
63
+ *
64
+ * Note this is NOT the live-session list. A live worker builds Claude tasks
65
+ * through the SDK and everything else through `driveSubprocess`, so both belong
66
+ * here; `live` chooses the driver, it does not gate participation.
62
67
  */
63
- const LIVE_RUNTIMES = Object.values(RUNTIMES)
64
- .filter((r) => r.live === true && r.mcp && r.args)
65
- .map((r) => r.id);
68
+ const DRIVABLE_HERE = Object.values(RUNTIMES).filter(drivableHere).map((r) => r.id);
66
69
  // Short TTL + a heartbeat that re-asserts while the tunnel is alive. So a live
67
70
  // preview stays linked indefinitely (survives long reviews), but one whose
68
71
  // daemon DIED ungracefully (no more heartbeats) drops off the card within the
@@ -204,6 +207,36 @@ questions, delivery summaries, commits, or PRs — reference keys by NAME only
204
207
  (e.g. "set STRIPE_KEY"). Never screenshot a terminal or page that displays a
205
208
  credential, and never commit an env file.`;
206
209
 
210
+ /**
211
+ * The same contract, for a runtime that has no live session.
212
+ *
213
+ * A non-live runtime is driven as a SUBPROCESS: one headless turn, then the
214
+ * process exits and the daemon decides what happens next. That transport cannot
215
+ * see tool calls the way the SDK stream can — there is no `tool_use` block to
216
+ * read `complete` or `report_blocker` off — so the turn has to SAY how it ended.
217
+ * Hence the sentinels, which are the same three words the legacy poll path has
218
+ * always used; this is a transport detail bolted onto the contract, not a second
219
+ * contract, which is why it is SYSTEM_LIVE plus an epilogue rather than a
220
+ * parallel prompt that would drift from it.
221
+ *
222
+ * The claim instruction that opens SYSTEM_SINGLE is deliberately absent: the
223
+ * daemon already claimed this task before spawning, so a second claim would come
224
+ * back `active_run` and the turn would waste itself puzzling over it.
225
+ */
226
+ const SYSTEM_SUBPROCESS = `${SYSTEM_LIVE}
227
+
228
+ HOW THIS TURN ENDS. You are running as a one-shot process, not in a live session,
229
+ so the daemon can only see what you print. End your turn by printing EXACTLY ONE
230
+ of these on a line by itself, as the last thing you output:
231
+ DONE — the task is complete (you called complete, and opened the
232
+ PR unless placement is "patch")
233
+ BLOCKED:<blockerId> — you called report_blocker and are waiting on a human. Use
234
+ the id report_blocker returned. STOP after printing it;
235
+ you will be run again with the answer.
236
+ Print nothing else on that line. Do not print a sentinel you have not earned — a
237
+ DONE without a complete call strands the work, and the team is told the task
238
+ finished when it did not.`;
239
+
207
240
  /** The brief minus the parts rendered as prose below (conversations, the ask). */
208
241
  function briefWithoutThread(brief) {
209
242
  const {
@@ -568,6 +601,648 @@ async function landPatch({ mcpUrl, token, runId, intentId, repoRoot, cwd, patchB
568
601
  warn(`patch not applied: ${reason}`);
569
602
  }
570
603
 
604
+ /**
605
+ * The FORM a mediated runtime fills in instead of calling tools.
606
+ *
607
+ * Every field maps to one control-plane call the daemon makes on the agent's
608
+ * behalf, which is why the shape is this small: it is not a report, it is the
609
+ * arguments to `complete` / `report_blocker` / `attach_pr` with the runId taken
610
+ * out (the agent has no business naming a run it cannot see).
611
+ */
612
+ const MEDIATED_RESULT_SCHEMA = {
613
+ type: 'object',
614
+ required: ['outcome', 'summary'],
615
+ additionalProperties: false,
616
+ properties: {
617
+ outcome: { type: 'string', enum: ['done', 'blocked', 'failed'] },
618
+ summary: { type: 'string' },
619
+ prUrl: { type: 'string' },
620
+ branch: { type: 'string' },
621
+ blockerQuestion: { type: 'string' },
622
+ blockerOptions: { type: 'array', items: { type: 'string' } },
623
+ criteria: {
624
+ type: 'array',
625
+ items: {
626
+ type: 'object',
627
+ required: ['index', 'met'],
628
+ properties: {
629
+ index: { type: 'number' },
630
+ met: { type: 'boolean' },
631
+ note: { type: 'string' },
632
+ },
633
+ },
634
+ },
635
+ },
636
+ };
637
+
638
+ /**
639
+ * The contract for a runtime that cannot reach the flowviant MCP server.
640
+ *
641
+ * SYSTEM_LIVE tells the agent to call tools. This one tells it there are none —
642
+ * which has to be said explicitly, because the brief it is about to read is full
643
+ * of references to a control plane it cannot touch, and an agent that spends its
644
+ * turn hunting for `report_progress` is an agent that does not build anything.
645
+ */
646
+ const SYSTEM_MEDIATED = `You are a Flowviant build agent working ONE task, running FULLY AUTONOMOUSLY.
647
+ There is NO interactive user, NO terminal to ask in, and — importantly — NO
648
+ Flowviant tools available to you in this session. Do not look for them. A daemon
649
+ is watching this run and reports on your behalf: your file edits, commands and
650
+ progress are already visible to the team as you work.
651
+
652
+ Do the work described in the brief below, in the checkout you are running in.
653
+ Ship it exactly as the brief's "placement" says:
654
+ • placement "patch": commit your change with a one-line message and STOP. No
655
+ branch, no push, no PR — the daemon carries it into the owner's checkout.
656
+ • placement "branch" (the default): create the branch named in "branchName" (use
657
+ that exact name), push it, and open ONE draft pull request with
658
+ \`gh pr create --draft\`. If the brief has a "baseBranch", target it with
659
+ \`--base <baseBranch>\`. NEVER merge.
660
+
661
+ THEN RETURN THE RESULT FORM as your final answer, and nothing else — it is a
662
+ strict JSON schema and it is the only way anything you did gets recorded:
663
+ • outcome "done" — you finished. Include a plain-language "summary" for the
664
+ humans (it becomes your delivery card), the "prUrl" and "branch" if you opened
665
+ one, and a "criteria" self-report indexing into the brief's "done when" list.
666
+ • outcome "blocked" — you hit a decision only a human can make. Put the question
667
+ in "blockerQuestion" and any choices in "blockerOptions", and STOP. You will be
668
+ run again with the answer.
669
+ • outcome "failed" — you could not do it. Say why in "summary".
670
+ Do not invent a prUrl you did not open, and do not report "done" for work you did
671
+ not finish: the summary is shown to a person as a claim about what exists.
672
+ SECRETS: env files (.env, .dev.vars, …) hold the team's synced secrets. Their
673
+ VALUES must NEVER appear in the summary, in commits, or in a PR — reference keys
674
+ by NAME only. Never commit an env file.`;
675
+
676
+ /**
677
+ * Walk forward from an opening brace to its MATCHING close, or null.
678
+ *
679
+ * String-aware, because the thing being matched is JSON and this object's whole
680
+ * job is to carry human prose: a summary reading `fixed the {x} case` would
681
+ * otherwise close the object early, and an escaped quote inside it would end the
682
+ * string early. Depth counting alone is not enough.
683
+ */
684
+ function balancedSpan(text, start) {
685
+ let depth = 0;
686
+ let inStr = false;
687
+ let esc = false;
688
+ for (let i = start; i < text.length; i++) {
689
+ const ch = text[i];
690
+ if (inStr) {
691
+ if (esc) esc = false;
692
+ else if (ch === '\\') esc = true;
693
+ else if (ch === '"') inStr = false;
694
+ continue;
695
+ }
696
+ if (ch === '"') inStr = true;
697
+ else if (ch === '{') depth++;
698
+ else if (ch === '}' && --depth === 0) return text.slice(start, i + 1);
699
+ }
700
+ return null;
701
+ }
702
+
703
+ /** Pull the result object out of a turn's output. */
704
+ function parseMediatedResult(out) {
705
+ const text = String(out ?? '').trim();
706
+ if (!text) return null;
707
+ // The whole answer SHOULD be the object — that is what schema enforcement
708
+ // buys. Fall back to the last balanced {...} for a runtime that wraps it in a
709
+ // fence or adds a sentence, so one chatty model does not strand a finished
710
+ // build. Last rather than first: any preamble comes before the answer.
711
+ const direct = tryJson(text);
712
+ if (direct) return direct;
713
+ // Each candidate open brace gets its OWN close, found by scanning forward.
714
+ // The previous version anchored every attempt on `text.lastIndexOf('}')` —
715
+ // recomputed per iteration but loop-INVARIANT, so it was always the final `}`
716
+ // of the whole output. Only the start moved; the end never retreated. Any
717
+ // sentence after the object containing a brace (`Note: the } above closes it`)
718
+ // therefore made every slice unparseable, and a FINISHED build came back as
719
+ // `stalled` after two nudges. Reproduced before fixing.
720
+ let tried = 0;
721
+ for (let i = text.lastIndexOf('{'); i >= 0; i = text.lastIndexOf('{', i - 1)) {
722
+ // A candidate must OPEN ITS OWN LINE (whitespace aside). A form echoed
723
+ // mid-sentence is how a hypothetical became a delivery card: `I would
724
+ // return {"outcome":"done",…} once done. But I could not…` parsed as done
725
+ // and posted a completed card for a failed build (reproduced). A real form
726
+ // — bare, fenced, or followed by notes — opens at a line start, and a
727
+ // wrapper that inlines it gets the nudge, which asks for the bare object
728
+ // anyway. A wrong card has no recovery; a nudge does. Skipped candidates
729
+ // don't count against the bound, which also keeps a trailing prose brace
730
+ // from burning slots the real object needs.
731
+ const bol = text.lastIndexOf('\n', i - 1) + 1;
732
+ if (!text.slice(bol, i).trim()) {
733
+ // Bounded: an unbalanced brace scans to end-of-text, and a build's output
734
+ // can be very long. The real object is at the end — 200 candidates is far
735
+ // past any honest wrapper and keeps a pathological output from stalling
736
+ // the turn loop instead of the model.
737
+ if (++tried > 200) break;
738
+ const span = balancedSpan(text, i);
739
+ const cand = span && tryJson(span);
740
+ if (cand) return cand;
741
+ }
742
+ if (i === 0) break;
743
+ }
744
+ return null;
745
+ }
746
+ function tryJson(s) {
747
+ try {
748
+ const v = JSON.parse(s);
749
+ return v && typeof v === 'object' && typeof v.outcome === 'string' ? v : null;
750
+ } catch {
751
+ return null;
752
+ }
753
+ }
754
+
755
+ /**
756
+ * The server's own rule for a PR URL (`mcpAttachPrSchema`), checked BEFORE the
757
+ * call instead of discovered as a swallowed rejection after it.
758
+ *
759
+ * The result schema can only say `prUrl: string` — the model writes the value
760
+ * freehand — and the server's zod REJECTS a non-github or non-pull URL, so a
761
+ * plausible-looking mistake meant the PR was never linked and the task never
762
+ * moved to `review`, with nothing anywhere saying so. Deliberately NOT expressed
763
+ * as a `pattern` in MEDIATED_RESULT_SCHEMA: the mediated path is the one whose
764
+ * schema enforcement is a vendor flag we verified empirically on exactly one
765
+ * version, and adding a keyword that CLI may not implement risks the working
766
+ * case to defend the broken one. Validate on our side, where we know the rules.
767
+ */
768
+ const PR_URL_RE = /^https:\/\/github\.com\/[^/]+\/[^/]+\/pull\/\d+/;
769
+
770
+ /**
771
+ * Coerce the model's criteria self-report into the shape `complete` accepts.
772
+ *
773
+ * MEDIATED_RESULT_SCHEMA can only say `index: number`; the server says
774
+ * `int().min(0)`, note ≤500, array ≤50 — and ONE bad row makes the whole
775
+ * `complete` call throw, which on this path means no delivery card at all. So
776
+ * repairable rows are repaired and the rest dropped: a self-report missing an
777
+ * entry is worth far more than a card that never arrives.
778
+ */
779
+ function sanitizeCriteria(criteria) {
780
+ if (!Array.isArray(criteria)) return null;
781
+ const rows = criteria
782
+ // A negative index is DROPPED, not clamped: Math.max(0, …) would silently
783
+ // re-attribute the row to criterion 0, which is a wrong self-report rather
784
+ // than a missing one.
785
+ .filter((c) => c && Number.isFinite(c.index) && c.index >= 0 && typeof c.met === 'boolean')
786
+ .map((c) => ({
787
+ index: Math.trunc(c.index),
788
+ met: c.met,
789
+ ...(typeof c.note === 'string' && c.note ? { note: c.note.slice(0, 500) } : {}),
790
+ }))
791
+ .slice(0, 50);
792
+ return rows.length ? rows : null;
793
+ }
794
+
795
+ /**
796
+ * Post the delivery card, and get one honest retry at it.
797
+ *
798
+ * The retry drops `criteria` on purpose. runId, outcome and summary are all
799
+ * daemon-controlled and already clamped, so the only argument that can still be
800
+ * rejected is the one the model wrote — and dropping it also re-enters
801
+ * `complete`'s idempotent branch, which is what recovers the OTHER failure the
802
+ * server documents here (`task_status_failed`: the run row moved but the task's
803
+ * status write didn't, and the fix is to call again).
804
+ *
805
+ * Returns false only on an EXPLICIT `ok: false`. An unparseable or empty
806
+ * response is treated as success: this verdict decides whether the run is left
807
+ * for the stale sweep to roll back and rebuild, and a transient hiccup is not
808
+ * worth rebuilding a finished task over.
809
+ */
810
+ async function postComplete({ mcpUrl, token, runId, outcome, summary, criteria }) {
811
+ const rows = sanitizeCriteria(criteria);
812
+ const call = (args) =>
813
+ mcpCall(mcpUrl, token, 'complete', args).catch((e) => ({
814
+ ok: false,
815
+ reason: e?.message ?? String(e),
816
+ }));
817
+ const base = { runId, outcome, summary };
818
+ let res = await call(rows ? { ...base, criteria: rows } : base);
819
+ if (res?.ok === false && rows) {
820
+ warn(`complete rejected (${res.reason ?? 'unknown'}) — retrying without the criteria self-report`);
821
+ res = await call(base);
822
+ }
823
+ return res?.ok !== false;
824
+ }
825
+
826
+ /**
827
+ * Drive a task with a runtime that cannot reach the MCP server at all.
828
+ *
829
+ * THE CLI DOES THE WORK; THE DAEMON DOES THE PAPERWORK. Antigravity's server
830
+ * list is machine-wide (measured — a workspace-local config is never read), so
831
+ * handing it a per-lane worker token is impossible and handing it a shared one
832
+ * would make every lane indistinguishable to the control plane. Instead nothing
833
+ * is handed over: the agent gets a brief and returns a filled-in form, and every
834
+ * control-plane call below is made by the daemon with the lane's OWN token, over
835
+ * its own HTTP. Per-lane isolation is preserved by removing the need for the
836
+ * agent to have a credential at all.
837
+ *
838
+ * The cost, and it is real: NO ON-DEMAND CONTEXT. A direct-MCP agent can call
839
+ * search_wiki or get_module_files the moment it realises it does not understand
840
+ * a subsystem. A mediated one only knows what was in the brief. That is a
841
+ * genuine capability difference and it is why this is the fallback shape rather
842
+ * than the default — runtimes that CAN hold an MCP config keep the full tool
843
+ * surface.
844
+ *
845
+ * Also not yet carried: attach_evidence. A mediated agent cannot upload a
846
+ * screenshot, so its delivery card arrives without the proof a Claude lane's
847
+ * would have. Fixable (the agent writes files, the daemon uploads them) and
848
+ * deliberately not in this first pass.
849
+ */
850
+ async function driveMediated({
851
+ runtimeId,
852
+ mcpUrl,
853
+ token,
854
+ runId,
855
+ intentId,
856
+ title,
857
+ cwd,
858
+ brief,
859
+ isPatch,
860
+ patchBase,
861
+ repoRoot,
862
+ baseRef,
863
+ label,
864
+ seedText,
865
+ isAlive,
866
+ onChild,
867
+ markLanded,
868
+ }) {
869
+ const rt = runtimeById(runtimeId);
870
+ const dir = mkdtempSync(join(tmpdir(), 'flowviant-schema-'));
871
+ const schemaPath = join(dir, 'result.schema.json');
872
+
873
+ // The agent cannot call report_progress, so the daemon narrates for it off the
874
+ // parsed activity stream. Throttled: a build touches hundreds of files and the
875
+ // thread is for humans, not for a filesystem log.
876
+ let lastReport = 0;
877
+ const narrate = (activity) => {
878
+ if (!activity?.label) return;
879
+ const now = Date.now();
880
+ if (now - lastReport < 8000) return;
881
+ lastReport = now;
882
+ void mcpCall(mcpUrl, token, 'report_progress', {
883
+ runId,
884
+ kind: activity.kind === 'error' ? 'error' : 'progress',
885
+ message: envScrub(activity.label),
886
+ }).catch(() => {});
887
+ };
888
+
889
+ let prompt = seedText;
890
+ let resume = false;
891
+ let nudges = 0;
892
+ try {
893
+ // Inside the try so a failed write (disk full) still removes `dir` in the
894
+ // finally instead of leaking one temp directory per attempt.
895
+ writeFileSync(schemaPath, JSON.stringify(MEDIATED_RESULT_SCHEMA), { mode: 0o600 });
896
+ for (;;) {
897
+ if (!isAlive()) return { outcome: 'blocked', title, intentId };
898
+ let out = '';
899
+ try {
900
+ out = await runTurn({
901
+ prompt,
902
+ resume,
903
+ system: SYSTEM_MEDIATED,
904
+ cwd,
905
+ runtime: runtimeId,
906
+ // NO MCP. That is the entire point of this path.
907
+ resultSchemaArgs: rt.resultSchema?.(schemaPath) ?? [],
908
+ label,
909
+ model: brief.agentModel || undefined,
910
+ effort: brief.agentEffort || undefined,
911
+ onActivity: narrate,
912
+ onSpawn: (ch) => onChild?.(ch),
913
+ });
914
+ } catch (e) {
915
+ return { outcome: 'error', error: e?.message ?? String(e), title, intentId };
916
+ } finally {
917
+ onChild?.(null);
918
+ }
919
+ if (!isAlive()) return { outcome: 'blocked', title, intentId };
920
+ resume = true;
921
+
922
+ const rl = classifyRateLimit(String(out).slice(-4000));
923
+ const result = parseMediatedResult(out);
924
+ if (!result && rl.isRateLimit) {
925
+ await mcpCall(mcpUrl, token, 'report_paused', { runId, resetAt: rl.resetAt }).catch(() => {});
926
+ return { outcome: 'rate_limited', resetAt: rl.resetAt, runId, title, intentId };
927
+ }
928
+
929
+ if (!result) {
930
+ // No form came back. Same posture as a missing sentinel on the other
931
+ // paths: nudge, then give up rather than invent an outcome.
932
+ if (nudges < 2) {
933
+ nudges++;
934
+ prompt =
935
+ 'You did not return the result form. Return ONLY the JSON object described in your instructions, describing what you did.';
936
+ continue;
937
+ }
938
+ return { outcome: 'stalled', title, intentId };
939
+ }
940
+
941
+ if (result.outcome === 'blocked') {
942
+ // Clamp AND scrub, same discipline as postComplete below, and for the
943
+ // same reason: on this path the DAEMON is the caller, so the model never
944
+ // sees the server's zod rejection and cannot self-correct. The server
945
+ // caps question at 2000 and options at 10×500 (questionPayloadSchema) —
946
+ // an oversize value posted raw is a rejected post, i.e. a question that
947
+ // silently never reaches the human. And the question is model narration
948
+ // leaving the box, exactly what the uplink scrub exists for.
949
+ const q =
950
+ envScrub(String(result.blockerQuestion ?? result.summary ?? '').trim()).slice(0, 2000) ||
951
+ 'The agent stopped and did not say why.';
952
+ const options = (Array.isArray(result.blockerOptions) ? result.blockerOptions : [])
953
+ .filter((o) => typeof o === 'string' && o.trim())
954
+ .map((o) => envScrub(o.trim()).slice(0, 500))
955
+ .filter(Boolean)
956
+ .slice(0, 10);
957
+ const post = () =>
958
+ mcpCall(mcpUrl, token, 'report_blocker', {
959
+ runId,
960
+ taskId: intentId,
961
+ type: 'question',
962
+ payload: { question: q, ...(options.length ? { options } : {}) },
963
+ }).catch(() => null);
964
+ // One retry: reportBlockerOnce is idempotent server-side, and a dropped
965
+ // response is the documented reason it is.
966
+ let posted = await post();
967
+ if (!posted?.blockerId && !posted?.id) {
968
+ await sleep(2);
969
+ posted = await post();
970
+ }
971
+ const blockerId = posted?.blockerId ?? posted?.id ?? null;
972
+ if (!blockerId) {
973
+ // The question exists only in this process. Say why on the way out —
974
+ // silence here reads identically to a human who has not answered yet.
975
+ //
976
+ // 'error', NOT 'blocked': on every driver 'blocked' means "shutting
977
+ // down mid-park", and runLiveWorker BREAKS on it — a lane that ends
978
+ // its loop is never respawned (workers.delete fires only on roster
979
+ // removal), so returning it here turned one failed post into a lane
980
+ // that sat dead-but-listed until the daemon restarted. 'error' takes
981
+ // the refresh-token-and-retry path, and the shared finally's
982
+ // checkpoint keeps the work for whoever picks the task back up.
983
+ warn(`report_blocker did not return an id (${posted?.reason ?? posted?.raw ?? 'no response'}) — the question was not posted`);
984
+ return { outcome: 'error', error: 'report_blocker failed', title, intentId };
985
+ }
986
+ const res = await waitForResolution(mcpUrl, token, blockerId, isAlive);
987
+ if (res.status === 'resolved') {
988
+ prompt = `The human answered your blocker: ${JSON.stringify(res.answer)}\nApply it and continue, then return the result form.`;
989
+ nudges = 0;
990
+ continue;
991
+ }
992
+ if (res.status === 'timeout') return { outcome: 'parked', title, intentId };
993
+ return { outcome: 'blocked', title, intentId };
994
+ }
995
+
996
+ // done / failed — either way the turn is over and the thread gets a card.
997
+ //
998
+ // NOTHING FROM HERE DOWN IS BEST-EFFORT, and that is the difference this
999
+ // path has to make up for. On the direct-MCP paths the AGENT makes these
1000
+ // calls and sees the rejection, so it corrects and retries; a mediated
1001
+ // agent never learns that the daemon's call failed. Swallowing them (which
1002
+ // is what this shipped as) produced the worst available outcome: the PR
1003
+ // silently unlinked, the task never moved to `review`, no delivery card —
1004
+ // and `markLanded()` firing anyway, so the shared `finally` DELETED the WIP
1005
+ // checkpoint for work the control plane had never been told about.
1006
+ if (result.prUrl && !isPatch) {
1007
+ const prUrl = String(result.prUrl).trim();
1008
+ if (!PR_URL_RE.test(prUrl)) {
1009
+ // The model can fix this one, so ask it to — it already opened the PR.
1010
+ if (nudges < 2) {
1011
+ nudges++;
1012
+ prompt =
1013
+ `"${prUrl}" is not a GitHub pull request URL (expected https://github.com/<owner>/<repo>/pull/<number>). ` +
1014
+ 'Do NOT redo any work and do NOT open another PR. Return the result form again with the real URL of the ' +
1015
+ 'pull request you already opened, or omit prUrl entirely if you did not open one.';
1016
+ continue;
1017
+ }
1018
+ warn(`attach_pr skipped: unusable prUrl ${prUrl}`);
1019
+ } else {
1020
+ const attached = await mcpCall(mcpUrl, token, 'attach_pr', {
1021
+ runId,
1022
+ prUrl,
1023
+ ...(result.branch ? { branch: String(result.branch) } : {}),
1024
+ }).catch((e) => ({ ok: false, reason: e?.message ?? String(e) }));
1025
+ if (attached?.ok === false) warn(`attach_pr rejected: ${attached.reason ?? 'unknown'}`);
1026
+ }
1027
+ }
1028
+ clearTaskMarker(cwd);
1029
+ const done = result.outcome === 'done';
1030
+ if (done && isPatch) {
1031
+ await landPatch({ mcpUrl, token, runId, intentId, repoRoot, cwd, patchBase, baseRef });
1032
+ }
1033
+ const carded = await postComplete({
1034
+ mcpUrl,
1035
+ token,
1036
+ runId,
1037
+ outcome: done ? 'completed' : 'failed',
1038
+ summary: envScrub(String(result.summary ?? '').slice(0, 4000)),
1039
+ criteria: result.criteria,
1040
+ });
1041
+ if (!carded) {
1042
+ // No delivery card exists, so this run is not done however the work
1043
+ // ended. `landed` deliberately stays false: the shared finally takes one
1044
+ // last checkpoint instead of deleting the WIP ref, and the run is left
1045
+ // active for the stale sweep to roll back and re-dispatch — recoverable,
1046
+ // unlike reporting success into a thread that shows nothing.
1047
+ warn('complete failed — leaving the run for the server to reclaim');
1048
+ return { outcome: 'error', error: 'complete rejected', title, intentId };
1049
+ }
1050
+ if (done) markLanded();
1051
+ return { outcome: done ? 'done' : 'stalled', title, intentId };
1052
+ }
1053
+ } finally {
1054
+ rmSync(dir, { recursive: true, force: true });
1055
+ }
1056
+ }
1057
+
1058
+ /**
1059
+ * Drive a task with a runtime that has no live session.
1060
+ *
1061
+ * Same job, same outcomes, different transport. `runLiveTask` owns everything
1062
+ * around this — the claim, the worktree, the branch/patch/stack setup, the WIP
1063
+ * checkpoint timer, the diffstat sampler and the teardown — and calls one of two
1064
+ * drivers in the middle. That split is the whole point: routing non-live
1065
+ * runtimes at the WORKER level instead (the obvious shortcut, since the legacy
1066
+ * poll worker already spawns Codex) would have sent them down a path with no
1067
+ * patch landing, no WIP checkpoint/restore and no preview, so a `placement:
1068
+ * "patch"` task would follow its instructions to commit-and-stop and then wait
1069
+ * forever for a daemon that never picks it up.
1070
+ *
1071
+ * What is genuinely lost versus a live session, stated plainly rather than
1072
+ * discovered: a teammate's mid-task message cannot interrupt a running turn. It
1073
+ * lands between turns instead, which is the same place a poll-mode message has
1074
+ * always landed. Everything else — blockers, stop, teardown, release, patches,
1075
+ * checkpoints — behaves the same because it is the same surrounding code.
1076
+ */
1077
+ async function driveSubprocess({
1078
+ runtimeId,
1079
+ mcpUrl,
1080
+ token,
1081
+ runId,
1082
+ intentId,
1083
+ title,
1084
+ cwd,
1085
+ brief,
1086
+ isPatch,
1087
+ patchBase,
1088
+ repoRoot,
1089
+ baseRef,
1090
+ label,
1091
+ seedText,
1092
+ afterId,
1093
+ isAlive,
1094
+ onChild,
1095
+ markLanded,
1096
+ }) {
1097
+ let resume = false;
1098
+ let nudges = 0;
1099
+ let held = false;
1100
+ let prompt = seedText;
1101
+
1102
+ for (;;) {
1103
+ if (!isAlive()) return { outcome: 'blocked', title, intentId };
1104
+
1105
+ // A fresh token hand-off per turn: the worker token is minted per lane and
1106
+ // may rotate between turns, and for Codex it rides in the environment rather
1107
+ // than on disk, so there is nothing to clean up in that case (`dir` is null).
1108
+ const { dir, args: mcpArgs, env: mcpEnv } = mcpFor(runtimeId, token, mcpUrl);
1109
+ let out = '';
1110
+ try {
1111
+ out = await runTurn({
1112
+ prompt,
1113
+ resume,
1114
+ system: SYSTEM_SUBPROCESS,
1115
+ cwd,
1116
+ runtime: runtimeId,
1117
+ mcpArgs,
1118
+ mcpEnv,
1119
+ label,
1120
+ // Per-task first, this machine's default second — off the BRIEF, which is
1121
+ // the task we actually hold, never the roster's guess.
1122
+ model: brief.agentModel || undefined,
1123
+ effort: brief.agentEffort || undefined,
1124
+ onSpawn: (ch) => onChild?.(ch),
1125
+ });
1126
+ } catch (e) {
1127
+ // Defensive only. runTurn resolves rather than rejects on a failed child —
1128
+ // see the rate-limit note below — so this catches a throw from the
1129
+ // plumbing around it, not from the CLI.
1130
+ return { outcome: 'error', error: e?.message ?? String(e), title, intentId };
1131
+ } finally {
1132
+ if (dir) rmSync(dir, { recursive: true, force: true });
1133
+ onChild?.(null);
1134
+ }
1135
+ if (!isAlive()) return { outcome: 'blocked', title, intentId };
1136
+
1137
+ // A USAGE LIMIT reads differently here than it does in a live session, and
1138
+ // getting that wrong would show the user's own plan limit as a Flowviant
1139
+ // stall. The SDK THROWS on a 429, which is why the live path classifies an
1140
+ // exception; `runTurn` resolves with whatever the child printed no matter
1141
+ // how it exited, so the only evidence a subprocess leaves is text.
1142
+ //
1143
+ // Read the TAIL only, and only when the turn produced no sentinel. The whole
1144
+ // transcript is the model's narration, and an agent that writes "we should
1145
+ // handle rate limit errors" into a code comment would otherwise park a
1146
+ // perfectly healthy run. A fatal CLI error is the last thing printed. Both
1147
+ // ways of being wrong here are recoverable — a false park retries after the
1148
+ // reset, a missed limit reads as a stall and is re-dispatched — so the tail
1149
+ // heuristic buys the common case without risking the work.
1150
+ if (!sawSentinel(out, 'DONE') && !blockedId(out)) {
1151
+ const rl = classifyRateLimit(out.slice(-4000));
1152
+ if (rl.isRateLimit) {
1153
+ await mcpCall(mcpUrl, token, 'report_paused', { runId, resetAt: rl.resetAt }).catch(() => {});
1154
+ return { outcome: 'rate_limited', resetAt: rl.resetAt, runId, title, intentId };
1155
+ }
1156
+ }
1157
+
1158
+ // Every turn after the first continues the CLI's own session where the
1159
+ // runtime supports it (`--continue` / `resume --last`), so the agent keeps
1160
+ // its reasoning rather than re-reading the brief cold each time.
1161
+ resume = true;
1162
+
1163
+ const bid = blockedId(out);
1164
+ if (bid) {
1165
+ const res = await waitForResolution(mcpUrl, token, bid, isAlive);
1166
+ if (res.status === 'resolved') {
1167
+ prompt = `The human answered your blocker: ${JSON.stringify(res.answer)}\nApply it and continue.`;
1168
+ nudges = 0;
1169
+ continue;
1170
+ }
1171
+ if (res.status === 'timeout') return { outcome: 'parked', title, intentId };
1172
+ return { outcome: 'blocked', title, intentId };
1173
+ }
1174
+
1175
+ if (sawSentinel(out, 'DONE')) {
1176
+ // Identical to the live path's completion, and it must stay identical: the
1177
+ // marker clear is what stops this worktree being read as a resume of a
1178
+ // task that has finished (or been discarded and restarted).
1179
+ clearTaskMarker(cwd);
1180
+ if (isPatch) {
1181
+ await landPatch({ mcpUrl, token, runId, intentId, repoRoot, cwd, patchBase, baseRef });
1182
+ }
1183
+ markLanded();
1184
+ return { outcome: 'done', title, intentId };
1185
+ }
1186
+
1187
+ // No sentinel: the turn ended without saying how. Before nudging, find out
1188
+ // whether the RUN still exists — a restart or a release from the app tears
1189
+ // it down out from under us, and nudging a dead run just burns the user's
1190
+ // quota. Same three answers the live loop reads, for the same reasons.
1191
+ const poll = await mcpCall(mcpUrl, token, 'poll_channel', {
1192
+ runId,
1193
+ ...(afterId ? { afterId } : {}),
1194
+ }).catch(() => null);
1195
+ if (poll && poll.ok === false && poll.released) {
1196
+ return { outcome: 'released', title, intentId };
1197
+ }
1198
+ if (poll && poll.ok === false && poll.reason === 'run_not_active') {
1199
+ clearTaskMarker(cwd);
1200
+ try {
1201
+ git(['worktree', 'remove', '--force', cwd], repoRoot);
1202
+ } catch {
1203
+ resetWorktree(cwd, baseRef);
1204
+ }
1205
+ return { outcome: 'torn_down', title, intentId };
1206
+ }
1207
+
1208
+ const fresh = (poll?.messages ?? []).filter((x) => x.role === 'user');
1209
+ if (fresh.length) afterId = fresh[fresh.length - 1].id;
1210
+
1211
+ if (fresh.some((f) => STOP_RE.test(f.content))) {
1212
+ held = true;
1213
+ prompt =
1214
+ 'A teammate asked you to STOP. Halt, summarize where you are in one line, and wait for direction — do not continue until told.';
1215
+ continue;
1216
+ }
1217
+ if (fresh.length) {
1218
+ prompt = fresh
1219
+ .map((f) => (f.authorName ? `${f.authorName}: ` : '') + f.content)
1220
+ .join('\n');
1221
+ nudges = 0;
1222
+ held = false;
1223
+ continue;
1224
+ }
1225
+ if (held) {
1226
+ const next = await waitForMessage(mcpUrl, token, runId, afterId, isAlive);
1227
+ if (!next) return { outcome: 'parked', title, intentId };
1228
+ held = false;
1229
+ nudges = 0;
1230
+ afterId = next.id;
1231
+ prompt = (next.authorName ? `${next.authorName}: ` : '') + next.content;
1232
+ continue;
1233
+ }
1234
+
1235
+ if (nudges < 2) {
1236
+ nudges++;
1237
+ prompt = isPatch
1238
+ ? 'Continue until the task is complete: commit your change (no branch, no push, no PR) and call complete, then print DONE. Or report a blocker and print BLOCKED:<id>.'
1239
+ : 'Continue until the task is complete: open a draft PR and call complete, then print DONE. Or report a blocker and print BLOCKED:<id>.';
1240
+ continue;
1241
+ }
1242
+ return { outcome: 'stalled', title, intentId };
1243
+ }
1244
+ }
1245
+
571
1246
  export async function runLiveTask({
572
1247
  mcpUrl,
573
1248
  token,
@@ -592,12 +1267,11 @@ export async function runLiveTask({
592
1267
  // the only dispatch in this product, and silently answering it with a different
593
1268
  // CLI is the same class of bug as dispatching from the wrong surface.
594
1269
  //
595
- // A live session is Claude-only by construction, not by configuration — it is
596
- // an SDK, not a subprocess — so this list is exactly the runtimes marked
597
- // `live` in the registry. An older server ignores the argument and behaves as
598
- // before; that degrade is what `daemon:min` is for.
1270
+ // The list is every runtime this daemon can spawn or session, NOT just the
1271
+ // live ones — `driveSubprocess` below builds the rest. An older server ignores
1272
+ // the argument and behaves as before; that degrade is what `daemon:min` is for.
599
1273
  const claim = await mcpCall(mcpUrl, token, 'claim_next_task', {
600
- runtimes: LIVE_RUNTIMES,
1274
+ runtimes: DRIVABLE_HERE,
601
1275
  }).catch(() => null);
602
1276
  if (!claim || claim.claimed !== true) return { outcome: 'nothing' };
603
1277
  const { runId, intentId } = claim;
@@ -792,8 +1466,16 @@ export async function runLiveTask({
792
1466
  // already told us the real intent.
793
1467
  const stopDiffstat = sampleDiffstat?.(cwd, baseRef, intentId, agentId) ?? null;
794
1468
 
795
- const input = makeInput(seedPrompt(runId, brief, transcript, resumedInPlace));
796
- const session = query({
1469
+ // WHICH CLI builds this one, off the brief — the task we actually hold, not
1470
+ // the roster's prediction. Only Claude has a live session (it is an Anthropic
1471
+ // SDK, not a CLI contract); everything else is driven as a subprocess by
1472
+ // `driveSubprocess` below, which is what the registry's `live` flag has always
1473
+ // said would happen and what live mode never implemented.
1474
+ const rt = runtimeById(brief.agentRuntime ?? 'claude');
1475
+ const seedText = seedPrompt(runId, brief, transcript, resumedInPlace);
1476
+ const input = rt.live ? makeInput(seedText) : null;
1477
+ const session = rt.live
1478
+ ? query({
797
1479
  prompt: input.stream(),
798
1480
  options: {
799
1481
  cwd,
@@ -821,26 +1503,32 @@ export async function runLiveTask({
821
1503
  },
822
1504
  },
823
1505
  },
824
- });
1506
+ })
1507
+ : null;
825
1508
 
826
1509
  // Mark this worker BUSY for the daemon's reconcile loop: buildHave keeps the
827
1510
  // worker's token while a session is live (never rotate a credential out from
828
1511
  // under it), and teardown/agent-removal can interrupt the SDK session via this
829
1512
  // marker's kill(). Cleared in finally. Mirrors poll mode's onChild(child).
830
- onChild?.({
831
- kill: () => {
832
- try {
833
- session.interrupt?.();
834
- } catch {
835
- /* already ending */
836
- }
837
- try {
838
- session.return?.();
839
- } catch {
840
- /* already closed */
841
- }
842
- },
843
- });
1513
+ // The subprocess driver registers its own handle per turn (runTurn's onSpawn),
1514
+ // because there the killable thing is a child process and it only exists while
1515
+ // a turn is actually running.
1516
+ if (session) {
1517
+ onChild?.({
1518
+ kill: () => {
1519
+ try {
1520
+ session.interrupt?.();
1521
+ } catch {
1522
+ /* already ending */
1523
+ }
1524
+ try {
1525
+ session.return?.();
1526
+ } catch {
1527
+ /* already closed */
1528
+ }
1529
+ },
1530
+ });
1531
+ }
844
1532
 
845
1533
  let turnId = null;
846
1534
  let turnText = '';
@@ -860,6 +1548,25 @@ export async function runLiveTask({
860
1548
  lastBeat = Date.now();
861
1549
  void mcpCall(mcpUrl, token, 'heartbeat', { runId }).catch(() => {});
862
1550
  };
1551
+ // AND ON A TIMER, because "activity" is not a signal every driver has.
1552
+ //
1553
+ // `beat()` used to be called from exactly ONE place — the live session's
1554
+ // message loop, below — and the two other drivers return before they ever
1555
+ // reach it. So a mediated or subprocess turn renewed the task lease only
1556
+ // incidentally: `report_progress`, which fires only when the CLI happens to
1557
+ // emit a tool activity and is throttled to one per 8s. Meanwhile Antigravity
1558
+ // is handed `--print-timeout 60m` and AGENT_LEASE_TTL_MINUTES is 30, so a
1559
+ // quiet stretch INSIDE a turn we explicitly permitted made the task stale to
1560
+ // `isEligible` and claimable by another worker while it was still building it.
1561
+ // `heartbeat` renews the task lease server-side (refreshTaskLeaseRemote), not
1562
+ // just this token's last-seen, which is exactly the thing that goes stale.
1563
+ //
1564
+ // Fires at half the throttle window; `beat()`'s own guard is what rate-limits
1565
+ // the wire, so session traffic and this timer cannot double up. Started here
1566
+ // rather than in each driver for the same reason the checkpoint timer is
1567
+ // shared: three drivers with three answers is how this diverged once already.
1568
+ const heartbeatTimer = setInterval(beat, 30_000);
1569
+ heartbeatTimer.unref?.();
863
1570
 
864
1571
  const flush = async () => {
865
1572
  if (turnId && turnText.trim()) {
@@ -880,6 +1587,65 @@ export async function runLiveTask({
880
1587
  };
881
1588
 
882
1589
  try {
1590
+ // THE SEAM. Everything above prepared this task — the claim, the checkout,
1591
+ // the branch or patch base, the restored work in progress, the checkpoint
1592
+ // timer and the diffstat sampler — and everything in the `finally` below
1593
+ // tears it down. Only the middle differs by runtime, so only the middle
1594
+ // branches, and a non-live runtime inherits the other two thirds unchanged.
1595
+ if (!session && mediated(rt)) {
1596
+ // No MCP config this runtime can hold, so it is handed none: the daemon
1597
+ // makes every control-plane call itself with this lane's own token.
1598
+ return await driveMediated({
1599
+ runtimeId: rt.id,
1600
+ mcpUrl,
1601
+ token,
1602
+ runId,
1603
+ intentId,
1604
+ title,
1605
+ cwd,
1606
+ brief,
1607
+ isPatch,
1608
+ patchBase,
1609
+ repoRoot,
1610
+ baseRef,
1611
+ label: `[${rt.label}]`,
1612
+ seedText,
1613
+ isAlive,
1614
+ onChild,
1615
+ markLanded: () => {
1616
+ landed = true;
1617
+ },
1618
+ });
1619
+ }
1620
+
1621
+ if (!session) {
1622
+ return await driveSubprocess({
1623
+ runtimeId: rt.id,
1624
+ mcpUrl,
1625
+ token,
1626
+ runId,
1627
+ intentId,
1628
+ title,
1629
+ cwd,
1630
+ brief,
1631
+ isPatch,
1632
+ patchBase,
1633
+ repoRoot,
1634
+ baseRef,
1635
+ label: `[${rt.label}]`,
1636
+ seedText,
1637
+ afterId,
1638
+ isAlive,
1639
+ onChild,
1640
+ // `landed` decides whether the finally deletes this task's WIP ref or
1641
+ // takes one last checkpoint, so the driver has to be able to set it —
1642
+ // returning it would be too late, the finally runs first.
1643
+ markLanded: () => {
1644
+ landed = true;
1645
+ },
1646
+ });
1647
+ }
1648
+
883
1649
  for await (const m of session) {
884
1650
  if (!isAlive()) return { outcome: 'blocked', title, intentId };
885
1651
  beat(); // any session traffic = alive (throttled to 1/min)
@@ -1020,6 +1786,7 @@ export async function runLiveTask({
1020
1786
  return { outcome: 'error', error: e?.message ?? String(e), title, intentId };
1021
1787
  } finally {
1022
1788
  clearInterval(checkpointTimer);
1789
+ clearInterval(heartbeatTimer);
1023
1790
  // Same finally as the checkpoint: every path out of this task — done,
1024
1791
  // parked, rate-limited, thrown — must stop reporting a worktree that is
1025
1792
  // about to stop being this run's.
@@ -1038,14 +1805,17 @@ export async function runLiveTask({
1038
1805
  }
1039
1806
  onChild?.(null); // no longer busy — token may rotate between tasks
1040
1807
  onIntent?.(null);
1041
- input.close();
1808
+ // Only a live session has a streaming input to close or a generator to
1809
+ // return. The subprocess driver's children are already gone — runTurn awaits
1810
+ // each one — and it clears its own onChild handle per turn.
1811
+ input?.close();
1042
1812
  try {
1043
- await session.interrupt?.();
1813
+ await session?.interrupt?.();
1044
1814
  } catch {
1045
1815
  /* session already ended */
1046
1816
  }
1047
1817
  try {
1048
- await session.return?.();
1818
+ await session?.return?.();
1049
1819
  } catch {
1050
1820
  /* generator already closed */
1051
1821
  }