comfyui-mcp 0.49.3 → 0.49.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/orchestrator/agent-backend.js +8 -0
- package/dist/orchestrator/agent-backend.js.map +1 -1
- package/dist/orchestrator/codex-backend.js +49 -0
- package/dist/orchestrator/codex-backend.js.map +1 -1
- package/dist/orchestrator/grok-backend.js +5 -0
- package/dist/orchestrator/grok-backend.js.map +1 -1
- package/dist/orchestrator/index.js +317 -22
- package/dist/orchestrator/index.js.map +1 -1
- package/dist/orchestrator/panel-agent.js +591 -67
- package/dist/orchestrator/panel-agent.js.map +1 -1
- package/dist/orchestrator/panel-tools.js +505 -65
- package/dist/orchestrator/panel-tools.js.map +1 -1
- package/dist/orchestrator/run-completion-journal.js +914 -0
- package/dist/orchestrator/run-completion-journal.js.map +1 -0
- package/dist/orchestrator/session-store.js +11 -3
- package/dist/orchestrator/session-store.js.map +1 -1
- package/dist/services/download-jobs.js +178 -14
- package/dist/services/download-jobs.js.map +1 -1
- package/dist/services/download-progress.js +16 -18
- package/dist/services/download-progress.js.map +1 -1
- package/dist/services/extra-paths.js +61 -9
- package/dist/services/extra-paths.js.map +1 -1
- package/dist/services/hello-retarget.js +165 -0
- package/dist/services/hello-retarget.js.map +1 -0
- package/dist/services/manifest.js +92 -9
- package/dist/services/manifest.js.map +1 -1
- package/dist/services/model-resolver.js +616 -17
- package/dist/services/model-resolver.js.map +1 -1
- package/dist/services/output-dir.js +63 -15
- package/dist/services/output-dir.js.map +1 -1
- package/dist/services/panel-pin-guard.js +6 -62
- package/dist/services/panel-pin-guard.js.map +1 -1
- package/dist/services/ui-bridge.js +77 -8
- package/dist/services/ui-bridge.js.map +1 -1
- package/dist/services/workspace-env.js +179 -3
- package/dist/services/workspace-env.js.map +1 -1
- package/dist/tools/extra-paths.js +6 -4
- package/dist/tools/extra-paths.js.map +1 -1
- package/dist/tools/model-extras.js +26 -7
- package/dist/tools/model-extras.js.map +1 -1
- package/dist/tools/model-management.js +36 -8
- package/dist/tools/model-management.js.map +1 -1
- package/dist/tools/report-issue.js +12 -2
- package/dist/tools/report-issue.js.map +1 -1
- package/dist/tools/vocabulary.js +21 -0
- package/dist/tools/vocabulary.js.map +1 -1
- package/package.json +1 -1
- package/scripts/gen-tool-docs.ts +138 -4
- package/scripts/tool-doc-examples.ts +794 -0
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
// one Claude Agent SDK streaming session per panel tab (src/orchestrator/
|
|
9
9
|
// panel-agent.ts). Each agent runs on the user's Claude SUBSCRIPTION with no API
|
|
10
10
|
// key. See docs/design/panel-orchestrator.md.
|
|
11
|
-
import { existsSync, mkdirSync, writeFileSync, unlinkSync, readFileSync, readdirSync, rmSync } from "node:fs";
|
|
11
|
+
import { existsSync, mkdirSync, writeFileSync, unlinkSync, readFileSync, readdirSync, rmSync, appendFileSync, } from "node:fs";
|
|
12
12
|
import { execFileSync } from "node:child_process";
|
|
13
13
|
import { tmpdir, homedir, networkInterfaces } from "node:os";
|
|
14
14
|
import { dirname, join } from "node:path";
|
|
@@ -17,6 +17,7 @@ import { randomBytes } from "node:crypto";
|
|
|
17
17
|
import readline from "node:readline";
|
|
18
18
|
import { startUiBridge, isLoopbackBindHost, SESSION_EPOCH } from "../services/ui-bridge.js";
|
|
19
19
|
import { setupSecureBridge, resolveComfyuiPathForTarget } from "../services/secure-bridge.js";
|
|
20
|
+
import { judgeHelloRetarget, canonComfyuiTargetUrl } from "../services/hello-retarget.js";
|
|
20
21
|
import { startQuickTunnel } from "../services/tunnel.js";
|
|
21
22
|
import { detectInstallMode } from "../services/self-update.js";
|
|
22
23
|
import { performPanelSync } from "../services/panel-sync.js";
|
|
@@ -61,6 +62,7 @@ import { resolveHttpLaneComfyToolMode } from "./http-backend-tools.js";
|
|
|
61
62
|
import { startPanelConsoleHttpServer } from "./panel-console-http.js";
|
|
62
63
|
import { readComfyuiCrashLog, formatCrashNote } from "../services/crash-log.js";
|
|
63
64
|
import { QueueMonitor } from "../services/queue-monitor.js";
|
|
65
|
+
import { RunCompletions, describe as describeCorrelation, } from "./run-completion-journal.js";
|
|
64
66
|
import { initRunpodWatcher, getRunpodWatcher } from "../services/runpod-watch.js";
|
|
65
67
|
import { getPod } from "../services/runpod-client.js";
|
|
66
68
|
import { listTargetChangeRequests, consumeTargetChange, ackTargetChange, setProgressDir, CONTROL_PREFIX, newestAttemptEpochs, isSupersededAttempt, downloadAttemptKey } from "../services/download-progress.js";
|
|
@@ -669,6 +671,10 @@ export async function runPanelOrchestrator() {
|
|
|
669
671
|
// reattached to it. Exit so the pack respawns a clean orchestrator (Node's own
|
|
670
672
|
// default is to crash on uncaughtException anyway).
|
|
671
673
|
logger.error(`[panel-orchestrator] FATAL uncaught exception — exiting so a fresh orchestrator can take over: ${err.stack ?? err.message}`);
|
|
674
|
+
// #468 — this path exits without any teardown, so record whatever run
|
|
675
|
+
// completions die with it. Log-only by construction (see the function): a
|
|
676
|
+
// crash is no time to await a bridge write.
|
|
677
|
+
reportLostCompletionsOnExit();
|
|
672
678
|
process.exit(1);
|
|
673
679
|
});
|
|
674
680
|
// Self-exit seam. Wired to the real clean shutdown once it's defined below; until
|
|
@@ -682,6 +688,13 @@ export async function runPanelOrchestrator() {
|
|
|
682
688
|
return;
|
|
683
689
|
selfExiting = true;
|
|
684
690
|
logger.error(`[panel-orchestrator] self-exit (${why}) — closing the bridge so a fresh orchestrator can take over.`);
|
|
691
|
+
// #468 — a fatal self-exit deliberately BYPASSES the idle gate (the whole
|
|
692
|
+
// point is to collapse a wedged orchestrator), and the journal is in-memory,
|
|
693
|
+
// so any undelivered run completion dies here. It must not die SILENTLY:
|
|
694
|
+
// tell each affected tab, in the panel chat, exactly which runs it will never
|
|
695
|
+
// be told about, so the user (and the agent that resumes after the respawn)
|
|
696
|
+
// treats them as UNDETERMINED instead of still-pending.
|
|
697
|
+
reportLostCompletionsOnExit();
|
|
685
698
|
if (runShutdown) {
|
|
686
699
|
runShutdown();
|
|
687
700
|
}
|
|
@@ -1856,6 +1869,30 @@ export async function runPanelOrchestrator() {
|
|
|
1856
1869
|
onAgentFatal: (tabId, reason) => {
|
|
1857
1870
|
requestSelfExit(`tab ${tabId.slice(0, 8)} ${reason}`);
|
|
1858
1871
|
},
|
|
1872
|
+
// #468 — run-completion journal acks. `key` is the composite agent key; the
|
|
1873
|
+
// journal is keyed by the PANEL TAB, so a provider switch (which retires
|
|
1874
|
+
// `tab::old` and spawns `tab::new`) can never strand a completion.
|
|
1875
|
+
onEventDelivered: (_key, tokens) => {
|
|
1876
|
+
for (const token of tokens)
|
|
1877
|
+
RunCompletions.ack(token);
|
|
1878
|
+
},
|
|
1879
|
+
onEventUndelivered: (key, tokens, opts) => {
|
|
1880
|
+
for (const token of tokens)
|
|
1881
|
+
RunCompletions.release(token, { carried: opts?.carried === true });
|
|
1882
|
+
const panelTab = panelTabOf(key);
|
|
1883
|
+
logger.warn(`[panel-orchestrator] tab ${panelTab.slice(0, 8)} handed back ${tokens.length} undelivered run completion(s) — journaled for replay (#468)`);
|
|
1884
|
+
// Try again immediately. When the agent is still alive (a stall-abandoned
|
|
1885
|
+
// turn, a plain Stop) this re-queues the completion into its next turn;
|
|
1886
|
+
// when it is going away, injectEvent refuses and the entry simply stays
|
|
1887
|
+
// pending for the next spawn. The refusal path returns false WITHOUT
|
|
1888
|
+
// calling back here, so this cannot recurse.
|
|
1889
|
+
flushRunCompletions(panelTab);
|
|
1890
|
+
},
|
|
1891
|
+
// A fresh agent for this key can take mail now — replay whatever the
|
|
1892
|
+
// previous one never delivered.
|
|
1893
|
+
onAgentReady: (key) => {
|
|
1894
|
+
flushRunCompletions(panelTabOf(key));
|
|
1895
|
+
},
|
|
1859
1896
|
sessionStore,
|
|
1860
1897
|
// #570 P0 — bind each persisted exact session to its tab's FULL trusted workflow
|
|
1861
1898
|
// identity (server-observed origin + per-instance uuid), so a saved workflow
|
|
@@ -1870,6 +1907,115 @@ export async function runPanelOrchestrator() {
|
|
|
1870
1907
|
// Let refreshEnvCapabilities() feed a freshly-gathered env block into agents
|
|
1871
1908
|
// spawned after a ComfyUI restart/reconnect.
|
|
1872
1909
|
liveManager = manager;
|
|
1910
|
+
// #468 — let the journal pull a still-unread completion back off an agent's
|
|
1911
|
+
// queue when it has to WEAKEN that completion's correlation (a reused prompt
|
|
1912
|
+
// id, a replaced conversation). The wording is baked in at queue time, so
|
|
1913
|
+
// without this the stale copy would still reach the agent claiming to be the
|
|
1914
|
+
// run it queued. Only ever removes an injected `completionOnly` item.
|
|
1915
|
+
RunCompletions.setRevoker((panelTabId, token) => manager.revokeEvent(agentKeyFor(panelTabId), token), (panelTabId) => flushRunCompletions(panelTabId));
|
|
1916
|
+
/**
|
|
1917
|
+
* Deliver every journaled run completion for a panel tab (#468).
|
|
1918
|
+
*
|
|
1919
|
+
* Called on arrival, and again at every later delivery opportunity (a fresh
|
|
1920
|
+
* agent spawn). Order is preserved and the loop STOPS at the first refusal, so
|
|
1921
|
+
* a completion can never overtake an older one that is still stuck.
|
|
1922
|
+
*
|
|
1923
|
+
* Nothing here re-correlates: each entry carries the verdict computed when it
|
|
1924
|
+
* ARRIVED, so a replay can never be re-attributed to a run that started after
|
|
1925
|
+
* it landed. `injectEvent` returning true means only that the agent took it
|
|
1926
|
+
* onto its queue — the entry stays in the journal until the turn that carried
|
|
1927
|
+
* it ends (onEventDelivered), which is what makes it survive a restart.
|
|
1928
|
+
*/
|
|
1929
|
+
/**
|
|
1930
|
+
* Last-ditch disclosure before the process dies (#468).
|
|
1931
|
+
*
|
|
1932
|
+
* The self-exit paths (agent-fatal, a never-handshaking probe) skip the idle
|
|
1933
|
+
* gate on purpose, and the journal does not survive the process — so a
|
|
1934
|
+
* completion still journaled here is genuinely lost. Say so, per tab, naming
|
|
1935
|
+
* the run: an UNDETERMINED outcome the user can act on beats a promise that
|
|
1936
|
+
* silently evaporates. Best-effort and never throws — this runs on the way out.
|
|
1937
|
+
*/
|
|
1938
|
+
let lostCompletionsReported = false;
|
|
1939
|
+
function reportLostCompletionsOnExit() {
|
|
1940
|
+
if (lostCompletionsReported)
|
|
1941
|
+
return; // every exit path calls this; report once
|
|
1942
|
+
lostCompletionsReported = true;
|
|
1943
|
+
const byTab = new Map();
|
|
1944
|
+
try {
|
|
1945
|
+
for (const entry of RunCompletions.allOutstanding()) {
|
|
1946
|
+
const list = byTab.get(entry.key) ?? [];
|
|
1947
|
+
list.push(describeCorrelation(entry.correlation));
|
|
1948
|
+
byTab.set(entry.key, list);
|
|
1949
|
+
}
|
|
1950
|
+
}
|
|
1951
|
+
catch {
|
|
1952
|
+
return; // nothing readable — nothing to report
|
|
1953
|
+
}
|
|
1954
|
+
// LOG FIRST, unconditionally, and SYNCHRONOUSLY. This is the durable half of
|
|
1955
|
+
// the disclosure: the chat push below can only reach a CONNECTED tab (an
|
|
1956
|
+
// offline one's frame lands in the bridge's missedFrames buffer, which dies
|
|
1957
|
+
// with the process moments later), and `bridge` may not even exist yet on a
|
|
1958
|
+
// very early fatal.
|
|
1959
|
+
//
|
|
1960
|
+
// `writeSync` on fd 2, not the logger: `process.stderr.write` is async when
|
|
1961
|
+
// stderr is a pipe (the normal case under the ComfyUI launcher), so a
|
|
1962
|
+
// `process.exit()` immediately after can terminate with the record still
|
|
1963
|
+
// queued and never written. Since the whole in-memory-journal tradeoff rests
|
|
1964
|
+
// on "the disclosure always happens", the write it rests on must block.
|
|
1965
|
+
// ONE write for every tab, not one per tab: a synchronous write to a full
|
|
1966
|
+
// pipe blocks until its reader drains, so the smallest possible number of
|
|
1967
|
+
// bytes is the right shape. (Blocking here is the deliberate cost of a
|
|
1968
|
+
// guaranteed record — a stderr reader that has stopped consuming is a broken
|
|
1969
|
+
// environment, and losing the disclosure would undercut the whole
|
|
1970
|
+
// in-memory-journal tradeoff.)
|
|
1971
|
+
const record = [...byTab]
|
|
1972
|
+
.map(([panelTab, runs]) => `[panel-orchestrator] tab ${panelTab.slice(0, 8)} — exiting with ${runs.length} undelivered run completion(s), outcome UNDETERMINED: ${runs.join("; ")}`)
|
|
1973
|
+
.join("\n");
|
|
1974
|
+
// A FILE is the ONLY synchronous sink. A sync write to a PIPE can block
|
|
1975
|
+
// indefinitely when its reader has stalled, and blocking here blocks the
|
|
1976
|
+
// event loop — so Node can never dispatch the repeated SIGTERM that is
|
|
1977
|
+
// supposed to force the exit, and the process becomes unkillable through its
|
|
1978
|
+
// handled signals. A regular file always makes progress.
|
|
1979
|
+
//
|
|
1980
|
+
// There is deliberately NO synchronous fallback. Ranking the two outcomes:
|
|
1981
|
+
// an unkillable process is worse than a missing log line, and a stderr
|
|
1982
|
+
// `writeSync` in the fallback would reintroduce exactly the hang the file
|
|
1983
|
+
// sink exists to avoid. If the file write fails we accept losing the durable
|
|
1984
|
+
// record and fall back to the async logger, which can never wedge the exit.
|
|
1985
|
+
let recorded = false;
|
|
1986
|
+
try {
|
|
1987
|
+
appendFileSync(`${lockPath}.lost-completions.log`, `${new Date().toISOString()} ${record}\n`);
|
|
1988
|
+
recorded = true;
|
|
1989
|
+
}
|
|
1990
|
+
catch {
|
|
1991
|
+
// No usable file (a very early fatal, a read-only dir) — console only.
|
|
1992
|
+
}
|
|
1993
|
+
// Console visibility, always, and always non-blocking.
|
|
1994
|
+
logger.error(record);
|
|
1995
|
+
if (!recorded) {
|
|
1996
|
+
logger.error("[panel-orchestrator] …and no durable sink accepted that record (it may not survive)");
|
|
1997
|
+
}
|
|
1998
|
+
try {
|
|
1999
|
+
for (const [panelTab, runs] of byTab) {
|
|
2000
|
+
bridge.push({
|
|
2001
|
+
type: "say",
|
|
2002
|
+
text: `⚠️ The agent backend is being restarted, and ${runs.length} finished render result(s) could not be delivered ` +
|
|
2003
|
+
`(${runs.join("; ")}). Their outcome is UNDETERMINED from the agent's point of view — ask it to check ` +
|
|
2004
|
+
`\`get_history\` for those runs once it reconnects rather than assuming it saw them.`,
|
|
2005
|
+
}, panelTab);
|
|
2006
|
+
}
|
|
2007
|
+
}
|
|
2008
|
+
catch (err) {
|
|
2009
|
+
logger.warn(`[panel-orchestrator] could not push the lost-completion notice (the log above is the record): ${err instanceof Error ? err.message : String(err)}`);
|
|
2010
|
+
}
|
|
2011
|
+
}
|
|
2012
|
+
function flushRunCompletions(panelTabId) {
|
|
2013
|
+
const key = agentKeyFor(panelTabId);
|
|
2014
|
+
const { blockedOn } = RunCompletions.deliverPending(panelTabId, (payload, token) => manager.injectEvent(key, payload, { eventToken: token }));
|
|
2015
|
+
if (blockedOn) {
|
|
2016
|
+
logger.warn(`[panel-orchestrator] tab ${panelTabId.slice(0, 8)} has no live agent for ${describeCorrelation(blockedOn.correlation)} — journaled, replayed when one comes back (#468)`);
|
|
2017
|
+
}
|
|
2018
|
+
}
|
|
1873
2019
|
// Flag the mobile mirror picker's "session attached" (green) dot from live agents.
|
|
1874
2020
|
bridge.setHasSessionPredicate((tabId) => manager.hasLiveAgent(agentKeyFor(tabId)));
|
|
1875
2021
|
// #570 — stamp each dispatched command with the tab's trusted per-instance workflow uuid so
|
|
@@ -1968,17 +2114,9 @@ export async function runPanelOrchestrator() {
|
|
|
1968
2114
|
// http://h:443 and http://h are NOT the same endpoint (codex finding).
|
|
1969
2115
|
// Shared by applyComfyuiUrl's dedupe and the control-channel ack check
|
|
1970
2116
|
// (runpodProxyUrl omits :443 while getComfyUIBaseUrl includes it — codex).
|
|
1971
|
-
|
|
1972
|
-
|
|
1973
|
-
|
|
1974
|
-
if ((p.protocol === "https:" && p.port === "443") || (p.protocol === "http:" && p.port === "80"))
|
|
1975
|
-
p.port = "";
|
|
1976
|
-
return p.toString().replace(/\/+$/, "");
|
|
1977
|
-
}
|
|
1978
|
-
catch {
|
|
1979
|
-
return u.replace(/\/+$/, "");
|
|
1980
|
-
}
|
|
1981
|
-
};
|
|
2117
|
+
// ONE implementation with the hello-retarget judge's same-target check so
|
|
2118
|
+
// the dedupe and the veto can never disagree (canonComfyuiTargetUrl).
|
|
2119
|
+
const canonTargetUrl = (u) => canonComfyuiTargetUrl(u);
|
|
1982
2120
|
const applyComfyuiUrl = (rawUrl) => {
|
|
1983
2121
|
if (typeof rawUrl !== "string")
|
|
1984
2122
|
return false;
|
|
@@ -2374,17 +2512,38 @@ export async function runPanelOrchestrator() {
|
|
|
2374
2512
|
// BEFORE the readiness probe so the "ready" ack reflects the right instance —
|
|
2375
2513
|
// but a hello can arrive from a STALE browser tab on a DEAD instance (E2E
|
|
2376
2514
|
// finding: a zombie :8189 tab kept retargeting the orchestrator to a corpse
|
|
2377
|
-
// and silently breaking every tool that probes the target).
|
|
2378
|
-
//
|
|
2379
|
-
//
|
|
2515
|
+
// and silently breaking every tool that probes the target). The veto
|
|
2516
|
+
// (judgeHelloRetarget, #303) protects only a HEALTHY current target: when the
|
|
2517
|
+
// current target reads dead too — the ComfyUI restart window — the live
|
|
2518
|
+
// tab's hello is trusted instead, so the reconnect correction back to a
|
|
2519
|
+
// LOCAL target can never be vetoed into keeping a stale REMOTE one (#756).
|
|
2520
|
+
// TRUST: hello.comfyui_url is page-JS-writable, so every apply path is
|
|
2521
|
+
// gated on the SERVER-OBSERVED handshake origin (tabServerOrigin — the
|
|
2522
|
+
// browser sets it, page JS can't forge it; #509's trusted source, codex
|
|
2523
|
+
// gate). Only a corroborated claim gets the no-probe shortcuts and the
|
|
2524
|
+
// both-dead recovery; an uncorroborated claim must earn its retarget by
|
|
2525
|
+
// answering its probe, and a dead one fails closed. RunPod proxies still
|
|
2526
|
+
// skip the probe for corroborated claims (booting pods answer late —
|
|
2527
|
+
// readiness is the connector's job).
|
|
2380
2528
|
const helloUrl = event.comfyui_url;
|
|
2381
2529
|
void (async () => {
|
|
2382
|
-
|
|
2383
|
-
|
|
2384
|
-
|
|
2385
|
-
|
|
2386
|
-
|
|
2387
|
-
|
|
2530
|
+
const verdict = await judgeHelloRetarget({
|
|
2531
|
+
helloUrl,
|
|
2532
|
+
currentUrl: comfyuiUrl,
|
|
2533
|
+
observedOrigin: bridge.tabServerOrigin(panelTab),
|
|
2534
|
+
probe: (u) => probeOk(u, 3_000),
|
|
2535
|
+
});
|
|
2536
|
+
if (!verdict.apply) {
|
|
2537
|
+
logger.warn(verdict.reason === "vetoed-untrusted"
|
|
2538
|
+
? `[panel-orchestrator] refusing hello retarget to ${verdict.base}: the tab's handshake origin does not ` +
|
|
2539
|
+
`corroborate the claimed URL and the claimed instance is unreachable — NOT retargeting on an ` +
|
|
2540
|
+
`unverifiable claim (keeping ${comfyuiUrl})`
|
|
2541
|
+
: `[panel-orchestrator] ignoring hello retarget to unreachable ${verdict.base} (stale tab on a dead instance?) — keeping ${comfyuiUrl}`);
|
|
2542
|
+
return;
|
|
2543
|
+
}
|
|
2544
|
+
if (verdict.reason === "current-also-unreachable") {
|
|
2545
|
+
logger.info(`[panel-orchestrator] hello target ${verdict.base} and current target ${comfyuiUrl} are BOTH unreachable ` +
|
|
2546
|
+
`(ComfyUI restart window?) — trusting the live tab's origin-corroborated hello rather than pinning a stale target (#756)`);
|
|
2388
2547
|
}
|
|
2389
2548
|
applyComfyuiUrl(helloUrl);
|
|
2390
2549
|
})();
|
|
@@ -2475,6 +2634,12 @@ export async function runPanelOrchestrator() {
|
|
|
2475
2634
|
// under migratedFrom would still resolve the retired workflow's tool call after the
|
|
2476
2635
|
// switch (#570 P0). The socket is unchanged, so only queued WORK is dropped.
|
|
2477
2636
|
bridge.dropQueuedDeliveries(migratedFrom);
|
|
2637
|
+
// #468 — the old id's journaled run completions belong to the workflow
|
|
2638
|
+
// being switched AWAY from. Carrying them onto an unproven-different
|
|
2639
|
+
// workflow would deliver one workflow's render as another's, which is
|
|
2640
|
+
// exactly the misattribution the journal exists to prevent. Drop them
|
|
2641
|
+
// (logged per entry) rather than migrate them.
|
|
2642
|
+
RunCompletions.forget(migratedFrom);
|
|
2478
2643
|
manager.retire(migratedFrom + AGENT_KEY_SEP + prevBackend);
|
|
2479
2644
|
logger.info(`[panel-orchestrator] same-socket re-hello ${migratedFrom.slice(0, 12)} → ${panelTab.slice(0, 12)} without proven workflow continuity — old agent retired (NOT rebound); each workflow keeps its own conversation`);
|
|
2480
2645
|
// Retire the old id's routing/prefs — do NOT carry them to a different
|
|
@@ -2519,6 +2684,32 @@ export async function runPanelOrchestrator() {
|
|
|
2519
2684
|
// have a LIVE agent (a provider switch retires the others), so at most one rebind is a
|
|
2520
2685
|
// live-agent move; the rest are durable-only.
|
|
2521
2686
|
let destinationCollision = false;
|
|
2687
|
+
// #468 — the journal purge must happen BEFORE the loop, not after it.
|
|
2688
|
+
// manager.reset() inside the loop hands back any run-completion tokens
|
|
2689
|
+
// parked in that provider's held mail, and the hand-back callback
|
|
2690
|
+
// flushes immediately — into whatever agent currently owns panelTab,
|
|
2691
|
+
// which on a collision is the SUPERSEDED destination tab's agent on a
|
|
2692
|
+
// different provider. Purging first means there is nothing to hand
|
|
2693
|
+
// back. Read-only pre-pass over the same predicate the loop uses, so
|
|
2694
|
+
// the two can't disagree about what counts as a collision.
|
|
2695
|
+
// #468 CRITICAL — a tab holding ONLY a journal entry is NOT empty. Its
|
|
2696
|
+
// agent and durable session may both be gone (New chat) while an
|
|
2697
|
+
// undelivered completion for its render is still addressed to it;
|
|
2698
|
+
// without counting that the destination reads as unoccupied, the source
|
|
2699
|
+
// is rebound onto its id, and the next flush hands the DESTINATION
|
|
2700
|
+
// tab's render to the SOURCE tab's conversation.
|
|
2701
|
+
const destinationJournaled = RunCompletions.outstanding(panelTab).length;
|
|
2702
|
+
const destinationHadState = [...KNOWN_BACKENDS].some((b) => destinationHasCollisionState({
|
|
2703
|
+
hasManagerState: manager.hasAnyState(panelTab + AGENT_KEY_SEP + b),
|
|
2704
|
+
hasDurableSession: sessionStore.get(panelTab + AGENT_KEY_SEP + b) !== undefined,
|
|
2705
|
+
renderHeldCount: heldDuringGen.get(panelTab + AGENT_KEY_SEP + b)?.length ?? 0,
|
|
2706
|
+
journaledCompletionCount: destinationJournaled,
|
|
2707
|
+
}));
|
|
2708
|
+
// PURGE, never inherit: the destination's completions belong to the tab
|
|
2709
|
+
// being superseded, and there is no agent left to deliver them to.
|
|
2710
|
+
// forget() logs each one, so the loss is disclosed, not silent.
|
|
2711
|
+
if (destinationHadState)
|
|
2712
|
+
RunCompletions.forget(panelTab);
|
|
2522
2713
|
for (const b of KNOWN_BACKENDS) {
|
|
2523
2714
|
const srcKey = migratedFrom + AGENT_KEY_SEP + b;
|
|
2524
2715
|
const newKey = panelTab + AGENT_KEY_SEP + b;
|
|
@@ -2540,6 +2731,11 @@ export async function runPanelOrchestrator() {
|
|
|
2540
2731
|
hasManagerState: manager.hasAnyState(newKey),
|
|
2541
2732
|
hasDurableSession: sessionStore.get(newKey) !== undefined,
|
|
2542
2733
|
renderHeldCount: heldDuringGen.get(newKey)?.length ?? 0,
|
|
2734
|
+
// #468 — journal state is per PANEL TAB, not per backend, so the
|
|
2735
|
+
// same count applies to every provider's key. Counted BEFORE the
|
|
2736
|
+
// pre-pass purge above so a journal-only destination still drives
|
|
2737
|
+
// the per-backend reset + the bridge-queue drop below.
|
|
2738
|
+
journaledCompletionCount: destinationJournaled,
|
|
2543
2739
|
})) {
|
|
2544
2740
|
destinationCollision = true;
|
|
2545
2741
|
manager.reset(newKey);
|
|
@@ -2558,6 +2754,14 @@ export async function runPanelOrchestrator() {
|
|
|
2558
2754
|
// at send time) and is untouched, so the proven migration's own continuity is preserved.
|
|
2559
2755
|
if (destinationCollision)
|
|
2560
2756
|
bridge.dropQueuedDeliveries(panelTab);
|
|
2757
|
+
// #468 — the superseded destination's journal state was already purged
|
|
2758
|
+
// above (before anything could hand tokens back); now re-address the
|
|
2759
|
+
// INCOMING tab's own completions + run tickets onto the new id.
|
|
2760
|
+
RunCompletions.moveKey(migratedFrom, panelTab);
|
|
2761
|
+
// rebindAgent MOVES the agent rather than spawning one, so onAgentReady
|
|
2762
|
+
// never fires for the new id — flush explicitly or the re-addressed
|
|
2763
|
+
// entries would sit pending until some unrelated later trigger.
|
|
2764
|
+
flushRunCompletions(panelTab);
|
|
2561
2765
|
// #570 — carry the PROVEN source identity forward as the tab's prior identity. The
|
|
2562
2766
|
// rebound agent belongs to it (prevIdentity === newIdentity by sameWorkflow), but the
|
|
2563
2767
|
// new tab id has no prior identity and the agent may have no durable record yet
|
|
@@ -2654,6 +2858,7 @@ export async function runPanelOrchestrator() {
|
|
|
2654
2858
|
}
|
|
2655
2859
|
}
|
|
2656
2860
|
const prev = tabBackends.get(panelTab);
|
|
2861
|
+
let providerSwitched = false;
|
|
2657
2862
|
if (prev && prev !== backend) {
|
|
2658
2863
|
// #570 — Provider switch via re-hello: RETIRE (not reset) the previous provider's
|
|
2659
2864
|
// agent so it stops lingering but its identity-bound durable session is PRESERVED. A
|
|
@@ -2663,8 +2868,16 @@ export async function runPanelOrchestrator() {
|
|
|
2663
2868
|
// still starts fresh (the panel replays the transcript as context on its first message).
|
|
2664
2869
|
manager.retire(panelTab + AGENT_KEY_SEP + prev);
|
|
2665
2870
|
bridge.broadcastTabList(); // live agent dropped on backend switch → refresh dot
|
|
2871
|
+
providerSwitched = true;
|
|
2666
2872
|
}
|
|
2667
2873
|
tabBackends.set(panelTab, backend);
|
|
2874
|
+
// #468 — retire() handed back any run completion the OLD provider held, but
|
|
2875
|
+
// the flush it triggered ran while agentKeyFor() still resolved the OLD
|
|
2876
|
+
// backend, so it could only re-journal it. Re-address it now that the tab
|
|
2877
|
+
// points at the NEW provider, so the completion reaches the conversation
|
|
2878
|
+
// the user actually switched to instead of waiting for their next message.
|
|
2879
|
+
if (providerSwitched)
|
|
2880
|
+
flushRunCompletions(panelTab);
|
|
2668
2881
|
// A headless client (mobile/remote pseudo-panel, no browser canvas) advertises
|
|
2669
2882
|
// itself in the hello frame so its agent gets the in-turn-delivery directive.
|
|
2670
2883
|
if (event.headless === true)
|
|
@@ -2746,6 +2959,13 @@ export async function runPanelOrchestrator() {
|
|
|
2746
2959
|
heldDuringGen.delete(bKey); // render-held work belonging to the torn-down provider
|
|
2747
2960
|
}
|
|
2748
2961
|
bridge.dropQueuedDeliveries(panelTab);
|
|
2962
|
+
// #468 — this tab id is no longer provably serving the workflow that
|
|
2963
|
+
// queued its outstanding runs (replaced in place). Close those tickets
|
|
2964
|
+
// so a late completion is reported to whoever holds the tab now as
|
|
2965
|
+
// UNDETERMINED instead of "the run YOU queued". Journal ENTRIES are
|
|
2966
|
+
// kept: a completion that already arrived is still real news and is
|
|
2967
|
+
// still delivered — just no longer as this conversation's own.
|
|
2968
|
+
RunCompletions.closeRuns(panelTab);
|
|
2749
2969
|
}
|
|
2750
2970
|
}
|
|
2751
2971
|
// Reload restore: the panel re-sends the last session id it saw. HONOR IT ONLY
|
|
@@ -3137,6 +3357,13 @@ export async function runPanelOrchestrator() {
|
|
|
3137
3357
|
tabStableKey.delete(panelTab);
|
|
3138
3358
|
}
|
|
3139
3359
|
tabBackends.set(panelTab, reqBackend);
|
|
3360
|
+
// #468 — the retire() above handed back any run completion the outgoing
|
|
3361
|
+
// provider's agent held, but that flush ran while agentKeyFor() still
|
|
3362
|
+
// resolved the OLD backend. Re-address it now that the tab points at the
|
|
3363
|
+
// new one, so the completion reaches the conversation the user switched to
|
|
3364
|
+
// instead of sitting journaled until their next message.
|
|
3365
|
+
if (prev !== reqBackend)
|
|
3366
|
+
flushRunCompletions(panelTab);
|
|
3140
3367
|
// Leaving a LOCAL provider frees its VRAM (no other tab still on it) —
|
|
3141
3368
|
// the point of switching to Claude/hosted is usually reclaiming the GPU.
|
|
3142
3369
|
if (prev !== reqBackend) {
|
|
@@ -3643,6 +3870,20 @@ export async function runPanelOrchestrator() {
|
|
|
3643
3870
|
// but a mirror viewer (mobile has no Blind concept) can inject
|
|
3644
3871
|
// agent_event frames with images onto a blinded desktop tab.
|
|
3645
3872
|
const evForTab = blindTabs.has(event.tab_id) ? { ...ev, images: [] } : ev;
|
|
3873
|
+
// #468 — a RUN COMPLETION is a promise `panel_run` made ("end your turn,
|
|
3874
|
+
// you WILL be notified"), so it goes through the journal: correlated by
|
|
3875
|
+
// exact prompt id ONCE, here, and replayed until the turn that carries it
|
|
3876
|
+
// ends. Other agent_event kinds (download_done) keep the old best-effort
|
|
3877
|
+
// path — nothing is waiting on them the way a render is.
|
|
3878
|
+
if (ev.kind === "executed") {
|
|
3879
|
+
// Journal the BLIND-STRIPPED copy: a replay must not resurrect pixels
|
|
3880
|
+
// the blind gate removed on arrival. `null` = the panel re-sent a
|
|
3881
|
+
// completion this tab was already given; suppressed, never duplicated.
|
|
3882
|
+
const entry = RunCompletions.record(event.tab_id, evForTab);
|
|
3883
|
+
logger.info(`[panel-orchestrator] tab ${event.tab_id.slice(0, 8)} run completion for ${describeCorrelation(entry.correlation)}${entry.possibleRepeat ? " (flagged as a possible repeat)" : ""}`);
|
|
3884
|
+
flushRunCompletions(event.tab_id);
|
|
3885
|
+
return;
|
|
3886
|
+
}
|
|
3646
3887
|
const delivered = manager.injectEvent(agentKeyFor(event.tab_id), evForTab);
|
|
3647
3888
|
if (delivered) {
|
|
3648
3889
|
logger.info(`[panel-orchestrator] tab ${event.tab_id.slice(0, 8)} event → agent: ${event.kind}`);
|
|
@@ -3678,6 +3919,12 @@ export async function runPanelOrchestrator() {
|
|
|
3678
3919
|
// reset() is synchronous (map cleared now), so no concurrent send() can
|
|
3679
3920
|
// spawn an agent before we report the cleared session.
|
|
3680
3921
|
manager.reset(agentKeyFor(tabId));
|
|
3922
|
+
// #468 — the conversation that queued this tab's outstanding renders is
|
|
3923
|
+
// gone. Close its run tickets so a render it queued, finishing after the
|
|
3924
|
+
// New chat, is reported to the replacement agent as UNDETERMINED rather
|
|
3925
|
+
// than as "the run YOU queued". Already-arrived completions keep the
|
|
3926
|
+
// verdict frozen at their arrival and are still delivered.
|
|
3927
|
+
RunCompletions.closeRuns(tabId);
|
|
3681
3928
|
// reset() clears the exact-tab store; the stable resume index (#570) is the
|
|
3682
3929
|
// manager's blind spot, so drop it here too — a deliberate NEW chat must not
|
|
3683
3930
|
// be resurrected by the unsaved-workflow fallback on the next reload.
|
|
@@ -3720,6 +3967,10 @@ export async function runPanelOrchestrator() {
|
|
|
3720
3967
|
const sid = typeof event.session_id === "string" ? event.session_id : undefined;
|
|
3721
3968
|
const key = agentKeyFor(tabId);
|
|
3722
3969
|
manager.reset(key);
|
|
3970
|
+
// #468 — same as New chat: the conversation being replaced owns the open
|
|
3971
|
+
// runs, so a completion landing after the switch is UNDETERMINED, not the
|
|
3972
|
+
// historical session's own render.
|
|
3973
|
+
RunCompletions.closeRuns(tabId);
|
|
3723
3974
|
if (sid)
|
|
3724
3975
|
manager.setResume(key, sid);
|
|
3725
3976
|
bridge.push({ type: "ack", ok: true, kind: "resume_session" }, tabId);
|
|
@@ -4679,6 +4930,12 @@ export async function runPanelOrchestrator() {
|
|
|
4679
4930
|
return;
|
|
4680
4931
|
shuttingDown = true;
|
|
4681
4932
|
logger.info("[panel-orchestrator] shutting down — stopping agents…");
|
|
4933
|
+
// #468 — EVERY exit path discloses, not just the fatal self-exit: an ordinary
|
|
4934
|
+
// SIGTERM/SIGINT teardown destroys the in-memory journal just as thoroughly.
|
|
4935
|
+
// Runs BEFORE stopAll() so the still-live agents' tabs are still routable for
|
|
4936
|
+
// the chat notice. Idempotent (reportLostCompletionsOnExit no-ops the second
|
|
4937
|
+
// time), so the self-exit path calling it first is harmless.
|
|
4938
|
+
reportLostCompletionsOnExit();
|
|
4682
4939
|
selfRestarter?.stop();
|
|
4683
4940
|
clearInterval(downloadTimer);
|
|
4684
4941
|
clearInterval(queueStatusTimer);
|
|
@@ -4719,8 +4976,41 @@ export async function runPanelOrchestrator() {
|
|
|
4719
4976
|
// No lockfile / unreadable — nothing to clean up.
|
|
4720
4977
|
}
|
|
4721
4978
|
};
|
|
4979
|
+
/** The single in-flight teardown, so repeated signals queue behind it. */
|
|
4980
|
+
let teardownOnce = null;
|
|
4981
|
+
/** How long a REPEATED shutdown signal waits for the in-flight teardown before
|
|
4982
|
+
* forcing the exit. Long enough for a healthy teardown to finish, short enough
|
|
4983
|
+
* that a hung one can't make the process unkillable via SIGINT/SIGTERM. */
|
|
4984
|
+
const FORCED_SHUTDOWN_GRACE_MS = 3000;
|
|
4722
4985
|
const shutdown = async () => {
|
|
4723
|
-
|
|
4986
|
+
// RE-ENTRANT SAFE (#468). teardownCore's `shuttingDown` flag makes a second
|
|
4987
|
+
// call return IMMEDIATELY, so a repeated SIGTERM used to race straight past
|
|
4988
|
+
// the in-flight teardown to process.exit() — skipping the undelivered-
|
|
4989
|
+
// completion disclosure the first one had not reached yet. Memoize the
|
|
4990
|
+
// teardown promise so every later signal AWAITS the first one instead.
|
|
4991
|
+
const first = teardownOnce === null;
|
|
4992
|
+
teardownOnce ??= teardownCore();
|
|
4993
|
+
// …but BOUNDED. A repeated signal is a user asking harder, and awaiting an
|
|
4994
|
+
// unbounded teardown would make the process unkillable through its handled
|
|
4995
|
+
// signals if anything in it hangs. The first signal waits; a later one gives
|
|
4996
|
+
// the in-flight teardown a short grace and then forces the exit. The
|
|
4997
|
+
// completion disclosure is unaffected either way — it runs at the TOP of
|
|
4998
|
+
// teardownCore, before anything that could block.
|
|
4999
|
+
if (first) {
|
|
5000
|
+
await teardownOnce;
|
|
5001
|
+
}
|
|
5002
|
+
else {
|
|
5003
|
+
await Promise.race([
|
|
5004
|
+
teardownOnce,
|
|
5005
|
+
new Promise((resolve) => {
|
|
5006
|
+
const t = setTimeout(() => {
|
|
5007
|
+
logger.warn(`[panel-orchestrator] shutdown did not finish within ${FORCED_SHUTDOWN_GRACE_MS}ms of a repeated signal — forcing exit`);
|
|
5008
|
+
resolve();
|
|
5009
|
+
}, FORCED_SHUTDOWN_GRACE_MS);
|
|
5010
|
+
t.unref?.();
|
|
5011
|
+
}),
|
|
5012
|
+
]);
|
|
5013
|
+
}
|
|
4724
5014
|
process.exit(0);
|
|
4725
5015
|
};
|
|
4726
5016
|
process.on("SIGINT", shutdown);
|
|
@@ -4741,6 +5031,11 @@ export async function runPanelOrchestrator() {
|
|
|
4741
5031
|
// the gate reads as the full "nothing queued or held" contract.)
|
|
4742
5032
|
!manager.hasHeldMail() &&
|
|
4743
5033
|
![...heldDuringGen.values()].some((msgs) => msgs.length > 0) &&
|
|
5034
|
+
// Undelivered run completions (#468) are in-memory like the held mail
|
|
5035
|
+
// above, so teardown erases them too. A restart while one is journaled
|
|
5036
|
+
// would silently drop the render result the agent was promised — exactly
|
|
5037
|
+
// the failure this whole path exists to prevent. Wait for it to land.
|
|
5038
|
+
!RunCompletions.hasOutstanding() &&
|
|
4744
5039
|
!QueueMonitor.isBusy(),
|
|
4745
5040
|
announce: (text) => void bridge.push({ type: "say", text }),
|
|
4746
5041
|
teardown: teardownCore,
|