comfyui-mcp 0.52.2 → 0.52.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +13 -0
  2. package/dist/comfyui/client.js +35 -8
  3. package/dist/comfyui/client.js.map +1 -1
  4. package/dist/comfyui/fetch.js +44 -0
  5. package/dist/comfyui/fetch.js.map +1 -1
  6. package/dist/comfyui/json-guard.js +47 -3
  7. package/dist/comfyui/json-guard.js.map +1 -1
  8. package/dist/config.js +7 -0
  9. package/dist/config.js.map +1 -1
  10. package/dist/orchestrator/panel-tools.js +1511 -332
  11. package/dist/orchestrator/panel-tools.js.map +1 -1
  12. package/dist/orchestrator/promoted-widget.js +101 -0
  13. package/dist/orchestrator/promoted-widget.js.map +1 -0
  14. package/dist/services/download-cache.js +10 -0
  15. package/dist/services/download-cache.js.map +1 -1
  16. package/dist/services/manager-node-search.js +190 -0
  17. package/dist/services/manager-node-search.js.map +1 -0
  18. package/dist/services/manifest.js +27 -60
  19. package/dist/services/manifest.js.map +1 -1
  20. package/dist/services/mid-command-remedy.js +31 -4
  21. package/dist/services/mid-command-remedy.js.map +1 -1
  22. package/dist/services/model-resolver.js +217 -49
  23. package/dist/services/model-resolver.js.map +1 -1
  24. package/dist/services/node-authoring.js +25 -12
  25. package/dist/services/node-authoring.js.map +1 -1
  26. package/dist/services/process-control.js +230 -3
  27. package/dist/services/process-control.js.map +1 -1
  28. package/dist/services/queue-monitor.js +51 -2
  29. package/dist/services/queue-monitor.js.map +1 -1
  30. package/dist/services/trainer-bootstrap.js +51 -5
  31. package/dist/services/trainer-bootstrap.js.map +1 -1
  32. package/dist/services/ui-bridge.js +89 -17
  33. package/dist/services/ui-bridge.js.map +1 -1
  34. package/dist/services/workspace-env.js +40 -0
  35. package/dist/services/workspace-env.js.map +1 -1
  36. package/dist/tools/node-pack.js +20 -4
  37. package/dist/tools/node-pack.js.map +1 -1
  38. package/dist/tools/process-control.js +1 -1
  39. package/dist/tools/process-control.js.map +1 -1
  40. package/dist/tools/vocabulary.js +350 -0
  41. package/dist/tools/vocabulary.js.map +1 -1
  42. package/dist/utils/origin.js +12 -2
  43. package/dist/utils/origin.js.map +1 -1
  44. package/package.json +1 -1
@@ -34,12 +34,17 @@ import { fileURLToPath } from "node:url";
34
34
  import { comfyuiFetch } from "../comfyui/fetch.js";
35
35
  import { assertPanelNotTargetedUnverifiable } from "../services/panel-pin-guard.js";
36
36
  import { nodesInstallCommandArgs } from "../services/node-management.js";
37
+ import { searchPanelNodes } from "../services/manager-node-search.js";
37
38
  import { isPanelAnsweredError } from "../services/panel-answered.js";
38
39
  import { isPreExecutorRefusal } from "../services/panel-refusal.js";
39
40
  import { createSdkMcpServer, tool } from "@anthropic-ai/claude-agent-sdk";
40
41
  import { parse as parseYaml } from "yaml";
42
+ import { SEMVER_RE } from "../services/ui-bridge.js";
43
+ import { compareSemver } from "../services/self-update.js";
44
+ import { primePanelBase, verifiedPanelDiskVersion, } from "../services/panel-workspace.js";
41
45
  import { conversationOfScopeAddress, isScopeAddress, shortTabId } from "../services/session-scope.js";
42
46
  import { NODE_ID_MESSAGE, NODE_ID_PATTERN, normalizeNodeId } from "./node-id.js";
47
+ import { parseContradictoryPromotedWidgetRefusal, resolveInnerPromotedTarget, } from "./promoted-widget.js";
43
48
  import { clearSwitchHold, describeSwitchHold, recordSwitchHold, successProvesSwitchCleared, } from "./switch-hold.js";
44
49
  import { NO_ORIGIN_REMEDY } from "./fence-refusal.js";
45
50
  /** #884 — journal TICKETS (run completions #468, ask answers #486) must be
@@ -80,7 +85,7 @@ import { getNsfwConsent, setNsfwConsent } from "../services/panel-settings.js";
80
85
  import { QueueMonitor } from "../services/queue-monitor.js";
81
86
  import { RunCompletions } from "./run-completion-journal.js";
82
87
  import { AskAnswers, askFingerprint, PANEL_ASK_ID_PREFIX, } from "./ask-answer-journal.js";
83
- import { getObjectInfo, backfillObjectInfo, resetClient, resetObjectInfoCache, } from "../comfyui/client.js";
88
+ import { getObjectInfo, backfillObjectInfo, getQueueVerified, resetClient, resetObjectInfoCache, } from "../comfyui/client.js";
84
89
  import { convertUiToApi, collectNodeTypes } from "../services/workflow-converter.js";
85
90
  import { restartComfyUI, preflightLocalRestart, readServingArgv, describeArgvDrift, recordRestartDispatch, clearRestartDispatch, getRestartDispatchRecord, RESTART_DISPATCH_CAUSATION_WINDOW_MS, PROCESS_WIDE_RESTART_DISPATCH_TOKEN, __processControlTestHooks, } from "../services/process-control.js";
86
91
  import { resetManagerApiCache } from "../services/manager-api-cache.js";
@@ -159,6 +164,14 @@ function fail(err) {
159
164
  const msg = err instanceof Error ? err.message : String(err);
160
165
  return { content: [{ type: "text", text: `Error: ${msg}` }], isError: true };
161
166
  }
167
+ /** A refusal that also carries {@link FenceRepairDiagnosis}. The text is unchanged by
168
+ * this wrapper: the field is ADDITIVE, so a client that ignores structuredContent
169
+ * reads exactly what it read before. */
170
+ function failWithFenceDiagnosis(text, diagnosis) {
171
+ const res = fail(text);
172
+ res.structuredContent = { panel_fence: diagnosis };
173
+ return res;
174
+ }
162
175
  /**
163
176
  * #971 — the AMBIGUOUS-rebind refusal, worded so it can be acted on.
164
177
  *
@@ -543,6 +556,11 @@ export const __panelToolsTestHooks = {
543
556
  setDeclineProbeTiming(timing) {
544
557
  declineProbeTimingOverride = timing;
545
558
  },
559
+ /** Inject a fake #1249 server-side /free so the frozen-tab settle can be
560
+ * driven without real HTTP. null restores the live freeVramDirect. */
561
+ setFreeVramDirect(fn) {
562
+ freeVramDirectOverride = fn;
563
+ },
546
564
  /** Direct access to the #742 decline recheck loop so its hard-deadline
547
565
  * guarantee (codex gate r2) can be unit-tested with a custom deadline. */
548
566
  probeDeclineRecovery,
@@ -625,6 +643,43 @@ function sleep(ms) {
625
643
  // is not mistaken for a dead tab — still capped (never Infinity) so a genuinely
626
644
  // frozen/backgrounded tab fails in bounded time instead of hanging forever.
627
645
  const OBJECT_INFO_REFRESH_ACK_TIMEOUT_MS = 30_000;
646
+ // #1639 — while a ComfyUI prompt is running the frontend main thread often
647
+ // cannot service graph_* at all (reads included). Waiting out the 20/30 s ack
648
+ // bound only surfaces "tab may be backgrounded or frozen" with an unknown
649
+ // mutation outcome. Fail closed BEFORE dispatch for canvas-touching graph
650
+ // commands so the agent gets an explicit QUEUE BUSY instead. `graph_run` is
651
+ // excluded: queuing behind an in-flight job is the documented sweep path, and
652
+ // panel_run already has its own duplicate fence.
653
+ function queueBusySnapshotNote() {
654
+ const snap = QueueMonitor.snapshot();
655
+ if (!snap.running)
656
+ return "";
657
+ const prompt = snap.runningPromptId ? ` (running prompt ${snap.runningPromptId}` : "";
658
+ const node = snap.currentNode ? `, currently at node ${snap.currentNode}` : "";
659
+ const close = snap.runningPromptId ? ")" : "";
660
+ return `${prompt}${node}${close}`;
661
+ }
662
+ function graphCmdBlockedByRunningPrompt(cmd) {
663
+ const name = typeof cmd.cmd === "string" ? cmd.cmd : "";
664
+ if (!name.startsWith("graph_") || name === "graph_run")
665
+ return null;
666
+ const snap = QueueMonitor.snapshot();
667
+ if (!snap.running)
668
+ return null;
669
+ return (`${name} was NOT sent — nothing was applied. QUEUE BUSY: a ComfyUI prompt is running` +
670
+ `${queueBusySnapshotNote()}. The panel tab typically cannot answer graph_* commands ` +
671
+ `(including read-only graph_query / graph_outline) while a prompt is executing — ` +
672
+ `waiting out the ack timeout would only surface a generic "tab may be backgrounded ` +
673
+ `or frozen" with an unknown outcome. Retry after queue (action:"list") shows running: 0.`);
674
+ }
675
+ function queueBusyTimeoutNote() {
676
+ if (!QueueMonitor.snapshot().running)
677
+ return "";
678
+ return (`\n\nQUEUE BUSY: a ComfyUI prompt is still running${queueBusySnapshotNote()}. ` +
679
+ `The panel tab typically cannot answer graph_* (including read-only queries) while a ` +
680
+ `prompt is executing — this is not a backgrounded or frozen tab. Retry after queue ` +
681
+ `(action:"list") shows running: 0.`);
682
+ }
628
683
  const RETRY_SAFE_CMDS = new Set([
629
684
  // Idempotent reads (mirror UiBridge.READONLY_CMDS + list/status probes).
630
685
  "graph_serialize",
@@ -707,6 +762,26 @@ function isMutatingGraphCmd(cmd) {
707
762
  const name = typeof cmd.cmd === "string" ? cmd.cmd : "";
708
763
  return MUTATING_GRAPH_EDIT_CMDS.has(name);
709
764
  }
765
+ /**
766
+ * #1519 — a graph command the panel FENCES that does not mutate the canvas.
767
+ *
768
+ * The panel's `activeWorkflowFenceApplies` fences every `graph_*` command, reads
769
+ * included, and exempts the recovery probe `workflow_list`
770
+ * (`commandIsCanvasTargetless`, panel #759). So "fenced, and not a mutation" is
771
+ * exactly the `graph_*` names that are not in MUTATING_GRAPH_EDIT_CMDS — derived
772
+ * from that one allowlist rather than kept as a second one, so a newly added edit
773
+ * command cannot drift into being classified as a read.
774
+ *
775
+ * The `graph_` prefix is load-bearing for a second reason: the diagnosis this
776
+ * gates runs `workflow_list`, which flows back through this same catch. Keying on
777
+ * the prefix keeps the probe OUT of the branch that launched it, so a panel that
778
+ * fences the probe too (a build predating the #759 exemption) surfaces its own
779
+ * refusal instead of recursing.
780
+ */
781
+ function isFencedGraphRead(cmd) {
782
+ const name = typeof cmd.cmd === "string" ? cmd.cmd : "";
783
+ return name.startsWith("graph_") && !MUTATING_GRAPH_EDIT_CMDS.has(name);
784
+ }
710
785
  /** True when an error is a TRANSIENT transport/reconnect drop (the tab went away
711
786
  * or was replaced), NOT a genuine command error or a live-but-frozen reply
712
787
  * timeout. Deliberately EXCLUDES "did not reply within N ms" (a backgrounded/
@@ -1542,6 +1617,49 @@ function captureRebootHealthBase(ctx) {
1542
1617
  // it carries) can therefore never cross to a different-family instance.
1543
1618
  return loopbackProbeUrl(base);
1544
1619
  }
1620
+ /**
1621
+ * #1671 — the configured LOCAL boot instance, when that is a known loopback
1622
+ * process this orchestrator can account for without a live panel tab.
1623
+ *
1624
+ * `captureRebootHealthBase` requires a live tab handshake. After a crash that
1625
+ * takes the panel bridge offline that proof is gone — the tab is the component
1626
+ * that disappeared. The configured boot URL is still known, and it is the same
1627
+ * target `restart_comfyui` would act on. Returning it is NOT a claim that the
1628
+ * vanished tab fronted this instance; callers must still refuse a proven
1629
+ * mismatch (see offlineRestartHealthBase).
1630
+ */
1631
+ function configuredBootRestartBase() {
1632
+ if (isCloudMode() || isRemoteMode())
1633
+ return null;
1634
+ const bootBase = getBootLocalComfyUIBaseUrl();
1635
+ if (!bootBase || !isLoopbackOrigin(bootBase))
1636
+ return null;
1637
+ const base = bootBase.replace(/\/+$/, "");
1638
+ if (!sameHttpBase(getComfyUIBaseUrl(), base))
1639
+ return null;
1640
+ return loopbackProbeUrl(base);
1641
+ }
1642
+ /**
1643
+ * #1671 — which base, if any, a panel-offline crash recovery may restart.
1644
+ *
1645
+ * Prefer a still-provable tab binding. If the tab is gone, fall back to the
1646
+ * configured boot instance UNLESS the last-known handshake Origin proves the
1647
+ * panel was on a DIFFERENT server (#851/#1593: never restart the wrong one).
1648
+ */
1649
+ function offlineRestartHealthBase(ctx) {
1650
+ const bound = captureRebootHealthBase(ctx);
1651
+ if (bound != null && sameHttpBase(getComfyUIBaseUrl(), bound))
1652
+ return bound;
1653
+ const observed = ctx.bridge?.tabServerOrigin?.(ctx.tabId) ?? null;
1654
+ const verdict = classifyRestartFallbackTarget({
1655
+ headlessBase: getComfyUIBaseUrl(),
1656
+ panelBase: bound,
1657
+ observedOrigin: observed,
1658
+ });
1659
+ if (verdict.kind === "different")
1660
+ return null;
1661
+ return configuredBootRestartBase();
1662
+ }
1545
1663
  let healthProbeOverride = null;
1546
1664
  /** Test injection for the #742 refuse-safe restart preflight (the real one is
1547
1665
  * preflightLocalRestart in process-control). null → the live preflight. */
@@ -1904,6 +2022,167 @@ async function settleExitSubgraphAfterAckTimeout(ctx, timedOut) {
1904
2022
  }
1905
2023
  return timedOut;
1906
2024
  }
2025
+ // ---- panel_free_vram: verifiable when the canvas tab is frozen (#1249) -----
2026
+ // `free_vram` is a purely SERVER-SIDE operation — the panel's handler is a plain
2027
+ // `POST /free` against the ComfyUI the tab fronts — yet the tool treated the
2028
+ // frozen tab's ACK as the only source of truth: a reply timeout left the caller
2029
+ // with "MUTATES … may have been applied" and no way to verify recovery. The
2030
+ // settle below takes the path the tab was only proxying: when the tab PROVABLY
2031
+ // fronts the orchestrator's local boot instance (the captureRebootHealthBase
2032
+ // gate — loopback, server-trusted, handshake-origin-matched), issue /free
2033
+ // DIRECTLY and read /system_stats around it.
2034
+ //
2035
+ // Why re-issuing is safe HERE and nowhere else on the timeout path: /free is
2036
+ // IDEMPOTENT. Unloading already-unloaded models and freeing an already-empty
2037
+ // cache is a no-op, so the copy the frozen tab may still execute when it wakes
2038
+ // cannot double-apply — the exact hazard the bridge's "do not blind-retry"
2039
+ // disclosure guards against for every other mutation does not exist for this
2040
+ // one command. This is a per-command exception, argued per command; it is NOT a
2041
+ // precedent for settling other mutations this way.
2042
+ /** Per-request bound for the direct /free + /system_stats round-trips, so a
2043
+ * wedged server degrades to the honest outcome-unknown instead of hanging the
2044
+ * tool call. ComfyUI applies /free synchronously before answering, so a large
2045
+ * unload can take seconds — 10s matches the bound probeComfyEndpoint callers
2046
+ * already pay on this same server. */
2047
+ const FREE_VRAM_DIRECT_TIMEOUT_MS = 10_000;
2048
+ /** GET `${base}/system_stats` and return its device VRAM counters, or null when
2049
+ * the read cannot answer (unreachable, non-2xx, non-ComfyUI body). Never
2050
+ * throws: an unreadable stat is "no numbers", never evidence in either
2051
+ * direction — the POST's own status is what certifies the free. */
2052
+ async function readVramDevices(base, timeoutMs) {
2053
+ const controller = new AbortController();
2054
+ const timer = setTimeout(() => controller.abort(), Math.max(1, timeoutMs));
2055
+ timer.unref?.();
2056
+ try {
2057
+ const res = await comfyuiFetch(`${base}/system_stats`, {
2058
+ signal: controller.signal,
2059
+ redirect: "manual",
2060
+ });
2061
+ if (res.status < 200 || res.status >= 300)
2062
+ return null;
2063
+ let body;
2064
+ try {
2065
+ body = await res.json();
2066
+ }
2067
+ catch {
2068
+ return null; // 2xx but not JSON — up, but not a /system_stats we trust
2069
+ }
2070
+ if (!looksLikeSystemStats(body))
2071
+ return null;
2072
+ const devices = body.devices;
2073
+ if (!Array.isArray(devices))
2074
+ return null;
2075
+ return devices.map((d) => {
2076
+ const dev = (d ?? {});
2077
+ const sample = {};
2078
+ if (typeof dev.name === "string")
2079
+ sample.name = dev.name;
2080
+ if (typeof dev.vram_total === "number")
2081
+ sample.vram_total = dev.vram_total;
2082
+ if (typeof dev.vram_free === "number")
2083
+ sample.vram_free = dev.vram_free;
2084
+ return sample;
2085
+ });
2086
+ }
2087
+ catch {
2088
+ return null; // unreachable/timed out — no numbers to report
2089
+ }
2090
+ finally {
2091
+ clearTimeout(timer);
2092
+ }
2093
+ }
2094
+ /** Issue ComfyUI's /free DIRECTLY against a proven-local base and read the
2095
+ * VRAM counters around it. Never throws — every failure is a value, so the
2096
+ * settle can degrade to the honest outcome-unknown instead of masking the
2097
+ * original timeout behind a new error. */
2098
+ async function freeVramDirect(base) {
2099
+ const before = await readVramDevices(base, FREE_VRAM_DIRECT_TIMEOUT_MS);
2100
+ const controller = new AbortController();
2101
+ const timer = setTimeout(() => controller.abort(), FREE_VRAM_DIRECT_TIMEOUT_MS);
2102
+ timer.unref?.();
2103
+ try {
2104
+ const res = await comfyuiFetch(`${base}/free`, {
2105
+ method: "POST",
2106
+ headers: { "Content-Type": "application/json" },
2107
+ body: JSON.stringify({ unload_models: true, free_memory: true }),
2108
+ signal: controller.signal,
2109
+ redirect: "manual",
2110
+ });
2111
+ if (res.status < 200 || res.status >= 300) {
2112
+ return { ok: false, reason: `POST ${base}/free answered HTTP ${res.status}` };
2113
+ }
2114
+ }
2115
+ catch (err) {
2116
+ const msg = controller.signal.aborted
2117
+ ? `POST ${base}/free did not answer within ${FREE_VRAM_DIRECT_TIMEOUT_MS} ms`
2118
+ : `POST ${base}/free failed: ${err instanceof Error ? err.message : String(err)}`;
2119
+ return { ok: false, reason: msg };
2120
+ }
2121
+ finally {
2122
+ clearTimeout(timer);
2123
+ }
2124
+ const after = await readVramDevices(base, FREE_VRAM_DIRECT_TIMEOUT_MS);
2125
+ return { ok: true, before, after };
2126
+ }
2127
+ /** Test injection for the direct server-side /free, so the settle can be
2128
+ * driven without real HTTP. null restores the live path. */
2129
+ let freeVramDirectOverride = null;
2130
+ /**
2131
+ * After an ack timeout on `free_vram`, settle the outcome against the one
2132
+ * channel a frozen tab cannot block: the ComfyUI server itself.
2133
+ *
2134
+ * Returns the timeout UNTOUCHED in every case where nothing was verified —
2135
+ * no provable local server (remote/cloud tab, ambiguous origin, untrusted
2136
+ * socket), or a direct /free that itself failed. #1473's rule: an unknown
2137
+ * answer claims nothing in either direction, and the bridge's original
2138
+ * outcome-unknown disclosure is already the honest verdict there.
2139
+ */
2140
+ async function settleFreeVramAfterAckTimeout(ctx, timedOut) {
2141
+ // The same gate the restart certification uses: null unless the tab PROVABLY
2142
+ // fronts THIS orchestrator's local boot instance (loopback + server-trusted +
2143
+ // handshake-origin match). Without that proof a direct /free could aim at a
2144
+ // DIFFERENT server than the one the tab was asked to free — reporting that as
2145
+ // this command's success would be the wrong-target success the gate exists to
2146
+ // prevent, which is worse than the honest unknown being fixed.
2147
+ const base = captureRebootHealthBase(ctx);
2148
+ if (!base)
2149
+ return timedOut;
2150
+ const direct = await (freeVramDirectOverride ?? freeVramDirect)(base);
2151
+ if (!direct.ok) {
2152
+ // The tab never answered AND the server-side path failed. The outcome stays
2153
+ // unknown — say so, and name what was tried, without claiming either way.
2154
+ const text = timedOut.content?.find((c) => c.type === "text")?.text ?? "";
2155
+ return {
2156
+ ...timedOut,
2157
+ content: [
2158
+ {
2159
+ type: "text",
2160
+ text: `${text}\n\nThe server-side fallback also could not reach ComfyUI's /free ` +
2161
+ `(${direct.reason ?? "no detail"}), so whether VRAM was freed remains UNVERIFIED — ` +
2162
+ `check with get_system_stats (action:"health") once the server answers.`,
2163
+ },
2164
+ ],
2165
+ };
2166
+ }
2167
+ const statsNote = direct.before != null && direct.after != null
2168
+ ? "vram_before/vram_after are the server's own /system_stats counters around the free."
2169
+ : "The /system_stats read around it did not answer, so no VRAM counters are reported — the 2xx from /free is the verification, not a measured delta.";
2170
+ return ok({
2171
+ freed: true,
2172
+ unload_models: true,
2173
+ free_memory: true,
2174
+ acknowledged: false,
2175
+ verified: "server-side",
2176
+ via: `POST ${base}/free`,
2177
+ ...(direct.before != null ? { vram_before: direct.before } : {}),
2178
+ ...(direct.after != null ? { vram_after: direct.after } : {}),
2179
+ note: `The panel tab never acknowledged (frozen or backgrounded), so the free was issued ` +
2180
+ `DIRECTLY to the ComfyUI server this tab provably fronts — the same /free endpoint the ` +
2181
+ `panel would have called — and the server confirmed it. /free is idempotent: if the tab's ` +
2182
+ `queued copy still executes when the tab wakes, it is a no-op, so nothing was applied ` +
2183
+ `twice. ${statsNote}`,
2184
+ });
2185
+ }
1907
2186
  // ---- panel_install_node: accepted-but-never-enqueued (#1129) ---------------
1908
2187
  // #1143 fixed the pre-queue REFUSAL (403/404 → direct clone). This is the other
1909
2188
  // half of the same family: legacy Manager 3.x answers the install POST with
@@ -3274,49 +3553,53 @@ export function readOpenActiveAgainstTarget(active, path, activeConfirmed) {
3274
3553
  (typeof a.key === "string" && a.key !== "");
3275
3554
  return identified ? "different" : "indeterminate";
3276
3555
  }
3277
- /**
3278
- * #887 — observe what is active after an open, WITHOUT adopting anything.
3279
- *
3280
- * Split out of `refreshOpenWorkflowUuid` because the two questions have different
3281
- * preconditions. Adoption is gated on the open reply corroborating the requested
3282
- * identity — a reply that cannot prove which workflow it opened must never
3283
- * authorize a fence refresh. Pure observation needs none of that: "what does the
3284
- * panel say is active right now" is answerable regardless of what the open replied,
3285
- * and on the path that matters most the open reply is an ERROR carrying no JSON at
3286
- * all. Requiring corroboration there is what kept the reporter's own case
3287
- * unexamined.
3288
- */
3289
- /**
3290
- * #1337 — CLEAR THE FENCE ON THE VERDICT THAT PROVED IDENTITY.
3291
- *
3292
- * A reporter's session lost the canvas permanently: every graph call refused with
3293
- * `workflow instance mismatch`, and the one fence-EXEMPT recovery could not clear it.
3294
- *
3295
- * `workflow_open` re-derives the fence only on its success branch. The verdict
3296
- *
3297
- * "workflow_open RAN and the canvas IS bound to X — that much was proven — but the
3298
- * graph on it does not match the state that was loaded … You are NOT on the wrong
3299
- * workflow: X IS the active one."
3300
- *
3301
- * is delivered as an ERROR, so control never reaches the re-derivation. That mismatches
3302
- * two different levels: the verdict asserts IDENTITY was proven, and what it withholds
3303
- * is CONTENT (frontend normalisation vs a partial load). Applying content-level
3304
- * uncertainty to an identity-level fence is what leaves the session with no in-band
3305
- * recovery — the only remaining move is a browser reload, which destroys unsaved work.
3306
- *
3307
- * MEASURED, because the comment that justified the old behaviour asserts otherwise:
3308
- * • the panel really does publish NO workflow_uuid on this reply (its own
3309
- * FENCE_NOT_REFRESHED text), so there is nothing to adopt FROM THE REPLY — true;
3310
- * • but `workflow_list` IS fence-exempt (commandIsCanvasTargetless in the panel's
3311
- * workflow-chat-identity.js), and the panel's own recovery text names
3312
- * panel_list_workflows as "exempt from the fence" and says it "republishes the
3313
- * active identity". So the fence CAN be re-derived here; nothing tried.
3314
- *
3315
- * This runs that exempt re-derivation and reports what it achieved. The open still
3316
- * FAILS — the content warning is preserved verbatim, because "re-read before editing"
3317
- * is still the right instruction — and nothing is adopted on the UNPROVEN verdict,
3318
- * where identity itself is in doubt.
3319
- */
3556
+ async function probeLiveGraphUnderCurrentFence(ctx) {
3557
+ // `fields:"ids", limit:1` is the cheapest shape that still has to pass the
3558
+ // instance fence — we only need whether the canvas STILL ACCEPTS this session's
3559
+ // stamp, not the graph itself.
3560
+ let res;
3561
+ try {
3562
+ res = await ctx.call({ cmd: "graph_query", fields: "ids", limit: 1 }, 8000);
3563
+ }
3564
+ catch (err) {
3565
+ if (isWorkflowInstanceMismatch(err))
3566
+ return { status: "mismatch_refused" };
3567
+ return { status: "unanswered", detail: err instanceof Error ? err.message : String(err) };
3568
+ }
3569
+ if (!res?.isError)
3570
+ return { status: "answered" };
3571
+ const text = toolResultText(res);
3572
+ if (isWorkflowInstanceMismatch(text))
3573
+ return { status: "mismatch_refused" };
3574
+ // An acked executor error still means the fence passed — the canvas is the
3575
+ // one this session was already bound to.
3576
+ if (isPanelAnsweredResult(res))
3577
+ return { status: "answered" };
3578
+ return { status: "unanswered", detail: text };
3579
+ }
3580
+ function identityClaimedContentUnverifiedNote(detail) {
3581
+ const busy = queueBusyTimeoutNote();
3582
+ return (`\n\nFENCE: NOT cleared (live graph unread). The panel asserted the canvas IS bound to the ` +
3583
+ `requested workflow, but a live graph read did not come back (${detail || "no reason was reported"}), ` +
3584
+ `so content is UNVERIFIED. Identity-matched is not content-matched: do NOT trust ` +
3585
+ `"you are on the right workflow" / "You are NOT on the wrong workflow". Do NOT edit or save ` +
3586
+ `expecting the opened file. Retry the graph read (panel_graph_outline) once the tab answers.` +
3587
+ busy);
3588
+ }
3589
+ function identityClaimedButLiveGraphUnchangedNote(ctx) {
3590
+ const fence = currentWorkflowFence(ctx);
3591
+ const fenceTxt = fence.known && fence.uuid
3592
+ ? `under this session's existing fence (${fence.uuid})`
3593
+ : `without being refused by a workflow-instance fence`;
3594
+ return (`\n\nFENCE: NOT cleared (live graph still answers). CONTENT MISMATCH: the panel asserted the ` +
3595
+ `canvas IS bound to the requested workflow, but a live graph read still answers ${fenceTxt} — ` +
3596
+ `the graph on screen is the PREVIOUS workflow, not the one just opened. That is the failure ` +
3597
+ `the fence exists to prevent: clearing it here would let later reads of this graph succeed ` +
3598
+ `as if they were the opened file. Do NOT trust "you are on the right workflow" / ` +
3599
+ `"You are NOT on the wrong workflow". Do NOT edit or save expecting the opened file. ` +
3600
+ `Read the graph (panel_graph_outline) to see what is actually open, then retry ` +
3601
+ `panel_open_workflow or panel_load_workflow if the canvas did not switch.`);
3602
+ }
3320
3603
  async function clearFenceOnIdentityProvenOpen(ctx, res) {
3321
3604
  const text = toolResultText(res);
3322
3605
  // ONLY the class that states identity was proven. The UNPROVEN verdict ("could not
@@ -3324,6 +3607,13 @@ async function clearFenceOnIdentityProvenOpen(ctx, res) {
3324
3607
  // fence onto a canvas we cannot identify is how an edit lands on the wrong graph.
3325
3608
  if (!/the canvas IS bound to/i.test(text))
3326
3609
  return { res, repaired: false };
3610
+ const canvas = await probeLiveGraphUnderCurrentFence(ctx);
3611
+ if (canvas.status === "answered") {
3612
+ return { res: appendToolResultText(res, identityClaimedButLiveGraphUnchangedNote(ctx)), repaired: false };
3613
+ }
3614
+ if (canvas.status === "unanswered") {
3615
+ return { res: appendToolResultText(res, identityClaimedContentUnverifiedNote(canvas.detail)), repaired: false };
3616
+ }
3327
3617
  let note;
3328
3618
  // #1560 — reported STRUCTURALLY, never re-read out of the sentence below. The caller
3329
3619
  // uses this to decide whether a "the channel is not answering" note would contradict
@@ -3672,7 +3962,8 @@ function corroborateActiveForFence(parsed) {
3672
3962
  // Same tri-state primitive the pin path uses. `false` = they name DIFFERENT
3673
3963
  // canvases (the stale/mixed case). `undefined` = they share no comparable
3674
3964
  // identity field, so agreement was never established — which is not agreement.
3675
- const verdict = identityVerdict(flaggedActive[0], active);
3965
+ const flagged = flaggedActive[0];
3966
+ const verdict = identityVerdict(flagged, active);
3676
3967
  if (verdict === false) {
3677
3968
  return {
3678
3969
  ok: false,
@@ -3683,6 +3974,20 @@ function corroborateActiveForFence(parsed) {
3683
3974
  };
3684
3975
  }
3685
3976
  if (verdict !== true) {
3977
+ // #1650 — unsaved (`tmp:`) tabs never have path/filename. After a reconnect
3978
+ // the top-level `active` record historically omitted `key`/`routing_key`
3979
+ // whenever the panel had not yet established a workflow identity, while the
3980
+ // unique flagged-active list entry still carried the per-tab `tmp:` handle
3981
+ // from `workflowTabId()`. Those two records describe the same canvas; they
3982
+ // just do not share a field that `identityVerdict` can pair. The reverse
3983
+ // (handle on `active`, omitted on the list row) is the same gap.
3984
+ //
3985
+ // Restricted to BOTH sides being unsaved and exactly one flagged-active
3986
+ // (already checked above). A saved path on either side is a different
3987
+ // canvas, not a missing field.
3988
+ if (unsavedTmpHandleCorroborates(flagged, active)) {
3989
+ return { ok: true, active: fenceRecordForAdoption(flagged, active) };
3990
+ }
3686
3991
  return {
3687
3992
  ok: false,
3688
3993
  seenUuid,
@@ -3692,7 +3997,7 @@ function corroborateActiveForFence(parsed) {
3692
3997
  settles: false,
3693
3998
  };
3694
3999
  }
3695
- return { ok: true, active: active };
4000
+ return { ok: true, active: fenceRecordForAdoption(flagged, active) };
3696
4001
  }
3697
4002
  /**
3698
4003
  * Re-derive this session's command fence from the panel's live active canvas.
@@ -3773,6 +4078,46 @@ WORTH CHECKING — THE PANEL'S VERSION IS UNKNOWN HERE: this session's panel has
3773
4078
  }
3774
4079
  if (!v?.tooOld)
3775
4080
  return "";
4081
+ // #1229 — IS THE INSTALL EVEN BEHIND, OR ONLY WHAT COMFYUI IS SERVING?
4082
+ //
4083
+ // This branch compares the RUNNING panel against the minimum and concludes
4084
+ // "pack is out of date" — but the reporter's pack on disk was 0.14.37 while
4085
+ // the session ran 0.11.38: ComfyUI-Manager had updated the pack IN PLACE
4086
+ // after ComfyUI started, and ComfyUI keeps serving the web assets it
4087
+ // registered at startup. Prescribing `sync` there is a no-op remedy — the
4088
+ // disk already clears the floor — and it cost the reporter a full
4089
+ // pack-version investigation to discover that. The actual fix is a RESTART
4090
+ // plus a hard-refresh, and a hard-refresh ALONE provably does not work,
4091
+ // because the new assets are registered server-side at startup, not re-read
4092
+ // from disk on reload.
4093
+ //
4094
+ // Same proof discipline as resolveStaleBundleSkew (#774): only a disk
4095
+ // version re-read NOW from the observed install dir may override the update
4096
+ // advice. Anything unproven — no observation, an unparseable version, or a
4097
+ // disk version genuinely below the floor — falls through to the sync remedy
4098
+ // unchanged, which is correct for a pack that really is behind.
4099
+ const disk = verifiedPanelDiskVersion()?.trim();
4100
+ if (!disk) {
4101
+ // The observation is missing or stale (most often the live-base
4102
+ // resolution lapsed and this refusal is the first thing to ask in a
4103
+ // while). Refresh it in the background — never awaited, since building
4104
+ // an error message must not block on I/O — so a retry can answer.
4105
+ void primePanelBase().catch(() => { });
4106
+ }
4107
+ if (disk && SEMVER_RE.test(disk) && compareSemver(disk, v.needed) >= 0) {
4108
+ return (`
4109
+
4110
+ WHY THIS READ WAS NEEDED AT ALL: this session's RUNNING panel is ${v.version}, ` +
4111
+ `and a panel only reports the new workflow's identity ON THE REPLY from ` +
4112
+ `${v.needed} onwards — but DO NOT SYNC THE PANEL: the pack ON DISK is ` +
4113
+ `${disk}, which already meets ${v.needed}, so a sync would change nothing. ` +
4114
+ `What is stale is what ComfyUI is SERVING: the pack was updated after ` +
4115
+ `ComfyUI started, and ComfyUI keeps serving the web assets it registered ` +
4116
+ `at startup. Restart ComfyUI so it serves ${disk}, then HARD-REFRESH the ` +
4117
+ `browser tab (Ctrl+Shift+R) — a hard refresh ALONE does not fix this, ` +
4118
+ `because the assets are registered server-side at startup, not re-read ` +
4119
+ `from disk on reload.`);
4120
+ }
3776
4121
  return (`
3777
4122
 
3778
4123
  WHY THIS READ WAS NEEDED AT ALL: this session's panel is ${v.version}, and a ` +
@@ -3860,7 +4205,7 @@ NOTE: an API-format load CAN re-mint the canvas workflow instance. If your next
3860
4205
  `Clear it with panel_set_workflow_target({mode:"current"}), which re-derives the fence ` +
3861
4206
  `from the live canvas, then retry. If the next command is not refused, nothing needs doing.`);
3862
4207
  }
3863
- async function rebindWorkflowFence(ctx) {
4208
+ async function rebindWorkflowFence(ctx, opts) {
3864
4209
  const tabAtStart = ctx.tabId;
3865
4210
  let before = currentWorkflowFence(ctx);
3866
4211
  // `before` describes the tab we are ABOUT to compare against — but ctx.call can
@@ -3952,6 +4297,13 @@ async function rebindWorkflowFence(ctx) {
3952
4297
  // the stamp already matched without ever having read it.
3953
4298
  if (before.known && before.uuid === uuid)
3954
4299
  return { status: "already_current", uuid, before };
4300
+ // #1646 — a READ-ONLY probe never moves the fence: the live canvas naming a
4301
+ // DIFFERENT workflow is reported, not adopted. Only a deliberate rebind
4302
+ // (panel_set_workflow_target, open/new) may replace the fence — a mismatch
4303
+ // diagnosis that re-pointed the session on its own authority routed the
4304
+ // caller's NEXT edits onto the very canvas the refusal named as the wrong one.
4305
+ if (opts?.adopt === false)
4306
+ return { status: "diverged", uuid, before };
3955
4307
  // refreshWorkflowUuid routes through the orchestrator's validator, which
3956
4308
  // re-checks reachability and the uuid's shape/origin binding. A `false` here is
3957
4309
  // a REFUSAL, not a no-op, so it gets its own status rather than being reported
@@ -4095,6 +4447,21 @@ panelGapNote = "") {
4095
4447
  `(#803).`
4096
4448
  : "";
4097
4449
  switch (r.status) {
4450
+ case "diverged":
4451
+ // #1646 — produced ONLY by a read-only probe (`adopt:false`), which the
4452
+ // mismatch diagnosis uses; the deliberate rebinds this renderer serves
4453
+ // never pass it. Handled anyway, because an unhandled union member would
4454
+ // silently render as `undefined` — say exactly what happened if a future
4455
+ // caller ever routes one here.
4456
+ return {
4457
+ binding: "not_recovered",
4458
+ note: ` The live canvas is a DIFFERENT workflow instance (${r.uuid}) than this session's ` +
4459
+ `fence, and it was deliberately NOT adopted — a diagnosis must never re-point ` +
4460
+ `mutation routing on its own.` +
4461
+ (r.before.known && r.before.uuid ? ` The fence still names ${r.before.uuid}.` : "") +
4462
+ `\n\nWHAT TO DO: re-open the workflow you mean with panel_open_workflow, or re-target ` +
4463
+ `the live canvas deliberately by calling this tool with mode:"current".`,
4464
+ };
4098
4465
  case "refreshed":
4099
4466
  return {
4100
4467
  binding: okBinding,
@@ -4864,11 +5231,67 @@ function computeIsActive(rec, activeObj) {
4864
5231
  return identityVerdict(rec, activeObj);
4865
5232
  }
4866
5233
  /**
4867
- * Stable-identity (key/path/routing_key) verdict between a record and the active object.
4868
- * Returns `true` on a positive match, `false` only when the two expose a COMPARABLE field
4869
- * (both non-empty) that DISAGREES, and `undefined` when they share no comparable field at
4870
- * all (so the caller cannot conclude "background" — stay lenient). Filename is never used
4871
- * (it collides across tabs).
5234
+ * Per-tab unsaved handle (`tmp:<id>`). Unsaved tabs have no path/filename; this
5235
+ * is the only unique identity they publish. Accepts any non-empty `tmp:` token
5236
+ * (not only RFC-uuid suffixes) so a panel that mints a shorter handle still
5237
+ * corroborates — `canonicalUnsavedWorkflowIdentity` stays strict for OPEN,
5238
+ * which is a caller-supplied selector.
5239
+ */
5240
+ function recordTmpHandle(value) {
5241
+ if (!value || typeof value !== "object")
5242
+ return null;
5243
+ const rec = value;
5244
+ for (const v of [rec.routing_key, rec.key]) {
5245
+ if (typeof v === "string" && /^tmp:\S+$/.test(v))
5246
+ return v;
5247
+ }
5248
+ return null;
5249
+ }
5250
+ /** Canonical saved path, or null when the record is unsaved / has no path. */
5251
+ function recordSavedPath(value) {
5252
+ if (!value || typeof value !== "object")
5253
+ return null;
5254
+ return canonicalSavedWorkflowPath(value.path);
5255
+ }
5256
+ /**
5257
+ * #1650 — the unique flagged-active list entry and the top-level `active`
5258
+ * record describe the same UNSAVED canvas even when they do not share a
5259
+ * pairable field. True only when BOTH sides lack a saved path and at least
5260
+ * one carries a `tmp:` handle. A saved path on either side is a different
5261
+ * canvas (or a mixed reply), not a missing field.
5262
+ */
5263
+ function unsavedTmpHandleCorroborates(listRec, activeObj) {
5264
+ if (recordSavedPath(listRec) || recordSavedPath(activeObj))
5265
+ return false;
5266
+ if (!(recordTmpHandle(listRec) || recordTmpHandle(activeObj)))
5267
+ return false;
5268
+ const listUuid = responseWorkflowUuid(listRec);
5269
+ const activeUuid = responseWorkflowUuid(activeObj);
5270
+ // Two published uuids that disagree are a mixed reply, not a missing field.
5271
+ if (listUuid && activeUuid && listUuid !== activeUuid)
5272
+ return false;
5273
+ return true;
5274
+ }
5275
+ /**
5276
+ * Record to adopt a fence uuid from. Prefer the top-level `active` object
5277
+ * (it is the one that historically carries `workflow_uuid`); fall back to
5278
+ * the flagged list row when only that row published one.
5279
+ */
5280
+ function fenceRecordForAdoption(listRec, activeObj) {
5281
+ const active = activeObj;
5282
+ if (responseWorkflowUuid(active))
5283
+ return active;
5284
+ const list = listRec;
5285
+ if (responseWorkflowUuid(list))
5286
+ return list;
5287
+ return active;
5288
+ }
5289
+ /**
5290
+ * Stable-identity (key/path/routing_key/tmp: handle) verdict between a record and the
5291
+ * active object. Returns `true` on a positive match, `false` only when the two expose a
5292
+ * COMPARABLE field (both non-empty) that DISAGREES, and `undefined` when they share no
5293
+ * comparable field at all (so the caller cannot conclude "background" — stay lenient).
5294
+ * Filename is never used (it collides across tabs).
4872
5295
  */
4873
5296
  function identityVerdict(rec, activeObj) {
4874
5297
  if (!activeObj || typeof activeObj !== "object")
@@ -4882,6 +5305,12 @@ function identityVerdict(rec, activeObj) {
4882
5305
  [r.routing_key, a.routing_key],
4883
5306
  [r.key, a.routing_key],
4884
5307
  [r.routing_key, a.key],
5308
+ // #1650 — a tmp: handle is a per-tab identity, not a saved path. Pair it
5309
+ // the same way key↔routing_key is paired so an unsaved canvas is not
5310
+ // treated as "no comparable field" when one side published `key` and the
5311
+ // other published `routing_key` (or vice versa).
5312
+ [recordTmpHandle(r), recordTmpHandle(a)],
5313
+ [r.workflow_uuid, a.workflow_uuid],
4885
5314
  ];
4886
5315
  // A CONTRADICTION OUTRANKS AN AGREEMENT (codex gate P0). Returning `true` on
4887
5316
  // the first equal pair meant a mixed reply — matching `key`, conflicting
@@ -4912,6 +5341,12 @@ function identityVerdict(rec, activeObj) {
4912
5341
  [r.key, a.key, (v) => (nonEmpty(v) ? v : null)],
4913
5342
  [r.path, a.path, canonicalSavedWorkflowPath],
4914
5343
  [r.routing_key, a.routing_key, canonicalSavedWorkflowRoutingIdentity],
5344
+ [recordTmpHandle(r), recordTmpHandle(a), (v) => (nonEmpty(v) ? v : null)],
5345
+ [
5346
+ r.workflow_uuid,
5347
+ a.workflow_uuid,
5348
+ (v) => (typeof v === "string" && WORKFLOW_UUID_RE.test(v) ? v : null),
5349
+ ],
4915
5350
  ];
4916
5351
  for (const [x, y, canon] of sameField) {
4917
5352
  if (!nonEmpty(x) || !nonEmpty(y))
@@ -5968,6 +6403,12 @@ export function makePanelToolCtx(bridge, tabId, workflowTargets) {
5968
6403
  await awaitReachable();
5969
6404
  }
5970
6405
  ensureReachable();
6406
+ // #1639 — a running prompt freezes the tab's graph_* channel. Refuse
6407
+ // BEFORE dispatch so a mutation is known-not-applied rather than
6408
+ // delivered-into-a-frozen-tab with a 20/30s unknown-outcome timeout.
6409
+ const blocked = graphCmdBlockedByRunningPrompt(cmd);
6410
+ if (blocked)
6411
+ return fail(blocked);
5971
6412
  const firstTry = ok(await sendRouted(cmd, timeoutMs, observeRid));
5972
6413
  // panel#1097 — a guard-domain command that SUCCEEDS is the evidence that the
5973
6414
  // switch is over, whichever attempt lands it. Without this an ordinary
@@ -6067,7 +6508,7 @@ export function makePanelToolCtx(bridge, tabId, workflowTargets) {
6067
6508
  // Leaving it unmarked fails closed (nothing is settled, no false success)
6068
6509
  // but silently switches the settle off for a real sequence, which is the
6069
6510
  // kind of gap that reads as "the fix does not work" much later.
6070
- return carryReplyTimeoutMark(err2, fail(err2));
6511
+ return carryReplyTimeoutMark(err2, fail(`${err2 instanceof Error ? err2.message : String(err2)}${queueBusyTimeoutNote()}`));
6071
6512
  }
6072
6513
  }
6073
6514
  // #442 defect 4: a MUTATING command (deliberately excluded from RETRY_SAFE_CMDS)
@@ -6111,35 +6552,198 @@ export function makePanelToolCtx(bridge, tabId, workflowTargets) {
6111
6552
  // The bridge cannot tell them apart — from its side both are "no identity". The
6112
6553
  // orchestrator can, with the same read-only re-derivation the documented recovery
6113
6554
  // performs. So it is measured, once, and the answer names the remedy that fits.
6555
+ // panel#1339 — `refreshed` AND `already_current` ARE NOT THE SAME ANSWER.
6556
+ //
6557
+ // The branch below used to return ONE sentence for both: "the live canvas DOES
6558
+ // carry an identity (<uuid>) and this session's fence has been re-derived onto
6559
+ // it. RETRY THIS EXACT CALL ONCE". The reporter read that as a contradiction —
6560
+ // told the identity had *already* been re-derived, yet refused anyway — and the
6561
+ // two states it covers want opposite next moves:
6562
+ //
6563
+ // refreshed the tab had NO fence and THIS CALL installed one, derived
6564
+ // from the live canvas. The refusal you are holding is the
6565
+ // repair. A bare retry is the right move and works because
6566
+ // of it — which is why it looked "transient": the first call
6567
+ // did the work and reported failure.
6568
+ // already_current the fence read back as ALREADY PRESENT AND EQUAL. It cannot
6569
+ // have been that at dispatch — the refusal is proof the stamp
6570
+ // was empty, and currentWorkflowFence reads the same resolver
6571
+ // the bridge consulted — so a fence appeared between the two
6572
+ // reads, and NOT because of this call: `already_current`
6573
+ // returns BEFORE any adoption. Two things produce it and this
6574
+ // code cannot tell them apart, which is why it asserts
6575
+ // neither: the session was moved onto a DIFFERENT tab while
6576
+ // the check ran (rebindWorkflowFence's workflow_list round
6577
+ // trip is retry-safe, its retry runs ensureReachable, and
6578
+ // `before` is then re-read for the new tab), or a fence for
6579
+ // THIS tab was installed in that window by something else (a
6580
+ // concurrent rebind, or the panel's own mismatch re-hello,
6581
+ // #1043/#932). Either way NOTHING WAS REPAIRED for the caller
6582
+ // and the uuid quoted may not be the one they were refused
6583
+ // against — so the remedy is to confirm the target, not to
6584
+ // name a mechanism nobody measured.
6585
+ //
6586
+ // Splitting the sentence is half the fix. The other half is that the answer must
6587
+ // be readable WITHOUT parsing the sentence: an agent deciding to re-run a
6588
+ // mutation off matched error prose is how a write gets double-applied. See
6589
+ // FenceRepairDiagnosis — the verdict, and the bridge-owned dispatch flag the
6590
+ // "nothing was applied" claim rests on, ride in structuredContent.
6591
+ //
6592
+ // What does NOT change: every branch still REFUSES. The call really did not
6593
+ // perform the widget write — it performed the repair — and a refusal that
6594
+ // repairs and REPORTS is a different risk from one that repairs and PROCEEDS
6595
+ // (#1646 removed exactly that from the neighbouring branch). Nothing here
6596
+ // auto-continues the mutation.
6114
6597
  if (isNoTrustedIdentityRefusal(err) && isMutatingGraphCmd(cmd)) {
6115
6598
  const raw = err instanceof Error ? err.message : String(err);
6599
+ // The TYPED flag, not the text predicate above: `isNoTrustedIdentityRefusal`
6600
+ // is a phrase match and would fire on anything that merely quotes the phrase.
6601
+ // Only the bridge can say whether the frame reached the socket.
6602
+ const dispatchFlag = dispatchOutcomeOf(err);
6603
+ const dispatched = dispatchFlag === false ? "no" : dispatchFlag === true ? "yes" : "unknown";
6604
+ const retrySafe = dispatched === "no" ? "yes" : dispatched === "yes" ? "no" : "unknown";
6605
+ // THREE-WAY, because the flag is three-way. A two-way ternary here printed
6606
+ // "this refusal carries no dispatch flag" for a refusal whose flag was
6607
+ // present and said `true` — the same collapse this whole branch exists to
6608
+ // remove, reintroduced one level up (review of this PR). Each arm states the
6609
+ // observation it actually has.
6610
+ const nothingApplied = dispatched === "no"
6611
+ ? ` Nothing was applied (the bridge reports this frame was never written to the ` +
6612
+ `socket), so re-issuing cannot double-apply.`
6613
+ : dispatched === "yes"
6614
+ ? ` CAUTION — the bridge reports this frame WAS written to the socket, so the ` +
6615
+ `mutation may ALREADY have been applied. Do not re-issue it blindly; establish ` +
6616
+ `what landed first (read the node back with panel_query_graph).`
6617
+ : ` Whether anything was applied is NOT established here — this refusal carries no ` +
6618
+ `dispatch flag — so do not re-issue on the strength of this message alone.`;
6619
+ // A retry may only be ORDERED when the bridge proved nothing was written.
6620
+ // Otherwise the instruction contradicts `retry_safe`, which is exactly what a
6621
+ // caller keys on: "RETRY THIS EXACT CALL ONCE" beside `retry_safe:"no"` is a
6622
+ // self-contradictory verdict, and the prose is the half an agent obeys.
6623
+ const mayOrderRetry = dispatched === "no";
6624
+ const retryOrder = mayOrderRetry
6625
+ ? `RETRY THIS EXACT CALL ONCE.`
6626
+ : `The fence is repaired, so the same call should now pass it — but re-issue only ` +
6627
+ `after settling the question below.`;
6628
+ // …and the machine-readable step follows the same rule.
6629
+ const retryAction = (fallback) => mayOrderRetry ? fallback : "verify_applied_then_decide";
6630
+ const tabBefore = ctx.tabId;
6116
6631
  try {
6117
6632
  const rebind = await rebindWorkflowFence(ctx);
6633
+ // WHAT WAS THERE BEFORE, reported as the tri-state it is rather than folded
6634
+ // into an absence. `refreshed` is returned for every `before` that is not a
6635
+ // known-equal fence, which is THREE different priors: definitively none, a
6636
+ // read that FAILED, and a fence naming a DIFFERENT workflow that was then
6637
+ // replaced. FenceRead's own contract forbids collapsing the second into the
6638
+ // first ("an absence nobody observed"), and a first draft of this message did
6639
+ // exactly that by saying "this session had NO fence for it" on all three.
6640
+ const priorFence = !rebind.before.known
6641
+ ? "unreadable"
6642
+ : rebind.before.uuid
6643
+ ? "present"
6644
+ : "absent";
6645
+ const base = {
6646
+ dispatched,
6647
+ retry_safe: retrySafe,
6648
+ rebind_status: rebind.status,
6649
+ prior_fence: priorFence,
6650
+ ...(rebind.before.known && rebind.before.uuid
6651
+ ? { prior_fence_uuid: rebind.before.uuid }
6652
+ : {}),
6653
+ tab_before: tabBefore,
6654
+ tab_after: ctx.tabId,
6655
+ };
6118
6656
  if (rebind.status === "no_identity") {
6119
- return fail(`${raw}\n\nCHECKED, so this is not a guess: the live canvas was re-read and it ` +
6657
+ return failWithFenceDiagnosis(`${raw}\n\nCHECKED, so this is not a guess: the live canvas was re-read and it ` +
6120
6658
  `carries no workflow identity either (${rebind.why}). ` +
6121
6659
  `panel_set_workflow_target({mode:"current"}) will NOT clear this — it chooses ` +
6122
6660
  `WHICH workflow to fence against and cannot mint an identity for one that has ` +
6123
6661
  `none, so it reports success while every mutation keeps failing. RECOVERY: ` +
6124
6662
  `panel_open_workflow(<path>) on this same workflow; re-opening it is what gives ` +
6125
6663
  `it an identity. If it has never been saved there is no path to re-open — save ` +
6126
- `it first with panel_save_workflow, which also gives it a stable identity.`);
6664
+ `it first with panel_save_workflow, which also gives it a stable identity.`, {
6665
+ ...base,
6666
+ fence_repaired_by_this_call: "no",
6667
+ retry_clears_refusal: "no",
6668
+ next_action: "open_or_save_workflow",
6669
+ });
6127
6670
  }
6128
- if (rebind.status === "refreshed" || rebind.status === "already_current") {
6129
- return fail(`${raw}\n\nCHECKED: the live canvas DOES carry an identity (${rebind.uuid}) and ` +
6130
- `this session's fence has been re-derived onto it. RETRY THIS EXACT CALL ONCE — ` +
6131
- `nothing was applied, so a retry cannot double-apply.`);
6671
+ if (rebind.status === "refreshed") {
6672
+ // Only what was OBSERVED about the prior fence. Each arm is a different
6673
+ // fact with a different implication, and "no fence" is true for exactly
6674
+ // one of them.
6675
+ const wasBefore = priorFence === "absent"
6676
+ ? `this session had NO fence for it`
6677
+ : priorFence === "present"
6678
+ ? `this session's fence named a DIFFERENT workflow (${rebind.before.known ? rebind.before.uuid : ""}), which has been REPLACED`
6679
+ : `this session's prior fence could not be read, so whether there was one is ` +
6680
+ `not claimed here`;
6681
+ return failWithFenceDiagnosis(`${raw}\n\nCHECKED: the live canvas DOES carry an identity (${rebind.uuid}), ` +
6682
+ `${wasBefore}, and THIS CALL installed one derived from that canvas — the ` +
6683
+ `refusal you are reading is what repaired it, which is why the same call ` +
6684
+ `refused now and passes the fence next. ${retryOrder}${nothingApplied}`, {
6685
+ ...base,
6686
+ workflow_uuid: rebind.uuid,
6687
+ fence_repaired_by_this_call: "yes",
6688
+ retry_clears_refusal: "yes",
6689
+ next_action: retryAction("retry_same_call"),
6690
+ });
6691
+ }
6692
+ if (rebind.status === "already_current") {
6693
+ return failWithFenceDiagnosis(`${raw}\n\nCHECKED, and THIS CALL REPAIRED NOTHING: the re-read found a fence that ` +
6694
+ `was ALREADY present and already named the live canvas (${rebind.uuid}). That ` +
6695
+ `cannot be the fence your command was refused against — the refusal is proof ` +
6696
+ `that one was missing — so a fence appeared between the two reads, and not ` +
6697
+ `through this call. It is EITHER a fence for a different tab this session was ` +
6698
+ `moved onto while the check ran, OR one installed for this tab by something ` +
6699
+ `else in that window; nothing here can tell which, so ${rebind.uuid} is not ` +
6700
+ `claimed to be the identity you asked for. CONFIRM THE TARGET BEFORE RETRYING: ` +
6701
+ `panel_set_workflow_target({mode:"current"}) if you mean the canvas that is live ` +
6702
+ `now, or panel_open_workflow(<path>) for the workflow you actually meant; then ` +
6703
+ `re-issue.${nothingApplied} A bare retry is not refused by this message — it is ` +
6704
+ `simply not aimed at anything this check verified.`, {
6705
+ ...base,
6706
+ workflow_uuid: rebind.uuid,
6707
+ fence_repaired_by_this_call: "no",
6708
+ // NOT "yes". A retry would carry the fence this read saw, but nothing
6709
+ // here establishes that it belongs to the tab the caller addressed.
6710
+ retry_clears_refusal: "unknown",
6711
+ next_action: retryAction("confirm_target_then_retry"),
6712
+ });
6132
6713
  }
6133
6714
  // unreadable / uncorroborated — say so rather than picking a remedy.
6134
- return fail(`${raw}\n\nCHECKED, and the answer is UNKNOWN: the live canvas could not be re-read ` +
6715
+ return failWithFenceDiagnosis(`${raw}\n\nCHECKED, and the answer is UNKNOWN: the live canvas could not be re-read ` +
6135
6716
  `well enough to say whether it has an identity. Try ` +
6136
6717
  `panel_open_workflow(<path>) on the workflow you mean — it is the only recovery ` +
6137
6718
  `that works in BOTH states, because it gives the workflow an identity rather than ` +
6138
- `adopting one that may not exist.`);
6719
+ `adopting one that may not exist.`, {
6720
+ ...base,
6721
+ ...(("uuid" in rebind) ? { workflow_uuid: rebind.uuid } : {}),
6722
+ // `adopt_error` is the one status that cannot say which side of the write
6723
+ // it threw on, so it is the one that may not claim "nothing was repaired".
6724
+ fence_repaired_by_this_call: rebind.status === "adopt_error" ? "unknown" : "no",
6725
+ retry_clears_refusal: "unknown",
6726
+ next_action: "open_workflow",
6727
+ });
6139
6728
  }
6140
6729
  catch {
6141
- // The diagnosis must never change how the call failed.
6142
- return fail(raw);
6730
+ // The diagnosis must never change how the call failed — so the TEXT is the
6731
+ // bare cause, exactly as before. The field is still emitted, saying unknown:
6732
+ // a caller that has to distinguish "no field" from "field says unknown" is
6733
+ // back to inferring, which is the defect this fix is about.
6734
+ return failWithFenceDiagnosis(raw, {
6735
+ dispatched,
6736
+ retry_safe: retrySafe,
6737
+ rebind_status: "check_threw",
6738
+ // The check threw, so it never reported a `before` — that is not an
6739
+ // absence, it is an unmade observation.
6740
+ prior_fence: "unreadable",
6741
+ tab_before: tabBefore,
6742
+ tab_after: ctx.tabId,
6743
+ fence_repaired_by_this_call: "unknown",
6744
+ retry_clears_refusal: "unknown",
6745
+ next_action: "unknown",
6746
+ });
6143
6747
  }
6144
6748
  }
6145
6749
  // #1330 — CORROBORATE A FENCE MISMATCH INSTEAD OF LETTING IT REPEAT.
@@ -6152,9 +6756,16 @@ export function makePanelToolCtx(bridge, tabId, workflowTargets) {
6152
6756
  // because nothing in the refusal distinguishes "the canvas really is a different
6153
6757
  // workflow" from "the identity flipped for a moment while you were building".
6154
6758
  //
6155
- // The fence is NOT weakened and nothing is auto-applied. This performs the same
6156
- // read-only re-derivation the documented recovery performs, then says which of the
6157
- // two states was found. One informed retry replaces fourteen blind ones.
6759
+ // The fence is NOT weakened, nothing is auto-applied, and — #1646 — the
6760
+ // probe is READ-ONLY. The first version of this check re-derived the fence
6761
+ // onto the live canvas when the two genuinely differed ("AUTO-REBIND"), so
6762
+ // every later mutation in the caller's sequence was silently re-pointed at
6763
+ // the very canvas the refusal had just named as the wrong one — the exact
6764
+ // corruption the fence exists to prevent, delivered as recovery. Now the
6765
+ // check says which of the two states it found and the fence moves ONLY on
6766
+ // an explicit rebind: panel_set_workflow_target({mode:"current"}) or a
6767
+ // successful open. Until then every write stays refused against the target
6768
+ // the caller actually named. One informed retry replaces fourteen blind ones.
6158
6769
  //
6159
6770
  // Safe to recommend a retry because a fence refusal is checked BEFORE the handler
6160
6771
  // runs — "Nothing was applied" is structural here, not an echoed claim.
@@ -6170,36 +6781,209 @@ export function makePanelToolCtx(bridge, tabId, workflowTargets) {
6170
6781
  const stamped = /issued for workflow instance ([0-9a-f-]{36})/i.exec(err instanceof Error ? err.message : String(err))?.[1] ?? null;
6171
6782
  let verdict;
6172
6783
  try {
6173
- const rebind = await rebindWorkflowFence(ctx);
6784
+ const probe = await rebindWorkflowFence(ctx, { adopt: false });
6174
6785
  verdict =
6175
- rebind.status === "already_current" && stamped && rebind.uuid === stamped
6176
- ? `\n\nAUTO-REBIND: ATTEMPTED, and the live canvas now reports the SAME workflow ` +
6177
- `instance this command carried (${stamped}) — nothing needed repairing. The ` +
6178
- `mismatch was TRANSIENT: the identity flipped and settled back, which happens ` +
6179
- `while a new unsaved workflow is still materialising. RETRY THIS EXACT CALL ONCE. ` +
6180
- `Nothing was applied, so a retry cannot double-apply, and re-issuing the whole ` +
6181
- `build would duplicate the work that already succeeded.`
6182
- : rebind.status === "already_current"
6183
- ? `\n\nAUTO-REBIND: ATTEMPTED; the fence already named the live canvas ` +
6184
- `(${rebind.uuid}), so it was not the stale side. Retry once — if it refuses ` +
6786
+ probe.status === "already_current" && stamped && probe.uuid === stamped
6787
+ ? `\n\nCHECKED: the live canvas now reports the SAME workflow instance this ` +
6788
+ `command carried (${stamped}), so the mismatch was TRANSIENT: the identity ` +
6789
+ `flipped and settled back, which happens while a new unsaved workflow is still ` +
6790
+ `materialising. RETRY THIS EXACT CALL ONCE. Nothing was applied, so a retry ` +
6791
+ `cannot double-apply, and re-issuing the whole build would duplicate the work ` +
6792
+ `that already succeeded.`
6793
+ : probe.status === "already_current"
6794
+ ? `\n\nCHECKED: the session's fence already names the live canvas ` +
6795
+ `(${probe.uuid}), so it was not the stale side. Retry once — if it refuses ` +
6185
6796
  `again with the same pair, the two identities are genuinely disagreeing and ` +
6186
6797
  `panel_open_workflow is the way to settle which one you mean.`
6187
- : rebind.status === "refreshed"
6188
- ? `\n\nAUTO-REBIND: ATTEMPTED and the fence was RE-DERIVED onto the live canvas ` +
6189
- `(now ${rebind.uuid}). Retry once. If you meant the EARLIER workflow, re-select ` +
6190
- `it with panel_open_workflow first — this session now points at the live one.`
6191
- : `\n\nAUTO-REBIND: ATTEMPTED and did NOT succeed (${rebind.status}), so the fence ` +
6192
- `is unchanged and a bare retry will fail the same way. Re-select the workflow ` +
6193
- `you mean with panel_open_workflow, then retry.`;
6798
+ : probe.status === "diverged"
6799
+ ? `\n\nCHECKED, and this session was NOT re-pointed: the live canvas is a ` +
6800
+ `DIFFERENT workflow (${probe.uuid}) than the one this command was issued ` +
6801
+ `for${stamped ? ` (${stamped})` : ""}. The fence is unchanged, so later ` +
6802
+ `edits in this sequence keep being refused rather than landing on the ` +
6803
+ `wrong canvas. WHAT TO DO: to edit the workflow you issued for, bring it ` +
6804
+ `back with panel_open_workflow; to follow the live canvas instead, ` +
6805
+ `re-target deliberately with panel_set_workflow_target({mode:"current"}). ` +
6806
+ `Either way the move is explicit — it is never made for you off a refused ` +
6807
+ `mutation.`
6808
+ : probe.status === "healed_by_panel"
6809
+ ? `\n\nCHECKED, and the answer CHANGED while it was being read: the panel ` +
6810
+ `re-advertised its identity and this session's fence moved to the live ` +
6811
+ `canvas (${probe.uuid}) — through the panel's own repair, not this ` +
6812
+ `check. If you meant the EARLIER workflow, re-select it with ` +
6813
+ `panel_open_workflow before any further edits; they now target the live one.`
6814
+ : `\n\nCHECKED, but the live canvas could not be established ` +
6815
+ `(${probe.status}), so the fence is unchanged and a bare retry will fail ` +
6816
+ `the same way. Re-select the workflow you mean with panel_open_workflow, ` +
6817
+ `then retry.`;
6194
6818
  }
6195
- catch (rebindErr) {
6819
+ catch (probeErr) {
6196
6820
  // Never let the diagnosis fail the call differently than it already failed.
6197
6821
  verdict =
6198
- `\n\nAUTO-REBIND: ATTEMPTED and threw, so the fence state is UNKNOWN — this refusal ` +
6199
- `stands on its own terms. (${rebindErr instanceof Error ? rebindErr.message : String(rebindErr)})`;
6822
+ `\n\nCHECKED, and the check itself threw, so the live canvas is UNKNOWN — this ` +
6823
+ `refusal stands on its own terms and the fence is unchanged. ` +
6824
+ `(${probeErr instanceof Error ? probeErr.message : String(probeErr)})`;
6200
6825
  }
6201
6826
  return fail(`${name} was NOT applied — nothing changed. ${raw}${verdict}`);
6202
6827
  }
6828
+ // #1519 — THE SAME REFUSAL ON A READ, AND AN ABSENT STAMP IS NOT A WRONG ONE.
6829
+ //
6830
+ // The reporter's session resumed onto a different workflow and the very first
6831
+ // live-canvas read came back
6832
+ //
6833
+ // workflow instance mismatch: this command carries no workflow-instance
6834
+ // stamp, and the active canvas reports 2b3f4684-…. Nothing was applied.
6835
+ //
6836
+ // Measured on current main before this branch existed: the panel's refusal IS
6837
+ // the entire tool result. The corroboration above is gated on
6838
+ // `isMutatingGraphCmd`, so a READ refused by the very same fence fell through
6839
+ // every branch here and this side added nothing at all. #1480's guard already
6840
+ // extends its diagnosis to reads "on purpose: `panel_graph_outline` refusing
6841
+ // was half of the reported dead end"; this is the same reasoning for the stamp
6842
+ // fence.
6843
+ //
6844
+ // What the panel says on its own is NOT nothing, and the difference is the
6845
+ // point. Since panel 0.11.83 its refusal ends "Re-target with
6846
+ // panel_set_workflow_target({mode:"current"}), or re-select the intended
6847
+ // workflow with panel_open_workflow, then retry" — both exits, offered as
6848
+ // interchangeable, with nothing said about which one this refusal calls for.
6849
+ // That is right for the panel, which by design "observed only that the two
6850
+ // identities differ" and refuses to infer a cause; it is not enough for the
6851
+ // caller, and in the #1331 state the first of the two cannot work at all — it
6852
+ // reports success while the read keeps failing. Only this side can take the
6853
+ // read that decides, so this side takes it.
6854
+ //
6855
+ // TWO REFUSALS, NOT ONE. `isWorkflowInstanceMismatch` matches both of the
6856
+ // panel's states, and they are different facts with OPPOSITE remedies:
6857
+ //
6858
+ // "carries no workflow-instance stamp" → this session has NO workflow
6859
+ // identity. Nothing was compared; the command was refused for arriving
6860
+ // bare. Deriving a fence from the live canvas is what fixes it.
6861
+ // "issued for workflow instance <uuid>" → this session HAS an identity and
6862
+ // the canvas disagrees with it. Deriving a fence from the live canvas
6863
+ // ABANDONS the workflow the caller named — the right move only if that
6864
+ // is what they meant.
6865
+ //
6866
+ // Collapsing them would hand the second case the first case's remedy, which is
6867
+ // the retarget #1646 removed for cause. So the shape is read from the panel's
6868
+ // own words and, when it matches NEITHER wording, the answer is UNKNOWN and is
6869
+ // said to be — never guessed into one of the two.
6870
+ //
6871
+ // Read from the REFUSAL, never from `cmd.workflow_uuid`: the stamp is applied
6872
+ // downstream of here, so that field is undefined at this point for BOTH states
6873
+ // (the trap that silently disabled #1330's transient branch one block up).
6874
+ //
6875
+ // NOTHING IS ADOPTED. The probe is the same read-only one (`adopt:false`), so
6876
+ // this reports which state it found and the fence moves only on an explicit
6877
+ // rebind. The refusal itself is preserved verbatim and the call still fails.
6878
+ if (isWorkflowInstanceMismatch(err) && isFencedGraphRead(cmd)) {
6879
+ const name = typeof cmd.cmd === "string" ? cmd.cmd : "panel command";
6880
+ const raw = err instanceof Error ? err.message : String(err);
6881
+ const stamped = /issued for workflow instance ([0-9a-f-]{36})/i.exec(raw)?.[1] ?? null;
6882
+ const unstamped = /carries no workflow-instance stamp/i.test(raw);
6883
+ // Three-valued on purpose. A panel whose wording matches neither is not
6884
+ // evidence for either state, and this branch must not manufacture one.
6885
+ const shape = unstamped
6886
+ ? "unstamped"
6887
+ : stamped
6888
+ ? "stamped"
6889
+ : "unstated";
6890
+ // Naming `mode:"current"` to a PINNED session is naming something that also
6891
+ // RELEASES the pin. Say so where it applies rather than letting the caller
6892
+ // discover it by losing their target.
6893
+ const pin = ctx.workflowTarget?.get(ctx.tabId);
6894
+ const pinNote = pin?.mode === "pinned" && pin.path
6895
+ ? ` NOTE: this session is PINNED to ${pin.filename ?? pin.path}, and mode:"current" ` +
6896
+ `RELEASES that pin. To keep it, bring that workflow back to the canvas with ` +
6897
+ `panel_open_workflow(${JSON.stringify(pin.path)}) and retry instead.`
6898
+ : "";
6899
+ const RETRY_IS_FREE = `RETRY THIS EXACT CALL ONCE — this is a read, so re-issuing it cannot double-apply ` +
6900
+ `anything.`;
6901
+ let verdict;
6902
+ try {
6903
+ const probe = await rebindWorkflowFence(ctx, { adopt: false });
6904
+ const live = "uuid" in probe ? probe.uuid : null;
6905
+ verdict =
6906
+ probe.status === "no_identity"
6907
+ // Worded without reference to the session's own side, because this
6908
+ // state is reachable from BOTH shapes: an unstamped session and a
6909
+ // stamped one can each face a canvas with no readable identity, and
6910
+ // "no identity EITHER" would be false for the second.
6911
+ ? `\n\nCHECKED, so this is not a guess: the live canvas was re-read and it carries ` +
6912
+ `no workflow identity of its own (${probe.why}). ` +
6913
+ `panel_set_workflow_target({mode:"current"}) will NOT clear this — it chooses ` +
6914
+ `WHICH workflow to fence against and cannot mint an identity for one that has ` +
6915
+ `none, so it reports success while this read keeps being refused. RECOVERY: ` +
6916
+ `panel_open_workflow(<path>) on this same workflow; re-opening it is what gives ` +
6917
+ `it an identity. If it has never been saved there is no path to re-open — save ` +
6918
+ `it first with panel_save_workflow, which also gives it a stable identity.`
6919
+ : probe.status === "healed_by_panel"
6920
+ ? `\n\nCHECKED, and the answer CHANGED while it was being read: the panel ` +
6921
+ `re-advertised its identity and this session's fence moved to the live canvas ` +
6922
+ `(${live}) — through the panel's own repair, not this check. ${RETRY_IS_FREE}`
6923
+ : probe.status === "already_current"
6924
+ ? shape === "unstamped"
6925
+ ? // The COMPARISON, not a cause — the panel's own discipline. An
6926
+ // identity established between dispatch and this read explains
6927
+ // it, and so does this session having been re-bound onto another
6928
+ // tab in between; neither was witnessed, so neither is asserted.
6929
+ `\n\nCHECKED, and the two readings disagree: this session's fence NOW ` +
6930
+ `names the live canvas (${live}), yet this command reached the panel ` +
6931
+ `carrying no stamp at all. What produced that gap was not observed from ` +
6932
+ `here. ${RETRY_IS_FREE}`
6933
+ : `\n\nCHECKED: this session's fence already names the live canvas ` +
6934
+ `(${live}), so it was not the stale side and the disagreement is gone by ` +
6935
+ `the time it was looked at. ${RETRY_IS_FREE} If it refuses again with the ` +
6936
+ `same pair, the two identities are genuinely disagreeing and ` +
6937
+ `panel_open_workflow is the way to settle which one you mean.`
6938
+ : probe.status === "diverged"
6939
+ ? shape === "unstamped"
6940
+ ? // Says only what the panel reported and what the probe read. It
6941
+ // does NOT assert that this session holds no fence right now:
6942
+ // `diverged` is also reached with a fence naming some third
6943
+ // workflow, and that reading was never taken.
6944
+ `\n\nCHECKED, and this is a MISSING stamp rather than a wrong one: the ` +
6945
+ `panel refused it for arriving with NO stamp, so no two identities were ` +
6946
+ `compared — this is not the case where you are pointed at another ` +
6947
+ `workflow. The live canvas DOES have an identity ` +
6948
+ `(${live}). Nothing was adopted here; this check is read-only and the ` +
6949
+ `fence is unchanged. RECOVERY: ` +
6950
+ `panel_set_workflow_target({mode:"current"}) derives this session's ` +
6951
+ `fence from the live canvas, after which this read carries a stamp and ` +
6952
+ `runs.${pinNote}`
6953
+ : shape === "stamped"
6954
+ ? // "was issued for", not "is fenced to": the uuid comes from the
6955
+ // panel's account of what the COMMAND carried, and the session's
6956
+ // fence may have moved since. The mutation branch above words
6957
+ // it the same way for the same reason.
6958
+ `\n\nCHECKED, and this session was NOT re-pointed: this command was ` +
6959
+ `issued for ${stamped} and the live canvas is a DIFFERENT workflow ` +
6960
+ `(${live}). ` +
6961
+ `This is a WRONG stamp, not a missing one, so the two exits are not ` +
6962
+ `interchangeable: to read the workflow you issued for, bring it back ` +
6963
+ `with panel_open_workflow; to read the live canvas instead, re-target ` +
6964
+ `deliberately with panel_set_workflow_target({mode:"current"}) — that ` +
6965
+ `also re-points every later EDIT in this session, which is why it is ` +
6966
+ `never done for you off a refusal.${pinNote}`
6967
+ : `\n\nCHECKED, and the live canvas reports ${live}. Which side is ` +
6968
+ `stale is NOT known from here: this panel's refusal states neither ` +
6969
+ `that the command was unstamped nor which instance it was issued ` +
6970
+ `for, so no remedy is named for it — read panel_list_workflows (the ` +
6971
+ `panel exempts it from this fence) and decide which workflow you mean.`
6972
+ : `\n\nCHECKED, but the live canvas could not be established ` +
6973
+ `(${probe.status}), so the answer is UNKNOWN and the fence is unchanged. ` +
6974
+ `Try panel_list_workflows — the panel exempts that read from this fence ` +
6975
+ `(it is the recovery probe) — and re-select the workflow you mean with ` +
6976
+ `panel_open_workflow.`;
6977
+ }
6978
+ catch (probeErr) {
6979
+ // A diagnosis must never change how the call failed.
6980
+ verdict =
6981
+ `\n\nCHECKED, and the check itself threw, so the live canvas is UNKNOWN — this ` +
6982
+ `refusal stands on its own terms and the fence is unchanged. ` +
6983
+ `(${probeErr instanceof Error ? probeErr.message : String(probeErr)})`;
6984
+ }
6985
+ return fail(`${name} was refused before it ran — no graph data was read. ${raw}${verdict}`);
6986
+ }
6203
6987
  // #1480 — NAME A REMEDY THE TAB CAN ACTUALLY ACCEPT.
6204
6988
  //
6205
6989
  // The panel's own remedy for this verdict is `panel_open_workflow(<path>)`, which
@@ -6299,9 +7083,11 @@ export function makePanelToolCtx(bridge, tabId, workflowTargets) {
6299
7083
  RETRY_TOKEN_CMDS.has(typeof cmd.cmd === "string" ? cmd.cmd : "") &&
6300
7084
  (dispatchOutcomeOf(err) === true || isReplyTimeoutTagged(err))) {
6301
7085
  const cause = err instanceof Error ? err.message : String(err);
6302
- return carryReplyTimeoutMark(err, fail(`${cause}\n\nTo retry this exact mutation, re-issue identical args plus retry_of:"${dispatchedRid}"; otherwise call normally.`));
7086
+ return carryReplyTimeoutMark(err, fail(`${cause}\n\nTo retry this exact mutation, re-issue identical args plus retry_of:"${dispatchedRid}"; otherwise call normally.` +
7087
+ queueBusyTimeoutNote()));
6303
7088
  }
6304
- return carryReplyTimeoutMark(err, fail(err));
7089
+ const timeoutCause = err instanceof Error ? err.message : String(err);
7090
+ return carryReplyTimeoutMark(err, fail(`${timeoutCause}${queueBusyTimeoutNote()}`));
6305
7091
  }
6306
7092
  };
6307
7093
  /**
@@ -8465,7 +9251,65 @@ export function buildPanelToolDefs() {
8465
9251
  // a single revalidation, #338/#458) — that authoritative fetch can outlast
8466
9252
  // the 6000 ms default ack on a large install and return a FALSE timeout.
8467
9253
  // Give the guarded write the bounded refresh ack budget.
8468
- return ctx.call({ cmd: "graph_set_widget", node_id: args.node_id, widget: args.widget, value }, OBJECT_INFO_REFRESH_ACK_TIMEOUT_MS);
9254
+ const write = (nodeId, widget) => ctx.call({ cmd: "graph_set_widget", node_id: nodeId, widget, value }, OBJECT_INFO_REFRESH_ACK_TIMEOUT_MS);
9255
+ const first = await write(args.node_id, args.widget);
9256
+ if (!first.isError)
9257
+ return first;
9258
+ // #1655 — the panel listed this widget as promoted while refusing it as
9259
+ // not promoted. The listing is node.widgets; the write looks up host
9260
+ // inputs. When those disagree, resolve the displayed name to the unique
9261
+ // inner mapping and set it there (the issue's own workaround), then
9262
+ // leave the subgraph so the caller's scope is unchanged.
9263
+ const refusal = parseContradictoryPromotedWidgetRefusal(textOfToolResult(first), args.widget);
9264
+ if (!refusal || String(refusal.nodeId) !== String(args.node_id))
9265
+ return first;
9266
+ if (refusal.widget !== args.widget) {
9267
+ const remapped = await write(args.node_id, refusal.widget);
9268
+ if (!remapped.isError)
9269
+ return remapped;
9270
+ if (!parseContradictoryPromotedWidgetRefusal(textOfToolResult(remapped), refusal.widget)) {
9271
+ return remapped;
9272
+ }
9273
+ }
9274
+ const sub = await ctx.call({ cmd: "graph_get_subgraph", node_id: args.node_id });
9275
+ if (sub.isError) {
9276
+ return appendToolResultText(first, `\n\n(The panel listed "${refusal.widget}" as promoted while refusing it. ` +
9277
+ `Tried to resolve that name to the inner widget and graph_get_subgraph FAILED: ` +
9278
+ `${textOfToolResult(sub)})`);
9279
+ }
9280
+ const inner = resolveInnerPromotedTarget(parseToolResultJson(sub), refusal.widget);
9281
+ if (!inner) {
9282
+ return appendToolResultText(first, `\n\n(The panel listed "${refusal.widget}" as promoted while refusing it. ` +
9283
+ `graph_get_subgraph did not uniquely identify an inner node that owns that widget, ` +
9284
+ `so the write was not retried — guessing among several inners, or acting on a ` +
9285
+ `truncated inner list, would target the wrong node.)`);
9286
+ }
9287
+ const entered = await ctx.call({ cmd: "graph_enter_subgraph", node_id: args.node_id }, 15000);
9288
+ if (entered.isError) {
9289
+ return appendToolResultText(first, `\n\n(The panel listed "${refusal.widget}" as promoted while refusing it. ` +
9290
+ `Resolved it to inner node ${inner.innerNodeId} but panel_enter_subgraph FAILED: ` +
9291
+ `${textOfToolResult(entered)})`);
9292
+ }
9293
+ const written = await write(inner.innerNodeId, inner.widget);
9294
+ const exited = await ctx.call({ cmd: "graph_exit_subgraph" }, 15000);
9295
+ if (!written.isError) {
9296
+ const via = `\n\n(Applied via the inner widget this promotion lists: node ${inner.innerNodeId} ` +
9297
+ `"${inner.widget}". The panel listed "${refusal.widget}" as promoted while refusing ` +
9298
+ `it as not promoted; the displayed name was resolved to that inner mapping.)`;
9299
+ if (exited.isError) {
9300
+ return appendToolResultText(written, `${via} panel_exit_subgraph then FAILED — the canvas may still be inside the ` +
9301
+ `subgraph. Call panel_exit_subgraph. (${textOfToolResult(exited)})`);
9302
+ }
9303
+ return appendToolResultText(written, via);
9304
+ }
9305
+ if (exited.isError) {
9306
+ return appendToolResultText(first, `\n\n(Tried the inner mapping node ${inner.innerNodeId} "${inner.widget}" and it ` +
9307
+ `FAILED: ${textOfToolResult(written)} panel_exit_subgraph also FAILED — the ` +
9308
+ `canvas may still be inside the subgraph. Call panel_exit_subgraph. ` +
9309
+ `(${textOfToolResult(exited)}))`);
9310
+ }
9311
+ return appendToolResultText(first, `\n\n(Tried the inner mapping node ${inner.innerNodeId} "${inner.widget}" and it ` +
9312
+ `FAILED: ${textOfToolResult(written)})`);
8469
9313
  }),
8470
9314
  def("panel_remove_widget", "Remove ONE dynamic widget row from a node — the rows custom nodes add themselves, like the rgthree Power Lora Loader's `lora_1`, `lora_2`, … or an Impact/Inspire list node's entries. Their add/remove affordance is a canvas-drawn button you cannot click, so this is the only way to delete a row; panel_set_widget can only overwrite a row's value, and panel_remove_node deletes the whole node. REFUSES, with the reason, when the widget is an input the BACKEND declares (removing it would change what is sent at queue time — set it with panel_set_widget instead), when it is a frontend-generated control widget (control_after_generate, which the frontend re-creates), when its input slot currently has a link (disconnect it first), and when the node definitions cannot be read at all — an unreadable definition is reported as unknown, never treated as 'not declared'. The remaining rows are deliberately NOT renumbered: `lora_N` is a monotonic id, not a position, and the backend matches rows by name prefix, so gaps are harmless — the reply lists the remaining widget names, and those are the names to use next. Undoable with Ctrl+Z.", {
8471
9315
  node_id: nodeId().describe("Node id from panel_graph_outline / panel_query_graph."),
@@ -8596,7 +9440,7 @@ export function buildPanelToolDefs() {
8596
9440
  : "") +
8597
9441
  `Any values in this result are the canvas's actual state.`);
8598
9442
  }),
8599
- def("panel_run", "Queue the workflow the user has OPEN — exactly like them pressing Queue Prompt (current widget values, the live graph they can see). On success it confirms the run was queued; if ComfyUI REFUSES the prompt (validation failure on either channel — per-node node_errors OR a top-level error like a missing node type) it returns a FAILURE with that rejection detail, never a false 'queued'. Pass to_node_id to RUN ONLY ONE BRANCH ('run to node'): ComfyUI renders just that output node plus everything upstream of it and SKIPS every other output branch — handy for previewing or debugging part of a big graph without rendering the whole thing. to_node_id MUST be an OUTPUT node (SaveImage, PreviewImage, SaveVideo, …) — pick the one at the END of the branch you want; nodes are tagged is_output:true in panel_query_graph's detail rows. The output node may be NESTED inside a subgraph — just pass its id (resolved in the scope you're currently viewing, then anywhere in the workflow); the tool builds the nested execution path for you. Omit it to run the whole graph. DUPLICATE FENCE (#862): if a render this session cannot account for is already in flight (after a reconnect this is usually YOUR earlier render still running — the queue record does not survive a restart), the run is REFUSED before anything is queued and the in-flight prompt is named; inspect queue (action:'list') first, or pass allow_duplicate:true only to deliberately stack behind it. Use this so the render runs on THEIR canvas and they see the result.", {
9443
+ def("panel_run", "Queue the workflow the user has OPEN — exactly like them pressing Queue Prompt (current widget values, the live graph they can see). On success it confirms the run was queued; if ComfyUI REFUSES the prompt (validation failure on either channel — per-node node_errors OR a top-level error like a missing node type) it returns a FAILURE with that rejection detail, never a false 'queued'. Pass to_node_id to RUN ONLY ONE BRANCH ('run to node'): ComfyUI renders just that output node plus everything upstream of it and SKIPS every other output branch — handy for previewing or debugging part of a big graph without rendering the whole thing. to_node_id MUST be an OUTPUT node (SaveImage, PreviewImage, SaveVideo, …) — pick the one at the END of the branch you want; nodes are tagged is_output:true in panel_query_graph's detail rows. The output node may be NESTED inside a subgraph — just pass its id (resolved in the scope you're currently viewing, then anywhere in the workflow); the tool builds the nested execution path for you. Omit it to run the whole graph. DUPLICATE FENCE (#862): if a render this session cannot account for is already in flight (after a reconnect this is usually YOUR earlier render still running — the queue record does not survive a restart), the run is REFUSED before anything is queued and the in-flight prompt is named; inspect queue (action:'list') first, then pass allow_duplicate:true once you have decided it is fine to run behind what is there — a scoped to_node_id preview after a reconnect is the ordinary case for it, a deliberate sweep/batch the other. Use this so the render runs on THEIR canvas and they see the result.", {
8600
9444
  batch_count: z
8601
9445
  .number()
8602
9446
  .int()
@@ -8612,7 +9456,7 @@ export function buildPanelToolDefs() {
8612
9456
  allow_duplicate: z
8613
9457
  .boolean()
8614
9458
  .optional()
8615
- .describe("Queue even when a render this session cannot account for is already in flight (default false). When work is in flight that this session has no record of queueing — e.g. YOUR OWN earlier render still running after a reconnect, whose record does not survive the restart — panel_run REFUSES to stack a duplicate and names the in-flight prompt instead. Pass true only to deliberately queue behind it (a sweep/batch)."),
9459
+ .describe("Queue even when a render this session cannot account for is already in flight (default false). When work is in flight that this session has no record of queueing — e.g. YOUR OWN earlier render still running after a reconnect, whose record does not survive the restart — panel_run REFUSES to stack a duplicate and names the in-flight prompt instead. Pass true once you have LOOKED at what is in flight (queue action:'list') and decided it is fine to run behind it. After a reconnect that is the ordinary case, not an exotic one: you confirmed the in-flight job is your own earlier render or the user's, and you still want the next run — a scoped to_node_id preview, the next step of the task. Deliberately stacking a sweep/batch uses the same override."),
8616
9460
  }, async (args, ctx) => {
8617
9461
  // BACKPRESSURE: the agent can't see ComfyUI's queue, so re-queuing while a
8618
9462
  // render is already running silently stacks behind it (this is how a stuck
@@ -8678,9 +9522,12 @@ export function buildPanelToolDefs() {
8678
9522
  `no prompt id) even YOUR OWN earlier render reads as unconfirmable, and queueing now ` +
8679
9523
  `would stack a DUPLICATE behind it (#862). Nothing was queued. Inspect with queue ` +
8680
9524
  `(action:"list"): if the in-flight job is the render you already started, wait for it ` +
8681
- `and confirm the outcome with get_history instead of re-running it. If you genuinely ` +
8682
- `intend to stack another render behind it (a deliberate sweep/batch), re-call panel_run ` +
8683
- `with allow_duplicate:true. If the in-flight job is actually wedged, queue ` +
9525
+ `and confirm the outcome with get_history instead of re-running it. Once you HAVE ` +
9526
+ `looked and decided it is fine to run behind what is there, re-call panel_run with ` +
9527
+ `allow_duplicate:true — after a reconnect that is the ORDINARY case, not an exotic ` +
9528
+ `one: the in-flight job is your own earlier render or the user's, and you still want ` +
9529
+ `the next run (a scoped to_node_id preview, the next step of the task). Deliberately ` +
9530
+ `stacking a sweep/batch uses the same override. If the in-flight job is actually wedged, queue ` +
8684
9531
  `(action:"cancel") with clear_pending:true interrupts it AND drops everything pending.`);
8685
9532
  }
8686
9533
  const runCmd = { cmd: "graph_run", batch_count: args.batch_count, to_node_id: args.to_node_id };
@@ -9836,66 +10683,97 @@ export function buildPanelToolDefs() {
9836
10683
  let rebindNote = "";
9837
10684
  let deferredBind = false;
9838
10685
  let fenceRebind;
10686
+ // panel#1292 — a scope ctx stays scope-bound, so ctx.tabId never changes
10687
+ // on a successful turn-pin recovery. Track that separately from the
10688
+ // real-tab rebind note below.
10689
+ let currentModeTurnRepinned = false;
9839
10690
  if (mode === "current" && ctx.rebindToActiveTab) {
9840
10691
  const before = ctx.tabId;
9841
- // Give an in-flight reconnect (a ComfyUI restart / panel reload still
9842
- // settling) a brief chance to bind immediately, since this IS the recovery
9843
- // signal the agent reaches for in exactly that window (#474). awaitReachable
9844
- // rebinds via ensureReachable when a tab is (re)connected.
9845
- if (ctx.awaitReachable)
9846
- await ctx.awaitReachable();
9847
- try {
9848
- // completes the rebind if awaitReachable didn't. mode:"current" is
9849
- // THE explicit scope-recovery consent (#884 gate 3) — the only
9850
- // caller that may escape a DEAD scope pin (a healthy pin still
9851
- // stays put; see rebindToActiveTab's double gate).
9852
- const rebind = ctx.rebindToActiveTab({ scopeRecoveryConsent: true });
9853
- // #1077 Finding 2 — a scope repin that declined now says WHY, and
9854
- // this is where the user reads it. The refusal used to be a bare
9855
- // boolean, so a session stuck in the one state that repeats forever
9856
- // (the active tab belongs to another backend's conversation while
9857
- // this one has several eligible tabs) saw no difference from a
9858
- // healthy pin being correctly left alone.
9859
- if (rebind?.repinRefusal) {
9860
- rebindNote += ` The session pin was NOT moved: ${rebind.repinRefusal}.`;
10692
+ const recoveringScope = isScopeAddress(before);
10693
+ // Hold the send() wait BEFORE the first await so a same-batch sibling
10694
+ // that already hit the null pin waits instead of minting #884.
10695
+ if (recoveringScope)
10696
+ ctx.bridge.beginScopeRecovery?.(before);
10697
+ const tryRebind = (deferIfNoTabs) => {
10698
+ try {
10699
+ // mode:"current" is THE explicit scope-recovery consent (#884 gate 3)
10700
+ // — the only caller that may escape a DEAD scope pin (a healthy pin
10701
+ // still stays put; see rebindToActiveTab's double gate).
10702
+ const rebind = ctx.rebindToActiveTab({ scopeRecoveryConsent: true });
10703
+ // #1077 Finding 2 — a scope repin that declined now says WHY.
10704
+ if (rebind?.repinRefusal && !/pin was NOT moved/.test(rebindNote)) {
10705
+ rebindNote += ` The session pin was NOT moved: ${rebind.repinRefusal}.`;
10706
+ }
10707
+ if (rebind?.rebound)
10708
+ currentModeTurnRepinned = true;
9861
10709
  }
9862
- }
9863
- catch (err) {
9864
- // #474: with 2+ live tabs the rebind is AMBIGUOUS — fail so the user picks.
9865
- // But with ZERO tabs connected (the "Connected: none" window right after a
9866
- // restart/reload where the old tmp: tab is gone) the recovery call must NOT
9867
- // hard-fail: clear the stale binding and record the current-mode intent so
9868
- // the session binds onto the tab the moment one reconnects, instead of
9869
- // stranding the agent with no way to recover.
9870
- const live = typeof ctx.bridge.tabs === "function" ? ctx.bridge.tabs() : undefined;
9871
- let noTabsConnected;
9872
- if (Array.isArray(live)) {
9873
- // Count only INTERACTIVE (canvas-owning) tabs: a headless-only reconnect is
9874
- // NOT a usable graph binding, so it defers (binds once a real canvas tab
9875
- // connects) rather than failing as if a tab were pickable. Call isHeadless
9876
- // THROUGH the bridge (it reads `this.conns`) — a detached reference would
9877
- // lose `this` and throw "reading 'conns'" (the same #478 unbound-method bug).
9878
- const isHeadlessTab = (id) => typeof ctx.bridge.isHeadless === "function" && ctx.bridge.isHeadless(id);
9879
- const interactive = live.filter((t) => !isHeadlessTab(t.tab_id));
9880
- noTabsConnected = interactive.length === 0;
10710
+ catch (err) {
10711
+ // #474: with 2+ live tabs the rebind is AMBIGUOUS — fail so the user picks.
10712
+ // But with ZERO tabs connected (the "Connected: none" window right after a
10713
+ // restart/reload where the old tmp: tab is gone) the recovery call must NOT
10714
+ // hard-fail: clear the stale binding and record the current-mode intent so
10715
+ // the session binds onto the tab the moment one reconnects, instead of
10716
+ // stranding the agent with no way to recover.
10717
+ const live = typeof ctx.bridge.tabs === "function" ? ctx.bridge.tabs() : undefined;
10718
+ let noTabsConnected;
10719
+ if (Array.isArray(live)) {
10720
+ // Count only INTERACTIVE (canvas-owning) tabs: a headless-only reconnect is
10721
+ // NOT a usable graph binding, so it defers (binds once a real canvas tab
10722
+ // connects) rather than failing as if a tab were pickable. Call isHeadless
10723
+ // THROUGH the bridge (it reads `this.conns`) — a detached reference would
10724
+ // lose `this` and throw "reading 'conns'" (the same #478 unbound-method bug).
10725
+ const isHeadlessTab = (id) => typeof ctx.bridge.isHeadless === "function" && ctx.bridge.isHeadless(id);
10726
+ const interactive = live.filter((t) => !isHeadlessTab(t.tab_id));
10727
+ noTabsConnected = interactive.length === 0;
10728
+ }
10729
+ else {
10730
+ // No tab enumeration — classify by the resolve error: only "nothing
10731
+ // connected" defers; an AMBIGUOUS multi-tab error must still fail so the
10732
+ // user picks (never silently defer a routable-but-ambiguous session).
10733
+ const msg = err instanceof Error ? err.message : String(err ?? "");
10734
+ noTabsConnected =
10735
+ /no panel connected|not reachable|connected:\s*none|no connected tab/i.test(msg) &&
10736
+ !/multiple|last active|pass tab_id/i.test(msg);
10737
+ }
10738
+ if (!noTabsConnected)
10739
+ return fail(ambiguousRebindGuidance(ctx, err));
10740
+ // The first pass is BEFORE awaitReachable. Deferring here skipped the
10741
+ // wait, so a tab that reconnects mid-call (#474) was never adopted.
10742
+ if (!deferIfNoTabs)
10743
+ return undefined;
10744
+ deferredBind = true;
10745
+ rebindNote =
10746
+ " No panel tab is connected yet — cleared the stale binding; this session will " +
10747
+ "follow (bind onto) the tab as soon as one reconnects. Retry your graph tool in a " +
10748
+ "moment; if nothing reconnects shortly, ask the user to refresh (reload) the ComfyUI " +
10749
+ "browser tab, which reconnects the Agent panel after a restart (not an install issue).";
9881
10750
  }
9882
- else {
9883
- // No tab enumeration — classify by the resolve error: only "nothing
9884
- // connected" defers; an AMBIGUOUS multi-tab error must still fail so the
9885
- // user picks (never silently defer a routable-but-ambiguous session).
9886
- const msg = err instanceof Error ? err.message : String(err ?? "");
9887
- noTabsConnected =
9888
- /no panel connected|not reachable|connected:\s*none|no connected tab/i.test(msg) &&
9889
- !/multiple|last active|pass tab_id/i.test(msg);
10751
+ return undefined;
10752
+ };
10753
+ try {
10754
+ // panel#1292 hole 1 — recover the turn pin SYNCHRONOUSLY, before
10755
+ // awaitReachable yields to same-batch siblings.
10756
+ const failed = tryRebind(false);
10757
+ if (failed)
10758
+ return failed;
10759
+ // Give an in-flight reconnect (a ComfyUI restart / panel reload still
10760
+ // settling) a brief chance to bind immediately, since this IS the recovery
10761
+ // signal the agent reaches for in exactly that window (#474). awaitReachable
10762
+ // rebinds via ensureReachable when a tab is (re)connected.
10763
+ if (ctx.awaitReachable)
10764
+ await ctx.awaitReachable();
10765
+ // A first attempt that found no canvas (or a dead pin that is still
10766
+ // dead after the wait) gets one more recovery now that a tab may exist.
10767
+ const pinStillDead = typeof ctx.bridge.canReach === "function" && !ctx.bridge.canReach(ctx.tabId);
10768
+ if (!currentModeTurnRepinned && pinStillDead) {
10769
+ const failed2 = tryRebind(true);
10770
+ if (failed2)
10771
+ return failed2;
9890
10772
  }
9891
- if (!noTabsConnected)
9892
- return fail(ambiguousRebindGuidance(ctx, err));
9893
- deferredBind = true;
9894
- rebindNote =
9895
- " No panel tab is connected yet — cleared the stale binding; this session will " +
9896
- "follow (bind onto) the tab as soon as one reconnects. Retry your graph tool in a " +
9897
- "moment; if nothing reconnects shortly, ask the user to refresh (reload) the ComfyUI " +
9898
- "browser tab, which reconnects the Agent panel after a restart (not an install issue).";
10773
+ }
10774
+ finally {
10775
+ if (recoveringScope)
10776
+ ctx.bridge.endScopeRecovery?.(before);
9899
10777
  }
9900
10778
  // Detect the rebind regardless of whether awaitReachable or rebindToActiveTab
9901
10779
  // performed it (either mutates ctx.tabId), so the note is never swallowed.
@@ -9907,6 +10785,12 @@ export function buildPanelToolDefs() {
9907
10785
  // was whether the retarget did anything, it read as a no-op.
9908
10786
  rebindNote = ` Rebound this session from tab ${shortTabId(before)} onto the active tab ${shortTabId(ctx.tabId)}.`;
9909
10787
  }
10788
+ if (currentModeTurnRepinned) {
10789
+ rebindNote +=
10790
+ ` This session's turn routing was AMBIGUOUS (a reconnect delivered messages from ` +
10791
+ `several workflows at once) and is now pinned to the active tab, so graph tools ` +
10792
+ `will resolve deterministically.`;
10793
+ }
9910
10794
  }
9911
10795
  // PIN: bind to the EXACT open-workflow identity from the authoritative
9912
10796
  // workflow_list, canonicalizing to its stable `key`, FAILING CLOSED when the
@@ -10104,6 +10988,20 @@ export function buildPanelToolDefs() {
10104
10988
  const fence = fenceRebind
10105
10989
  ? describeFenceRebind(fenceRebind, canMutateNow, refusalCause)
10106
10990
  : undefined;
10991
+ // panel#1292 hole 2 — `graph_binding:"bound"` is a fence verdict, not a
10992
+ // statement that the turn-origin pin was recovered. A null pin still
10993
+ // mints the #884 refusal on the next scope-addressed graph call.
10994
+ const turnPinStillAmbiguous = () => isScopeAddress(ctx.tabId) &&
10995
+ typeof ctx.bridge.resolveFailure === "function" &&
10996
+ ctx.bridge.resolveFailure(ctx.tabId) === "ambiguous";
10997
+ const refuseBoundWhileAmbiguous = () => fail(`panel_set_workflow_target({mode:"current"}) did NOT restore this session's turn ` +
10998
+ `routing.\n\nAPPLIED (do not repeat this part): the workflow target is now ` +
10999
+ `mode:"current"${rebindNote ? `.${rebindNote}` : "."}\n\nNOT APPLIED: the ` +
11000
+ `workflow-instance fence could be described as bound, but the turn-origin pin ` +
11001
+ `is still ambiguous, so the next graph call would fail with "issued from ` +
11002
+ `multiple workflows at once". Name a workflow with ` +
11003
+ `panel_set_workflow_target({mode:"pinned", path:...}) or wait for the next ` +
11004
+ `single-origin message.`);
10107
11005
  // #1473 — TAKE THE ADVICE THIS MESSAGE GIVES, instead of assigning it as homework.
10108
11006
  //
10109
11007
  // The reporter restarted ComfyUI, called this, was told the binding was NOT
@@ -10208,10 +11106,13 @@ CHECKED FOR YOU: the graph read this message prescribes was just run, and it ` +
10208
11106
  const heldUuid = fenceRebind && fenceRebind.status === "no_identity" && fenceRebind.before.known
10209
11107
  ? fenceRebind.before.uuid
10210
11108
  : undefined;
11109
+ if (turnPinStillAmbiguous())
11110
+ return refuseBoundWhileAmbiguous();
10211
11111
  return ok({
10212
11112
  ...target,
10213
11113
  graph_binding: "bound",
10214
11114
  ...(fenceRebind ? { graph_binding_status: fenceRebind.status } : {}),
11115
+ ...(currentModeTurnRepinned ? { turn_routing: "repinned" } : {}),
10215
11116
  note: hint +
10216
11117
  rebindNote +
10217
11118
  ` The graph binding was NOT re-derived (the panel's active reply could ` +
@@ -10242,12 +11143,17 @@ CHECKED FOR YOU: the graph read this message prescribes was just run, and it ` +
10242
11143
  : scopeRepin && typeof scopeRepin === "object" && scopeRepin.reason
10243
11144
  ? ` NOTE — the workflow target was set, but this session's turn routing was NOT re-pinned: ${scopeRepin.reason}.`
10244
11145
  : "";
11146
+ if (fence?.binding === "bound" && turnPinStillAmbiguous()) {
11147
+ return refuseBoundWhileAmbiguous();
11148
+ }
10245
11149
  return ok({
10246
11150
  ...target,
10247
11151
  ...(deferredBind ? { deferred: true } : {}),
10248
11152
  ...(fence ? { graph_binding: fence.binding } : {}),
10249
11153
  ...(fenceRebind ? { graph_binding_status: fenceRebind.status } : {}),
10250
- ...(typeof scopeRepin === "string" ? { turn_routing: "repinned" } : {}),
11154
+ ...(typeof scopeRepin === "string" || currentModeTurnRepinned
11155
+ ? { turn_routing: "repinned" }
11156
+ : {}),
10251
11157
  note: hint + rebindNote + (fence?.note ?? "") + scopeRepinNote,
10252
11158
  });
10253
11159
  }),
@@ -10493,7 +11399,19 @@ CHECKED FOR YOU: the graph read this message prescribes was just run, and it ` +
10493
11399
  name: args.name,
10494
11400
  }, 15000)),
10495
11401
  def("panel_unpack_subgraph", "EXPAND / DISSOLVE a subgraph node on the user's open graph — inline its interior nodes back into the PARENT graph, rewire all external links to those now-inlined nodes, and remove the subgraph wrapper. This is the frontend's \"Unpack Subgraph\" (litegraph LGraph.unpackSubgraph) and the exact INVERSE of panel_create_subgraph. Use it to flatten a stage that was over-nested, or to edit interior nodes directly at the parent level. The interior nodes reappear on the parent canvas with their connections preserved. Undoable with Ctrl+Z.", { node_id: nodeId().describe("Subgraph node id to unpack/dissolve (is_subgraph=true, from panel_graph_outline / panel_query_graph).") }, async (args, ctx) => ctx.call({ cmd: "graph_unpack_subgraph", node_id: args.node_id }, 15000)),
10496
- def("panel_search_nodes", "Search installable custom-node packs via the user's BUILT-IN ComfyUI Manager (the same source the Manager UI uses). Returns matching packs {id, title, description}. Use the `id` with panel_install_node. Prefer this over the headless search_custom_nodes tool — it works against the user's actual (Desktop) Manager.", { query: z.string().describe("Search text, e.g. 'kjnodes', 'controlnet', 'ipadapter'."), limit: z.number().int().min(1).max(40).optional().describe("Max results to return, 1-40 (default 15). Requests above 40 are refused; the panel also clamps to 40 and discloses limit_cap.") }, async (args, ctx) => ctx.call({ cmd: "nodes_search", query: args.query, limit: args.limit }, 20000)),
11402
+ def("panel_search_nodes", "Search installable custom-node packs via the user's BUILT-IN ComfyUI Manager (the same source the Manager UI uses). Returns matching packs {id, title, description}. Use the `id` with panel_install_node. Prefer this over the headless search_custom_nodes tool — it works against the user's actual (Desktop) Manager. If Manager's cache mappings endpoint returns HTTP 5xx, this retries remote/local and still searches; a remaining 5xx is a Manager outage, not proof the pack is missing.", { query: z.string().describe("Search text, e.g. 'kjnodes', 'controlnet', 'ipadapter'."), limit: z.number().int().min(1).max(40).optional().describe("Max results to return, 1-40 (default 15). Requests above 40 are refused; the panel also clamps to 40 and discloses limit_cap.") }, async (args, ctx) => {
11403
+ // #1669 — the panel asks getmappings?mode=cache and used to fail the
11404
+ // whole search on HTTP 500. Degrade: retry remote/local, or name the
11405
+ // 500 as a Manager outage (not a missing pack).
11406
+ const query = String(args.query ?? "");
11407
+ const limit = typeof args.limit === "number" ? args.limit : undefined;
11408
+ const out = await searchPanelNodes({
11409
+ panelSearch: () => ctx.call({ cmd: "nodes_search", query, limit }, 20000),
11410
+ query,
11411
+ limit,
11412
+ });
11413
+ return out.via === "panel" ? out.value : ok(out.value);
11414
+ }),
10497
11415
  def("panel_list_nodes", "List the custom-node packs currently installed in the user's ComfyUI (via the built-in Manager). Read-only.", {}, async (_args, ctx) => ctx.call({ cmd: "nodes_list" }, 20000)),
10498
11416
  def("panel_install_node", "Install a custom-node pack into the user's ComfyUI via the BUILT-IN Manager (queues the install). Pass `id` (registry id like 'comfyui-kjnodes' or 'author/repo') from panel_search_nodes, or `repository` (git URL) to request a nightly/from-source install — see the v4 limit below before relying on it. A search result whose `id` IS a git URL (legacy/repository-style entries) is auto-routed to a from-source 'nightly' install — 'latest' cannot resolve for those. " +
10499
11417
  "⚠️ ON MANAGER v4, `repository` IS NOT THE URL THAT GETS CLONED (#1539). Read out of ComfyUI-Manager V4.2.2's own source and confirmed on a live V4.2.2: a 'nightly' install resolves the pack by its BARE REPO NAME against the CHANNEL's custom-node-list, then clones the URL recorded in THAT entry; the `repository` you pass is stored in the task params and never read. So what decides success is whether the repo is listed in the channel this call asks for — and a miss does NOT simply stop: on 'nightly' v4 falls back to the COMFY REGISTRY entry whose id is that same bare name and clones whatever repository it is registered to, so an unlisted name can still install someone else's code. Only when the registry lacks the id too do you get \"Node '<name>@nightly' not found in [ManagerChannel.<channel>, ManagerDatabaseSource.<mode>]\", naming that channel. " +
@@ -10559,7 +11477,7 @@ CHECKED FOR YOU: the graph read this message prescribes was just run, and it ` +
10559
11477
  return ctx.call({ cmd: "graph_update_node", id: args.id, version: args.version, channel: args.channel, mode: args.mode }, 30000);
10560
11478
  }),
10561
11479
  def("panel_node_queue_status", "Check the built-in Manager's install/update queue status (to see if a queued install finished). Read-only.", {}, async (_args, ctx) => ctx.call({ cmd: "nodes_queue_status" }, 20000)),
10562
- def("panel_restart_comfyui", "Restart the user's ComfyUI server via the built-in Manager — needed to load newly installed/updated custom nodes. CALL THIS DIRECTLY when a restart is needed: it pops a confirm card and only restarts on a yes (don't ask separately first). ComfyUI and this agent go down briefly, then the panel auto-reconnects and you resume. ⚠️ BUSY GUARD: a restart ABORTS any in-progress or queued generation — if ComfyUI is generating, this tool REFUSES and tells you (it does NOT restart). When that happens, tell the user a render is running and WAIT for it (poll panel_node_queue_status), or pass force:true ONLY if the user explicitly confirms they want to kill the running generation. Best practice: before restarting after an install, check the queue is idle first. Only call when a restart is actually needed. On an externally-managed install whose relaunch can't be proven from here (e.g. Pinokio), the restart is REFUSED before anything is stopped — restart from the launcher that owns the server instead.", { force: z.boolean().optional() }, async ({ force }, ctx) => {
11480
+ def("panel_restart_comfyui", "Restart the user's ComfyUI server via the built-in Manager — needed to load newly installed/updated custom nodes. CALL THIS DIRECTLY when a restart is needed: it pops a confirm card and only restarts on a yes (don't ask separately first). ComfyUI and this agent go down briefly, then the panel auto-reconnects and you resume. ⚠️ BUSY GUARD: a restart ABORTS any in-progress or queued generation — if ComfyUI is generating, this tool REFUSES and tells you (it does NOT restart). When that happens, tell the user a render is running and WAIT for it (poll panel_node_queue_status), or pass force:true ONLY if the user explicitly confirms they want to kill the running generation. Best practice: before restarting after an install, check the queue is idle first. Only call when a restart is actually needed. If a crash takes the panel bridge offline so the confirmation card cannot be shown, this tool falls back to a headless restart of the configured local process (or COMFYUI_RESTART_COMMAND) instead of depending on the dead bridge — it still refuses a readable busy queue without force:true, and still refuses when a relaunch cannot be proven. On an externally-managed install whose relaunch can't be proven from here (e.g. Pinokio), the restart is REFUSED before anything is stopped — restart from the launcher that owns the server instead, or set COMFYUI_RESTART_COMMAND to the exact command that restarts the instance (e.g. `docker restart <container>`): the restart then runs through that command (the busy guard above still applies) instead of needing the launch path.", { force: z.boolean().optional() }, async ({ force }, ctx) => {
10563
11481
  // Whole-handler budget (#536): confirm + dispatch + readiness — INCLUDING
10564
11482
  // the legacy path's UNPREEMPTIBLE synchronous execSync blocks — must ALL finish
10565
11483
  // under the outer ~300s tools/call limit. 255s + the legacy admission rule below
@@ -10612,6 +11530,12 @@ CHECKED FOR YOU: the graph read this message prescribes was just run, and it ` +
10612
11530
  "previous restart, so the confirmation card wasn't answered. Tell me to restart it " +
10613
11531
  `and I'll re-ask. ${fallback}`);
10614
11532
  }
11533
+ // #1671 — set when the confirmation card was UNREACHABLE and ComfyUI is
11534
+ // not healthy: the crash took the panel bridge offline, so recovery
11535
+ // must not depend on asking that bridge. The headless path below runs
11536
+ // instead. An explicit decline, a still-healthy server, and remote/
11537
+ // cloud keep the existing reports (confirmation + busy-queue stay).
11538
+ let recoverWithoutPanel = false;
10615
11539
  if (decision !== "yes") {
10616
11540
  // #742: NEVER claim "not restarted" while the server is actually DOWN —
10617
11541
  // and NEVER declare a loss from ONE probe (codex gate): a genuinely
@@ -10671,7 +11595,20 @@ CHECKED FOR YOU: the graph read this message prescribes was just run, and it ` +
10671
11595
  clearSessionRestartDispatchIfSame(ctx, declineHeldToken);
10672
11596
  }
10673
11597
  }
10674
- if (outcome.status === "down") {
11598
+ // #1671: the recovery command must not depend on the component a crash
11599
+ // takes offline. UNREACHABLE + not-healthy (down, or ambiguous — the
11600
+ // reporter's empty-body HTTP 502) on a LOCAL target falls through to
11601
+ // the headless restart. An explicit decline still reports and does
11602
+ // NOT restart (confirmation). A still-healthy server still does not
11603
+ // auto-restart (#1332). Remote/cloud have no local process to cycle.
11604
+ const crashTookBridgeOffline = decision === "unreachable" &&
11605
+ (outcome.status === "down" || outcome.status === "ambiguous") &&
11606
+ !isRemoteMode() &&
11607
+ !isCloudMode();
11608
+ if (crashTookBridgeOffline) {
11609
+ recoverWithoutPanel = true;
11610
+ }
11611
+ else if (outcome.status === "down") {
10675
11612
  const secs = Math.max(1, Math.round(outcome.waited_ms / 1000));
10676
11613
  if (boundToRestartTarget) {
10677
11614
  // r4: causation may be named ONLY against a RECORDED restart
@@ -10712,7 +11649,12 @@ CHECKED FOR YOU: the graph read this message prescribes was just run, and it ` +
10712
11649
  "restart would have cycled. Check ComfyUI on its host and start it " +
10713
11650
  "manually if it is down, then reload the panel tab so it reconnects.");
10714
11651
  }
10715
- if (outcome.status === "recovered" && boundToRestartTarget) {
11652
+ if (recoverWithoutPanel) {
11653
+ // Fall through to the headless path once runHeadlessManagedRestart
11654
+ // is defined. Do not claim the server is reachable, and do not ask
11655
+ // the dead panel to reboot it.
11656
+ }
11657
+ else if (outcome.status === "recovered" && boundToRestartTarget) {
10716
11658
  // r14: the recovery CLAIM ("a restart initiated earlier appears to
10717
11659
  // have completed") passes the SAME causation gate as the DOWN
10718
11660
  // report — a session-held, bound-confirmed record, recent, and
@@ -10731,26 +11673,37 @@ CHECKED FOR YOU: the graph read this message prescribes was just run, and it ` +
10731
11673
  : "Cancelled — no new restart was dispatched. ComfyUI was briefly " +
10732
11674
  "unreachable but is healthy again.");
10733
11675
  }
10734
- // #1332 — the reporter's exact string, and it was FALSE. They accepted the
10735
- // restart, ComfyUI restarted (a fresh startup in the server log), and this
10736
- // said it had not — because the restart dropped the socket the answer had to
10737
- // travel back on, and a transport failure used to arrive here as "no".
10738
- //
10739
- // The probes above already refuse to claim "not restarted" while the server
10740
- // is DOWN. This is the remaining case: the server is HEALTHY, which is
10741
- // equally true of "nothing happened" and of "it restarted and came back".
10742
- // With an explicit decline we know which; without one we do not, and the
10743
- // sentence must stop asserting it.
10744
- return ok(decision === "unreachable"
10745
- ? "This call did NOT dispatch a restart. Whether ComfyUI restarted for some " +
10746
- "other reason cannot be told from here: the panel could not be reached to " +
10747
- "ask for confirmation — the question never appeared — so no decision was " +
10748
- "made either way, and the server is reachable now, which looks the same " +
10749
- "whether it never went down or went down and came back. If you asked for a " +
10750
- "restart and one has already happened, this is that transport loss, not a " +
10751
- "cancellation. Check the ComfyUI log for a fresh startup line before " +
10752
- "restarting again."
10753
- : "Cancelled — ComfyUI was not restarted.");
11676
+ else {
11677
+ // #1332 — the reporter's exact string, and it was FALSE. They accepted the
11678
+ // restart, ComfyUI restarted (a fresh startup in the server log), and this
11679
+ // said it had not — because the restart dropped the socket the answer had to
11680
+ // travel back on, and a transport failure used to arrive here as "no".
11681
+ //
11682
+ // The probes above already refuse to claim "not restarted" while the server
11683
+ // is DOWN. This is the remaining case: the server is HEALTHY, which is
11684
+ // equally true of "nothing happened" and of "it restarted and came back".
11685
+ // With an explicit decline we know which; without one we do not, and the
11686
+ // sentence must stop asserting it.
11687
+ const fallback = decision === "unreachable"
11688
+ ? " " +
11689
+ restartTimeoutFallbackAdvice({
11690
+ headlessBase: getComfyUIBaseUrl(),
11691
+ panelBase: declineBootBase,
11692
+ observedOrigin: ctx.bridge?.tabServerOrigin?.(ctx.tabId) ?? null,
11693
+ })
11694
+ : "";
11695
+ return ok(decision === "unreachable"
11696
+ ? "This call did NOT dispatch a restart. Whether ComfyUI restarted for some " +
11697
+ "other reason cannot be told from here: the panel could not be reached to " +
11698
+ "ask for confirmation — the question never appeared — so no decision was " +
11699
+ "made either way, and the server is reachable now, which looks the same " +
11700
+ "whether it never went down or went down and came back. If you asked for a " +
11701
+ "restart and one has already happened, this is that transport loss, not a " +
11702
+ "cancellation. Check the ComfyUI log for a fresh startup line before " +
11703
+ "restarting again." +
11704
+ fallback
11705
+ : "Cancelled — ComfyUI was not restarted.");
11706
+ }
10754
11707
  }
10755
11708
  // Heal an orphaned session onto the live tab FIRST, then bind the reboot dispatch
10756
11709
  // to that ONE tab id (no await between capture and dispatch, so JS run-to-
@@ -10758,6 +11711,366 @@ CHECKED FOR YOU: the graph read this message prescribes was just run, and it ` +
10758
11711
  // server-authorized + immutable, bound to the exact host FAMILY the reboot goes
10759
11712
  // to (null unless the bound tab provably fronts our boot instance).
10760
11713
  ctx.ensureReachable?.();
11714
+ // Run the HEADLESS managed restart (restartComfyUI) from inside this tool and
11715
+ // report its outcome against OUR OWN independent boot-endpoint observation
11716
+ // (never restartComfyUI's self-reported readiness, which a first-healthy
11717
+ // no-op would flunk). Shared by three call sites that must not dispatch the
11718
+ // tab reboot: the legacy no-endpoint fallback below (#425), a configured
11719
+ // COMFYUI_RESTART_COMMAND (panel#1262), and #1671 crash recovery when the
11720
+ // panel bridge is already gone. restartComfyUI acts on the orchestrator's
11721
+ // GLOBAL config target, so the first two require a BOUND-CONFIRMED local
11722
+ // tab; #1671 may also use the configured boot instance when the tab is gone.
11723
+ const runHeadlessManagedRestart = async (args) => {
11724
+ const { healthBase, preRestartPanelIdentity, why, mechanism, noteHealthyLead, noteRanLead } = args;
11725
+ // The managed kill+relaunch does UNPREEMPTIBLE synchronous execSync work — PID
11726
+ // discovery (~5+8s) + termination (~10s) + first port-free lookup (~13s) ≈ 40s
11727
+ // worst case (Windows) — that a Promise.race CANNOT interrupt, and it BLOCKS the
11728
+ // observer during that window. Admit it ONLY with enough budget for that sync
11729
+ // work AND a full cold-start observation AFTER it, and give the observer a
11730
+ // deadline that spans BOTH (coordinator P1: the proof deadline must start after,
11731
+ // not before, the restart's synchronous work — otherwise a genuine cold start
11732
+ // that finishes at sync+coldStart false-times-out).
11733
+ const LEGACY_SYNC_WORST_CASE_MS = 40_000; // execSync PID lookup + kill + port-free
11734
+ const LEGACY_COLD_START_OBS_MS = 100_000; // cold-start observation AFTER the sync
11735
+ const LEGACY_RESTART_MIN_BUDGET_MS = LEGACY_SYNC_WORST_CASE_MS + LEGACY_COLD_START_OBS_MS;
11736
+ if (overallDeadline - Date.now() < LEGACY_RESTART_MIN_BUDGET_MS) {
11737
+ return ok({
11738
+ rebooting: false,
11739
+ ready: false,
11740
+ confirmed_cycle: false,
11741
+ note: `${why}, and there isn't enough remaining time to safely run ${mechanism}. ` +
11742
+ "ComfyUI was NOT restarted — retry panel_restart_comfyui " +
11743
+ "(a fresh call gets the full budget).",
11744
+ });
11745
+ }
11746
+ // A managed kill+relaunch restarts ComfyUI out-of-band, so drop the memoized
11747
+ // caches. The observer watches the boot endpoint itself with a deadline spanning
11748
+ // the ~40s blocking sync + a full cold-start window, and certifies ONLY on an
11749
+ // OBSERVED down→up — a never-restarted healthy endpoint (a Desktop first-healthy
11750
+ // Manager-reboot / preflight no-op) is honestly couldn't-confirm (coordinator P1).
11751
+ resetClient();
11752
+ resetObjectInfoCache();
11753
+ resetManagerApiCache("panel managed restart");
11754
+ const headlessTiming = getPanelRebootTiming();
11755
+ // The observation window spans the ~40s blocking sync + a full cold-start
11756
+ // window. (Under a test timing override, use the injected budget instead so the
11757
+ // never-certify cases don't wait the real ~140s.)
11758
+ const legacyProofWindow = panelRebootTimingOverride
11759
+ ? headlessTiming.settleMs + headlessTiming.budgetMs
11760
+ : LEGACY_RESTART_MIN_BUDGET_MS;
11761
+ const proofDeadline = Math.min(Date.now() + legacyProofWindow, overallDeadline);
11762
+ const proofPromise = observeRecovery(headlessTiming, proofDeadline, { healthBase });
11763
+ const restartBudget = Math.max(1, overallDeadline - Date.now());
11764
+ let restart;
11765
+ let restartTimer;
11766
+ try {
11767
+ restart = await Promise.race([
11768
+ restartComfyUI(),
11769
+ new Promise((resolve) => {
11770
+ restartTimer = setTimeout(() => resolve(undefined), restartBudget);
11771
+ restartTimer.unref?.();
11772
+ }),
11773
+ ]);
11774
+ }
11775
+ catch (err) {
11776
+ clearTimeout(restartTimer);
11777
+ void proofPromise.catch(() => { }); // self-terminates at proofDeadline
11778
+ return fail(`${why}, and ${mechanism} also failed: ` +
11779
+ (err instanceof Error ? err.message : String(err)) +
11780
+ " — restart ComfyUI on the host, then reconnect.");
11781
+ }
11782
+ clearTimeout(restartTimer);
11783
+ // #742 r5/r6: the managed restart stopped the process — record the
11784
+ // dispatch with THIS session holding the token, stamped with the
11785
+ // BOUND-CONFIRMED base (this path only runs when the instance
11786
+ // binding held, so healthBase is non-null here). restartComfyUI
11787
+ // also stamped its own process-wide record, which never grounds
11788
+ // causation. Only a PROVEN stop is recorded; a refusal/timeout
11789
+ // (restart undefined, or stopped!==true) records nothing. The
11790
+ // token is kept so the recovery clear below is CLEAR-IF-SAME (r15).
11791
+ let headlessDispatchToken;
11792
+ if (restart?.stopped === true) {
11793
+ headlessDispatchToken = stampSessionRestartDispatch(ctx, healthBase);
11794
+ }
11795
+ // DEFINITIVE no-restart: a spawn failure, OR restartComfyUI refused before
11796
+ // stopping anything (no process found / unsafe relaunch → stopped:false &&
11797
+ // started:false). The process was NOT cycled, so the still-healthy endpoint is
11798
+ // the OLD one — fail clearly rather than certify a no-op (coordinator P1).
11799
+ if (restart?.spawn_error ||
11800
+ (restart != null && restart.stopped !== true && restart.started !== true)) {
11801
+ void proofPromise.catch(() => { });
11802
+ return fail(`${why}. Tried ${mechanism}, but it did not restart ` +
11803
+ `ComfyUI: ${restart?.message ?? "unknown error"} ` +
11804
+ "Restart ComfyUI on the host, then reconnect.");
11805
+ }
11806
+ // Otherwise (the process WAS stopped/started, or restartComfyUI's own readiness
11807
+ // poll merely expired — neither terminal) DEFER to OUR OWN observed DOWN→UP.
11808
+ const recovery = await proofPromise;
11809
+ // #742 r4/r5/r15: the managed restart was observed back — clear THIS
11810
+ // session's record, CLEAR-IF-SAME: only when the session still holds
11811
+ // the token THIS restart stamped (a concurrent dispatch's newer
11812
+ // record survives). restartComfyUI also clears its own process-wide
11813
+ // record on success; this covers only-observer-saw-it recoveries.
11814
+ if (recovery.ready && headlessDispatchToken != null) {
11815
+ clearSessionRestartDispatchIfSame(ctx, headlessDispatchToken);
11816
+ }
11817
+ const observed = recovery.via === "observed-cycle";
11818
+ // The headless path restarts ComfyUI out-of-band too. Server recovery alone
11819
+ // is not graph-tool readiness: wait for the browser tab to reconnect, then verify
11820
+ // the same workflow-stamp capability the bridge requires before it dispatches a
11821
+ // mutation. Without this, updating the panel pack followed by a headless restart can
11822
+ // falsely report ready while the browser is still running stale panel JS (#709).
11823
+ const tabBack = recovery.ready
11824
+ ? ctx.awaitPostRestartReachable
11825
+ ? await ctx.awaitPostRestartReachable(preRestartPanelIdentity, Math.max(0, overallDeadline - Date.now()))
11826
+ : ctx.awaitReachable
11827
+ ? await ctx.awaitReachable(Math.max(0, overallDeadline - Date.now()))
11828
+ : true
11829
+ : false;
11830
+ const graphToolsReady = tabBack && (ctx.tabCanMutateGraph ? ctx.tabCanMutateGraph() : true);
11831
+ // #654 — `ready`/`graph_tools_ready` still come from the BOOLEAN, so an
11832
+ // undetermined reconnect withholds graph tools exactly as before. Only
11833
+ // the reported observation changes.
11834
+ const tabReconnect = classifyTabReconnect({
11835
+ serverReady: recovery.ready,
11836
+ baselineCaptured: preRestartPanelIdentity != null,
11837
+ tabBack,
11838
+ });
11839
+ return ok({
11840
+ rebooting: true,
11841
+ ready: graphToolsReady,
11842
+ graph_tools_ready: graphToolsReady,
11843
+ server_ready: recovery.ready,
11844
+ panel_tab_reconnected: tabReconnect,
11845
+ confirmed_cycle: observed, // true = we directly observed the down→up cycle
11846
+ recovered_ms: recovery.waited_ms,
11847
+ probes: recovery.attempts,
11848
+ saw_down: recovery.sawDown,
11849
+ via: recovery.ready ? recovery.via : undefined,
11850
+ note: recovery.ready && !graphToolsReady
11851
+ ? `${noteHealthyLead} came back healthy in ${(recovery.waited_ms / 1000).toFixed(1)}s, but ` +
11852
+ (!tabBack
11853
+ ? tabReconnect === "unknown"
11854
+ ? "whether the panel tab reconnected could NOT be determined — no pre-restart " +
11855
+ "baseline was captured for it, so nothing was watched. WHY is not established " +
11856
+ "here, and it is one of: the tab's socket was not open at the instant the " +
11857
+ "restart was dispatched; the panel advertised no tab session id (an older " +
11858
+ "build, or its browser-tab lease was refused because a duplicate tab holds " +
11859
+ "it); or the tab did not resolve at all. The tab may well be back. Graph " +
11860
+ "tools are withheld (ready:false) because that is unproven, NOT because the " +
11861
+ 'tab is known to be gone: call panel_list_workflows or panel_set_workflow_target({mode:"current"}) ' +
11862
+ "to find out, and only refresh the browser if those also fail. If this " +
11863
+ "repeats on every restart, the panel is probably too old to advertise a tab " +
11864
+ "session id — update it."
11865
+ : "the panel tab has NOT reconnected yet (ready:false). Wait a moment then retry, or " +
11866
+ 'rebind with panel_set_workflow_target({mode:"current"}) before issuing graph tools.'
11867
+ : "the panel tab reconnected but cannot safely run graph mutations (ready:false), usually " +
11868
+ "because it is still running a stale panel bundle. Hard-refresh the ComfyUI browser tab " +
11869
+ "(Ctrl+Shift+R) before issuing graph tools; if that does not restore it, update the panel " +
11870
+ "and open/reload a saved workflow with a stable identity.")
11871
+ : `${noteRanLead} ` +
11872
+ (recovery.ready
11873
+ ? `and it came back healthy in ${(recovery.waited_ms / 1000).toFixed(1)}s` +
11874
+ (observed ? " (observed it go down then come back)." : " (cycle not directly observed).")
11875
+ : `but it did NOT become healthy within ${Math.round(recovery.waited_ms / 1000)}s — verify with get_system_stats (action:"health") / panel_node_queue_status before assuming it restarted.`),
11876
+ });
11877
+ };
11878
+ // #1671: panel-offline crash recovery. The confirmation card could not
11879
+ // be shown because the bridge is gone, and ComfyUI is not healthy
11880
+ // (down, or an empty-body 502). Restart the configured local process
11881
+ // through the same headless path the Manager-missing and
11882
+ // COMFYUI_RESTART_COMMAND cases already use. Do NOT send comfy_reboot
11883
+ // — that is the dead bridge. A readable busy queue still refuses
11884
+ // without force:true; an unreadable queue on an already-unhealthy
11885
+ // server is the crash, not "idle", and is not a reason to refuse.
11886
+ // A proven-different panel origin, or a process we cannot relaunch,
11887
+ // fails with the true cause instead of claiming success.
11888
+ if (recoverWithoutPanel) {
11889
+ const healthBase = offlineRestartHealthBase(ctx);
11890
+ if (healthBase == null || !sameHttpBase(getComfyUIBaseUrl(), healthBase)) {
11891
+ return ok({
11892
+ rebooting: false,
11893
+ ready: false,
11894
+ confirmed_cycle: false,
11895
+ refused: true,
11896
+ note: "This call did NOT dispatch a restart. The panel could not be reached to " +
11897
+ "ask for confirmation (the crash took the panel bridge offline) and I " +
11898
+ "cannot identify a local ComfyUI process I can account for, so I will " +
11899
+ "not stop a server I cannot prove I can bring back. Nothing was " +
11900
+ "stopped. " +
11901
+ restartRefusalHandoffAdvice({
11902
+ headlessBase: getComfyUIBaseUrl(),
11903
+ panelBase: captureRebootHealthBase(ctx),
11904
+ observedOrigin: ctx.bridge?.tabServerOrigin?.(ctx.tabId) ?? null,
11905
+ }),
11906
+ });
11907
+ }
11908
+ if (force !== true) {
11909
+ let busyCount = null;
11910
+ try {
11911
+ const queue = await getQueueVerified();
11912
+ busyCount = queue.queue_running.length + queue.queue_pending.length;
11913
+ }
11914
+ catch {
11915
+ // Unreadable queue + already-unhealthy server is the crash itself.
11916
+ // Unlike the configured-command YES path, "cannot check" is not a
11917
+ // reason to refuse: the generation that might have been running
11918
+ // is the one that took the bridge down.
11919
+ busyCount = null;
11920
+ }
11921
+ if (busyCount != null && busyCount !== 0) {
11922
+ return ok({
11923
+ rebooting: false,
11924
+ ready: false,
11925
+ confirmed_cycle: false,
11926
+ refused: true,
11927
+ note: `Refusing to restart ComfyUI: ${busyCount} generation(s) are in progress ` +
11928
+ "or queued, and a restart ABORTS them. The panel could not be reached " +
11929
+ "to ask, so nothing was stopped. Wait for the queue to drain, or retry " +
11930
+ "with force:true ONLY if the user explicitly confirms they want to kill " +
11931
+ "the running generation.",
11932
+ });
11933
+ }
11934
+ }
11935
+ const configuredCmd = config.comfyuiRestartCommand;
11936
+ if (!configuredCmd) {
11937
+ const preflight = await (localRestartPreflightOverride ?? preflightLocalRestart)();
11938
+ if (!preflight.ok) {
11939
+ return ok({
11940
+ rebooting: false,
11941
+ ready: false,
11942
+ confirmed_cycle: false,
11943
+ refused: true,
11944
+ note: "The panel could not be reached to ask for confirmation (the crash " +
11945
+ "took the panel bridge offline). Refusing to restart ComfyUI: " +
11946
+ `${preflight.reason} A restart from here would STOP ComfyUI and ` +
11947
+ "nothing would bring it back automatically, so it was refused " +
11948
+ "BEFORE anything was stopped. Restart it from whatever launches it, " +
11949
+ "or set COMFYUI_RESTART_COMMAND to the exact command that restarts " +
11950
+ "the instance.",
11951
+ });
11952
+ }
11953
+ }
11954
+ const postHealthBase = offlineRestartHealthBase(ctx);
11955
+ if (postHealthBase == null || !sameHttpBase(healthBase, postHealthBase)) {
11956
+ return ok({
11957
+ rebooting: false,
11958
+ ready: false,
11959
+ confirmed_cycle: false,
11960
+ refused: true,
11961
+ note: "Refusing to restart ComfyUI: the ComfyUI target changed while the " +
11962
+ "offline recovery was being prepared, so I can no longer confirm the " +
11963
+ "headless restart would act on the instance this session accounts for. " +
11964
+ "Nothing was stopped.",
11965
+ });
11966
+ }
11967
+ return runHeadlessManagedRestart({
11968
+ healthBase,
11969
+ preRestartPanelIdentity: ctx.panelConnectionIdentity?.(),
11970
+ why: "The panel could not be reached to ask for confirmation (the crash took " +
11971
+ "the panel bridge offline)",
11972
+ mechanism: configuredCmd
11973
+ ? "the configured restart command"
11974
+ : "the headless managed restart (kill + relaunch)",
11975
+ noteHealthyLead: configuredCmd
11976
+ ? "The panel bridge was offline after the crash, so the restart ran " +
11977
+ "through COMFYUI_RESTART_COMMAND; ComfyUI"
11978
+ : "The panel bridge was offline after the crash, so the restart ran " +
11979
+ "through the headless managed restart; ComfyUI",
11980
+ noteRanLead: configuredCmd
11981
+ ? "The panel bridge was offline after the crash, so the restart ran " +
11982
+ "through COMFYUI_RESTART_COMMAND (not a panel confirmation card)"
11983
+ : "The panel bridge was offline after the crash, so the restart ran " +
11984
+ "through the headless managed restart (not a panel confirmation card)",
11985
+ });
11986
+ }
11987
+ // panel#1262: A CONFIGURED RESTART COMMAND REPLACES THE TAB REBOOT.
11988
+ //
11989
+ // On an externally-managed local install (a container, a systemd unit, a
11990
+ // launcher) the tab reboot is the WRONG mechanism twice over: the
11991
+ // refuse-safe preflight below cannot prove a relaunch from a bare
11992
+ // `main.py` argv that anchors only inside the instance's own namespace
11993
+ // (so it refuses and the wedge wins), and a wedged server answers no
11994
+ // Manager reboot anyway. COMFYUI_RESTART_COMMAND is the user's explicit
11995
+ // statement of what cycles the instance, so when it is set the restart
11996
+ // runs through it (headless restartComfyUI honors it) instead of the
11997
+ // tab dispatch. The BUSY GUARD the server-side reboot would have
11998
+ // enforced is re-implemented here against a VERIFIED queue read, with
11999
+ // the same force:true contract; a queue that cannot be read at all (the
12000
+ // wedge itself) refuses without force, because "cannot check" is not
12001
+ // "idle" and a restart aborts whatever is running.
12002
+ const configuredRestartCommand = config.comfyuiRestartCommand;
12003
+ if (configuredRestartCommand && !isRemoteMode() && !isCloudMode()) {
12004
+ const commandHealthBase = captureRebootHealthBase(ctx);
12005
+ if (commandHealthBase != null && sameHttpBase(getComfyUIBaseUrl(), commandHealthBase)) {
12006
+ if (force !== true) {
12007
+ let busyCount = null;
12008
+ try {
12009
+ const queue = await getQueueVerified();
12010
+ busyCount = queue.queue_running.length + queue.queue_pending.length;
12011
+ }
12012
+ catch {
12013
+ // unknown-ok: an UNREADABLE queue is the wedge case itself — null
12014
+ // below refuses without force (fail closed), it never reads as idle.
12015
+ busyCount = null;
12016
+ }
12017
+ if (busyCount !== 0) {
12018
+ return ok({
12019
+ rebooting: false,
12020
+ ready: false,
12021
+ confirmed_cycle: false,
12022
+ refused: true,
12023
+ note: busyCount === null
12024
+ ? "Refusing to restart ComfyUI: COMFYUI_RESTART_COMMAND is set, so the " +
12025
+ "restart runs the configured command — which ABORTS any in-progress or " +
12026
+ "queued generation — and the queue could not be read to confirm it is " +
12027
+ "idle (the server is not answering, which may be the very wedge you are " +
12028
+ "restarting to escape). Nothing was stopped. If the user confirms any " +
12029
+ "running render may be killed, retry with force:true."
12030
+ : `Refusing to restart ComfyUI: ${busyCount} generation(s) are in progress ` +
12031
+ "or queued, and the configured restart command ABORTS them. Nothing was " +
12032
+ "stopped. Wait for the queue to drain (poll panel_node_queue_status), or " +
12033
+ "retry with force:true ONLY if the user explicitly confirms they want to " +
12034
+ "kill the running generation.",
12035
+ });
12036
+ }
12037
+ }
12038
+ // The busy-check AWAIT makes the pre-await binding capture stale (r7's
12039
+ // own rule): a retarget or tab rebind landing during it would run the
12040
+ // command against an instance this tab no longer provably fronts.
12041
+ // Re-heal and re-verify at the point of action, exactly as the
12042
+ // dispatch path below does.
12043
+ ctx.ensureReachable?.();
12044
+ const postCheckHealthBase = captureRebootHealthBase(ctx);
12045
+ if (postCheckHealthBase == null ||
12046
+ !sameHttpBase(commandHealthBase, postCheckHealthBase)) {
12047
+ return ok({
12048
+ rebooting: false,
12049
+ ready: false,
12050
+ confirmed_cycle: false,
12051
+ refused: true,
12052
+ note: "Refusing to restart ComfyUI: the panel connection or target changed " +
12053
+ "while the queue was being checked, so I can no longer confirm the " +
12054
+ "configured restart command would act on the instance this tab fronts. " +
12055
+ "Nothing was stopped. Retry once the panel has settled.",
12056
+ });
12057
+ }
12058
+ return runHeadlessManagedRestart({
12059
+ healthBase: commandHealthBase,
12060
+ preRestartPanelIdentity: ctx.panelConnectionIdentity?.(),
12061
+ why: "COMFYUI_RESTART_COMMAND is set",
12062
+ mechanism: "the configured restart command",
12063
+ noteHealthyLead: "COMFYUI_RESTART_COMMAND is set, so the restart ran through the configured " +
12064
+ "command; ComfyUI",
12065
+ noteRanLead: "COMFYUI_RESTART_COMMAND is set, so the restart ran through the configured " +
12066
+ "command (not the Manager reboot)",
12067
+ });
12068
+ }
12069
+ // UNBOUND local target: the configured command acts on the orchestrator's
12070
+ // CONFIGURED target, which is not provably the instance this tab fronts —
12071
+ // fall through to the normal preflight/refusal machinery below (its refusal
12072
+ // names restart_comfyui, the non-tab-scoped entry point that CAN use it).
12073
+ }
10761
12074
  // #742 REFUSE-SAFE PREFLIGHT: a Manager reboot stops ComfyUI OUT-OF-BAND —
10762
12075
  // it never goes through our validated kill+relaunch — so before dispatching
10763
12076
  // anything, the stop must be provable survivable (#368/#370: losing a restart
@@ -10957,7 +12270,10 @@ CHECKED FOR YOU: the graph read this message prescribes was just run, and it ` +
10957
12270
  "from whatever launches it (its own launcher — e.g. Pinokio's own controls — " +
10958
12271
  "the Desktop app, or your terminal); for an externally-managed install you " +
10959
12272
  "can also point COMFYUI_PATH " +
10960
- "at the live install so a relaunch can be proven and use restart_comfyui.",
12273
+ "at the live install so a relaunch can be proven and use restart_comfyui, " +
12274
+ "or set COMFYUI_RESTART_COMMAND to the exact command that restarts the " +
12275
+ "instance (e.g. `docker restart <container>`) — both restart tools then run " +
12276
+ "that command instead of needing the launch path resolvable from here.",
10961
12277
  });
10962
12278
  }
10963
12279
  // Otherwise: a PASS with a stable config (proven safe for THE
@@ -11178,161 +12494,14 @@ CHECKED FOR YOU: the graph read this message prescribes was just run, and it ` +
11178
12494
  // and tracked in #871; this gate narrows the window, it does not close it.
11179
12495
  healthBase != null &&
11180
12496
  sameHttpBase(getComfyUIBaseUrl(), healthBase)) {
11181
- // The managed kill+relaunch does UNPREEMPTIBLE synchronous execSync work — PID
11182
- // discovery (~5+8s) + termination (~10s) + first port-free lookup (~13s) ≈ 40s
11183
- // worst case (Windows) — that a Promise.race CANNOT interrupt, and it BLOCKS the
11184
- // observer during that window. Admit it ONLY with enough budget for that sync
11185
- // work AND a full cold-start observation AFTER it, and give the observer a
11186
- // deadline that spans BOTH (coordinator P1: the proof deadline must start after,
11187
- // not before, the restart's synchronous work — otherwise a genuine cold start
11188
- // that finishes at sync+coldStart false-times-out).
11189
- const LEGACY_SYNC_WORST_CASE_MS = 40_000; // execSync PID lookup + kill + port-free
11190
- const LEGACY_COLD_START_OBS_MS = 100_000; // cold-start observation AFTER the sync
11191
- const LEGACY_RESTART_MIN_BUDGET_MS = LEGACY_SYNC_WORST_CASE_MS + LEGACY_COLD_START_OBS_MS;
11192
- if (overallDeadline - Date.now() < LEGACY_RESTART_MIN_BUDGET_MS) {
11193
- return ok({
11194
- rebooting: false,
11195
- ready: false,
11196
- confirmed_cycle: false,
11197
- note: "The built-in Manager exposed no reboot endpoint (legacy Manager 3.x), and " +
11198
- "there isn't enough remaining time to safely run the headless managed restart " +
11199
- "(kill + relaunch). ComfyUI was NOT restarted — retry panel_restart_comfyui " +
11200
- "(a fresh call gets the full budget).",
11201
- });
11202
- }
11203
- // A managed kill+relaunch restarts ComfyUI out-of-band, so drop the memoized
11204
- // caches. The observer watches the boot endpoint itself with a deadline spanning
11205
- // the ~40s blocking sync + a full cold-start window, and certifies ONLY on an
11206
- // OBSERVED down→up — a never-restarted healthy endpoint (a Desktop first-healthy
11207
- // Manager-reboot / preflight no-op) is honestly couldn't-confirm (coordinator P1).
11208
- resetClient();
11209
- resetObjectInfoCache();
11210
- resetManagerApiCache("panel managed restart");
11211
- // The observation window spans the ~40s blocking sync + a full cold-start
11212
- // window. (Under a test timing override, use the injected budget instead so the
11213
- // never-certify cases don't wait the real ~140s.)
11214
- const legacyProofWindow = panelRebootTimingOverride
11215
- ? timing.settleMs + timing.budgetMs
11216
- : LEGACY_RESTART_MIN_BUDGET_MS;
11217
- const proofDeadline = Math.min(Date.now() + legacyProofWindow, overallDeadline);
11218
- const proofPromise = observeRecovery(timing, proofDeadline, { healthBase });
11219
- const restartBudget = Math.max(1, overallDeadline - Date.now());
11220
- let restart;
11221
- let restartTimer;
11222
- try {
11223
- restart = await Promise.race([
11224
- restartComfyUI(),
11225
- new Promise((resolve) => {
11226
- restartTimer = setTimeout(() => resolve(undefined), restartBudget);
11227
- restartTimer.unref?.();
11228
- }),
11229
- ]);
11230
- }
11231
- catch (err) {
11232
- clearTimeout(restartTimer);
11233
- void proofPromise.catch(() => { }); // self-terminates at proofDeadline
11234
- return fail("The built-in Manager exposed no reboot endpoint (legacy Manager 3.x), " +
11235
- "and the headless managed restart also failed: " +
11236
- (err instanceof Error ? err.message : String(err)) +
11237
- " — restart ComfyUI on the host, then reconnect.");
11238
- }
11239
- clearTimeout(restartTimer);
11240
- // #742 r5/r6: the managed restart stopped the process — record the
11241
- // dispatch with THIS session holding the token, stamped with the
11242
- // BOUND-CONFIRMED base (this fallback only runs when the instance
11243
- // binding held, so healthBase is non-null here). restartComfyUI
11244
- // also stamped its own process-wide record, which never grounds
11245
- // causation. Only a PROVEN stop is recorded; a refusal/timeout
11246
- // (restart undefined, or stopped!==true) records nothing. The
11247
- // token is kept so the recovery clear below is CLEAR-IF-SAME (r15).
11248
- let legacyDispatchToken;
11249
- if (restart?.stopped === true) {
11250
- legacyDispatchToken = stampSessionRestartDispatch(ctx, healthBase);
11251
- }
11252
- // DEFINITIVE no-restart: a spawn failure, OR restartComfyUI refused before
11253
- // stopping anything (no process found / unsafe relaunch → stopped:false &&
11254
- // started:false). The process was NOT cycled, so the still-healthy endpoint is
11255
- // the OLD one — fail clearly rather than certify a no-op (coordinator P1).
11256
- if (restart?.spawn_error ||
11257
- (restart != null && restart.stopped !== true && restart.started !== true)) {
11258
- void proofPromise.catch(() => { });
11259
- return fail("The built-in Manager exposed no reboot endpoint (legacy Manager 3.x). " +
11260
- "Tried the headless managed restart (kill + relaunch), but it did not restart " +
11261
- `ComfyUI: ${restart?.message ?? "unknown error"} ` +
11262
- "Restart ComfyUI on the host, then reconnect.");
11263
- }
11264
- // Otherwise (the process WAS stopped/started, or restartComfyUI's own readiness
11265
- // poll merely expired — neither terminal) DEFER to OUR OWN observed DOWN→UP.
11266
- const recovery = await proofPromise;
11267
- // #742 r4/r5/r15: the managed restart was observed back — clear THIS
11268
- // session's record, CLEAR-IF-SAME: only when the session still holds
11269
- // the token THIS restart stamped (a concurrent dispatch's newer
11270
- // record survives). restartComfyUI also clears its own process-wide
11271
- // record on success; this covers only-observer-saw-it recoveries.
11272
- if (recovery.ready && legacyDispatchToken != null) {
11273
- clearSessionRestartDispatchIfSame(ctx, legacyDispatchToken);
11274
- }
11275
- const observed = recovery.via === "observed-cycle";
11276
- // The legacy Manager path restarts ComfyUI out-of-band too. Server recovery alone
11277
- // is not graph-tool readiness: wait for the browser tab to reconnect, then verify
11278
- // the same workflow-stamp capability the bridge requires before it dispatches a
11279
- // mutation. Without this, updating the panel pack followed by a legacy restart can
11280
- // falsely report ready while the browser is still running stale panel JS (#709).
11281
- const tabBack = recovery.ready
11282
- ? ctx.awaitPostRestartReachable
11283
- ? await ctx.awaitPostRestartReachable(preRestartPanelIdentity, Math.max(0, overallDeadline - Date.now()))
11284
- : ctx.awaitReachable
11285
- ? await ctx.awaitReachable(Math.max(0, overallDeadline - Date.now()))
11286
- : true
11287
- : false;
11288
- const graphToolsReady = tabBack && (ctx.tabCanMutateGraph ? ctx.tabCanMutateGraph() : true);
11289
- // #654 — `ready`/`graph_tools_ready` still come from the BOOLEAN, so an
11290
- // undetermined reconnect withholds graph tools exactly as before. Only
11291
- // the reported observation changes.
11292
- const tabReconnect = classifyTabReconnect({
11293
- serverReady: recovery.ready,
11294
- baselineCaptured: preRestartPanelIdentity != null,
11295
- tabBack,
11296
- });
11297
- return ok({
11298
- rebooting: true,
11299
- ready: graphToolsReady,
11300
- graph_tools_ready: graphToolsReady,
11301
- server_ready: recovery.ready,
11302
- panel_tab_reconnected: tabReconnect,
11303
- confirmed_cycle: observed, // true = we directly observed the down→up cycle
11304
- recovered_ms: recovery.waited_ms,
11305
- probes: recovery.attempts,
11306
- saw_down: recovery.sawDown,
11307
- via: recovery.ready ? recovery.via : undefined,
11308
- note: recovery.ready && !graphToolsReady
11309
- ? "ComfyUI-Manager (legacy 3.x) had no reboot endpoint; the headless managed restart " +
11310
- `came back healthy in ${(recovery.waited_ms / 1000).toFixed(1)}s, but ` +
11311
- (!tabBack
11312
- ? tabReconnect === "unknown"
11313
- ? "whether the panel tab reconnected could NOT be determined — no pre-restart " +
11314
- "baseline was captured for it, so nothing was watched. WHY is not established " +
11315
- "here, and it is one of: the tab's socket was not open at the instant the " +
11316
- "restart was dispatched; the panel advertised no tab session id (an older " +
11317
- "build, or its browser-tab lease was refused because a duplicate tab holds " +
11318
- "it); or the tab did not resolve at all. The tab may well be back. Graph " +
11319
- "tools are withheld (ready:false) because that is unproven, NOT because the " +
11320
- 'tab is known to be gone: call panel_list_workflows or panel_set_workflow_target({mode:"current"}) ' +
11321
- "to find out, and only refresh the browser if those also fail. If this " +
11322
- "repeats on every restart, the panel is probably too old to advertise a tab " +
11323
- "session id — update it."
11324
- : "the panel tab has NOT reconnected yet (ready:false). Wait a moment then retry, or " +
11325
- 'rebind with panel_set_workflow_target({mode:"current"}) before issuing graph tools.'
11326
- : "the panel tab reconnected but cannot safely run graph mutations (ready:false), usually " +
11327
- "because it is still running a stale panel bundle. Hard-refresh the ComfyUI browser tab " +
11328
- "(Ctrl+Shift+R) before issuing graph tools; if that does not restore it, update the panel " +
11329
- "and open/reload a saved workflow with a stable identity.")
11330
- : "ComfyUI-Manager (legacy 3.x) had no reboot endpoint; ran the headless managed " +
11331
- "restart (kill + relaunch) " +
11332
- (recovery.ready
11333
- ? `and it came back healthy in ${(recovery.waited_ms / 1000).toFixed(1)}s` +
11334
- (observed ? " (observed it go down then come back)." : " (cycle not directly observed).")
11335
- : `but it did NOT become healthy within ${Math.round(recovery.waited_ms / 1000)}s — verify with get_system_stats (action:"health") / panel_node_queue_status before assuming it restarted.`),
12497
+ return runHeadlessManagedRestart({
12498
+ healthBase,
12499
+ preRestartPanelIdentity,
12500
+ why: "The built-in Manager exposed no reboot endpoint (legacy Manager 3.x)",
12501
+ mechanism: "the headless managed restart (kill + relaunch)",
12502
+ noteHealthyLead: "ComfyUI-Manager (legacy 3.x) had no reboot endpoint; the headless managed restart",
12503
+ noteRanLead: "ComfyUI-Manager (legacy 3.x) had no reboot endpoint; ran the headless managed " +
12504
+ "restart (kill + relaunch)",
11336
12505
  });
11337
12506
  }
11338
12507
  // Genuine refusal (busy guard / security / no eligible fallback) — return
@@ -11582,7 +12751,17 @@ CHECKED FOR YOU: the graph read this message prescribes was just run, and it ` +
11582
12751
  ".") + argvNote + (preflightNote ? ` ${preflightNote}` : ""),
11583
12752
  });
11584
12753
  }),
11585
- def("panel_free_vram", "Unload all loaded models and free VRAM (ComfyUI /free). Use to unwedge a stuck/OOM ComfyUI when a cancel didn't free memory — before retrying or, last resort, restarting (panel_restart_comfyui). Does NOT restart ComfyUI; it just drops resident models and frees cached memory.", {}, async (_args, ctx) => ctx.call({ cmd: "free_vram" }, 15000)),
12754
+ def("panel_free_vram", "Unload all loaded models and free VRAM (ComfyUI /free). Use to unwedge a stuck/OOM ComfyUI when a cancel didn't free memory — before retrying or, last resort, restarting (panel_restart_comfyui). Does NOT restart ComfyUI; it just drops resident models and frees cached memory. If the panel tab is frozen and cannot acknowledge, the free is instead issued DIRECTLY to the ComfyUI server and verified there (same /free, idempotent) whenever the tab provably fronts the local server — otherwise the outcome is reported unknown rather than claimed.", {}, async (_args, ctx) => {
12755
+ const res = await ctx.call({ cmd: "free_vram" }, 15000);
12756
+ // #1249 — ONLY a no-reply is settled server-side. An acked executor error
12757
+ // (the panel's own "Failed to free VRAM: …") is a reply the bridge
12758
+ // received and relayed; it already says what failed, and re-issuing from
12759
+ // out here would fire a second mutation behind a verdict the caller was
12760
+ // given. A tagged reply-timeout is the one case where nothing answered.
12761
+ if (!isReplyTimeoutResult(res))
12762
+ return res;
12763
+ return settleFreeVramAfterAckTimeout(ctx, res);
12764
+ }),
11586
12765
  def("panel_show_media", "Display one or more images or videos directly in the panel chat. Use this whenever the user asks to SEE or SHOW a file — a disk path you composited/downloaded/generated (absolute path on the orchestrator host) OR a ComfyUI output ref ({ filename, subfolder?, type? }). Items are rendered as media cards in the agent chat area; supply optional captions. Max 8 items per call. NEVER describe an image with emoji or text placeholders — call this tool instead.", {
11587
12766
  items: z
11588
12767
  .array(z.object({