@melaya/runner 1.0.105 → 1.0.108
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assistantHost.py +34 -6
- package/dist/connection.js +48 -9
- package/package.json +1 -1
package/dist/assistantHost.py
CHANGED
|
@@ -72,24 +72,49 @@ def _memory_watermark(agent) -> int:
|
|
|
72
72
|
return 0
|
|
73
73
|
|
|
74
74
|
|
|
75
|
+
def _sync_ollama_memory_budget(agent) -> None:
|
|
76
|
+
"""P2-6 (persistent host): the Assistant builds ONE agent + BoundedMemory, so an
|
|
77
|
+
OOM num_ctx downgrade that shrinks the ollama context cache would otherwise leave
|
|
78
|
+
the live memory budgeting against the OLD (too-large) window — packing more than
|
|
79
|
+
the GPU can hold and re-OOMing every turn. Re-resolve the budget from the (now
|
|
80
|
+
downgraded) cache each turn and LOWER max_tokens to match. Ratchets DOWN only;
|
|
81
|
+
best-effort + ollama-only (a cloud/CLI provider never touches this)."""
|
|
82
|
+
if (os.environ.get("MEL_ASSISTANT_PROVIDER", "") or "").lower() != "ollama":
|
|
83
|
+
return
|
|
84
|
+
try:
|
|
85
|
+
from shared.runtime.agent_factory import _resolve_memory_budget
|
|
86
|
+
budget = _resolve_memory_budget("ollama", os.environ.get("MEL_ASSISTANT_MODEL", "") or "")
|
|
87
|
+
mem = _agent_memory(agent)
|
|
88
|
+
if mem is not None and hasattr(mem, "max_tokens") and budget > 0:
|
|
89
|
+
cur = int(getattr(mem, "max_tokens", 0) or 0)
|
|
90
|
+
if cur <= 0 or budget < cur:
|
|
91
|
+
mem.max_tokens = budget
|
|
92
|
+
_log(f"ollama memory budget re-synced {cur}->{budget} (post-OOM downgrade)")
|
|
93
|
+
except Exception:
|
|
94
|
+
pass
|
|
95
|
+
|
|
96
|
+
|
|
75
97
|
def _config_hash() -> str:
|
|
76
98
|
"""PR4 config-drift: a stable hash over the config env that determines host
|
|
77
99
|
behaviour, canonicalized IDENTICALLY to the server (runnerNamespace.ts
|
|
78
100
|
_assistantConfigHash): pipe-delimited provider|model|language|connectors|
|
|
79
|
-
|
|
80
|
-
|
|
101
|
+
phoneReady, language default 'en', connectors lowercased+sorted, phoneReady
|
|
102
|
+
1/0. Reported in `ready` so the server can detect env divergence.
|
|
103
|
+
|
|
104
|
+
NOTE: hitlMode is DELIBERATELY EXCLUDED (it must match the server byte-for-byte,
|
|
105
|
+
and the server dropped it — it is a PER-TURN parameter carried on every
|
|
106
|
+
assistant_turn frame, so a safe↔autonomous flip takes effect without a reboot).
|
|
107
|
+
Including it here made the host hash NEVER match the server's, so config-drift
|
|
108
|
+
detection was permanently fail-open (a stale-connector host was accepted)."""
|
|
81
109
|
import hashlib
|
|
82
110
|
raw_conn = os.environ.get("MEL_ASSISTANT_CONNECTORS", "") or ""
|
|
83
111
|
connectors = ",".join(sorted(c.lower() for c in raw_conn.split(",") if c.strip()))
|
|
84
|
-
hitl = os.environ.get("MEL_ASSISTANT_HITL_MODE", "safe") or "safe"
|
|
85
|
-
if hitl not in ("safe", "autonomous", "payments_only"):
|
|
86
|
-
hitl = "safe"
|
|
87
112
|
phone = "1" if (os.environ.get("MEL_ASSISTANT_PHONE_READY", "") or "") else "0"
|
|
88
113
|
canon = "|".join([
|
|
89
114
|
os.environ.get("MEL_ASSISTANT_PROVIDER", "") or "",
|
|
90
115
|
os.environ.get("MEL_ASSISTANT_MODEL", "") or "",
|
|
91
116
|
os.environ.get("MEL_ASSISTANT_LANGUAGE", "en") or "en",
|
|
92
|
-
connectors,
|
|
117
|
+
connectors, phone,
|
|
93
118
|
])
|
|
94
119
|
return hashlib.sha256(canon.encode("utf-8")).hexdigest()[:32]
|
|
95
120
|
|
|
@@ -677,6 +702,9 @@ def main() -> int:
|
|
|
677
702
|
if not message:
|
|
678
703
|
_emit(turn_id, "done")
|
|
679
704
|
continue
|
|
705
|
+
# P2-6: nudge the live memory budget down if a prior turn's OOM downgraded
|
|
706
|
+
# the ollama context (no-op for every other provider / when unchanged).
|
|
707
|
+
_sync_ollama_memory_budget(agent)
|
|
680
708
|
_run_turn(agent, turn_id, message)
|
|
681
709
|
|
|
682
710
|
|
package/dist/connection.js
CHANGED
|
@@ -898,17 +898,34 @@ export async function connect(opts) {
|
|
|
898
898
|
return;
|
|
899
899
|
const generation = Number(payload.generation ?? 0) || 0;
|
|
900
900
|
const emitEv = (ev) => socket.emit("runner:assistant_event", { sessionId: sid, turnId: "", ...ev });
|
|
901
|
-
// A live host already exists — report ready WITH its last-known memory
|
|
902
|
-
// watermark so the server doesn't needlessly re-offer a rehydrate (a bare
|
|
903
|
-
//
|
|
901
|
+
// A live host already exists — normally report ready WITH its last-known memory
|
|
902
|
+
// watermark so the server doesn't needlessly re-offer a rehydrate (a bare ready
|
|
903
|
+
// would read as watermark 0) and NO second spawn. BUT first check config drift:
|
|
904
904
|
const live = activeAssistants.get(sid);
|
|
905
|
-
|
|
906
|
-
// server that restarted (its in-memory session map wiped) can DETECT a host
|
|
907
|
-
// that booted under an older generation and reboot it instead of serving the
|
|
908
|
-
// next turn from stale memory/config. A bare ready hid this.
|
|
905
|
+
const expectedConfig = String(payload.configHash || "");
|
|
909
906
|
if (live) {
|
|
910
|
-
|
|
911
|
-
|
|
907
|
+
// SECURITY (config-drift fail-open fix): the server may have RESTARTED and
|
|
908
|
+
// lost its session map, then sent assistant_start with the CURRENT config
|
|
909
|
+
// (e.g. a connector the user has since removed). If the live host booted under
|
|
910
|
+
// a DIFFERENT config, returning it would serve the removed connector's tools +
|
|
911
|
+
// credentials. Kill the stale host and fall through to spawn a FRESH one under
|
|
912
|
+
// the new config. Same-config → reuse (the fast path). An empty live.configHash
|
|
913
|
+
// (a pre-configHash host) is treated as stale so it can't linger fail-open.
|
|
914
|
+
const drifted = expectedConfig ? (live.configHash !== expectedConfig) : false;
|
|
915
|
+
if (drifted || (expectedConfig && !live.configHash)) {
|
|
916
|
+
console.log(chalk.yellow(` ◆ Assistant host ${sid.slice(0, 10)}… config drift (host=${live.configHash || "∅"} ≠ server=${expectedConfig}) — rebooting`));
|
|
917
|
+
live.killed = true; // suppress its session_closed (see proc.exit)
|
|
918
|
+
activeAssistants.delete(sid); // stop returning it immediately
|
|
919
|
+
try {
|
|
920
|
+
live.proc.kill();
|
|
921
|
+
}
|
|
922
|
+
catch { /* already dead */ } // SIGTERM → Python atexit cleanup
|
|
923
|
+
// fall through to the fresh-boot path below (bootingAssistants guards races)
|
|
924
|
+
}
|
|
925
|
+
else {
|
|
926
|
+
emitEv({ kind: "ready", memoryWatermark: live.memoryWatermark || 0, generation: live.generation, hostInstanceId: live.hostInstanceId, configHash: live.configHash, protocol: ASSISTANT_PROTOCOL_VERSION });
|
|
927
|
+
return;
|
|
928
|
+
}
|
|
912
929
|
}
|
|
913
930
|
// A boot is already in flight for this sid (concurrent start / server restart
|
|
914
931
|
// mid-boot). Drop this duplicate — the in-flight boot will emit `ready`.
|
|
@@ -1013,6 +1030,13 @@ export async function connect(opts) {
|
|
|
1013
1030
|
activeAssistants.set(sid, session);
|
|
1014
1031
|
console.log(chalk.hex("#7c6ff0")(` ◆ Assistant session ${sid.slice(0, 10)}… (${payload.provider})`));
|
|
1015
1032
|
proc.stdout?.on("data", (data) => {
|
|
1033
|
+
// ANY host output means the turn is actively working — refresh the idle
|
|
1034
|
+
// clock. Without this, lastActivity was set ONLY at turn start (assistant_turn),
|
|
1035
|
+
// so a SINGLE long turn (a phone task is one turn with ~100 tool calls streaming
|
|
1036
|
+
// for >15 min) was reaped mid-action by the idle sweep despite continuous
|
|
1037
|
+
// activity ("idle-swept" while properly working). Now only a genuinely silent
|
|
1038
|
+
// host (no output for ASSISTANT_IDLE_MS) is swept.
|
|
1039
|
+
session.lastActivity = Date.now();
|
|
1016
1040
|
session.stdoutBuf += data.toString();
|
|
1017
1041
|
let idx;
|
|
1018
1042
|
while ((idx = session.stdoutBuf.indexOf("\n")) >= 0) {
|
|
@@ -1066,6 +1090,13 @@ export async function connect(opts) {
|
|
|
1066
1090
|
// would orphan the newcomer and make the next turn "session_not_found".
|
|
1067
1091
|
if (activeAssistants.get(sid) === session)
|
|
1068
1092
|
activeAssistants.delete(sid);
|
|
1093
|
+
// A host we INTENTIONALLY killed for a config-drift reboot must NOT emit
|
|
1094
|
+
// session_closed — that frame would fail the fresh boot's ready waiter
|
|
1095
|
+
// (the server can't pre-mark it superseded here, the kill is runner-side).
|
|
1096
|
+
if (session.killed) {
|
|
1097
|
+
console.log(chalk.gray(` ■ Assistant host ${sid.slice(0, 10)}… replaced (config drift, exit ${code})`));
|
|
1098
|
+
return;
|
|
1099
|
+
}
|
|
1069
1100
|
const detail = code !== 0 && stderrTail.length ? stderrTail.slice(-12).join("\n") : "";
|
|
1070
1101
|
socket.emit("runner:assistant_event", { sessionId: sid, turnId: "", kind: "session_closed", code, detail });
|
|
1071
1102
|
console.log(chalk.gray(` ■ Assistant session ${sid.slice(0, 10)}… closed (exit ${code})`));
|
|
@@ -1146,6 +1177,14 @@ export async function connect(opts) {
|
|
|
1146
1177
|
activeAssistants.delete(String(payload.sessionId));
|
|
1147
1178
|
}
|
|
1148
1179
|
});
|
|
1180
|
+
// Keepalive: the server is blocked waiting on the user's behalf (a HITL approval
|
|
1181
|
+
// poll, a slow phone action) so the host emits no stdout, but the turn is ALIVE.
|
|
1182
|
+
// Refresh the idle clock so the idle sweep doesn't reap a host mid-approval.
|
|
1183
|
+
socket.on("runner:assistant_keepalive", (payload) => {
|
|
1184
|
+
const s = activeAssistants.get(String(payload.sessionId || ""));
|
|
1185
|
+
if (s)
|
|
1186
|
+
s.lastActivity = Date.now();
|
|
1187
|
+
});
|
|
1149
1188
|
// Cooperative cancel: the user pressed STOP in chat. Write a cancel line to the
|
|
1150
1189
|
// host's stdin — its stdin reader thread flips a flag the running turn's
|
|
1151
1190
|
// hooks poll, so the agent loop aborts BETWEEN steps (right where the next
|