grix-connector 4.5.0 → 4.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/dsh-bridge/grix-dsh-bridge-4.6.0.tgz +0 -0
- package/dist/assets/dsh-bridge/manifest.json +6 -6
- package/dist/bridge/bridge.js +4 -4
- package/dist/bridge/local-action-toolbar.js +1 -1
- package/dist/bridge/toolbar-meta.js +1 -1
- package/dist/bridge/toolbar-selection.js +1 -0
- package/dist/core/installer/npm-registry.js +1 -1
- package/dist/core/mcp/tools.js +1 -1
- package/dist/core/upgrade/npm-upgrader.js +2 -2
- package/dist/core/upgrade/upgrade-checker.js +1 -1
- package/dist/core/upgrade/version-store.js +2 -2
- package/dist/default-skills/grix-scheduled-trigger/SKILL.md +162 -0
- package/dist/grix.js +7 -7
- package/dist/launcher.mjs +98 -35
- package/dist/manager.js +1 -1
- package/dist/mcp/stream-http/security.js +1 -1
- package/openclaw-plugin/index.js +70 -0
- package/package.json +1 -1
- package/scripts/launcher.mjs +98 -35
- package/dist/assets/dsh-bridge/grix-dsh-bridge-4.5.0.tgz +0 -0
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: grix-scheduled-trigger
|
|
3
|
+
description: Set up human- or timer-driven triggers for this agent — anything that must fire on a schedule, repeat on an interval, or be kicked off later by a person — by reusing (grix_webhook_list) or creating (grix_webhook_create) the session's standing webhook and registering a local scheduler job (macOS launchd/cron, Linux cron/systemd timer, Windows Task Scheduler) that POSTs to it; grix_webhook_delete tears it down. Not for agent-to-agent dispatch callbacks. Trigger when the user asks for a scheduled, recurring, periodic, daily/hourly, cron, reminder, polling, or "wake me up at X" behaviour.
|
|
4
|
+
trigger: When the user wants this agent to run on a schedule, repeat periodically, poll something, or be woken later by a timer or a person instead of by another agent
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
# Grix Scheduled Trigger
|
|
8
|
+
|
|
9
|
+
Your process only runs while handling a chat turn. You cannot sleep, loop, or
|
|
10
|
+
wait in the background and "come back later". The reliable way to be woken at
|
|
11
|
+
a given time is a **session webhook** plus an **OS-level scheduler job**:
|
|
12
|
+
|
|
13
|
+
1. The session has one standing webhook URL (`grix_webhook_list` finds it,
|
|
14
|
+
`grix_webhook_create` makes it if missing).
|
|
15
|
+
2. Any HTTP client POSTing `{"content": "..."}` to that URL sends the text into
|
|
16
|
+
the session **as the owner** — exactly like the owner typing it — so you are
|
|
17
|
+
triggered and receive it as a normal message.
|
|
18
|
+
3. A scheduler on this machine (launchd/cron/systemd timer/Task Scheduler) does
|
|
19
|
+
the POST at the right times. The scheduler survives reboots and does not
|
|
20
|
+
depend on your process being alive.
|
|
21
|
+
|
|
22
|
+
## When to use / not use
|
|
23
|
+
|
|
24
|
+
Use this mechanism whenever the trigger comes from **a timer or a human**:
|
|
25
|
+
|
|
26
|
+
- "every morning at 8 summarise X", "check the deploy every 10 minutes", "remind
|
|
27
|
+
me in 2 hours", "run the daily report", polling an external state, recurring
|
|
28
|
+
health checks, any one-shot or repeating wake-up.
|
|
29
|
+
- Anything a person wants to fire manually later (they can `curl` the URL or
|
|
30
|
+
wire it into their own tooling).
|
|
31
|
+
|
|
32
|
+
Do **not** use it for **agent-driven** callbacks: dispatch results
|
|
33
|
+
(`report_dispatch_result`), replies to another agent, or anything already
|
|
34
|
+
covered by `grix_session_send` / `grix_dispatch_agent`. Those flows have their
|
|
35
|
+
own contract; a webhook there is redundant and leaks a secret URL for nothing.
|
|
36
|
+
|
|
37
|
+
## Step 1 — reuse the session's standing webhook (create only if missing)
|
|
38
|
+
|
|
39
|
+
**One webhook per session, shared by every schedule targeting that session.**
|
|
40
|
+
Never create a second one while an active one exists.
|
|
41
|
+
|
|
42
|
+
1. `grix_webhook_list` with `session_id` = the **current** `chat_id` from the
|
|
43
|
+
channel metadata. It returns the session's endpoints with full `url`,
|
|
44
|
+
`status`, `expires_at`, `last_used_at`.
|
|
45
|
+
2. If there is an entry with `status: "active"` and no `expires_at`, use its
|
|
46
|
+
`url`. If several exist, use the newest and `grix_webhook_delete` the rest.
|
|
47
|
+
3. Only if none is usable: `grix_webhook_create` with the same `session_id`
|
|
48
|
+
and **no `expires_at`** — the standing endpoint must not expire; one-shot
|
|
49
|
+
reminders are made one-shot by the scheduler job (Step 3), not by the
|
|
50
|
+
webhook. The result carries the `url`.
|
|
51
|
+
|
|
52
|
+
Rules:
|
|
53
|
+
|
|
54
|
+
- You may only list/create/delete webhooks for sessions you are a member of;
|
|
55
|
+
the platform checks that and that the owner is a member too.
|
|
56
|
+
- Permission is granted per agent by the owner ("Create Session Webhook"). A
|
|
57
|
+
permission error means it is not enabled — ask the owner to enable it in the
|
|
58
|
+
agent's permission settings in the App. Do not retry.
|
|
59
|
+
- The platform caps active endpoints at 20 per session and rejects
|
|
60
|
+
`expires_at` in the past; if you ever hit "too many active webhooks", list
|
|
61
|
+
and delete duplicates instead of creating more.
|
|
62
|
+
|
|
63
|
+
Local registry: `~/.grix/scheduled/<session_id>.json` (directory mode 700,
|
|
64
|
+
file mode 600) stores `session_id`, the webhook `id` + `url`, and the list of
|
|
65
|
+
scheduler jobs you registered. If the file is lost, `grix_webhook_list`
|
|
66
|
+
recovers the URL; the jobs are recovered from the scheduler itself.
|
|
67
|
+
|
|
68
|
+
## Step 2 — the trigger script
|
|
69
|
+
|
|
70
|
+
Write a small script per job under `~/.grix/scheduled/jobs/` that POSTs to the
|
|
71
|
+
URL. Keep the URL in the script (mode 600), never in the scheduler's visible
|
|
72
|
+
arguments or in chat.
|
|
73
|
+
|
|
74
|
+
Payload: JSON `{"content": "<message>", "msg_type": "text", "client_msg_id": "<unique>"}`.
|
|
75
|
+
`content` is what you will receive as the owner's message — make it
|
|
76
|
+
self-describing so a future turn knows what to do without context, e.g.
|
|
77
|
+
`[scheduled] daily-report 08:00 — generate yesterday's sales summary`.
|
|
78
|
+
`client_msg_id` must be unique per firing (job name + timestamp) so a retry
|
|
79
|
+
never posts twice.
|
|
80
|
+
|
|
81
|
+
**Group sessions**: the message is fanned out to every agent in the group and
|
|
82
|
+
each connector applies its own group policy, so start `content` with an
|
|
83
|
+
@-mention of yourself (your own agent display name, e.g. `@四喜 Claude`) to
|
|
84
|
+
make sure this agent — and only this agent — is woken. In a private
|
|
85
|
+
owner↔agent session no mention is needed.
|
|
86
|
+
|
|
87
|
+
Reliability rules for the script:
|
|
88
|
+
|
|
89
|
+
- Retry on network failure / HTTP 5xx / 429 with backoff (e.g. 3 attempts,
|
|
90
|
+
5s → 20s → 60s). Rate limit is 60 requests per minute per endpoint+IP.
|
|
91
|
+
- Treat 404 (`WEBHOOK_NOT_FOUND`), 410 (`WEBHOOK_EXPIRED`), 403
|
|
92
|
+
(`WEBHOOK_FORBIDDEN`) as **permanent**: do not retry, log it, and on the next
|
|
93
|
+
turn tell the owner the schedule is broken and needs a new webhook.
|
|
94
|
+
- Append one line per attempt to `~/.grix/scheduled/logs/<job>.log`
|
|
95
|
+
(timestamp, HTTP status, message_id) so failures are diagnosable.
|
|
96
|
+
|
|
97
|
+
Minimal POSIX example (`~/.grix/scheduled/jobs/daily-report.sh`):
|
|
98
|
+
|
|
99
|
+
```sh
|
|
100
|
+
#!/bin/sh
|
|
101
|
+
URL='https://example.com/v1/webhook/incoming/whk_...'
|
|
102
|
+
JOB='daily-report'
|
|
103
|
+
LOG="$HOME/.grix/scheduled/logs/$JOB.log"
|
|
104
|
+
mkdir -p "$(dirname "$LOG")"
|
|
105
|
+
for delay in 5 20 60 0; do
|
|
106
|
+
code=$(curl -sS -o /tmp/grix-$JOB.out -w '%{http_code}' -X POST "$URL" \
|
|
107
|
+
-H 'Content-Type: application/json' \
|
|
108
|
+
-d "{\"content\":\"[scheduled] $JOB $(date '+%F %T') — generate yesterday's sales summary\",\"msg_type\":\"text\",\"client_msg_id\":\"$JOB-$(date +%s)\"}")
|
|
109
|
+
echo "$(date '+%F %T') $code $(cat /tmp/grix-$JOB.out)" >> "$LOG"
|
|
110
|
+
case "$code" in
|
|
111
|
+
200) exit 0 ;;
|
|
112
|
+
403|404|410) exit 1 ;; # permanent — needs a new webhook
|
|
113
|
+
esac
|
|
114
|
+
[ "$delay" = 0 ] || sleep "$delay"
|
|
115
|
+
done
|
|
116
|
+
exit 1
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
On Windows write the equivalent as PowerShell (`Invoke-RestMethod`, same
|
|
120
|
+
retry/permanent-error rules).
|
|
121
|
+
|
|
122
|
+
## Step 3 — register the scheduler job
|
|
123
|
+
|
|
124
|
+
Pick the native scheduler; register with the user's own account (no sudo):
|
|
125
|
+
|
|
126
|
+
- **macOS**: prefer `launchd` — write
|
|
127
|
+
`~/Library/LaunchAgents/im.grix.scheduled.<job>.plist` with
|
|
128
|
+
`StartCalendarInterval` (fixed times) or `StartInterval` (every N seconds),
|
|
129
|
+
`ProgramArguments` = the script, then
|
|
130
|
+
`launchctl bootstrap gui/$(id -u) <plist>` (fallback `launchctl load`).
|
|
131
|
+
`crontab -e` also works but launchd runs missed jobs after sleep; cron does
|
|
132
|
+
not.
|
|
133
|
+
- **Linux**: `crontab` entry (`0 8 * * * $HOME/.grix/scheduled/jobs/daily-report.sh`)
|
|
134
|
+
or a user `systemd` timer (`systemctl --user enable --now <job>.timer`;
|
|
135
|
+
add `Persistent=true` to catch up missed runs).
|
|
136
|
+
- **Windows**: `schtasks /Create /SC DAILY /ST 08:00 /TN "Grix\<job>" /TR "powershell -NoProfile -ExecutionPolicy Bypass -File <script.ps1>"`
|
|
137
|
+
(use `/SC MINUTE /MO N` for intervals). Run as the current user, no elevation.
|
|
138
|
+
|
|
139
|
+
One-shot reminders: reuse the standing webhook and make the **job** one-shot —
|
|
140
|
+
`launchd` `StartCalendarInterval` with a full date, `at` /
|
|
141
|
+
`systemd-run --user --on-calendar`, or `schtasks /SC ONCE` — and have the
|
|
142
|
+
script remove its own scheduler entry and script file after a successful 200.
|
|
143
|
+
Do not create a separate expiring webhook for it.
|
|
144
|
+
|
|
145
|
+
## Step 4 — verify before you report done
|
|
146
|
+
|
|
147
|
+
1. Run the script once by hand and confirm HTTP 200 **and** that the message
|
|
148
|
+
actually arrived in this session (you will see it on your next turn, or
|
|
149
|
+
check `grix_query` message history).
|
|
150
|
+
2. Confirm the scheduler accepted the job (`launchctl print gui/$(id -u)/<label>`,
|
|
151
|
+
`crontab -l`, `systemctl --user list-timers`, `schtasks /Query /TN`).
|
|
152
|
+
3. Record the job in the registry file. Report to the owner: what fires, when,
|
|
153
|
+
the log path, and how to remove it. Never paste the webhook URL into chat.
|
|
154
|
+
|
|
155
|
+
## Removing schedules
|
|
156
|
+
|
|
157
|
+
- Removing one job: unload/delete its scheduler entry, delete the script,
|
|
158
|
+
update the registry. Keep the webhook — other jobs (or future ones) reuse it.
|
|
159
|
+
- Removing the session's scheduling entirely (owner asks to "stop all timers
|
|
160
|
+
here"): remove every job, then `grix_webhook_delete` the endpoint `id` from
|
|
161
|
+
the registry (or from `grix_webhook_list`) so the URL stops working, and
|
|
162
|
+
delete the registry file.
|
package/dist/grix.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import D from"node:path";import{writeFileSync as re}from"node:fs";import{Manager as xe}from"./manager.js";import{ensureGrixDirs as be,initLogger as Pe,log as i,installProcessLogRotation as Ie,setConsoleOutput as ne}from"./core/log/index.js";import{HealthServer as
|
|
3
|
-
Run \`grix-connector --help\` to see the supported commands and options.`),process.exit(1)),R.version&&(console.log(
|
|
2
|
+
import D from"node:path";import{writeFileSync as re}from"node:fs";import{Manager as xe}from"./manager.js";import{ensureGrixDirs as be,initLogger as Pe,log as i,installProcessLogRotation as Ie,setConsoleOutput as ne}from"./core/log/index.js";import{HealthServer as ke,bindPortOrFail as oe}from"./core/runtime/index.js";import{writePidFile as Ce,removePidFile as O,readDaemonPid as Te}from"./core/runtime/index.js";import{resolveRuntimePaths as W}from"./core/config/index.js";import{supportsNonMitmRelayConfig as _e}from"./core/config/provider-env.js";import{ServiceManager as ae}from"./service/service-manager.js";import{RESTART_EXIT_CODE as ie,markManagedStop as Ue,resolveLauncherStore as $e}from"./core/upgrade/version-store.js";import{migrateWindowsWrapper as K}from"./service/windows-wrapper-migration.js";const Le=300*1e3;import{killProcessesByCommandLine as Fe,isWindowsElevated as Oe,shouldSweepWindowsWrappers as He}from"./service/process-control.js";import{buildServiceID as Me}from"./service/service-paths.js";import{acquireDaemonLock as Ge,isLockHolderSameProcess as Ne,readDaemonLock as X,releaseDaemonLock as H}from"./runtime/daemon-lock.js";import{writeDaemonStatus as N,removeDaemonStatus as Be}from"./runtime/service-state.js";import{AdminServer as je,generateToken as We,writeTokenFile as Ke}from"./core/admin/index.js";import{initSentry as Xe,closeSentry as V,reportFatal as q}from"./core/observability/sentry.js";import{initProxyManager as Ve,getProxyManager as f,relayHostsForClientType as se,RelayStateStore as qe,SecretStore as Ye}from"./core/proxy/index.js";import{stopProxyIfNobodyNeedsIt as le,switchAgentRelayMode as Y,reconcileRelayStates as ce}from"./core/proxy/relay-orchestration.js";import{setAgentRelayCredential as de,RelayCredentialError as pe}from"./core/proxy/relay-credential.js";import{RelayFetchError as ge,RelayToggleCoordinator as Je}from"./core/proxy/relay-credential-fetch.js";import{applyHermesProfileRelay as ze,clearHermesProfileRelay as Qe,discoverHermesProfiles as Ze}from"./core/proxy/hermes-profile-relay.js";import{disableRelayCredential as et,enableRelayCredential as tt,RelayRouteError as ue}from"./core/proxy/relay-credential-router.js";import{upgradeStatusSnapshot as rt}from"./core/upgrade/npm-upgrader.js";import{resolveClientVersion as nt}from"./core/util/client-version.js";import{parseCliArgs as ot}from"./core/util/cli-args.js";const{command:x,flags:R,unknownFlags:J}=ot(process.argv.slice(2));if(J.length>0&&(console.error(`Unknown option${J.length>1?"s":""}: ${J.map(r=>`--${r}`).join(", ")}
|
|
3
|
+
Run \`grix-connector --help\` to see the supported commands and options.`),process.exit(1)),R.version&&(console.log(nt()),process.exit(0)),R.help&&(console.log(`grix-connector \u2014 Unified AI Agent Bridge
|
|
4
4
|
|
|
5
5
|
Usage: grix-connector <command> [options]
|
|
6
6
|
|
|
@@ -34,9 +34,9 @@ Examples:
|
|
|
34
34
|
grix-connector start # Start as system service
|
|
35
35
|
grix-connector status # Check service status
|
|
36
36
|
grix-connector restart # Restart the service
|
|
37
|
-
`),process.exit(0)),x==="reload"){const r=Te();r||(console.error("reload failed: daemon is not running (no pid file)"),process.exit(1));try{process.kill(r,0)}catch{console.error(`reload failed: daemon process ${r} is not running (stale pid file)`),process.exit(1)}{const
|
|
37
|
+
`),process.exit(0)),x==="reload"){const r=Te();r||(console.error("reload failed: daemon is not running (no pid file)"),process.exit(1));try{process.kill(r,0)}catch{console.error(`reload failed: daemon process ${r} is not running (stale pid file)`),process.exit(1)}{const p=X(W().daemonLockFile);p&&p.pid===r&&!Ne(p)&&(console.error(`reload failed: pid ${r} has been reused by another process (stale pid file)`),process.exit(1))}try{process.kill(r,"SIGHUP"),console.log(JSON.stringify({ok:!0,signaled:r},null,2)),process.exit(0)}catch(p){console.error(`reload failed: ${p instanceof Error?p.message:p}`),process.exit(1)}}const fe=["start","stop","restart","status"];if(x&&fe.includes(x)){process.platform==="win32"&&["start","stop","restart"].includes(x)&&!Oe()&&console.warn(`Warning: Not running as administrator. Task Scheduler registration is skipped;
|
|
38
38
|
using Startup folder auto-start instead. For full Task Scheduler integration,
|
|
39
|
-
right-click the terminal and select "Run as administrator".`);const r=W(),
|
|
40
|
-
Valid commands: ${fe.join(", ")}`),process.exit(1));const s=W(),
|
|
41
|
-
`),i.error("main",r.message.replace(/\n/g," \u2014 "));const d=X(s.daemonLockFile);d?.pid===process.pid&&d.instance_id===k&&await N(s.daemonStatusFile,{state:"failed",pid:process.pid,instance_id:k,updated_at:Date.now(),reason:`port_bind_${r.kind}:${r.label}:${r.port}`}).catch(()=>{}),await V(),H(s.daemonLockFile).catch(()=>{}),O(),process.exit(1)}async function P(r){if(B)return;B=!0,i.info("main",`Received ${r}, shutting down...`),M.markShuttingDown();const d=r==="upgrade-switch"?ie:2,S=setTimeout(()=>{i.error("main","Shutdown timed out, forcing exit"),H(s.daemonLockFile).catch(()=>{}),O(),process.exit(d)},1e4);try{await n.stop(),await f()?.stop().catch(()=>{}),await b.stop(),await M.stop(),await V(),await H(s.daemonLockFile),await Ge(s.daemonStatusFile).catch(()=>{}),clearTimeout(S),O(),i.info("main","Shutdown complete"),process.exit(r==="upgrade-switch"?ie:0)}catch(u){i.error("main",`Shutdown error: ${u}`),H(s.daemonLockFile).catch(()=>{}),O(),process.exit(d)}}async function ot(){if(be(),Pe(),Ie(s.stdoutLogFile,s.stderrLogFile),ne(!1),Fe()){const e=Oe(s.rootDir,"win32");await $e(e,{platform:"win32"})}try{k=(await He(s.daemonLockFile,s.rootDir)).instance_id}catch(e){console.error(e instanceof Error?e.message:e),process.exit(1)}if(await We(),ke(),i.info("main",`grix-connector starting (PID ${process.pid})`),await N(s.daemonStatusFile,{state:"starting",pid:process.pid,instance_id:k,updated_at:Date.now()}),process.env.GRIX_LAUNCHER_MANAGED!=="1"&&await new ae({cliPath:D.resolve(process.argv[1]??""),nodePath:process.execPath}).adoptLauncher(s.rootDir).then(e=>{e&&i.info("main","Service definition now points at the version-store launcher")}).catch(e=>{i.warn("main",`Launcher adoption failed: ${e instanceof Error?e.message:e}`)}),(await K({rootDir:s.rootDir,requestGracefulShutdown:()=>{P("wrapper-migration")}}).catch(e=>(i.warn("main",`Windows wrapper migration failed: ${e instanceof Error?e.message:e}`),null)))?.handoffStarted)return;process.platform==="win32"&&process.env.GRIX_LAUNCHER_MANAGED!=="1"&&setInterval(()=>{B||K({rootDir:s.rootDir,requestGracefulShutdown:()=>{P("wrapper-migration")}}).then(o=>{o.handoffStarted&&i.info("main","Wrapper handoff started by the periodic migration retry")}).catch(o=>{i.warn("main",`Periodic wrapper migration failed: ${o instanceof Error?o.message:o}`)})},Ue).unref();const d=parseInt(R["health-port"]??process.env.GRIX_HEALTH_PORT??"19579",10);{const e=await oe({label:"health",port:d,envVar:"GRIX_HEALTH_PORT",cliFlag:"health-port",start:o=>M.start(o)});e&&await he(e)}const S=D.join(s.dataDir,"health-port");re(S,String(d),"utf-8"),process.on("SIGINT",()=>P("SIGINT")),process.on("SIGTERM",()=>P("SIGTERM")),process.on("SIGHUP",()=>{B||(i.info("main","Received SIGHUP, reloading config..."),n.reload().then(e=>i.info("main",`reload done: ${JSON.stringify(e)}`)).catch(e=>i.error("main",`reload failed: ${e instanceof Error?e.message:e}`)))});let u="",p=0,T;process.on("uncaughtException",e=>{const o=e instanceof Error?e.stack??e.message:String(e);o===u?(p++,(p<=3||p%100===0)&&i.error("main",`Uncaught exception (x${p}): ${o}`)):(p>3&&i.error("main",`Previous exception repeated ${p} times total`),u=o,p=1,i.error("main",`Uncaught exception: ${e instanceof Error?e.stack:e}`),T||(T=setTimeout(()=>{p>3&&i.error("main",`Previous exception repeated ${p} times total`),u="",p=0,T=void 0},1e4).unref())),!ye(e)&&(q(e,"uncaughtException"),P("uncaughtException"))}),process.on("unhandledRejection",e=>{i.error("main",`Unhandled rejection: ${e}`),!ye(e)&&(q(e,"unhandledRejection"),P("unhandledRejection"))});const _=Ke(s.dataDir);await _.clearRuntimeState();const I=new Xe(D.join(s.dataDir,"proxy","relay-state.json"),{warn:e=>i.warn("main",e)}),G=new Ve(D.join(s.dataDir,"secrets"));n.setRelaySecretResolver(e=>G.get(e));const U={hasDirectConfig:e=>n.hasDirectRelayConfig(e),clearDirectConfig:(e,o)=>n.setAgentProvider(e,void 0,{deferRestart:o?.deferRestart===!0})},j=()=>({pm:f(),agents:n,direct:U,store:I,secrets:G}),$=_.getRelayAgents();if($.length>0)try{await _.startIfAnyRelayAgent(),i.info("main",`mitm proxy started for relay agents: ${$.join(", ")} (per-agent; others stay direct)`)}catch(e){i.error("main",`mitm proxy FAILED to start (relay agents: ${$.join(", ")}): ${e}. These agents will REFUSE to start (they must not silently fall back to your own account). Fix the proxy or turn relay off for them. Other agents are unaffected.`)}else i.info("main",'mitm proxy idle (no agent has Grix relay enabled; enable: PUT /api/proxy/agents/<name>/enabled {"enabled":true})');let L=null;if(f()){const e=async()=>{const t=f();t&&await le(t,n)&&i.info("main","no agent uses Grix relay anymore; mitm proxy stopped")};n.setRelayEnvSettledHandler(()=>{const t=f();t&&ce(t,U,I),e().catch(a=>{i.warn("main",`failed to stop idle mitm proxy: ${a}`)})});const o=()=>{const t=f(),a=t.getRelayAgents(),l=t.getDegradedRelayAgents();return{enabled:a.length>0,relayAgents:a,staleRelayAgents:n.getAgentsWithStaleRelayEnv(),degradedRelayAgents:l,relayState:I.list(),runtime:t.getRuntimeInfo(),config:t.getConfigSnapshot()}},c=new qe,A={status:()=>o(),setAgentRelay:async(t,a,l)=>{const g=f(),w={info:h=>i.info("main",h),warn:h=>i.warn("main",h)},y=h=>n.getAgentsStatus().find(E=>E.name===h),m=y(t);if(!a){c.cancel(t);const h=await c.runExclusive(t,async()=>{const E=await Y(j(),w,{agentName:t,target:"off"});return{...o(),restarted:E.restarted,busy:E.busy,...E.pending?{pending:E.pending}:{}}});return n.reportRelayStateLocalChange(t),h}if(!m)throw Object.assign(new Error(`Agent "${t}" not found`),{code:"NOT_FOUND"});if(!_e(m.clientType)&&se(m.clientType).length===0)throw Object.assign(new Error(`client type "${m.clientType??""}" does not support Grix relay`),{code:"UNSUPPORTED_CLIENT_TYPE"});const F=await c.runExclusive(t,async()=>{const h=c.epoch(t),E=y(t);if(!E)throw Object.assign(new Error(`Agent "${t}" not found`),{code:"NOT_FOUND"});const Z=String(E.agentId??"").trim();if(!Z)throw new pe(`local agent "${t}" has no agent_id registered`,"MISSING_AGENT_ID");const ee=c.waitCancelled(t,h);let C;try{C=await Promise.race([n.fetchRelayCredential(t,{model:l}),ee.promise.then(()=>{throw new ge("relay enable cancelled by a later disable","CANCELLED")})])}finally{ee.dispose()}c.assertCurrent(t,h);const{restarted:Se,busy:Ae,pending:te}=await de(g,n,y,w,{localName:t,agentId:Z,virtualKey:C.apiKey,anthropicBaseUrl:C.anthropicBaseUrl,openaiBaseUrl:C.openaiBaseUrl,model:l,...C.directRelay!==void 0?{directRelay:C.directRelay}:{}},(Ee,ve,De)=>n.setAgentProvider(Ee,ve,{deferRestart:De?.deferRestart===!0}),{store:I,secrets:G,direct:U});if(c.epoch(t)!==h)throw await Y(j(),w,{agentName:t,target:"off"}),new ge("relay enable cancelled by a later disable","CANCELLED");return{...o(),restarted:Se,busy:Ae,...te?{pending:te}:{}}});return n.reportRelayStateLocalChange(t),F},setAgentRelayCredential:async(t,a)=>{const l=f(),g=n.getAgentsStatus(),{restarted:w,busy:y,pending:m}=await de(l,n,v=>g.find(F=>F.name===v),{info:v=>i.info("main",v),warn:v=>i.warn("main",v)},{localName:t,agentId:a.agentId,virtualKey:a.virtualKey,anthropicBaseUrl:a.anthropicBaseUrl,openaiBaseUrl:a.openaiBaseUrl,model:a.model,...a.directRelay!==void 0?{directRelay:a.directRelay}:{}},(v,F,h)=>n.setAgentProvider(v,F,{deferRestart:h?.deferRestart===!0}),{store:I,secrets:G,direct:U});return n.reportRelayStateLocalChange(t),{...o(),restarted:w,busy:y,...m?{pending:m}:{}}},disableAll:async()=>{const t=f(),a={info:g=>i.info("main",g),warn:g=>i.warn("main",g)},l=new Set([...t.getRelayAgents(),...n.getDirectRelayConfigAgentNames(),...Object.keys(I.list())]);for(const g of l)c.cancel(g);for(const g of l)await c.runExclusive(g,()=>Y(j(),a,{agentName:g,target:"off"})),n.reportRelayStateLocalChange(g);return await le(t,n),o()},setRoute:async(t,a)=>{const l=a,g=f();return g.setRoute({routeKey:t,targetBaseUrl:l.targetBaseUrl,...l.headers?{headers:l.headers}:{},...l.model?{model:l.model}:{},...l.modelMap?{modelMap:l.modelMap}:{},...l.passthrough?{passthrough:!0}:{},...l.relayOnly?{relayOnly:!0}:{},...l.codexResponsesToChat?{codexResponsesToChat:!0}:{}}),await g.persistConfig(),o()},deleteRoute:async t=>{const a=f();a.deleteRoute(t),await a.persistConfig()},setDefaultRoute:async t=>{const a=f();return a.setDefaultRouteKey(t),await a.persistConfig(),o()},setInterceptHosts:async t=>{const a=f();return a.setInterceptHosts(t),await a.persistConfig(),o()}};L=A,b.setProxyHandler(A),n.setRelayStateApplier({getLocalState:t=>{const a=f();if(a.getRelayAgents().includes(t)){const g=n.getAgentsStatus().find(m=>m.name===t),w=`grix-gateway-${String(g?.clientType??"").trim().toLowerCase()}`;return{enabled:!0,model:a.getConfigSnapshot().routes?.find(m=>m.routeKey===w)?.model?.trim()||void 0}}const l=n.getAgentProvider(t);return l?.api_key?.trim()?{enabled:!0,model:l.model?.trim()||void 0}:{enabled:!1}},applyEnable:async(t,a,l)=>{const g=n.getAgentsStatus().find(m=>m.name===t),w=String(g?.agentId??"").trim();if(!w)throw new pe(`local agent "${t}" has no agent_id registered`,"MISSING_AGENT_ID");const y=await A.setAgentRelayCredential(t,{agentId:w,virtualKey:a.apiKey,anthropicBaseUrl:a.anthropicBaseUrl,openaiBaseUrl:a.openaiBaseUrl,model:l??a.model,...a.directRelay!==void 0?{directRelay:a.directRelay}:{}});return{restarted:y.restarted===!0,busy:y.busy===!0,...y.pending?{pending:y.pending}:{}}},applyDisable:async t=>{const a=await A.setAgentRelay(t,!1);return{restarted:a.restarted===!0,busy:a.busy===!0,...a.pending?{pending:a.pending}:{}}}})}if(M.setStatusProvider(()=>n.getAgentsStatus()),M.setMetaProvider(()=>{const e=n.getAgentsStatus();return{ws:{connected:e.filter(o=>o.wsConnected===!0).length,total:e.length},upgrade:et()}}),await n.start(nt),$.length>0){const e=n.getAgentsStatus();for(const o of $){const c=e.find(t=>t.name===o);if(!c)continue;const A=se(c.clientType);A.length!==0&&await _.setAgentRelayEnabled(o,!0,{relayHosts:A}).catch(t=>{i.warn("main",`failed to reconcile relay hosts for "${o}": ${t}`)})}}ce(_,U,I);const z=parseInt(R["admin-port"]??process.env.GRIX_ADMIN_PORT??"19580",10);b.setAgentHandler({list:()=>n.getAgentsStatus(),add:e=>n.addAgent(e),remove:e=>n.removeAgent(e),restart:e=>n.restartAgent(e),reload:()=>n.reload()}),b.setUpgradeHandler({check:()=>n.checkUpgrade(),trigger:()=>n.triggerUpgrade()}),b.setProbeHandler({probeAll:e=>n.probeAll(e),probeOne:(e,o)=>n.probeOne(e,o)}),b.setInstallHandler({listInstallable:()=>n.listInstallable(),installAgent:e=>n.installAgent(e),getInstallProgress:e=>n.getInstallProgress(e)});const Q={listAgents:()=>n.getAgentsStatus(),enableManaged:(e,o,c)=>{if(!L)throw new ue(`agent "${e}" needs the MITM proxy for relay, but it is not running`,"PROXY_UNAVAILABLE");return L.setAgentRelayCredential(e,{agentId:o,virtualKey:c.virtualKey,anthropicBaseUrl:c.anthropicBaseUrl,openaiBaseUrl:c.openaiBaseUrl,model:c.model,...c.directRelay!==void 0?{directRelay:c.directRelay}:{}})},disableManaged:e=>{if(!L)throw new ue(`agent "${e}" needs the MITM proxy for relay, but it is not running`,"PROXY_UNAVAILABLE");return L.setAgentRelay(e,!1)},enableHermesProfile:(e,o)=>Ye({agentId:e,baseUrl:o.openaiBaseUrl??"",virtualKey:o.virtualKey,model:o.model??""}),disableHermesProfile:e=>Je(e)};b.setRelayCredentialHandler({listHermesProfiles:()=>ze(),enable:async(e,o)=>{const{target:c,result:A}=await Ze(Q,e,o);return{target:c,...A}},disable:async e=>{const{target:o,result:c}=await Qe(Q,e);return{target:o,...c}}});{const e=await oe({label:"admin",port:z,envVar:"GRIX_ADMIN_PORT",cliFlag:"admin-port",start:o=>b.start(o)});e&&await he(e)}const we=D.join(s.dataDir,"admin-token"),Re=D.join(s.dataDir,"admin-port");je(we,me),re(Re,String(z),"utf-8"),await N(s.daemonStatusFile,{state:"running",pid:process.pid,instance_id:k,updated_at:Date.now()}),process.send&&process.send("ready"),i.info("main","grix-connector ready")}ot().catch(async r=>{const d=`Fatal: ${r instanceof Error?r.stack??r.message:String(r)}`;try{process.stderr.write(`${d}
|
|
42
|
-
`)}catch{}ne(!1),i.error("main",
|
|
39
|
+
right-click the terminal and select "Run as administrator".`);const r=W(),p=R["config-dir"]??(R.profile?D.join(r.configDir,R.profile):void 0),w=D.resolve(process.argv[1]||`${r.rootDir}/dist/grix.js`),u=new ae({cliPath:w,nodePath:process.execPath});try{let c;switch(x){case"start":(await u.status({rootDir:r.rootDir})).installed?c=await u.start({rootDir:r.rootDir}):c=await u.install({rootDir:r.rootDir,configDir:p});break;case"stop":c=await u.stop({rootDir:r.rootDir});break;case"restart":(await u.status({rootDir:r.rootDir})).installed?c=await u.restart({rootDir:r.rootDir}):c=await u.install({rootDir:r.rootDir,configDir:p});break;case"status":c=await u.status({rootDir:r.rootDir});break}console.log(JSON.stringify(c,null,2)),process.exit(0)}catch(c){console.error(`${x} failed: ${c instanceof Error?c.message:c}`),process.exit(1)}}else x&&(console.error(`Unknown command: ${x}
|
|
40
|
+
Valid commands: ${fe.join(", ")}`),process.exit(1));const s=W(),at=R["config-dir"]??(R.profile?`${s.configDir}/${R.profile}`:void 0),n=new xe({requestGracefulShutdown:r=>{P(r)},onTransactionSettled:()=>{K({rootDir:s.rootDir,requestGracefulShutdown:()=>{P("wrapper-migration")}}).then(r=>{r.handoffStarted&&i.info("main","Wrapper handoff started after the upgrade transaction settled")}).catch(r=>{i.warn("main",`Post-upgrade wrapper migration failed: ${r instanceof Error?r.message:r}`)})}}),M=new ke,me=We(),b=new je(me);let B=!1,C;async function he(r){process.stderr.write(r.message+`
|
|
41
|
+
`),i.error("main",r.message.replace(/\n/g," \u2014 "));const p=X(s.daemonLockFile);p?.pid===process.pid&&p.instance_id===C&&await N(s.daemonStatusFile,{state:"failed",pid:process.pid,instance_id:C,updated_at:Date.now(),reason:`port_bind_${r.kind}:${r.label}:${r.port}`}).catch(()=>{}),await V(),H(s.daemonLockFile).catch(()=>{}),O(),process.exit(1)}async function P(r){if(B)return;B=!0,i.info("main",`Received ${r}, shutting down...`),M.markShuttingDown();const p=$e();p&&Ue(p);const w=r==="upgrade-switch"?ie:2,u=setTimeout(()=>{i.error("main","Shutdown timed out, forcing exit"),H(s.daemonLockFile).catch(()=>{}),O(),process.exit(w)},1e4);try{await n.stop(),await f()?.stop().catch(()=>{}),await b.stop(),await M.stop(),await V(),await H(s.daemonLockFile),await Be(s.daemonStatusFile).catch(()=>{}),clearTimeout(u),O(),i.info("main","Shutdown complete"),process.exit(r==="upgrade-switch"?ie:0)}catch(c){i.error("main",`Shutdown error: ${c}`),H(s.daemonLockFile).catch(()=>{}),O(),process.exit(w)}}async function it(){if(be(),Pe(),Ie(s.stdoutLogFile,s.stderrLogFile),ne(!1),He()){const e=Me(s.rootDir,"win32");await Fe(e,{platform:"win32"})}try{C=(await Ge(s.daemonLockFile,s.rootDir)).instance_id}catch(e){console.error(e instanceof Error?e.message:e),process.exit(1)}if(await Xe(),Ce(),i.info("main",`grix-connector starting (PID ${process.pid})`),await N(s.daemonStatusFile,{state:"starting",pid:process.pid,instance_id:C,updated_at:Date.now()}),process.env.GRIX_LAUNCHER_MANAGED!=="1"&&await new ae({cliPath:D.resolve(process.argv[1]??""),nodePath:process.execPath}).adoptLauncher(s.rootDir).then(e=>{e&&i.info("main","Service definition now points at the version-store launcher")}).catch(e=>{i.warn("main",`Launcher adoption failed: ${e instanceof Error?e.message:e}`)}),(await K({rootDir:s.rootDir,requestGracefulShutdown:()=>{P("wrapper-migration")}}).catch(e=>(i.warn("main",`Windows wrapper migration failed: ${e instanceof Error?e.message:e}`),null)))?.handoffStarted)return;process.platform==="win32"&&process.env.GRIX_LAUNCHER_MANAGED!=="1"&&setInterval(()=>{B||K({rootDir:s.rootDir,requestGracefulShutdown:()=>{P("wrapper-migration")}}).then(o=>{o.handoffStarted&&i.info("main","Wrapper handoff started by the periodic migration retry")}).catch(o=>{i.warn("main",`Periodic wrapper migration failed: ${o instanceof Error?o.message:o}`)})},Le).unref();const p=parseInt(R["health-port"]??process.env.GRIX_HEALTH_PORT??"19579",10);{const e=await oe({label:"health",port:p,envVar:"GRIX_HEALTH_PORT",cliFlag:"health-port",start:o=>M.start(o)});e&&await he(e)}const w=D.join(s.dataDir,"health-port");re(w,String(p),"utf-8"),process.on("SIGINT",()=>P("SIGINT")),process.on("SIGTERM",()=>P("SIGTERM")),process.on("SIGHUP",()=>{B||(i.info("main","Received SIGHUP, reloading config..."),n.reload().then(e=>i.info("main",`reload done: ${JSON.stringify(e)}`)).catch(e=>i.error("main",`reload failed: ${e instanceof Error?e.message:e}`)))});let u="",c=0,T;process.on("uncaughtException",e=>{const o=e instanceof Error?e.stack??e.message:String(e);o===u?(c++,(c<=3||c%100===0)&&i.error("main",`Uncaught exception (x${c}): ${o}`)):(c>3&&i.error("main",`Previous exception repeated ${c} times total`),u=o,c=1,i.error("main",`Uncaught exception: ${e instanceof Error?e.stack:e}`),T||(T=setTimeout(()=>{c>3&&i.error("main",`Previous exception repeated ${c} times total`),u="",c=0,T=void 0},1e4).unref())),!ye(e)&&(q(e,"uncaughtException"),P("uncaughtException"))}),process.on("unhandledRejection",e=>{i.error("main",`Unhandled rejection: ${e}`),!ye(e)&&(q(e,"unhandledRejection"),P("unhandledRejection"))});const _=Ve(s.dataDir);await _.clearRuntimeState();const I=new qe(D.join(s.dataDir,"proxy","relay-state.json"),{warn:e=>i.warn("main",e)}),G=new Ye(D.join(s.dataDir,"secrets"));n.setRelaySecretResolver(e=>G.get(e));const U={hasDirectConfig:e=>n.hasDirectRelayConfig(e),clearDirectConfig:(e,o)=>n.setAgentProvider(e,void 0,{deferRestart:o?.deferRestart===!0})},j=()=>({pm:f(),agents:n,direct:U,store:I,secrets:G}),$=_.getRelayAgents();if($.length>0)try{await _.startIfAnyRelayAgent(),i.info("main",`mitm proxy started for relay agents: ${$.join(", ")} (per-agent; others stay direct)`)}catch(e){i.error("main",`mitm proxy FAILED to start (relay agents: ${$.join(", ")}): ${e}. These agents will REFUSE to start (they must not silently fall back to your own account). Fix the proxy or turn relay off for them. Other agents are unaffected.`)}else i.info("main",'mitm proxy idle (no agent has Grix relay enabled; enable: PUT /api/proxy/agents/<name>/enabled {"enabled":true})');let L=null;if(f()){const e=async()=>{const t=f();t&&await le(t,n)&&i.info("main","no agent uses Grix relay anymore; mitm proxy stopped")};n.setRelayEnvSettledHandler(()=>{const t=f();t&&ce(t,U,I),e().catch(a=>{i.warn("main",`failed to stop idle mitm proxy: ${a}`)})});const o=()=>{const t=f(),a=t.getRelayAgents(),l=t.getDegradedRelayAgents();return{enabled:a.length>0,relayAgents:a,staleRelayAgents:n.getAgentsWithStaleRelayEnv(),degradedRelayAgents:l,relayState:I.list(),runtime:t.getRuntimeInfo(),config:t.getConfigSnapshot()}},d=new Je,A={status:()=>o(),setAgentRelay:async(t,a,l)=>{const g=f(),S={info:h=>i.info("main",h),warn:h=>i.warn("main",h)},y=h=>n.getAgentsStatus().find(E=>E.name===h),m=y(t);if(!a){d.cancel(t);const h=await d.runExclusive(t,async()=>{const E=await Y(j(),S,{agentName:t,target:"off"});return{...o(),restarted:E.restarted,busy:E.busy,...E.pending?{pending:E.pending}:{}}});return n.reportRelayStateLocalChange(t),h}if(!m)throw Object.assign(new Error(`Agent "${t}" not found`),{code:"NOT_FOUND"});if(!_e(m.clientType)&&se(m.clientType).length===0)throw Object.assign(new Error(`client type "${m.clientType??""}" does not support Grix relay`),{code:"UNSUPPORTED_CLIENT_TYPE"});const F=await d.runExclusive(t,async()=>{const h=d.epoch(t),E=y(t);if(!E)throw Object.assign(new Error(`Agent "${t}" not found`),{code:"NOT_FOUND"});const Z=String(E.agentId??"").trim();if(!Z)throw new pe(`local agent "${t}" has no agent_id registered`,"MISSING_AGENT_ID");const ee=d.waitCancelled(t,h);let k;try{k=await Promise.race([n.fetchRelayCredential(t,{model:l}),ee.promise.then(()=>{throw new ge("relay enable cancelled by a later disable","CANCELLED")})])}finally{ee.dispose()}d.assertCurrent(t,h);const{restarted:Re,busy:Ae,pending:te}=await de(g,n,y,S,{localName:t,agentId:Z,virtualKey:k.apiKey,anthropicBaseUrl:k.anthropicBaseUrl,openaiBaseUrl:k.openaiBaseUrl,model:l,...k.directRelay!==void 0?{directRelay:k.directRelay}:{}},(Ee,ve,De)=>n.setAgentProvider(Ee,ve,{deferRestart:De?.deferRestart===!0}),{store:I,secrets:G,direct:U});if(d.epoch(t)!==h)throw await Y(j(),S,{agentName:t,target:"off"}),new ge("relay enable cancelled by a later disable","CANCELLED");return{...o(),restarted:Re,busy:Ae,...te?{pending:te}:{}}});return n.reportRelayStateLocalChange(t),F},setAgentRelayCredential:async(t,a)=>{const l=f(),g=n.getAgentsStatus(),{restarted:S,busy:y,pending:m}=await de(l,n,v=>g.find(F=>F.name===v),{info:v=>i.info("main",v),warn:v=>i.warn("main",v)},{localName:t,agentId:a.agentId,virtualKey:a.virtualKey,anthropicBaseUrl:a.anthropicBaseUrl,openaiBaseUrl:a.openaiBaseUrl,model:a.model,...a.directRelay!==void 0?{directRelay:a.directRelay}:{}},(v,F,h)=>n.setAgentProvider(v,F,{deferRestart:h?.deferRestart===!0}),{store:I,secrets:G,direct:U});return n.reportRelayStateLocalChange(t),{...o(),restarted:S,busy:y,...m?{pending:m}:{}}},disableAll:async()=>{const t=f(),a={info:g=>i.info("main",g),warn:g=>i.warn("main",g)},l=new Set([...t.getRelayAgents(),...n.getDirectRelayConfigAgentNames(),...Object.keys(I.list())]);for(const g of l)d.cancel(g);for(const g of l)await d.runExclusive(g,()=>Y(j(),a,{agentName:g,target:"off"})),n.reportRelayStateLocalChange(g);return await le(t,n),o()},setRoute:async(t,a)=>{const l=a,g=f();return g.setRoute({routeKey:t,targetBaseUrl:l.targetBaseUrl,...l.headers?{headers:l.headers}:{},...l.model?{model:l.model}:{},...l.modelMap?{modelMap:l.modelMap}:{},...l.passthrough?{passthrough:!0}:{},...l.relayOnly?{relayOnly:!0}:{},...l.codexResponsesToChat?{codexResponsesToChat:!0}:{}}),await g.persistConfig(),o()},deleteRoute:async t=>{const a=f();a.deleteRoute(t),await a.persistConfig()},setDefaultRoute:async t=>{const a=f();return a.setDefaultRouteKey(t),await a.persistConfig(),o()},setInterceptHosts:async t=>{const a=f();return a.setInterceptHosts(t),await a.persistConfig(),o()}};L=A,b.setProxyHandler(A),n.setRelayStateApplier({getLocalState:t=>{const a=f();if(a.getRelayAgents().includes(t)){const g=n.getAgentsStatus().find(m=>m.name===t),S=`grix-gateway-${String(g?.clientType??"").trim().toLowerCase()}`;return{enabled:!0,model:a.getConfigSnapshot().routes?.find(m=>m.routeKey===S)?.model?.trim()||void 0}}const l=n.getAgentProvider(t);return l?.api_key?.trim()?{enabled:!0,model:l.model?.trim()||void 0}:{enabled:!1}},applyEnable:async(t,a,l)=>{const g=n.getAgentsStatus().find(m=>m.name===t),S=String(g?.agentId??"").trim();if(!S)throw new pe(`local agent "${t}" has no agent_id registered`,"MISSING_AGENT_ID");const y=await A.setAgentRelayCredential(t,{agentId:S,virtualKey:a.apiKey,anthropicBaseUrl:a.anthropicBaseUrl,openaiBaseUrl:a.openaiBaseUrl,model:l??a.model,...a.directRelay!==void 0?{directRelay:a.directRelay}:{}});return{restarted:y.restarted===!0,busy:y.busy===!0,...y.pending?{pending:y.pending}:{}}},applyDisable:async t=>{const a=await A.setAgentRelay(t,!1);return{restarted:a.restarted===!0,busy:a.busy===!0,...a.pending?{pending:a.pending}:{}}}})}if(M.setStatusProvider(()=>n.getAgentsStatus()),M.setMetaProvider(()=>{const e=n.getAgentsStatus();return{ws:{connected:e.filter(o=>o.wsConnected===!0).length,total:e.length},upgrade:rt()}}),await n.start(at),$.length>0){const e=n.getAgentsStatus();for(const o of $){const d=e.find(t=>t.name===o);if(!d)continue;const A=se(d.clientType);A.length!==0&&await _.setAgentRelayEnabled(o,!0,{relayHosts:A}).catch(t=>{i.warn("main",`failed to reconcile relay hosts for "${o}": ${t}`)})}}ce(_,U,I);const z=parseInt(R["admin-port"]??process.env.GRIX_ADMIN_PORT??"19580",10);b.setAgentHandler({list:()=>n.getAgentsStatus(),add:e=>n.addAgent(e),remove:e=>n.removeAgent(e),restart:e=>n.restartAgent(e),reload:()=>n.reload()}),b.setUpgradeHandler({check:()=>n.checkUpgrade(),trigger:()=>n.triggerUpgrade()}),b.setProbeHandler({probeAll:e=>n.probeAll(e),probeOne:(e,o)=>n.probeOne(e,o)}),b.setInstallHandler({listInstallable:()=>n.listInstallable(),installAgent:e=>n.installAgent(e),getInstallProgress:e=>n.getInstallProgress(e)});const Q={listAgents:()=>n.getAgentsStatus(),enableManaged:(e,o,d)=>{if(!L)throw new ue(`agent "${e}" needs the MITM proxy for relay, but it is not running`,"PROXY_UNAVAILABLE");return L.setAgentRelayCredential(e,{agentId:o,virtualKey:d.virtualKey,anthropicBaseUrl:d.anthropicBaseUrl,openaiBaseUrl:d.openaiBaseUrl,model:d.model,...d.directRelay!==void 0?{directRelay:d.directRelay}:{}})},disableManaged:e=>{if(!L)throw new ue(`agent "${e}" needs the MITM proxy for relay, but it is not running`,"PROXY_UNAVAILABLE");return L.setAgentRelay(e,!1)},enableHermesProfile:(e,o)=>ze({agentId:e,baseUrl:o.openaiBaseUrl??"",virtualKey:o.virtualKey,model:o.model??""}),disableHermesProfile:e=>Qe(e)};b.setRelayCredentialHandler({listHermesProfiles:()=>Ze(),enable:async(e,o)=>{const{target:d,result:A}=await tt(Q,e,o);return{target:d,...A}},disable:async e=>{const{target:o,result:d}=await et(Q,e);return{target:o,...d}}});{const e=await oe({label:"admin",port:z,envVar:"GRIX_ADMIN_PORT",cliFlag:"admin-port",start:o=>b.start(o)});e&&await he(e)}const we=D.join(s.dataDir,"admin-token"),Se=D.join(s.dataDir,"admin-port");Ke(we,me),re(Se,String(z),"utf-8"),await N(s.daemonStatusFile,{state:"running",pid:process.pid,instance_id:C,updated_at:Date.now()}),process.send&&process.send("ready"),i.info("main","grix-connector ready")}it().catch(async r=>{const p=`Fatal: ${r instanceof Error?r.stack??r.message:String(r)}`;try{process.stderr.write(`${p}
|
|
42
|
+
`)}catch{}ne(!1),i.error("main",p),q(r,"startup");const w=X(s.daemonLockFile);w?.pid===process.pid&&w.instance_id===C&&await N(s.daemonStatusFile,{state:"failed",pid:process.pid,instance_id:w.instance_id,updated_at:Date.now(),reason:"startup_error"}).catch(()=>{}),await V(),H(s.daemonLockFile).catch(()=>{}),O(),process.exit(1)});const st=new Set(["ECONNRESET","ECONNREFUSED","ETIMEDOUT","EPIPE","EAI_AGAIN","ENOTFOUND","EHOSTUNREACH","ENETUNREACH","EIO"]);function ye(r){return r instanceof Error&&"code"in r?st.has(r.code):!1}
|
package/dist/launcher.mjs
CHANGED
|
@@ -7,13 +7,16 @@
|
|
|
7
7
|
// - 退出码 75(RESTART_EXIT_CODE):升级切换 / 重载请求,立即重新读指针拉起;
|
|
8
8
|
// - 退出码 0 或被信号终止:daemon 是被要求停下的,launcher 随之退出,保活与否交给
|
|
9
9
|
// 上层 supervisor(launchd KeepAlive / Windows wrapper),语义与 launcher 出现前一致;
|
|
10
|
-
// 否则 `grix-connector stop` 的兜底强杀会被 launcher
|
|
10
|
+
// 否则 `grix-connector stop` 的兜底强杀会被 launcher 复活。不是 launcher 转发的信号
|
|
11
|
+
// (OOM/SIGKILL 等)仍按崩溃记账,让 supervisor 重拉后的 launcher 能接着数;daemon 进入
|
|
12
|
+
// 优雅关停时会写 `<store>/managed-stop`,关停卡住后被兜底强杀的不算崩溃。
|
|
11
13
|
// - 其他非零退出:按退避重启;若该版本从未写过 `<store>/health/<ver>.ok` 且 5 分钟内快崩
|
|
12
14
|
// 3 次,则把 `current` 切回 `previous`,写 `<store>/rollback.json` 供旧版本上报。
|
|
15
|
+
// 崩溃时间戳落在 `<store>/crashes.json`,launcher 自身被重拉也不会把计数归零。
|
|
13
16
|
// 收到 SIGTERM/SIGINT/SIGHUP 时转发给子进程并随其退出(服务停止语义)。
|
|
14
17
|
//
|
|
15
18
|
// 本文件由 daemon 在启动时按 LAUNCHER_TEMPLATE_VERSION 比对并原子重写,改动行为时必须 +1。
|
|
16
|
-
export const LAUNCHER_TEMPLATE_VERSION =
|
|
19
|
+
export const LAUNCHER_TEMPLATE_VERSION = 2;
|
|
17
20
|
|
|
18
21
|
import { spawn } from 'node:child_process';
|
|
19
22
|
import { existsSync, mkdirSync, readFileSync, realpathSync, renameSync, rmSync, writeFileSync, appendFileSync } from 'node:fs';
|
|
@@ -48,6 +51,79 @@ function writePointer(store, name, value) {
|
|
|
48
51
|
renameSync(tmp, file);
|
|
49
52
|
}
|
|
50
53
|
|
|
54
|
+
function crashFile(store) {
|
|
55
|
+
return join(store, 'crashes.json');
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
/** daemon 自己在优雅关停开始时写下的标记:随后的信号致死是管理性终止,不记崩溃。 */
|
|
59
|
+
function consumeManagedStop(store) {
|
|
60
|
+
const file = join(store, 'managed-stop');
|
|
61
|
+
const present = existsSync(file);
|
|
62
|
+
rmSync(file, { force: true });
|
|
63
|
+
return present;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/** 读取该版本在快崩窗口内的崩溃时间戳;文件属于别的版本或损坏时视为空。 */
|
|
67
|
+
function loadCrashes(store, version) {
|
|
68
|
+
try {
|
|
69
|
+
const parsed = JSON.parse(readFileSync(crashFile(store), 'utf8'));
|
|
70
|
+
if (parsed?.version !== version || !Array.isArray(parsed.at)) return [];
|
|
71
|
+
const now = Date.now();
|
|
72
|
+
return parsed.at.filter((at) => typeof at === 'number' && now - at < RAPID_CRASH_WINDOW_MS);
|
|
73
|
+
} catch {
|
|
74
|
+
return [];
|
|
75
|
+
}
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
function saveCrashes(store, version, crashes) {
|
|
79
|
+
try {
|
|
80
|
+
if (crashes.length === 0) {
|
|
81
|
+
rmSync(crashFile(store), { force: true });
|
|
82
|
+
return;
|
|
83
|
+
}
|
|
84
|
+
const tmp = `${crashFile(store)}.tmp-${process.pid}`;
|
|
85
|
+
writeFileSync(tmp, JSON.stringify({ version, at: crashes }), 'utf8');
|
|
86
|
+
renameSync(tmp, crashFile(store));
|
|
87
|
+
} catch { /* best effort */ }
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* 记一次未达健康的崩溃;达到阈值且有可用的 previous 时切回并写 rollback.json。
|
|
92
|
+
* 返回 true 表示已回滚(调用方应立即按新指针重拉)。
|
|
93
|
+
*/
|
|
94
|
+
function noteCrashAndMaybeRollback(store, version, code, startedAt) {
|
|
95
|
+
if (existsSync(join(store, 'health', `${version}.ok`))) {
|
|
96
|
+
saveCrashes(store, version, []);
|
|
97
|
+
return false;
|
|
98
|
+
}
|
|
99
|
+
const now = Date.now();
|
|
100
|
+
const crashes = loadCrashes(store, version);
|
|
101
|
+
crashes.push(now);
|
|
102
|
+
saveCrashes(store, version, crashes);
|
|
103
|
+
if (crashes.length < RAPID_CRASH_LIMIT) return false;
|
|
104
|
+
const previous = readPointer(store, 'previous');
|
|
105
|
+
if (!previous || previous === version || !existsSync(entryFor(store, previous))) {
|
|
106
|
+
log(store, `${version} keeps crashing and there is no previous version to roll back to`);
|
|
107
|
+
return false;
|
|
108
|
+
}
|
|
109
|
+
log(store, `${version} crashed ${crashes.length}x without becoming healthy; rolling back to ${previous}`);
|
|
110
|
+
const record = {
|
|
111
|
+
from_version: previous,
|
|
112
|
+
to_version: version,
|
|
113
|
+
exit_code: code,
|
|
114
|
+
crash_count: crashes.length,
|
|
115
|
+
uptime_ms: now - startedAt,
|
|
116
|
+
rolled_back_at: new Date().toISOString(),
|
|
117
|
+
};
|
|
118
|
+
const rollbackFile = join(store, 'rollback.json');
|
|
119
|
+
writeFileSync(`${rollbackFile}.tmp-${process.pid}`, JSON.stringify(record), 'utf8');
|
|
120
|
+
renameSync(`${rollbackFile}.tmp-${process.pid}`, rollbackFile);
|
|
121
|
+
writePointer(store, 'current', previous);
|
|
122
|
+
rmSync(join(store, 'previous'), { force: true });
|
|
123
|
+
saveCrashes(store, version, []);
|
|
124
|
+
return true;
|
|
125
|
+
}
|
|
126
|
+
|
|
51
127
|
function entryFor(store, version) {
|
|
52
128
|
return join(store, 'versions', version, 'dist', 'grix.js');
|
|
53
129
|
}
|
|
@@ -107,7 +183,6 @@ async function main() {
|
|
|
107
183
|
const store = resolveStoreDir();
|
|
108
184
|
const args = process.argv.slice(2);
|
|
109
185
|
mkdirSync(join(store, 'health'), { recursive: true });
|
|
110
|
-
let crashes = [];
|
|
111
186
|
let backoffMs = 1000;
|
|
112
187
|
log(store, `started node=${process.version} store=${store} template=${LAUNCHER_TEMPLATE_VERSION}`);
|
|
113
188
|
|
|
@@ -125,47 +200,35 @@ async function main() {
|
|
|
125
200
|
}
|
|
126
201
|
}
|
|
127
202
|
|
|
203
|
+
// launcher 自己被 supervisor 重拉时,上一轮落盘的崩溃可能已经够数(例如被 OOM 杀了三次)。
|
|
204
|
+
if (loadCrashes(store, version).length >= RAPID_CRASH_LIMIT
|
|
205
|
+
&& noteCrashAndMaybeRollback(store, version, null, Date.now())) {
|
|
206
|
+
continue;
|
|
207
|
+
}
|
|
208
|
+
|
|
128
209
|
const startedAt = Date.now();
|
|
210
|
+
consumeManagedStop(store);
|
|
129
211
|
log(store, `starting ${version}`);
|
|
130
212
|
const { code, signal, stopping } = await runChild(store, version, args);
|
|
131
213
|
log(store, `daemon ${version} exited code=${code} signal=${signal ?? '-'}`);
|
|
132
|
-
|
|
214
|
+
const managedStop = consumeManagedStop(store);
|
|
215
|
+
if (stopping || code === 0) process.exit(typeof code === 'number' ? code : 0);
|
|
216
|
+
if (signal) {
|
|
217
|
+
// 不是我们转发的信号:除非 daemon 已进入优雅关停(管理性终止的兜底强杀),
|
|
218
|
+
// 否则记一次崩溃;随后仍交还 supervisor,重拉后的 launcher 接着数。
|
|
219
|
+
if (managedStop) log(store, `${version} was terminated during a managed stop; not counting as a crash`);
|
|
220
|
+
else noteCrashAndMaybeRollback(store, version, code, startedAt);
|
|
221
|
+
process.exit(code);
|
|
222
|
+
}
|
|
133
223
|
if (code === RESTART_EXIT_CODE) {
|
|
134
|
-
|
|
224
|
+
saveCrashes(store, version, []);
|
|
135
225
|
backoffMs = 1000;
|
|
136
226
|
continue;
|
|
137
227
|
}
|
|
138
228
|
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
crashes = crashes.filter((at) => now - at < RAPID_CRASH_WINDOW_MS);
|
|
143
|
-
crashes.push(now);
|
|
144
|
-
if (crashes.length >= RAPID_CRASH_LIMIT) {
|
|
145
|
-
const previous = readPointer(store, 'previous');
|
|
146
|
-
if (previous && previous !== version && existsSync(entryFor(store, previous))) {
|
|
147
|
-
log(store, `${version} crashed ${crashes.length}x without becoming healthy; rolling back to ${previous}`);
|
|
148
|
-
const record = {
|
|
149
|
-
from_version: previous,
|
|
150
|
-
to_version: version,
|
|
151
|
-
exit_code: code,
|
|
152
|
-
crash_count: crashes.length,
|
|
153
|
-
uptime_ms: now - startedAt,
|
|
154
|
-
rolled_back_at: new Date().toISOString(),
|
|
155
|
-
};
|
|
156
|
-
const rollbackFile = join(store, 'rollback.json');
|
|
157
|
-
writeFileSync(`${rollbackFile}.tmp-${process.pid}`, JSON.stringify(record), 'utf8');
|
|
158
|
-
renameSync(`${rollbackFile}.tmp-${process.pid}`, rollbackFile);
|
|
159
|
-
writePointer(store, 'current', previous);
|
|
160
|
-
rmSync(join(store, 'previous'), { force: true });
|
|
161
|
-
crashes = [];
|
|
162
|
-
backoffMs = 1000;
|
|
163
|
-
continue;
|
|
164
|
-
}
|
|
165
|
-
log(store, `${version} keeps crashing and there is no previous version to roll back to`);
|
|
166
|
-
}
|
|
167
|
-
} else {
|
|
168
|
-
crashes = [];
|
|
229
|
+
if (noteCrashAndMaybeRollback(store, version, code, startedAt)) {
|
|
230
|
+
backoffMs = 1000;
|
|
231
|
+
continue;
|
|
169
232
|
}
|
|
170
233
|
|
|
171
234
|
await sleep(backoffMs);
|