@livx.cc/agentx 0.99.47 → 0.99.49

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -533,8 +533,336 @@ async function findSandboxWrapper(platform = process.platform) {
533
533
  return null;
534
534
  }
535
535
 
536
+ // src/tools.jobs.ts
537
+ var log5 = forComponent("jobs");
538
+ var JobRegistryOptions = class {
539
+ /** Tail buffer cap per job (chars). */
540
+ maxBuffer = 256 * 1024;
541
+ /** Idle window per kind: no output/progress/CPU for this long ⇒ stuck. Kinds without an activity channel
542
+ * (host, native) are absent on purpose — silence is all they can ever show, so it proves nothing; their
543
+ * bound is the hard cap. Explicit `background:true` shell jobs opt out per job (a quiet server is healthy).
544
+ * NOTE the asymmetry that default hides: for a HOST job silence really is uninformative (the caller holds
545
+ * the pending dispatch promise — independent proof it is alive), but a NATIVE job has no such proof AND it
546
+ * suppresses the stall watchdog, i.e. it turns the provider's own silence into permission to keep ignoring
547
+ * that provider's silence. So Agent passes an explicit per-job `idleMs` (nativeHoldMs) for `native`: the
548
+ * provider must keep re-proving liveness with `running` events or forfeit the hold. See Agent.nativeHoldMs. */
549
+ idleMs = { shell: 10 * 6e4, mcp: 10 * 6e4 };
550
+ /** Absolute ceiling per kind (0/absent = none). */
551
+ hardCapMs = { mcp: 30 * 6e4 };
552
+ /** Soft foreground timeout per kind: past it the caller stops waiting and gets a job id instead. */
553
+ foregroundMs = { mcp: 3e4 };
554
+ /** How often the reaper runs while reapable jobs exist. */
555
+ sweepMs = 15e3;
556
+ /** Grace between the kill signal and a forced kill (producers that escalate read this). */
557
+ killGraceMs = 5e3;
558
+ /** Finished jobs kept for inspection; older finished ones are pruned (every delegated tool call is a job). */
559
+ keepFinished = 200;
560
+ /** Clock (injectable for tests). Late-bound so a patched Date.now is honoured. */
561
+ now = () => Date.now();
562
+ /** Terminal-state notification for jobs with `notify` (the Agent wires this to `inject`). */
563
+ onExit;
564
+ };
565
+ var secs = (ms) => `${Math.round(ms / 1e3)}s`;
566
+ var JobRegistry = class {
567
+ options;
568
+ jobs = /* @__PURE__ */ new Map();
569
+ seq = 0;
570
+ timer;
571
+ constructor(options) {
572
+ this.options = { ...new JobRegistryOptions(), ...options };
573
+ }
574
+ /** Register externally-driven work (a process, a request, a native tool) and get a handle to feed it. */
575
+ track(opts = {}) {
576
+ const o = this.options;
577
+ const kind = opts.kind ?? "task";
578
+ const now = o.now();
579
+ let settle;
580
+ const done = new Promise((r) => {
581
+ settle = r;
582
+ });
583
+ const job = {
584
+ id: `job-${++this.seq}`,
585
+ kind,
586
+ label: opts.label ?? "",
587
+ status: "running",
588
+ startedAt: now,
589
+ lastActivityAt: now,
590
+ bytes: 0,
591
+ holds: !!opts.holds,
592
+ buf: "",
593
+ dropped: 0,
594
+ idleMs: opts.idleMs ?? o.idleMs[kind] ?? 0,
595
+ hardCapMs: opts.hardCapMs ?? o.hardCapMs[kind] ?? 0,
596
+ drain: !!opts.drain,
597
+ notify: !!opts.notify,
598
+ idleAfterActivity: !!opts.idleAfterActivity,
599
+ active: false,
600
+ cpuTime: opts.cpuTime,
601
+ onKill: opts.onKill,
602
+ onExit: opts.onExit,
603
+ done,
604
+ settle
605
+ };
606
+ this.jobs.set(job.id, job);
607
+ if (opts.seed) this.append(job, opts.seed);
608
+ this.arm();
609
+ return {
610
+ id: job.id,
611
+ chunk: (s) => {
612
+ if (s) {
613
+ this.append(job, s);
614
+ job.lastActivityAt = o.now();
615
+ job.active = true;
616
+ }
617
+ },
618
+ touch: () => {
619
+ job.lastActivityAt = o.now();
620
+ job.active = true;
621
+ },
622
+ finish: (status, result, reason) => this.end(job, status, result, reason),
623
+ detach: () => {
624
+ job.holds = false;
625
+ job.notify = true;
626
+ job.drain = true;
627
+ },
628
+ get status() {
629
+ return job.status;
630
+ }
631
+ };
632
+ }
633
+ /** Start `fn` in the background and return a job id immediately (sandbox/subagent semantics: drained at turn end). */
634
+ start(fn, opts = {}) {
635
+ return this.run(fn, { drain: true, ...opts }).handle.id;
636
+ }
637
+ /**
638
+ * Run `fn` in the FOREGROUND with a soft timeout: resolves `{done:true,value}` if it settles in time, else
639
+ * `{done:false,id}` — the work keeps going as a detached job whose completion is reported via `onExit`.
640
+ * A rejection before the timeout is rethrown (the caller's normal error path). `softTimeoutMs` ≤ 0 waits forever.
641
+ */
642
+ async runForeground(fn, opts = {}) {
643
+ const { handle, promise } = this.run(fn, opts);
644
+ const soft = opts.softTimeoutMs ?? this.options.foregroundMs[opts.kind ?? "task"] ?? 0;
645
+ let t;
646
+ const late = soft > 0 ? new Promise((r) => {
647
+ t = setTimeout(() => r("late"), soft);
648
+ }) : new Promise(() => {
649
+ });
650
+ try {
651
+ const r = await Promise.race([promise.then((v) => ({ v }), (e) => ({ e })), late]);
652
+ if (r !== "late" && handle.status !== "running") this.jobs.delete(handle.id);
653
+ if (r === "late") {
654
+ if (handle.status === "running") {
655
+ handle.detach();
656
+ return { done: false, id: handle.id };
657
+ }
658
+ const settled = await promise.then((v) => ({ v }), (e) => ({ e }));
659
+ if ("e" in settled) throw settled.e;
660
+ return { done: true, value: settled.v };
661
+ }
662
+ if ("e" in r) throw r.e;
663
+ return { done: true, value: r.v };
664
+ } finally {
665
+ if (t) clearTimeout(t);
666
+ }
667
+ }
668
+ run(fn, opts) {
669
+ const ctl = new AbortController();
670
+ const handle = this.track({ ...opts, onKill: (reason) => {
671
+ ctl.abort(Object.assign(new Error(reason), { name: "AbortError", code: "job_killed" }));
672
+ opts.onKill?.(reason);
673
+ } });
674
+ let promise;
675
+ try {
676
+ promise = fn({ signal: ctl.signal, onChunk: handle.chunk, touch: handle.touch });
677
+ } catch (e) {
678
+ promise = Promise.reject(e);
679
+ }
680
+ promise.then(
681
+ (res) => {
682
+ if (handle.status === "running" && typeof res === "string" && res) handle.chunk(res);
683
+ handle.finish("done", res);
684
+ },
685
+ (e) => {
686
+ if (handle.status === "running") handle.chunk(`
687
+ [error] ${e?.message ?? e}`);
688
+ handle.finish("error", void 0, e?.message ?? String(e));
689
+ }
690
+ );
691
+ return { handle, promise };
692
+ }
693
+ append(job, s) {
694
+ job.buf += s;
695
+ const over = job.buf.length - this.options.maxBuffer;
696
+ if (over > 0) {
697
+ job.buf = job.buf.slice(over);
698
+ job.dropped += over;
699
+ }
700
+ job.bytes = job.buf.length;
701
+ }
702
+ end(job, status, result, reason) {
703
+ if (job.status !== "running") return;
704
+ job.status = status;
705
+ job.endedAt = this.options.now();
706
+ job.result = result;
707
+ if (reason && status !== "done") job.reason = reason;
708
+ job.holds = false;
709
+ job.settle();
710
+ this.prune();
711
+ const onExit = job.onExit ?? this.options.onExit;
712
+ if (job.notify && onExit) {
713
+ try {
714
+ onExit({ id: job.id, kind: job.kind, label: job.label, status, reason: job.reason, tail: job.buf.slice(-4e3), result });
715
+ } catch (e) {
716
+ log5.warn(`job onExit handler threw for ${job.id}: ${e}`);
717
+ }
718
+ }
719
+ }
720
+ prune() {
721
+ const finished = [...this.jobs.values()].filter((j) => j.status !== "running");
722
+ for (const j of finished.slice(0, Math.max(0, finished.length - this.options.keepFinished))) this.jobs.delete(j.id);
723
+ }
724
+ /** Current tail output (null = no such job). */
725
+ output(id) {
726
+ return this.jobs.get(id)?.buf ?? null;
727
+ }
728
+ /** Incremental read: output from absolute char `offset` (clamped to what the ring still holds). */
729
+ read(id, offset = 0) {
730
+ const j = this.jobs.get(id);
731
+ if (!j) return null;
732
+ const start = Math.max(0, offset - j.dropped);
733
+ return { text: j.buf.slice(start), next: j.dropped + j.buf.length, truncated: offset < j.dropped };
734
+ }
735
+ get(id) {
736
+ const j = this.jobs.get(id);
737
+ return j ? this.info(j) : null;
738
+ }
739
+ status(id) {
740
+ const j = this.jobs.get(id);
741
+ return j ? { status: j.status, bytes: j.bytes, ...j.reason ? { reason: j.reason } : {} } : null;
742
+ }
743
+ list() {
744
+ return [...this.jobs.values()].map((j) => this.info(j));
745
+ }
746
+ info(j) {
747
+ const { id, kind, label, status, startedAt, lastActivityAt, endedAt, reason, bytes, holds, result } = j;
748
+ return { id, kind, label, status, startedAt, lastActivityAt, endedAt, reason, bytes, holds, result };
749
+ }
750
+ /** Wait (bounded) for a job to settle. Resolves the job's info either way; null for an unknown id. */
751
+ async wait(id, timeoutMs) {
752
+ const j = this.jobs.get(id);
753
+ if (!j) return null;
754
+ let t;
755
+ await Promise.race([j.done, new Promise((r) => {
756
+ t = setTimeout(r, Math.max(0, timeoutMs));
757
+ })]);
758
+ if (t) clearTimeout(t);
759
+ return this.info(j);
760
+ }
761
+ /** Stop a job: runs its `onKill` (abort/SIGTERM) and marks it killed with `reason`. */
762
+ kill(id, reason = "killed by request") {
763
+ const j = this.jobs.get(id);
764
+ if (!j) return false;
765
+ if (j.status === "running") {
766
+ this.end(j, "killed", void 0, reason);
767
+ try {
768
+ j.onKill?.(reason);
769
+ } catch (e) {
770
+ log5.warn(`onKill for ${id} threw: ${e}`);
771
+ }
772
+ }
773
+ return true;
774
+ }
775
+ /** Kill every running job — or, with `onlyDrainable`, only turn-scoped ones (sandbox/sub-agent/handed-off MCP),
776
+ * leaving jobs meant to outlive a run (a background dev server) alone. */
777
+ killAll(reason = "run ended", onlyDrainable = false) {
778
+ for (const j of this.jobs.values()) if (!onlyDrainable || j.drain) this.kill(j.id, reason);
779
+ }
780
+ /** Jobs something is synchronously waiting on (see `JobTrackOptions.holds`) that are still running. */
781
+ holding() {
782
+ return [...this.jobs.values()].filter((j) => j.status === "running" && j.holds).map((j) => this.info(j));
783
+ }
784
+ /** Await every drainable running job (turn end). Returns the ids that were still running when drain began. */
785
+ async drain(signal) {
786
+ const pending = [...this.jobs.values()].filter((j) => j.status === "running" && j.drain);
787
+ const all = Promise.all(pending.map((j) => j.done));
788
+ await (signal ? Promise.race([all, new Promise((r) => signal.aborted ? r() : signal.addEventListener("abort", () => r(), { once: true }))]) : all);
789
+ return pending.map((j) => j.id);
790
+ }
791
+ /**
792
+ * The stuck-job decision, synchronous part: kill every running job past its hard cap, or past its idle
793
+ * window when it has no CPU probe to argue otherwise. Cheap — the stall watchdog calls it on every tick.
794
+ * Returns the killed ids. Jobs WITH a probe are left to `sweep()`.
795
+ */
796
+ reap() {
797
+ const now = this.options.now();
798
+ const killed = [];
799
+ for (const j of this.jobs.values()) {
800
+ if (j.status !== "running") continue;
801
+ const verdict = this.verdict(j, now);
802
+ if (!verdict || verdict.idle && j.cpuTime) continue;
803
+ this.stuck(j, verdict.reason, now);
804
+ killed.push(j.id);
805
+ }
806
+ return killed;
807
+ }
808
+ /** Full sweep: sample CPU probes (advancing CPU = activity), then `reap()` including probed jobs. */
809
+ async sweep() {
810
+ const now = this.options.now();
811
+ for (const j of [...this.jobs.values()]) {
812
+ if (j.status !== "running" || !j.cpuTime || !j.idleMs) continue;
813
+ let cpu;
814
+ try {
815
+ cpu = await j.cpuTime();
816
+ } catch (e) {
817
+ log5.debug(`cpu probe for ${j.id} failed: ${e}`);
818
+ }
819
+ if (cpu != null && j.lastCpu != null && cpu > j.lastCpu) j.lastActivityAt = this.options.now();
820
+ const prev = j.lastCpu;
821
+ if (cpu != null) j.lastCpu = cpu;
822
+ const verdict = this.verdict(j, now);
823
+ if (!verdict?.idle || j.status !== "running") continue;
824
+ if (cpu == null) {
825
+ log5.debug(`not reaping ${j.id}: ${verdict.reason} but CPU unknown`);
826
+ continue;
827
+ }
828
+ if (prev == null) continue;
829
+ this.stuck(j, `${verdict.reason}, no CPU progress (${prev}s \u2192 ${cpu}s)`, now);
830
+ }
831
+ return this.reap();
832
+ }
833
+ verdict(j, now) {
834
+ if (j.hardCapMs && now - j.startedAt >= j.hardCapMs) return { reason: `stuck: exceeded hard cap ${secs(j.hardCapMs)}`, idle: false };
835
+ if (j.idleMs && (!j.idleAfterActivity || j.active) && now - j.lastActivityAt >= j.idleMs) return { reason: `stuck: idle ${secs(now - j.lastActivityAt)}`, idle: true };
836
+ return null;
837
+ }
838
+ stuck(j, reason, now) {
839
+ log5.warn(`reaping ${j.id} (${j.kind} "${j.label}"): ${reason} \u2014 age ${secs(now - j.startedAt)}, last activity ${secs(now - j.lastActivityAt)} ago, idleMs=${j.idleMs}, hardCapMs=${j.hardCapMs}, ${j.bytes} byte(s) output`);
840
+ if (j.onKill) j.notify = true;
841
+ this.kill(j.id, reason);
842
+ }
843
+ /** Keep a reaper interval alive only while a reapable job is running (unref'd: never holds the process open). */
844
+ arm() {
845
+ if (this.timer || !this.options.sweepMs) return;
846
+ this.timer = setInterval(() => {
847
+ const live = [...this.jobs.values()].some((j) => j.status === "running" && (j.idleMs || j.hardCapMs));
848
+ if (!live) {
849
+ clearInterval(this.timer);
850
+ this.timer = void 0;
851
+ return;
852
+ }
853
+ void this.sweep().catch((e) => log5.warn(`job sweep failed: ${e}`));
854
+ }, this.options.sweepMs);
855
+ this.timer.unref?.();
856
+ }
857
+ /** Stop the reaper timer (tests / teardown). */
858
+ dispose() {
859
+ if (this.timer) clearInterval(this.timer);
860
+ this.timer = void 0;
861
+ }
862
+ };
863
+
536
864
  // src/tools.shell.ts
537
- var log5 = forComponent("shell");
865
+ var log6 = forComponent("shell");
538
866
  var clean = (s) => truncateOutput(redactSecrets(s.replace(/\n+$/, "")));
539
867
  var DETACHED = { stdio: ["ignore", "pipe", "pipe"], detached: true };
540
868
  function killGroup(proc, signal) {
@@ -567,62 +895,122 @@ async function nodeSpawn() {
567
895
  return _spawn;
568
896
  }
569
897
  function formatJobExit(n) {
570
- return `[background job ${n.id} ${n.status}${n.exitCode != null ? ` exit ${n.exitCode}` : ""}] \`${n.command}\`
898
+ return `[background job ${n.id} ${n.status}${n.exitCode != null ? ` exit ${n.exitCode}` : ""}${n.reason ? `: ${n.reason}` : ""}] \`${n.command}\`
571
899
  ` + (n.tail ? `${n.tail}
572
900
  ` : "(no output)\n") + `Read the full output with ShellOutput({id:"${n.id}"}).`;
573
901
  }
902
+ function parsePsTime(t) {
903
+ const m = t.trim().match(/^(?:(\d+)-)?(?:(\d+):)?(\d+):(\d+(?:\.\d+)?)$/);
904
+ if (!m) return void 0;
905
+ return Number(m[1] ?? 0) * 86400 + Number(m[2] ?? 0) * 3600 + Number(m[3]) * 60 + Number(m[4]);
906
+ }
907
+ async function psGroupCpuSeconds(pgid) {
908
+ try {
909
+ const { execFile } = await import("child_process");
910
+ const out = await new Promise((res, rej) => execFile("ps", ["-A", "-o", "pid=,ppid=,pgid=,time="], { timeout: 5e3, env: { ...process.env, LC_ALL: "C" } }, (e, so) => e ? rej(e) : res(String(so))));
911
+ return sumTreeCpu(out, pgid);
912
+ } catch (e) {
913
+ log6.debug(`ps cpu probe failed for group ${pgid}: ${e}`);
914
+ return void 0;
915
+ }
916
+ }
917
+ function sumTreeCpu(psOut, root) {
918
+ const rows = psOut.split("\n").map((l) => l.trim().split(/\s+/)).filter((r) => r.length >= 4).map(([pid, ppid, pgid, time]) => ({ pid: Number(pid), ppid: Number(ppid), pgid: Number(pgid), sec: parsePsTime(time) }));
919
+ const inJob = new Set(rows.filter((r) => r.pgid === root || r.pid === root).map((r) => r.pid));
920
+ for (let grew = true; grew; ) {
921
+ grew = false;
922
+ for (const r of rows) if (!inJob.has(r.pid) && inJob.has(r.ppid)) {
923
+ inJob.add(r.pid);
924
+ grew = true;
925
+ }
926
+ }
927
+ let total;
928
+ for (const r of rows) if (inJob.has(r.pid) && r.sec != null) total = (total ?? 0) + r.sec;
929
+ return total;
930
+ }
574
931
  var ShellJobRegistry = class {
575
932
  constructor(cfg) {
576
933
  this.cfg = cfg;
934
+ this.jobs = cfg.jobs ?? new JobRegistry(cfg.maxBuffer ? { maxBuffer: cfg.maxBuffer } : {});
577
935
  if (cfg.killOnExit && typeof process !== "undefined") process.once("exit", () => this.killAll());
578
936
  }
579
937
  cfg;
580
- jobs = /* @__PURE__ */ new Map();
581
- seq = 0;
938
+ jobs;
939
+ meta = /* @__PURE__ */ new Map();
582
940
  async start(command) {
583
- const id = `job-${++this.seq}`;
584
- const max = this.cfg.maxBuffer ?? 256 * 1024;
585
- const job = { command, buf: "", status: "running" };
586
- const append = (chunk) => {
587
- const s = typeof chunk === "string" ? chunk : chunk?.toString?.("utf8") ?? "";
588
- job.buf = (job.buf + s).slice(-max);
589
- };
941
+ const h = this.track(command, { idleMs: 0 });
590
942
  try {
591
943
  const spawn = this.cfg.spawn ?? await nodeSpawn();
592
944
  const argv = this.cfg.osSandbox ? await spawnArgvFor(command, this.cfg.cwd, this.cfg.osSandbox) : { bin: "/bin/sh", args: ["-c", command] };
593
- const proc = spawn(argv.bin, argv.args, { cwd: this.cfg.cwd, env: childEnv(this.cfg), ...DETACHED });
594
- job.proc = proc;
595
- proc.stdout?.on("data", append);
596
- proc.stderr?.on("data", append);
597
- proc.on("error", (err) => {
598
- if (job.status === "running") {
599
- job.status = "error";
600
- append(`
601
- [error] ${err?.message ?? err}`);
602
- this.notifyExit(id, job);
603
- }
604
- });
605
- proc.on("close", (code) => {
606
- if (job.status === "running") {
607
- job.status = "exited";
608
- job.exitCode = code ?? void 0;
609
- this.notifyExit(id, job);
610
- }
611
- });
945
+ this.attach(h, spawn(argv.bin, argv.args, { cwd: this.cfg.cwd, env: childEnv(this.cfg), ...DETACHED }));
612
946
  } catch (e) {
613
- job.status = "error";
614
- job.buf = `failed to spawn: ${e?.message ?? e}`;
947
+ h.chunk(`failed to spawn: ${e?.message ?? e}`);
948
+ h.finish("error", void 0, e?.message ?? String(e));
949
+ }
950
+ return h.id;
951
+ }
952
+ track(command, extra) {
953
+ const meta = { command };
954
+ let id = "";
955
+ const h = this.jobs.track({
956
+ kind: "shell",
957
+ label: command,
958
+ seed: extra.seed,
959
+ idleMs: extra.idleMs,
960
+ cpuTime: extra.cpu ? async () => meta.proc?.pid ? (this.cfg.cpuTime ?? psGroupCpuSeconds)(meta.proc.pid) : void 0 : void 0,
961
+ onKill: () => this.signal(meta),
962
+ onExit: (n) => this.notifyExit(id, meta, n.status, n.reason),
963
+ notify: true
964
+ });
965
+ id = h.id;
966
+ this.meta.set(id, meta);
967
+ return h;
968
+ }
969
+ attach(h, proc) {
970
+ const meta = this.meta.get(h.id);
971
+ meta.proc = proc;
972
+ const append = (chunk) => h.chunk(typeof chunk === "string" ? chunk : chunk?.toString?.("utf8") ?? "");
973
+ proc.stdout?.on("data", append);
974
+ proc.stderr?.on("data", append);
975
+ proc.on("error", (err) => {
976
+ if (h.status === "running") {
977
+ append(`
978
+ [error] ${err?.message ?? err}`);
979
+ h.finish("error", void 0, err?.message ?? String(err));
980
+ }
981
+ });
982
+ proc.on("close", (code) => {
983
+ if (h.status === "running") {
984
+ meta.exitCode = code ?? void 0;
985
+ h.finish("done");
986
+ }
987
+ });
988
+ }
989
+ /** SIGTERM the whole group (bg children are detached group leaders — /bin/sh alone would orphan a forked
990
+ * server), then SIGKILL after the grace period if it is still there. Falls back to the pid for fakes. */
991
+ signal(meta) {
992
+ const proc = meta.proc;
993
+ if (!killGroup(proc, "SIGTERM")) {
994
+ try {
995
+ proc?.kill("SIGTERM");
996
+ } catch {
997
+ }
615
998
  }
616
- this.jobs.set(id, job);
617
- return id;
618
- }
619
- /** Fire `onExit` at most once per job, with the tail so the model can act without a second round-trip. */
620
- notified = /* @__PURE__ */ new Set();
621
- notifyExit(id, job) {
622
- if (this.notified.has(id) || !this.cfg.onExit) return;
623
- this.notified.add(id);
999
+ if (!proc?.pid) return;
1000
+ let closed = false;
1001
+ proc.on("close", () => {
1002
+ closed = true;
1003
+ });
1004
+ const t = setTimeout(() => {
1005
+ if (!closed) killGroup(proc, "SIGKILL");
1006
+ }, this.jobs.options.killGraceMs);
1007
+ t.unref?.();
1008
+ }
1009
+ /** Fire `onExit` at most once per job (the registry guarantees one terminal state), with the tail so the model can act without a second round-trip. */
1010
+ notifyExit(id, meta, status, reason) {
1011
+ if (!this.cfg.onExit || status === "killed" && !reason?.startsWith("stuck")) return;
624
1012
  try {
625
- this.cfg.onExit({ id, command: job.command, status: job.status, exitCode: job.exitCode, tail: clean(job.buf).slice(-4e3) });
1013
+ this.cfg.onExit({ id, command: meta.command, status: shellStatus(status), exitCode: meta.exitCode, tail: clean(this.jobs.output(id) ?? "").slice(-4e3), ...reason && status !== "done" ? { reason } : {} });
626
1014
  } catch {
627
1015
  }
628
1016
  }
@@ -641,14 +1029,20 @@ var ShellJobRegistry = class {
641
1029
  }
642
1030
  /** Current tail output for a job (null = no such job). */
643
1031
  output(id) {
644
- return this.jobs.get(id)?.buf ?? (this.jobs.has(id) ? "" : null);
1032
+ return this.meta.has(id) ? this.jobs.output(id) : null;
645
1033
  }
1034
+ // null too once the shared registry pruned it
646
1035
  status(id) {
647
- const j = this.jobs.get(id);
648
- return j ? { status: j.status, exitCode: j.exitCode, bytes: j.buf.length } : null;
1036
+ const m = this.meta.get(id);
1037
+ const j = m && this.jobs.get(id);
1038
+ if (!m || !j) {
1039
+ this.meta.delete(id);
1040
+ return null;
1041
+ }
1042
+ return { status: shellStatus(j.status), exitCode: m.exitCode, bytes: j.bytes, ...j.reason ? { reason: j.reason } : {} };
649
1043
  }
650
1044
  list() {
651
- return [...this.jobs].map(([id, j]) => ({ id, command: j.command, status: j.status }));
1045
+ return [...this.meta].filter(([id]) => this.jobs.get(id)).map(([id, m]) => ({ id, command: m.command, status: shellStatus(this.jobs.get(id).status) }));
652
1046
  }
653
1047
  /**
654
1048
  * Take over an ALREADY-RUNNING child as a background job. A foreground command that outruns its
@@ -656,53 +1050,22 @@ var ShellJobRegistry = class {
656
1050
  * worse, a launcher that spawned its own detached worker leaves that worker running with nothing
657
1051
  * tracking it. Adopting hands the model a handle instead: the command keeps going, its completion
658
1052
  * is reported like any other job, and `seed` carries the output produced before the handoff.
1053
+ * Unlike an explicit background job it IS liveness-checked (idle window + CPU probe): nobody asked for
1054
+ * a long-lived process here, so one that goes silent and CPU-idle is stuck, not serving.
659
1055
  */
660
1056
  adopt(command, proc, seed = "") {
661
- const id = `job-${++this.seq}`;
662
- const max = this.cfg.maxBuffer ?? 256 * 1024;
663
- const job = { command, buf: seed.slice(-max), status: "running", proc };
664
- const append = (chunk) => {
665
- const s = typeof chunk === "string" ? chunk : chunk?.toString?.("utf8") ?? "";
666
- job.buf = (job.buf + s).slice(-max);
667
- };
668
- proc.stdout?.on("data", append);
669
- proc.stderr?.on("data", append);
670
- proc.on("error", (err) => {
671
- if (job.status === "running") {
672
- job.status = "error";
673
- append(`
674
- [error] ${err?.message ?? err}`);
675
- this.notifyExit(id, job);
676
- }
677
- });
678
- proc.on("close", (code) => {
679
- if (job.status === "running") {
680
- job.status = "exited";
681
- job.exitCode = code ?? void 0;
682
- this.notifyExit(id, job);
683
- }
684
- });
685
- this.jobs.set(id, job);
686
- return id;
1057
+ const h = this.track(command, { seed, cpu: true });
1058
+ this.attach(h, proc);
1059
+ return h.id;
687
1060
  }
688
- kill(id) {
689
- const j = this.jobs.get(id);
690
- if (!j) return false;
691
- if (j.status === "running") {
692
- if (!killGroup(j.proc, "SIGTERM")) {
693
- try {
694
- j.proc?.kill("SIGTERM");
695
- } catch {
696
- }
697
- }
698
- j.status = "killed";
699
- }
700
- return true;
1061
+ kill(id, reason = "killed by ShellKill") {
1062
+ return this.meta.has(id) && this.jobs.kill(id, reason);
701
1063
  }
702
1064
  killAll() {
703
- for (const id of this.jobs.keys()) this.kill(id);
1065
+ for (const id of this.meta.keys()) this.jobs.kill(id, "agent exiting");
704
1066
  }
705
1067
  };
1068
+ var shellStatus = (s) => s === "done" ? "exited" : s === "running" ? "running" : s === "killed" ? "killed" : "error";
706
1069
  function makeRealShellTool(options) {
707
1070
  const defaultTimeoutMs = options.timeoutMs ?? 12e4;
708
1071
  const maxTimeoutMs = Math.max(options.maxTimeoutMs ?? 6e5, defaultTimeoutMs);
@@ -833,7 +1196,7 @@ function makeRealShellTool(options) {
833
1196
  proc.stderr?.on("data", collect);
834
1197
  proc.on("error", (err) => {
835
1198
  if (err?.name === "AbortError" || ctl.signal.aborted) return finish(reasonFor(timedOut, timeoutMs, clean(out)));
836
- log5.debug("shell spawn error", err);
1199
+ log6.debug("shell spawn error", err);
837
1200
  finish(`[exit 1] ${err?.message ?? err}${out ? "\n" + clean(out) : ""}`);
838
1201
  });
839
1202
  proc.on("close", (code) => {
@@ -853,7 +1216,7 @@ function makeRealShellTool(options) {
853
1216
  };
854
1217
  }
855
1218
  function handoffFor(timeoutMs, id, body, notifies) {
856
- const head = `[still running] exceeded ${timeoutMs}ms, so it was handed to background job ${id} \u2014 NOT killed, it is still going. Check on it with ShellOutput({id:"${id}"}) / ShellStatus({id:"${id}"}), stop it with ShellKill({id:"${id}"}). ` + (notifies ? "Its completion will be reported to you, so you may continue with other work meanwhile." : "Nothing will tell you when it finishes \u2014 poll it yourself before you rely on its result.");
1219
+ const head = `[still running] exceeded ${timeoutMs}ms, so it was handed to background job ${id} \u2014 NOT killed, it is still going (it is only stopped if it goes silent with no CPU activity for a long time, and then you are told why). Check on it with ShellOutput({id:"${id}"}) / ShellStatus({id:"${id}"}), stop it with ShellKill({id:"${id}"}). ` + (notifies ? "Its completion will be reported to you, so you may continue with other work meanwhile." : "Nothing will tell you when it finishes \u2014 poll it yourself before you rely on its result.");
857
1220
  return body ? `${head}
858
1221
  Output so far:
859
1222
  ${body}` : head;
@@ -878,7 +1241,7 @@ function makeShellJobTools(registry) {
878
1241
  const out = registry.output(String(id));
879
1242
  if (out == null) return NO_JOB(String(id));
880
1243
  const st = registry.status(String(id));
881
- return `[${st.status}${st.exitCode != null ? ` exit ${st.exitCode}` : ""}]
1244
+ return `[${st.status}${st.exitCode != null ? ` exit ${st.exitCode}` : ""}${st.reason ? `: ${st.reason}` : ""}]
882
1245
  ${clean(out) || "(no output yet)"}`;
883
1246
  }
884
1247
  },
@@ -893,13 +1256,13 @@ ${clean(out) || "(no output yet)"}`;
893
1256
  return jobs.length ? jobs.map((j) => `${j.id} ${j.status} ${j.command}`).join("\n") : "(no background jobs)";
894
1257
  }
895
1258
  const st = registry.status(String(id));
896
- return st ? `${st.status}${st.exitCode != null ? ` (exit ${st.exitCode})` : ""} \xB7 ${st.bytes} byte(s) buffered` : NO_JOB(String(id));
1259
+ return st ? `${st.status}${st.exitCode != null ? ` (exit ${st.exitCode})` : ""}${st.reason ? ` \u2014 ${st.reason}` : ""} \xB7 ${st.bytes} byte(s) buffered` : NO_JOB(String(id));
897
1260
  }
898
1261
  },
899
1262
  {
900
1263
  ...dropOnRebind,
901
1264
  name: "ShellKill",
902
- description: "Stop a running background Shell job by id (SIGTERM).",
1265
+ description: "Stop a running background Shell job by id (SIGTERM to its process group, SIGKILL if it ignores that).",
903
1266
  parameters: { type: "object", required: ["id"], properties: { id: { type: "string" } } },
904
1267
  async run({ id }) {
905
1268
  return registry.kill(String(id)) ? `Killed job ${id}.` : NO_JOB(String(id));
@@ -911,6 +1274,9 @@ export {
911
1274
  ShellJobRegistry,
912
1275
  formatJobExit,
913
1276
  makeRealShellTool,
914
- makeShellJobTools
1277
+ makeShellJobTools,
1278
+ parsePsTime,
1279
+ psGroupCpuSeconds,
1280
+ sumTreeCpu
915
1281
  };
916
1282
  //# sourceMappingURL=tools.shell.js.map