@link-assistant/hive-mind 2.15.1 → 2.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -25,7 +25,8 @@ import { formatSessionCompletionMessage, getSessionCompletionExitCode, classifyS
25
25
  import { notifySubscribers, getSubscriberCount } from './telegram-subscribers.lib.mjs';
26
26
  import { safeSendMessage, safeEditMessageText } from './telegram-safe-reply.lib.mjs';
27
27
  import { classifyExitStatus, normalizeExitCode } from './session-status.lib.mjs';
28
- import { readLastSessionIdFromLog, buildResumeCommand, formatResumeSection } from './session-resume.lib.mjs';
28
+ import { buildResumeCommand, formatResumeSection } from './session-resume.lib.mjs';
29
+ import { readLogMarkerLines, readLogTextBounded, scanLogTextChunks } from './log-bounded-read.lib.mjs';
29
30
  import { resolveFailedSessionPullRequestState } from './github-pr-state.lib.mjs';
30
31
  // Issue #2117: a docker terminal failure that no anchored log footer corroborates may be an exit code start-command fabricated from the command's own output.
31
32
  import { clearUnverifiedDockerTerminalMarker as clearUnverifiedDockerTerminalMarkerImpl, shouldDeferUnverifiedDockerTerminal as shouldDeferUnverifiedDockerTerminalImpl } from './session-monitor.docker-terminal.lib.mjs';
@@ -33,6 +34,8 @@ import { isDockerIsolation, sessionStartMs, resolveOomKilledState, resolveStaleE
33
34
  // Issue #2134: kill-cause diagnostics + the matching pull-request notice.
34
35
  import { buildKillCompletionSections, announceKillOnPullRequest } from './session-monitor.kill-sections.lib.mjs';
35
36
  import { runKillRecoveryForCompletion } from './session-kill-resume.lib.mjs';
37
+ // Issue #2189: the handled latch + the memoized last-tool-session-id read that keep a completed session from replaying its whole completion pipeline on every poll.
38
+ import { isCompletionHandled, markCompletionHandled, resolveCachedLastToolSessionId } from './session-completion-state.lib.mjs';
36
39
  import { createSessionRegistryQueries } from './session-monitor.queries.lib.mjs';
37
40
  export { formatSessionCompletionMessage, getSessionCompletionExitCode } from './work-session-formatting.lib.mjs';
38
41
  export { DOCKER_TERMINAL_FOOTER_GRACE_MS } from './session-monitor.docker-terminal.lib.mjs';
@@ -326,6 +329,8 @@ function normalizeSessionUrl(url) {
326
329
  return url.replace(/#.*$/, '').replace(/\/+$/, '').toLowerCase();
327
330
  }
328
331
  const GITHUB_PULL_REQUEST_URL_RE = /https:\/\/github\.com\/([A-Za-z0-9_.-]+)\/([A-Za-z0-9_.-]+)\/pull\/([0-9]+)/g;
332
+ /** Marker prefix parsed by `parseDiskMarkers` (issue #1945/#1988), used to collect just those lines. */
333
+ const DISK_MARKER_LINE_RE = /📊 \[DISK\] /;
329
334
  export function extractPullRequestUrlFromText(text, { owner = null, repo = null } = {}) {
330
335
  if (!text) return null;
331
336
  const expectedOwner = owner ? String(owner).toLowerCase() : null;
@@ -344,8 +349,12 @@ export function extractPullRequestUrlFromText(text, { owner = null, repo = null
344
349
  async function resolvePullRequestUrlFromSessionLog(logPath, ctx, { verbose = false, readFile = fs.readFile } = {}) {
345
350
  if (!logPath) return null;
346
351
  try {
347
- const logText = await readFile(logPath, 'utf8');
348
- const pullRequestUrl = extractPullRequestUrlFromText(logText, { owner: ctx.owner, repo: ctx.repo });
352
+ // Issue #2189: this used to be `readFile(logPath, 'utf8')`, run once per
353
+ // monitor tick for a session the bot never marked handled — a 134 MB
354
+ // transcript pulled into the bot's own heap over and over. The chunked scan
355
+ // still covers the whole log, one chunk at a time, and stops at the first
356
+ // matching URL (which is printed early, when the PR is created).
357
+ const pullRequestUrl = await scanLogTextChunks(logPath, text => extractPullRequestUrlFromText(text, { owner: ctx.owner, repo: ctx.repo }), { readFile, verbose });
349
358
  if (pullRequestUrl && verbose) {
350
359
  console.log(`[VERBOSE] Found PR ${pullRequestUrl} in completed session log ${logPath}`);
351
360
  }
@@ -370,7 +379,10 @@ export async function buildDiskDiagnosticsExtraSection(logPath, { verbose = fals
370
379
  let logText = '';
371
380
  if (logPath) {
372
381
  try {
373
- logText = await readFile(logPath, 'utf8');
382
+ // Issue #2189: `parseDiskMarkers` only ever looks at `📊 [DISK]` lines,
383
+ // so collect those lines instead of the whole transcript. Cost is one
384
+ // chunk plus a few kilobytes of markers, whatever the log's size.
385
+ logText = await readLogMarkerLines(logPath, DISK_MARKER_LINE_RE, { readFile, verbose });
374
386
  } catch (readError) {
375
387
  if (verbose) {
376
388
  console.log(`[VERBOSE] Could not read session log ${logPath} for disk diagnostics: ${readError?.message || readError}`);
@@ -402,7 +414,10 @@ export async function buildSubscriptionBlockedExtraSection(logPath, { verbose =
402
414
  try {
403
415
  let logText = '';
404
416
  try {
405
- logText = await readFile(logPath, 'utf8');
417
+ // Issue #2189: the blocked report is printed as the run stops, so the
418
+ // bounded head+tail excerpt always contains it; the middle of a multi-
419
+ // gigabyte transcript never has to be resident to find it.
420
+ logText = await readLogTextBounded(logPath, { readFile, verbose });
406
421
  } catch (readError) {
407
422
  if (verbose) {
408
423
  console.log(`[VERBOSE] Could not read session log ${logPath} for subscription block: ${readError?.message || readError}`);
@@ -439,6 +454,35 @@ async function getDockerContainerFilesystemSizeForSession(sessionName, sessionIn
439
454
  return null;
440
455
  }
441
456
  }
457
+ /**
458
+ * How long a docker writable-layer measurement stays good enough while the
459
+ * session is still running (issue #2189).
460
+ *
461
+ * `docker ps --size` walks the container's whole writable layer. On the 27 GB
462
+ * layer in the captured incident that is expensive, and the monitor was paying
463
+ * it once per session per 30-second poll for no gain: the number is only ever
464
+ * *reported* at completion. The completion measurement is never throttled, so
465
+ * the number in the report is still fresh.
466
+ */
467
+ export const DOCKER_FILESYSTEM_REFRESH_INTERVAL_MS = 5 * 60 * 1000;
468
+ /**
469
+ * Whether the writable-layer size should be measured on this poll.
470
+ *
471
+ * @param {Object} sessionInfo
472
+ * @param {Object} [options]
473
+ * @param {boolean} [options.stillRunning] - The session is still executing
474
+ * @param {number} [options.now] - Epoch ms (injectable for tests)
475
+ * @param {number} [options.intervalMs]
476
+ * @returns {boolean}
477
+ */
478
+ export function shouldRefreshDockerFilesystemSize(sessionInfo, { stillRunning = true, now = Date.now(), intervalMs = DOCKER_FILESYSTEM_REFRESH_INTERVAL_MS } = {}) {
479
+ if (sessionInfo?.isolationBackend !== 'docker') return false;
480
+ // The completion report quotes this number, so it is always measured fresh.
481
+ if (!stillRunning) return true;
482
+ const observedAt = Date.parse(sessionInfo?.containerFilesystemLastObservedAt || '');
483
+ if (!Number.isFinite(observedAt)) return true;
484
+ return now - observedAt >= intervalMs;
485
+ }
442
486
  async function refreshDockerContainerFilesystemSizeForSession(sessionName, sessionInfo, { verbose = false, sizeProvider = null } = {}) {
443
487
  const bytes = await getDockerContainerFilesystemSizeForSession(sessionName, sessionInfo, { verbose, sizeProvider });
444
488
  if (!Number.isFinite(bytes)) return null;
@@ -665,6 +709,20 @@ export async function monitorSessions(bot, verbose = false, options = {}) {
665
709
  console.log(`[VERBOSE] Checking ${sessions.length} active session(s)...`);
666
710
  }
667
711
  for (const { sessionName, sessionInfo } of sessions) {
712
+ // Issue #2189: a session whose completion notification was already
713
+ // delivered is terminal. Without this latch the monitor re-entered the whole
714
+ // completion pipeline on every poll — re-resolving the linked pull request,
715
+ // re-scanning a 134 MB log, re-walking a 27 GB writable layer and re-sending
716
+ // a notification the user already had — because a late failure in that
717
+ // pipeline (or a bot restart) left the session tracked. Finalize it here,
718
+ // before any status probe, and do none of that work again.
719
+ if (isCompletionHandled(sessionInfo)) {
720
+ if (verbose) {
721
+ console.log(`[VERBOSE] Session ${sessionName} was already reported at ${sessionInfo.completionNotifiedAt}; finalizing without repeating the completion work (issue #2189)`);
722
+ }
723
+ completeSession(sessionName, sessionInfo.completionExitCode ?? 0, verbose, sessionInfo.completionStatus ?? null);
724
+ continue;
725
+ }
668
726
  let stillRunning;
669
727
  let exitCode = null;
670
728
  let statusResult = null;
@@ -727,11 +785,13 @@ export async function monitorSessions(bot, verbose = false, options = {}) {
727
785
  }
728
786
  }
729
787
  }
730
- if (sessionInfo?.isolationBackend === 'docker') {
788
+ if (shouldRefreshDockerFilesystemSize(sessionInfo, { stillRunning })) {
731
789
  observedContainerFilesystemBytes = await refreshDockerContainerFilesystemSizeForSession(sessionName, sessionInfo, {
732
790
  verbose,
733
791
  sizeProvider: options.dockerContainerSizeProvider,
734
792
  });
793
+ } else if (sessionInfo?.isolationBackend === 'docker' && verbose) {
794
+ console.log(`[VERBOSE] Session ${sessionName}: reusing the writable-layer size observed at ${sessionInfo.containerFilesystemLastObservedAt} (issue #2189: not re-walking the layer every poll)`);
735
795
  }
736
796
  if (!stillRunning) {
737
797
  console.log(`Session ${sessionName} has finished. Sending notification to chat ${sessionInfo.chatId}`);
@@ -756,6 +816,10 @@ export async function monitorSessions(bot, verbose = false, options = {}) {
756
816
  statusResult,
757
817
  readFile: options.readFile,
758
818
  });
819
+ if (pullRequestUrl && sessionInfo.resolvedPullRequestUrl !== pullRequestUrl) {
820
+ sessionInfo.resolvedPullRequestUrl = pullRequestUrl;
821
+ persistSessionSnapshot(sessionName, sessionInfo);
822
+ }
759
823
  } catch (lookupError) {
760
824
  if (verbose) {
761
825
  console.log(`[VERBOSE] Pull request lookup failed for ${sessionName}: ${lookupError?.message || lookupError}`);
@@ -806,8 +870,20 @@ export async function monitorSessions(bot, verbose = false, options = {}) {
806
870
  }
807
871
  }
808
872
  }
873
+ // Issue #2189: the last tool session id is read through the session
874
+ // record's cache, so the working-session log is scanned once per session
875
+ // instead of once per poll per consumer (the resume section below and the
876
+ // automatic recovery further down both need it, and both used to scan).
877
+ const resolveLastToolSessionId = logPath => {
878
+ const resolution = resolveCachedLastToolSessionId({ sessionInfo, logPath, verbose });
879
+ if (resolution.scanned) persistSessionSnapshot(sessionName, sessionInfo);
880
+ return resolution.id;
881
+ };
809
882
  // Issue #1927: for a killed /solve, offer a command using the last tool
810
- // session ID in the log. Do not auto-relaunch work that may reliably OOM.
883
+ // session ID in the log. Issue #2189 additionally *starts* that command
884
+ // by default (`--on-session-kill=resume`), bounded by
885
+ // `--session-kill-resume-attempts`, so the work is not left for a human
886
+ // to notice hours later.
811
887
  const resumeExtraSections = [];
812
888
  let killResumeCommand = null;
813
889
  try {
@@ -821,7 +897,7 @@ export async function monitorSessions(bot, verbose = false, options = {}) {
821
897
  // `Session ID:` marker. Do not guess from neighboring UUID-named
822
898
  // logs: start-command stores unrelated tasks in the same backend
823
899
  // directory, which caused issue #2109's invalid resume id.
824
- const lastSessionId = readLastSessionIdFromLog(logPath, { verbose });
900
+ const lastSessionId = resolveLastToolSessionId(logPath);
825
901
  const resumeCommand = buildResumeCommand({ sessionInfo, lastSessionId });
826
902
  const resumeSection = formatResumeSection({ lastSessionId, command: resumeCommand });
827
903
  killResumeCommand = resumeCommand || null;
@@ -900,10 +976,21 @@ export async function monitorSessions(bot, verbose = false, options = {}) {
900
976
  runner: options.isolationRunner || null,
901
977
  trackSession: options.trackSession || trackSession,
902
978
  persistSnapshot: () => persistSessionSnapshot(sessionName, sessionInfo),
979
+ // Issue #2189: reuse the id already read above instead of scanning
980
+ // the same (possibly multi-gigabyte) log a second time.
981
+ readLastSessionId: logPath => resolveLastToolSessionId(logPath),
903
982
  locale: sessionInfo?.locale || null,
904
983
  verbose,
905
984
  });
906
985
  killRecovery = recovered.recovery;
986
+ if (killRecovery.resumed && killRecovery.sessionId) {
987
+ // Issue #2189: remember which session took over, so the durable
988
+ // history says what happened to this work and a restart cannot start
989
+ // a second recovery for the same kill.
990
+ sessionInfo.killRecoverySessionId = killRecovery.sessionId;
991
+ persistSessionSnapshot(sessionName, sessionInfo);
992
+ logEvent('session_kill_recovered', { sessionName, recoverySessionId: killRecovery.sessionId, attempt: killRecovery.attempt, maxAttempts: killRecovery.maxAttempts, policy: killRecovery.policy || null });
993
+ }
907
994
  if (recovered.section) killReport.sections.push(recovered.section);
908
995
  }
909
996
  const message = formatSessionCompletionMessage({
@@ -952,6 +1039,16 @@ export async function monitorSessions(bot, verbose = false, options = {}) {
952
1039
  notifyFromChatId = sent?.chat?.id || sessionInfo.chatId;
953
1040
  notifyMessageId = sent?.message_id || null;
954
1041
  }
1042
+ // Issue #2189: the user has now been told. Latch that fact durably
1043
+ // BEFORE the remaining best-effort work (subscriber fan-out, container
1044
+ // cleanup), because any failure after this point used to send the whole
1045
+ // completion pipeline — and this notification — round again on the next
1046
+ // poll. The snapshot write means even a bot restart in this window
1047
+ // finalizes the session silently instead of re-notifying.
1048
+ if (markCompletionHandled(sessionInfo, { exitCode: finalExitCode, status: resolvedStatus })) {
1049
+ persistSessionSnapshot(sessionName, sessionInfo);
1050
+ logEvent('session_completion_notified', { sessionName, exitCode: finalExitCode ?? null, status: resolvedStatus || null, notifiedAt: sessionInfo.completionNotifiedAt });
1051
+ }
955
1052
  // Issue #1688: forward the same completion message to every /subscribe-d user
956
1053
  // in their private chat with the bot. Failures are logged but don't block
957
1054
  // completion of the parent session.
@@ -1024,6 +1121,13 @@ async function resolvePullRequestUrlForSession(sessionInfo, { verbose = false, l
1024
1121
  if (!ctx || ctx.type !== 'issue' || !ctx.owner || !ctx.repo || !ctx.number) {
1025
1122
  return null;
1026
1123
  }
1124
+ // Issue #2189: a completion that has to be retried must not re-run the linked-PR
1125
+ // lookup (an API round trip, then a scan of the session log). The answer cannot
1126
+ // change for a session that has already finished, so remember it.
1127
+ if (typeof sessionInfo.resolvedPullRequestUrl === 'string' && sessionInfo.resolvedPullRequestUrl) {
1128
+ if (verbose) console.log(`[VERBOSE] Reusing resolved pull request ${sessionInfo.resolvedPullRequestUrl} for this session (not looked up again)`);
1129
+ return sessionInfo.resolvedPullRequestUrl;
1130
+ }
1027
1131
  if (typeof lookupLinkedPullRequest === 'function') {
1028
1132
  const linkedPullRequestUrl = await lookupLinkedPullRequest(ctx);
1029
1133
  if (linkedPullRequestUrl) return linkedPullRequestUrl;
@@ -1089,6 +1193,22 @@ export function startSessionMonitoring(bot, verbose = false, intervalMs = 30000,
1089
1193
  * record whose startTime is after the current bot start (it cannot belong to a
1090
1194
  * previous run), satisfying requirement #2's "started before bot start time".
1091
1195
  *
1196
+ * Issue #2189 asks for the other half of that sentence — "on startup resume all
1197
+ * still-running / interrupted commands". Both cases are handled here plus the
1198
+ * first monitor tick, which runs synchronously after this function:
1199
+ *
1200
+ * - a session that is **still running** keeps running; re-registering it is
1201
+ * exactly what resumes it, and the bot reports it when it ends;
1202
+ * - a session that was **interrupted** (its backend is gone, or its log footer
1203
+ * records a kill) is detected as finished on that first tick and, under the
1204
+ * now-default `--on-session-kill=resume`, a recovery working session is
1205
+ * started from its last tool session id — bounded by
1206
+ * `--session-kill-resume-attempts`, whose counter is persisted, so a job
1207
+ * that dies every time cannot be relaunched once per bot restart forever;
1208
+ * - a session that was already **reported** before the previous process died
1209
+ * carries the persisted `completionNotifiedAt` latch and is finalized
1210
+ * silently, so a restart never re-notifies.
1211
+ *
1092
1212
  * @param {object} [options]
1093
1213
  * @param {object} [options.store] - Session store to load from (default: the store set via setSessionStore).
1094
1214
  * @param {number} [options.botStartTime] - Epoch seconds; only sessions started strictly before this are resumed. Defaults to now.
@@ -1137,7 +1257,10 @@ export async function resumeTrackedSessions(options = {}) {
1137
1257
  }
1138
1258
  }
1139
1259
  if (resumed.length > 0) {
1140
- console.log(`♻️ Resumed monitoring of ${resumed.length} session(s) from durable store after restart`);
1260
+ // Issue #2189: say how many of these are already-reported leftovers, so the
1261
+ // startup line is not read as "N sessions are still working".
1262
+ const alreadyReported = resumed.filter(({ sessionInfo }) => isCompletionHandled(sessionInfo)).length;
1263
+ console.log(`♻️ Resumed monitoring of ${resumed.length} session(s) from durable store after restart${alreadyReported > 0 ? ` (${alreadyReported} already reported, will be finalized silently)` : ''}`);
1141
1264
  } else if (verbose) {
1142
1265
  console.log('[VERBOSE] resumeTrackedSessions: no eligible sessions to resume');
1143
1266
  }
@@ -38,7 +38,21 @@ import path from 'node:path';
38
38
  // `executionUuid` (#2154) is start-command's own identifier for the execution —
39
39
  // the one `$ --list` prints. It differs from `sessionId`, so persisting it is
40
40
  // what lets a restarted bot still correlate its sessions with the session list.
41
- const PERSISTABLE_FIELDS = ['chatId', 'messageId', 'startTime', 'url', 'command', 'commandAlias', 'isolationBackend', 'sessionId', 'executionUuid', 'containerFilesystemStartBytes', 'containerFilesystemLastBytes', 'containerFilesystemLastObservedAt', 'tool', 'infoBlock', 'urlContext', 'requesterUserId', 'showLimits', 'locale', 'logPath', 'args'];
41
+ // Issue #2189 adds the fields that keep a completed session terminal and its
42
+ // per-poll cost constant:
43
+ // - `completionNotifiedAt`/`completionExitCode`/`completionStatus` latch the
44
+ // one delivered notification, so a restart between "notified" and
45
+ // "untracked" finalizes silently instead of re-running the whole completion
46
+ // pipeline and notifying the user again.
47
+ // - `lastToolSessionId` caches the marker found by scanning the working
48
+ // session log, so that scan is never O(log size) per poll.
49
+ // - `killRecoveryAttempts`/`killRecoverySessionId`/`killRecoveryOfSession`
50
+ // bound and record automatic recovery across restarts — without them a
51
+ // reliably-crashing job could restart once per bot launch forever.
52
+ // - `stopRequestedByUser`/`stopRequestedBy` must survive a restart too: with
53
+ // `--on-session-kill=resume` now the default, forgetting that an operator
54
+ // asked for the stop would relaunch the very work they cancelled.
55
+ const PERSISTABLE_FIELDS = ['chatId', 'messageId', 'startTime', 'url', 'command', 'commandAlias', 'isolationBackend', 'sessionId', 'executionUuid', 'containerFilesystemStartBytes', 'containerFilesystemLastBytes', 'containerFilesystemLastObservedAt', 'tool', 'infoBlock', 'urlContext', 'requesterUserId', 'showLimits', 'locale', 'logPath', 'args', 'completionNotifiedAt', 'completionExitCode', 'completionStatus', 'lastToolSessionId', 'killRecoveryAttempts', 'killRecoverySessionId', 'killRecoveryOfSession', 'killRecoveryResumed', 'stopRequestedByUser', 'stopRequestedBy', 'onSessionKill', 'resolvedPullRequestUrl'];
42
56
 
43
57
  /**
44
58
  * Resolve the directory durable bot state is written to. Honors
@@ -159,9 +159,12 @@ export const SOLVE_OPTION_DEFINITIONS = {
159
159
  // solve honour the exact same value, so the report never differs by surface.
160
160
  'on-session-kill': {
161
161
  type: 'string',
162
- description: 'What to do when a working session is killed (out of memory, disk full, forced kill): "report" describes exactly what happened in the pull request and in Telegram, "resume" additionally starts a new working session to recover and says so in both places. Can also be set with HIVE_MIND_ON_SESSION_KILL.',
162
+ description: 'What to do when a working session is killed (out of memory, disk full, forced kill): "resume" (default) starts a new working session from the killed one\'s last tool session id and says so in the pull request and in Telegram, "report" only describes exactly what happened without restarting anything. Can also be set with HIVE_MIND_ON_SESSION_KILL.',
163
163
  choices: ['report', 'resume'],
164
- default: 'report',
164
+ // Issue #2189: a kill that is only ever *offered* for resume is a kill
165
+ // nobody recovers from — the offer in the captured incident reached its
166
+ // operator six hours late. Bounded by --session-kill-resume-attempts.
167
+ default: 'resume',
165
168
  },
166
169
  'session-kill-resume-attempts': {
167
170
  type: 'number',
@@ -7,8 +7,18 @@
7
7
  // issue comment so it's excluded from --auto-attach-solution-summary's check.
8
8
  import { REPOSITORY_INITIALIZATION_REQUIRED_MARKER, postTrackedComment } from './tool-comments.lib.mjs';
9
9
  import { QUIET_PROBE } from './quiet-probe.lib.mjs'; // issue #2130: keep read-only probe payloads out of the attached log
10
+ // Issue #2192: a credential helper is never consulted for a *public* clone
11
+ // (github.com answers 200, so git never asks), which is why an authenticated
12
+ // container still got throttled as anonymous. The token has to be sent
13
+ // preemptively, before the first git network call.
14
+ import { ensureAuthenticatedGitTransport } from './git-auth-transport.lib.mjs';
10
15
 
11
16
  export async function setupRepositoryAndClone({ argv, owner, repo, forkOwner, forkRepoName, tempDir, isContinueMode, issueUrl, log, $, needsClone = true }) {
17
+ // Issue #2192: authenticate git *before* the first clone/fetch. Doing this
18
+ // afterwards (as setupGitCredentialHelper does) is too late — the clone is the
19
+ // call GitHub rejected with "temporarily limiting some unauthenticated downloads".
20
+ await ensureAuthenticatedGitTransport({ $, log, reason: 'repository setup' });
21
+
12
22
  // Set up repository and handle forking
13
23
  const { repoToClone, forkedRepo, upstreamRemote, prForkOwner } = await setupRepository(argv, owner, repo, forkOwner, issueUrl, forkRepoName);
14
24
 
@@ -31,6 +31,9 @@ import { ensureAiToolScratchIgnored } from './ai-tool-scratch.lib.mjs';
31
31
  import { parseForkFullNameFromGhOutput } from './github-repository-names.lib.mjs';
32
32
  import { checkReplacementRepositoryBranchSafety } from './solve.repository-safety.lib.mjs';
33
33
  import { buildForkReplacementBlockedReason, buildForkReplacementSafetyCheckDescription } from './solve.repository-recovery-message.lib.mjs';
34
+ // Issue #2192: GitHub throttles *anonymous* git downloads; a token must be sent
35
+ // preemptively (a credential helper is never consulted for a public repository).
36
+ import { GIT_AUTH_TRANSPORT_DISABLE, ensureAuthenticatedGitTransport, isAnonymousDownloadLimit } from './git-auth-transport.lib.mjs';
34
37
 
35
38
  // Import GitHub utilities for permission checks
36
39
  const githubLib = await import('./github.lib.mjs');
@@ -949,6 +952,14 @@ export const classifyCloneError = errorOutput => {
949
952
  return { type: 'NETWORK', retryable: true, description: 'Network connectivity issue (interrupted transfer)' };
950
953
  }
951
954
 
955
+ // Issue #2192: GitHub refusing an *unauthenticated* download. Retryable, but
956
+ // waiting is not the remedy — the clone has to be authenticated. Checked
957
+ // before PERMISSION/NOT_FOUND/RATE_LIMIT because GitHub's wording ("limiting",
958
+ // "retry later or authenticate") overlaps all three.
959
+ if (isAnonymousDownloadLimit(errorOutput)) {
960
+ return { type: 'ANONYMOUS_RATE_LIMIT', retryable: true, description: 'GitHub is limiting unauthenticated downloads (this clone was not authenticated)' };
961
+ }
962
+
952
963
  // Authentication/permission errors - not retryable
953
964
  if (output.includes('error: 401') || output.includes('error: 403') || output.includes('authentication failed') || output.includes('permission denied')) {
954
965
  return { type: 'PERMISSION', retryable: false, description: 'Authentication or permission error' };
@@ -1075,6 +1086,9 @@ export const cloneRepository = async (repoToClone, tempDir, argv, owner, repo) =
1075
1086
  await log(' • Network connectivity issues');
1076
1087
  if (errorClassification.type === 'TRANSIENT') await log(' • GitHub server issues (temporary)');
1077
1088
  if (errorClassification.type === 'RATE_LIMIT') await log(' • API rate limiting exceeded');
1089
+ // Issue #2192: the request never carried an Authorization header, so
1090
+ // GitHub counted it against the anonymous budget regardless of `gh auth status`.
1091
+ if (errorClassification.type === 'ANONYMOUS_RATE_LIMIT') await log(' • The clone was sent anonymously — GitHub throttles unauthenticated downloads');
1078
1092
  // Issue #1957: the transfer started but was interrupted (e.g. the connection
1079
1093
  // dropped while reading the pack). The retries above were already exhausted.
1080
1094
  if (errorClassification.type === 'NETWORK') await log(' • Connection dropped mid-transfer (the clone was interrupted before completing)');
@@ -1087,6 +1101,11 @@ export const cloneRepository = async (repoToClone, tempDir, argv, owner, repo) =
1087
1101
  if (argv.fork) await log(` 4. Check fork: gh repo view ${repoToClone}`);
1088
1102
  if (errorClassification.type === 'TRANSIENT') await log(' 5. Wait and retry / check: https://www.githubstatus.com');
1089
1103
  if (errorClassification.type === 'RATE_LIMIT') await log(' 5. Wait for rate limit to reset or use --token with different token');
1104
+ if (errorClassification.type === 'ANONYMOUS_RATE_LIMIT') {
1105
+ await log(' 5. Make sure a token is available to git: gh auth token (or set GH_TOKEN)');
1106
+ await log(' 6. Repair the git/gh state non-interactively: gh-setup-git-identity --repair');
1107
+ await log(` 7. Hive Mind normally authenticates git itself; if that was turned off, unset ${GIT_AUTH_TRANSPORT_DISABLE}`);
1108
+ }
1090
1109
  if (errorClassification.type === 'NETWORK') {
1091
1110
  await log(' 5. Check your network connection / VPN / proxy, then re-run the command');
1092
1111
  await log(' 6. On slow or unstable links, a shallower history transfers faster and is less');
@@ -1100,6 +1119,13 @@ export const cloneRepository = async (repoToClone, tempDir, argv, owner, repo) =
1100
1119
  // Retryable error and we have attempts left
1101
1120
  const delay = baseDelay * Math.pow(2, attempt - 1); // Exponential backoff
1102
1121
  await log(`${formatAligned('⚠️', 'Clone failed:', errorClassification.description)}`);
1122
+ // Issue #2192: auto-recovery. GitHub rejected the download as anonymous, so a
1123
+ // plain retry would be rejected the same way. Authenticate the transport (and,
1124
+ // if no token is reachable, let `gh-setup-git-identity --repair` restore the
1125
+ // gh state non-interactively) before spending the next attempt.
1126
+ if (errorClassification.type === 'ANONYMOUS_RATE_LIMIT') {
1127
+ await ensureAuthenticatedGitTransport({ $, log, repair: true, reason: 'GitHub rejected the clone as unauthenticated' });
1128
+ }
1103
1129
  await log(`${formatAligned('⏳', 'Retrying:', `Waiting ${delay / 1000}s before attempt ${attempt + 1}/${maxRetries}...`)}`);
1104
1130
  if (errorClassification.type === 'RATE_LIMIT') {
1105
1131
  await log(' 💡 Tip: Rate limiting detected - using longer delay');
@@ -1,5 +1,6 @@
1
1
  import fs from 'node:fs';
2
2
  import os from 'node:os';
3
+ import v8 from 'node:v8';
3
4
 
4
5
  export const RESOURCE_MARKER_PREFIX = '📈 [RESOURCES]';
5
6
 
@@ -10,8 +11,19 @@ export const RESOURCE_PHASE_SOLVE_EXIT = 'solve_exit';
10
11
  export const RESOURCE_PHASE_RESTART_BEFORE = 'restart_before';
11
12
  export const RESOURCE_PHASE_RESTART_AFTER = 'restart_after';
12
13
  export const RESOURCE_PHASE_BOT_HEARTBEAT = 'bot_heartbeat';
14
+ // Issue #2189: the run that died of a V8 heap OOM inside the log sanitizer had
15
+ // its last resource sample at `after_agent` (RSS 373 MB), ten minutes before the
16
+ // fatal error — the whole log-upload phase was untelemetered, so the post-mortem
17
+ // could not tell a heap blow-up from an external kill. These phases bracket it.
18
+ export const RESOURCE_PHASE_LOG_UPLOAD_START = 'log_upload_start';
19
+ export const RESOURCE_PHASE_LOG_UPLOAD_END = 'log_upload_end';
13
20
 
14
- const RESOURCE_PHASES_BY_PREFERENCE = [RESOURCE_PHASE_SOLVE_EXIT, RESOURCE_PHASE_AFTER_AGENT, RESOURCE_PHASE_RESTART_AFTER, RESOURCE_PHASE_AFTER_CLONE, RESOURCE_PHASE_SOLVE_START, RESOURCE_PHASE_RESTART_BEFORE];
21
+ // A V8 heap this close to its own limit is the shape of an imminent
22
+ // "FATAL ERROR: Reached heap limit" abort; surface it while the process is
23
+ // still alive to print it.
24
+ export const HEAP_PRESSURE_WARN_PERCENT = 85;
25
+
26
+ const RESOURCE_PHASES_BY_PREFERENCE = [RESOURCE_PHASE_SOLVE_EXIT, RESOURCE_PHASE_LOG_UPLOAD_END, RESOURCE_PHASE_LOG_UPLOAD_START, RESOURCE_PHASE_AFTER_AGENT, RESOURCE_PHASE_RESTART_AFTER, RESOURCE_PHASE_AFTER_CLONE, RESOURCE_PHASE_SOLVE_START, RESOURCE_PHASE_RESTART_BEFORE];
15
27
 
16
28
  function finiteNumber(value) {
17
29
  return Number.isFinite(value) ? value : null;
@@ -139,7 +151,7 @@ export function formatExecutionContextForLog(context) {
139
151
  }
140
152
 
141
153
  export function captureResourceSnapshot(options = {}) {
142
- const { phase = 'snapshot', diskPath = '/', now = () => new Date(), osImpl = os, fsImpl = fs, processImpl = process } = options;
154
+ const { phase = 'snapshot', diskPath = '/', now = () => new Date(), osImpl = os, fsImpl = fs, processImpl = process, v8Impl = v8 } = options;
143
155
 
144
156
  const timestamp = (() => {
145
157
  try {
@@ -196,11 +208,26 @@ export function captureResourceSnapshot(options = {}) {
196
208
  return {
197
209
  rssBytes: finiteNumber(usage.rss),
198
210
  heapUsedBytes: finiteNumber(usage.heapUsed),
211
+ heapTotalBytes: finiteNumber(usage.heapTotal),
212
+ externalBytes: finiteNumber(usage.external),
199
213
  };
200
214
  } catch {
201
- return { rssBytes: null, heapUsedBytes: null };
215
+ return { rssBytes: null, heapUsedBytes: null, heapTotalBytes: null, externalBytes: null };
216
+ }
217
+ })();
218
+
219
+ // Issue #2189: the heap *limit* is the number that was missing. A process can
220
+ // die of "JavaScript heap out of memory" with 10 GB of the machine still free,
221
+ // so RSS against total RAM says nothing; used heap against `heap_size_limit`
222
+ // says everything.
223
+ const heapLimitBytes = (() => {
224
+ try {
225
+ return finiteNumber(v8Impl.getHeapStatistics().heap_size_limit);
226
+ } catch {
227
+ return null;
202
228
  }
203
229
  })();
230
+ const heapUsedPercent = Number.isFinite(processMemory.heapUsedBytes) && Number.isFinite(heapLimitBytes) && heapLimitBytes > 0 ? clampPercent((processMemory.heapUsedBytes / heapLimitBytes) * 100) : null;
204
231
 
205
232
  const disk = (() => {
206
233
  const path = String(diskPath || '/');
@@ -243,6 +270,10 @@ export function captureResourceSnapshot(options = {}) {
243
270
  usedBytes: usedMemoryBytes,
244
271
  processRssBytes: processMemory.rssBytes,
245
272
  processHeapUsedBytes: processMemory.heapUsedBytes,
273
+ processHeapTotalBytes: processMemory.heapTotalBytes,
274
+ processExternalBytes: processMemory.externalBytes,
275
+ processHeapLimitBytes: heapLimitBytes,
276
+ processHeapUsedPercent: heapUsedPercent,
246
277
  },
247
278
  disk,
248
279
  };
@@ -270,6 +301,27 @@ function numberField(name, value) {
270
301
  return Number.isFinite(value) ? `${name}=${value}` : `${name}=null`;
271
302
  }
272
303
 
304
+ /**
305
+ * Human-readable "used heap of the heap limit" summary. Issue #2189: this is the
306
+ * single line that would have made the incident self-diagnosing.
307
+ */
308
+ export function formatHeapUsage(memory) {
309
+ const m = memory || {};
310
+ if (!Number.isFinite(m.processHeapUsedBytes)) return 'unknown';
311
+ const limit = Number.isFinite(m.processHeapLimitBytes) ? ` of ${formatBytes(m.processHeapLimitBytes)} limit` : '';
312
+ const percent = Number.isFinite(m.processHeapUsedPercent) ? ` (${m.processHeapUsedPercent.toFixed(1)}%)` : '';
313
+ return `${formatBytes(m.processHeapUsedBytes)} used${limit}${percent}`;
314
+ }
315
+
316
+ /**
317
+ * True when the V8 heap is close enough to its own limit that the next big
318
+ * allocation can abort the process (issue #2189).
319
+ */
320
+ export function isHeapUnderPressure(memory, warnPercent = HEAP_PRESSURE_WARN_PERCENT) {
321
+ const percent = memory?.processHeapUsedPercent;
322
+ return Number.isFinite(percent) && percent >= warnPercent;
323
+ }
324
+
273
325
  export function buildResourceMarker(snapshot) {
274
326
  const s = snapshot || {};
275
327
  const cpu = s.cpu || {};
@@ -287,6 +339,11 @@ export function buildResourceMarker(snapshot) {
287
339
  numberField('memAvailableBytes', memory.availableBytes),
288
340
  numberField('memUsedBytes', memory.usedBytes),
289
341
  numberField('processRssBytes', memory.processRssBytes),
342
+ numberField('processHeapUsedBytes', memory.processHeapUsedBytes),
343
+ numberField('processHeapTotalBytes', memory.processHeapTotalBytes),
344
+ numberField('processExternalBytes', memory.processExternalBytes),
345
+ numberField('processHeapLimitBytes', memory.processHeapLimitBytes),
346
+ numberField('processHeapUsedPercent', memory.processHeapUsedPercent),
290
347
  `diskPath=${encodeValue(disk.path || '/')}`,
291
348
  numberField('diskTotalBytes', disk.totalBytes),
292
349
  numberField('diskAvailableBytes', disk.availableBytes),
@@ -294,6 +351,7 @@ export function buildResourceMarker(snapshot) {
294
351
  numberField('diskUsedPercent', disk.usedPercent),
295
352
  disk.error ? `error=${encodeValue(disk.error)}` : null,
296
353
  `mem=${encodeValue(`${formatBytes(memory.availableBytes)} available / ${formatBytes(memory.totalBytes)} total`)}`,
354
+ `heap=${encodeValue(formatHeapUsage(memory))}`,
297
355
  `disk=${encodeValue(`${formatBytes(disk.availableBytes)} available / ${formatBytes(disk.totalBytes)} total`)}`,
298
356
  ]
299
357
  .filter(Boolean)
@@ -332,6 +390,11 @@ function parseMarkerLine(line) {
332
390
  availableBytes: parseNumber(fields.memAvailableBytes),
333
391
  usedBytes: parseNumber(fields.memUsedBytes),
334
392
  processRssBytes: parseNumber(fields.processRssBytes),
393
+ processHeapUsedBytes: parseNumber(fields.processHeapUsedBytes),
394
+ processHeapTotalBytes: parseNumber(fields.processHeapTotalBytes),
395
+ processExternalBytes: parseNumber(fields.processExternalBytes),
396
+ processHeapLimitBytes: parseNumber(fields.processHeapLimitBytes),
397
+ processHeapUsedPercent: parseNumber(fields.processHeapUsedPercent),
335
398
  },
336
399
  disk: {
337
400
  path: decodeURIComponent(fields.diskPath || '/'),
@@ -376,7 +439,8 @@ export function formatResourceSnapshotForLog(snapshot, label = null) {
376
439
  const cpu = s.cpu || {};
377
440
  const memory = s.memory || {};
378
441
  const disk = s.disk || {};
379
- const lines = [`📈 Resource usage (${phaseLabel}):`, ` CPU load: ${formatNumber(cpu.load1)} ${formatNumber(cpu.load5)} ${formatNumber(cpu.load15)}${Number.isFinite(cpu.cpuCount) ? ` (${cpu.cpuCount} CPUs)` : ''}`, ` Memory: ${formatBytes(memory.availableBytes)} available / ${formatBytes(memory.totalBytes)} total (${formatBytes(memory.usedBytes)} used)`, ` Process RSS: ${formatBytes(memory.processRssBytes)}${Number.isFinite(memory.processHeapUsedBytes) ? `, heap ${formatBytes(memory.processHeapUsedBytes)}` : ''}`, ` Disk (${disk.path || '/'}): ${formatBytes(disk.availableBytes)} available / ${formatBytes(disk.totalBytes)} total${Number.isFinite(disk.usedPercent) ? ` (${disk.usedPercent.toFixed(1)}% used)` : ''}`];
442
+ const lines = [`📈 Resource usage (${phaseLabel}):`, ` CPU load: ${formatNumber(cpu.load1)} ${formatNumber(cpu.load5)} ${formatNumber(cpu.load15)}${Number.isFinite(cpu.cpuCount) ? ` (${cpu.cpuCount} CPUs)` : ''}`, ` Memory: ${formatBytes(memory.availableBytes)} available / ${formatBytes(memory.totalBytes)} total (${formatBytes(memory.usedBytes)} used)`, ` Process RSS: ${formatBytes(memory.processRssBytes)}, V8 heap: ${formatHeapUsage(memory)}`, ` Disk (${disk.path || '/'}): ${formatBytes(disk.availableBytes)} available / ${formatBytes(disk.totalBytes)} total${Number.isFinite(disk.usedPercent) ? ` (${disk.usedPercent.toFixed(1)}% used)` : ''}`];
443
+ if (isHeapUnderPressure(memory)) lines.push(` ⚠️ V8 heap is at ${memory.processHeapUsedPercent.toFixed(1)}% of its limit — a further allocation can abort the process with "JavaScript heap out of memory"`);
380
444
  if (disk.error) lines.push(` Disk probe error: ${disk.error}`);
381
445
  lines.push(buildResourceMarker(snapshot));
382
446
  return lines.join('\n');
@@ -422,6 +486,9 @@ export function summarizeResourceSnapshot(snapshot) {
422
486
  availableBytes: memory.availableBytes,
423
487
  usedBytes: memory.usedBytes,
424
488
  processRssBytes: memory.processRssBytes,
489
+ processHeapUsedBytes: memory.processHeapUsedBytes,
490
+ processHeapLimitBytes: memory.processHeapLimitBytes,
491
+ processHeapUsedPercent: memory.processHeapUsedPercent,
425
492
  },
426
493
  disk: {
427
494
  path: disk.path,
@@ -24,7 +24,8 @@ import path from 'path';
24
24
  import os from 'os';
25
25
  import fs from 'fs/promises';
26
26
  import { constants as fsConstants } from 'fs';
27
- import { sanitizeForPublication, writeSanitizedPublicationFile } from './token-sanitization.lib.mjs';
27
+ import { sanitizeForPublication } from './token-sanitization.lib.mjs';
28
+ import { sanitizeLogFileToFileBounded } from './log-sanitize-worker.lib.mjs';
28
29
  import { safeReply, safeReplyWithDocument, safeSendDocument } from './telegram-safe-reply.lib.mjs';
29
30
 
30
31
  const UUID_RE = /\b([0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})\b/i;
@@ -43,8 +44,11 @@ async function prepareSanitizedLogUpload(logPath, caption) {
43
44
  const sanitizedPath = path.join(tempDir, path.basename(logPath));
44
45
 
45
46
  try {
46
- const [rawLog, safeCaption] = await Promise.all([fs.readFile(logPath, 'utf8'), sanitizeForPublication(caption)]);
47
- await writeSanitizedPublicationFile(sanitizedPath, rawLog);
47
+ // Issue #2189: `/log` accepts files up to Telegram's 50 MB document limit,
48
+ // and this used to hold the log twice (raw string + sanitized string) plus a
49
+ // third copy inside the sanitizer. The streaming sanitizer writes the same
50
+ // artifact one block at a time, so a 50 MB log costs the same as a 50 kB one.
51
+ const [, safeCaption] = await Promise.all([sanitizeLogFileToFileBounded({ sourcePath: logPath, destPath: sanitizedPath }), sanitizeForPublication(caption)]);
48
52
  return {
49
53
  path: sanitizedPath,
50
54
  caption: safeCaption,
@@ -7,6 +7,7 @@
7
7
 
8
8
  import fs from 'fs/promises';
9
9
  import { extractSessionIdFromText, decideLogDestination, resolveLogPath } from './telegram-log-command.lib.mjs';
10
+ import { readLogTailText } from './log-bounded-read.lib.mjs';
10
11
  import { parseSessionExitFooter } from './isolation-runner.lib.mjs';
11
12
  import { safeReply, safeSendMessage, safeEditMessageText } from './telegram-safe-reply.lib.mjs';
12
13
  import { classifyExitStatus, isFailureSessionStatus } from './session-status.lib.mjs';
@@ -15,6 +16,11 @@ const DEFAULT_WIDTH = 120;
15
16
  const DEFAULT_HEIGHT = 25;
16
17
  const DEFAULT_INTERVAL_MS = 2500;
17
18
  const DEFAULT_MAX_CHARS = 3400;
19
+ // Issue #2189: the watch loop re-reads the log every `intervalMs`, but it only
20
+ // ever renders the last `height` lines and looks for the exit footer, both of
21
+ // which live at the very end. Reading the whole file made a poll tick cost
22
+ // O(log size) — a 134 MB session log allocated 134 MB of string per tick.
23
+ const TERMINAL_WATCH_TAIL_BYTES = 256 * 1024;
18
24
  const GITHUB_URL_RE = /https:\/\/github\.com\/[^\s"'`<>]+/i;
19
25
  const activeWatches = new Map();
20
26
 
@@ -141,7 +147,9 @@ export function formatTerminalWatchMessage({ sessionId, statusResult = null, log
141
147
 
142
148
  async function readLogFile(logPath) {
143
149
  try {
144
- return await fs.readFile(logPath, 'utf8');
150
+ const { size } = await fs.stat(logPath);
151
+ if (!size) return '';
152
+ return await readLogTailText(logPath, { maxBytes: TERMINAL_WATCH_TAIL_BYTES });
145
153
  } catch (error) {
146
154
  if (error?.code === 'ENOENT') return '';
147
155
  throw error;