pi-crew 0.9.64 → 0.9.66
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +31 -0
- package/README.md +46 -1
- package/dist/index.mjs +423 -330
- package/package.json +3 -2
- package/scripts/pty_probe.py +10 -8
- package/skills/real-test-pi-crew/SKILL.md +6 -6
- package/src/config/config.ts +19 -3
- package/src/config/types.ts +2 -0
- package/src/extension/team-tool/cancel.ts +34 -0
- package/src/extension/team-tool/dispatch/manage.ts +7 -4
- package/src/extension/team-tool/explain.ts +3 -1
- package/src/extension/team-tool/lifecycle-actions.ts +4 -1
- package/src/extension/team-tool-types.ts +2 -0
- package/src/observability/event-to-metric.ts +29 -0
- package/src/observability/metrics-primitives.ts +41 -3
- package/src/runtime/README.md +1 -1
- package/src/runtime/broker/crew-broker.ts +0 -16
- package/src/runtime/effectiveness.ts +23 -1
- package/src/runtime/merge-gate.ts +202 -0
- package/src/runtime/model/model-fallback.ts +11 -0
- package/src/runtime/model/provider-extensions.ts +31 -12
- package/src/runtime/output/output-validator.ts +34 -6
- package/src/runtime/output/progress-tracker.ts +3 -33
- package/src/runtime/scheduling/scheduler.ts +67 -19
- package/src/runtime/scratchpad/engine.ts +40 -2
- package/src/runtime/scratchpad/snapshot-hmac.ts +161 -0
- package/src/runtime/team-runner.ts +128 -203
- package/src/schema/team-tool-schema.ts +2 -0
- package/src/teams/discover-teams.ts +2 -0
- package/src/teams/team-config.ts +7 -0
- package/src/teams/team-serializer.ts +1 -0
- package/src/ui/mascot.ts +1 -14
- package/teams/default.team.md +1 -0
- package/teams/fast-fix.team.md +1 -0
- package/src/observability/event-bus.ts +0 -86
- package/src/plugins/plugin-define.ts +0 -6
- package/src/plugins/plugin-registry.ts +0 -32
- package/src/plugins/plugins/index.ts +0 -3
- package/src/plugins/plugins/nextjs.ts +0 -19
- package/src/plugins/plugins/vite.ts +0 -10
- package/src/plugins/plugins/vitest.ts +0 -9
- package/src/runtime/child-pi/child-pi-pool.ts +0 -68
- package/src/runtime/iteration-hooks.ts +0 -305
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pi-crew",
|
|
3
|
-
"version": "0.9.
|
|
3
|
+
"version": "0.9.66",
|
|
4
4
|
"description": "Pi extension for coordinated AI teams, workflows, worktrees, and async task orchestration",
|
|
5
5
|
"author": "baphuongna",
|
|
6
6
|
"license": "MIT",
|
|
@@ -67,11 +67,12 @@
|
|
|
67
67
|
],
|
|
68
68
|
"scripts": {
|
|
69
69
|
"check": "npm run ci",
|
|
70
|
-
"ci": "npm run typecheck && npm run lint && npm run format:check && npm run check:conflict-markers && npm run check:lazy-imports && npm run check:bundle-staleness && npm run build:bundle && npm run check:bundle-size && npm run test:bundle && npm test && npm pack --dry-run",
|
|
70
|
+
"ci": "npm run typecheck && npm run lint && npm run format:check && npm run check:conflict-markers && npm run check:decision-drift && npm run check:lazy-imports && npm run check:bundle-staleness && npm run build:bundle && npm run check:bundle-size && npm run test:bundle && npm test && npm pack --dry-run",
|
|
71
71
|
"check:lazy-imports": "node scripts/check-lazy-imports.mjs",
|
|
72
72
|
"check:bundle-staleness": "node scripts/check-bundle-staleness.mjs",
|
|
73
73
|
"check:bundle-size": "node scripts/check-bundle-size.mjs",
|
|
74
74
|
"check:conflict-markers": "node scripts/check-conflict-markers.mjs",
|
|
75
|
+
"check:decision-drift": "node scripts/check-decision-drift.mjs",
|
|
75
76
|
"typecheck": "tsc --noEmit && node --experimental-strip-types -e \"await import('./index.ts'); console.log('strip-types import ok')\"",
|
|
76
77
|
"lint": "biome check --linter-enabled=true --formatter-enabled=false --max-diagnostics=50 .",
|
|
77
78
|
"format:check": "biome format .",
|
package/scripts/pty_probe.py
CHANGED
|
@@ -1,20 +1,22 @@
|
|
|
1
1
|
#!/usr/bin/env python3
|
|
2
|
-
"""pty_probe.py — bulk-key
|
|
2
|
+
"""pty_probe.py — bulk-key probe for pi-crew TUI components.
|
|
3
3
|
|
|
4
4
|
Spawns a real `pi` session under a pty, sends a sequence of keys with
|
|
5
5
|
short sleeps, captures the resulting output. Useful for verifying that
|
|
6
6
|
keystrokes reached the component's handleInput after a ui/ change.
|
|
7
7
|
|
|
8
|
+
NOTE (2026-08-10): the per-keystroke diag env var (PI_CREW_BROKER_DIAG_UI)
|
|
9
|
+
was REMOVED in e3ee6fe2 (PR-B5/UI-8 — "remove TEMP DIAGNOSTIC from
|
|
10
|
+
run-dashboard"). There is no replacement in src/. Keystroke arrival is now
|
|
11
|
+
proven by SCREEN-CHANGE evidence: capture the output frames before/after
|
|
12
|
+
each key and diff them — a key that changes screen state reached the TUI.
|
|
13
|
+
|
|
8
14
|
Requires Python 3.x on PATH. Unix only (Linux + macOS) — uses POSIX `pty.fork`.
|
|
9
15
|
Does NOT work on native Windows (no `pty` module); use WSL or Tier 5 (tmux) instead.
|
|
10
16
|
|
|
11
17
|
Usage:
|
|
12
18
|
python3 scripts/pty_probe.py [--keys j,k,q] [--cwd /path/to/repo]
|
|
13
19
|
|
|
14
|
-
Env:
|
|
15
|
-
PI_CREW_BROKER_DIAG_UI=1 enable diag stderr writes from run-dashboard
|
|
16
|
-
(only component wired; see src/ui/run-dashboard.ts:831)
|
|
17
|
-
|
|
18
20
|
Examples:
|
|
19
21
|
# Default probe (vim nav + arrow keys + quit)
|
|
20
22
|
python3 scripts/pty_probe.py
|
|
@@ -22,8 +24,8 @@ Examples:
|
|
|
22
24
|
# Custom probe: only arrow keys
|
|
23
25
|
python3 scripts/pty_probe.py --keys '\x1bOA,\x1bOB,\x1bOC,\x1bOD,q,q'
|
|
24
26
|
|
|
25
|
-
# Capture to a file
|
|
26
|
-
python3 scripts/pty_probe.py 2>&1 | tee /tmp/
|
|
27
|
+
# Capture to a file (diff frames for screen-change evidence)
|
|
28
|
+
python3 scripts/pty_probe.py 2>&1 | tee /tmp/pty-probe.log
|
|
27
29
|
"""
|
|
28
30
|
import argparse
|
|
29
31
|
import os
|
|
@@ -99,7 +101,7 @@ def main() -> int:
|
|
|
99
101
|
pass # use as-is if decode fails
|
|
100
102
|
keys.append(k)
|
|
101
103
|
|
|
102
|
-
env =
|
|
104
|
+
env = dict(os.environ) # keystroke diag env var removed (see module docstring)
|
|
103
105
|
|
|
104
106
|
pid, fd = pty.fork()
|
|
105
107
|
if pid == 0:
|
|
@@ -303,7 +303,7 @@ tmux capture-pane -t pi -p > /tmp/screen-after-up.txt
|
|
|
303
303
|
import os, sys, time
|
|
304
304
|
|
|
305
305
|
CMD = ['pi']
|
|
306
|
-
ENV =
|
|
306
|
+
ENV = dict(os.environ) # keystroke diag env var REMOVED (see note below)
|
|
307
307
|
|
|
308
308
|
pid, fd = pty.fork()
|
|
309
309
|
if pid == 0:
|
|
@@ -331,14 +331,14 @@ else:
|
|
|
331
331
|
> ```
|
|
332
332
|
> The inline code works for a quick one-off but **leaks a zombie `pi` process** on exit.
|
|
333
333
|
|
|
334
|
-
|
|
334
|
+
**Keystroke diag env var REMOVED (2026-08-10)**: `PI_CREW_BROKER_DIAG_UI=1` made `run-dashboard`'s `handleInput` write a `[PI-CREW-DIAG]` line to stderr per keystroke. It was removed in `e3ee6fe2` (PR-B5: remove TEMP DIAGNOSTIC from run-dashboard, UI-8) — there is no replacement in `src/`. **To prove keystroke arrival now, rely on screen-change evidence** (Tier 5 tmux `capture-pane` before/after each key, or the pty output diff): a key that changes screen state reached the TUI; a key that does not was consumed or never arrived. Capture the probe output to a file with `2>&1 | tee /tmp/pty-probe.log` and diff the rendered frames.
|
|
335
335
|
|
|
336
336
|
**References**:
|
|
337
337
|
|
|
338
338
|
| What | Where |
|
|
339
339
|
|---|---|
|
|
340
|
-
|
|
|
341
|
-
| Reduced-noise commit | `00e8ba0 chore(broker): strip diagnostic noise from focused-field fix` — diag calls left in but no longer noisy |
|
|
340
|
+
| Keystroke diag env var | **REMOVED** — `e3ee6fe2` (PR-B5/UI-8). No replacement; use screen-change evidence |
|
|
341
|
+
| Reduced-noise commit | `00e8ba0 chore(broker): strip diagnostic noise from focused-field fix` — diag calls left in but no longer noisy (pre-removal) |
|
|
342
342
|
| Original probe | `84944f7 test(probe): add invalidate() to control object so typecheck passes` |
|
|
343
343
|
|
|
344
344
|
---
|
|
@@ -637,7 +637,7 @@ The skill mentions specific commits, line numbers, and version pins. As the code
|
|
|
637
637
|
|---|---|---|
|
|
638
638
|
| Verify line refs after each `src/` commit | Every commit touching the cited file | `git log -p -- src/extension/registration/lifecycle-handlers.ts \| grep effectiveEnabled` — if line moved, update the skill |
|
|
639
639
|
| Verify commit hashes still exist | Quarterly or before major edits | `git log --oneline -1 <hash>` — if gone, find the equivalent newer commit |
|
|
640
|
-
| Verify version pins (v0.9.46, etc.) | Each release | `git log --oneline -- src/ui/run-dashboard.ts \| head -5` —
|
|
640
|
+
| Verify version pins (v0.9.46, etc.) | Each release | `git log --oneline -- src/ui/run-dashboard.ts \| head -5` — confirm diag removal history (e3ee6fe2) still accurate |
|
|
641
641
|
| Verify `test:critical` still has 14 files | Each `src/runtime/crew-broker*.ts` edit | `cat package.json \| grep test:critical` — adjust the file list |
|
|
642
642
|
| Verify Tier 7 verifier prompts still say `test:critical` | Each workflow file edit | `grep "Run FAST checks" workflows/*.workflow.md` |
|
|
643
643
|
|
|
@@ -662,7 +662,7 @@ readlink ../node_modules/pi-crew # dev: → ../pi-crew
|
|
|
662
662
|
readlink "$(npm root -g)"/pi-crew # global install
|
|
663
663
|
# Tier 5 (tmux probe)
|
|
664
664
|
tmux -S /tmp/sock new-session -d -x 160 -y 50 -s pi \
|
|
665
|
-
"cd ${PWD} &&
|
|
665
|
+
"cd ${PWD} && exec pi 2>&1"
|
|
666
666
|
tmux send-keys -t pi '<key>' ; sleep 0.5
|
|
667
667
|
tmux capture-pane -t pi -p
|
|
668
668
|
# Tier 6 (pty probe)
|
package/src/config/config.ts
CHANGED
|
@@ -1249,12 +1249,20 @@ export function updateConfig(patch: PiTeamsConfig, options: UpdateConfigOptions
|
|
|
1249
1249
|
for (const unset of options.unsetPaths) unsetPath(raw, unset);
|
|
1250
1250
|
merged = parseConfig(raw);
|
|
1251
1251
|
}
|
|
1252
|
+
// Skip-if-unchanged: an empty/identical patch must not rewrite the file
|
|
1253
|
+
// (e.g. `team action='config'` with an empty patch — read-only path).
|
|
1254
|
+
// Both sides are parseConfig-normalized, so JSON.stringify key order is
|
|
1255
|
+
// deterministic (same construction path); no key sorting needed.
|
|
1256
|
+
const normalizedCurrent = parseConfig(current);
|
|
1257
|
+
if (JSON.stringify(merged) === JSON.stringify(normalizedCurrent)) {
|
|
1258
|
+
return { path: filePath, config: merged, written: false }; // unchanged — skip write
|
|
1259
|
+
}
|
|
1252
1260
|
fs.mkdirSync(path.dirname(filePath), { recursive: true });
|
|
1253
1261
|
atomicWriteFile(filePath, `${JSON.stringify(merged, null, 2)}\n`);
|
|
1254
1262
|
// (F16) Invalidate the loadConfig cache after a write — the next
|
|
1255
1263
|
// caller must see the new value, not a 0-2s stale snapshot.
|
|
1256
1264
|
invalidateConfigCache();
|
|
1257
|
-
return { path: filePath, config: merged };
|
|
1265
|
+
return { path: filePath, config: merged, written: true };
|
|
1258
1266
|
});
|
|
1259
1267
|
}
|
|
1260
1268
|
|
|
@@ -1273,10 +1281,18 @@ export function updateAutonomousConfig(patch: PiTeamsAutonomousConfig): SavedPiT
|
|
|
1273
1281
|
current.autonomous && typeof current.autonomous === "object" && !Array.isArray(current.autonomous)
|
|
1274
1282
|
? (current.autonomous as Record<string, unknown>)
|
|
1275
1283
|
: {};
|
|
1276
|
-
|
|
1284
|
+
// Skip-if-unchanged (raw shape): a no-op autonomous patch must not
|
|
1285
|
+
// rewrite the file. NOTE: compare the RAW on-disk record, NOT the
|
|
1286
|
+
// parseConfig-normalized shape — normalizing would add default keys and
|
|
1287
|
+
// false-positive the equality check.
|
|
1288
|
+
const next = { ...current, autonomous: { ...currentAutonomous, ...patch } };
|
|
1289
|
+
if (JSON.stringify(next) === JSON.stringify(current)) {
|
|
1290
|
+
return { path: filePath, config: parseConfig(current), written: false }; // unchanged — skip write
|
|
1291
|
+
}
|
|
1292
|
+
current.autonomous = next.autonomous;
|
|
1277
1293
|
atomicWriteFile(filePath, `${JSON.stringify(current, null, 2)}\n`);
|
|
1278
1294
|
// (F16) Invalidate the loadConfig cache after a write — see updateConfig.
|
|
1279
1295
|
invalidateConfigCache();
|
|
1280
|
-
return { path: filePath, config: parseConfig(current) };
|
|
1296
|
+
return { path: filePath, config: parseConfig(current), written: true };
|
|
1281
1297
|
});
|
|
1282
1298
|
}
|
package/src/config/types.ts
CHANGED
|
@@ -312,6 +312,8 @@ export interface ConfigValidationResult {
|
|
|
312
312
|
export interface SavedPiTeamsConfig {
|
|
313
313
|
config: PiTeamsConfig;
|
|
314
314
|
path: string;
|
|
315
|
+
/** Whether the file was actually rewritten. `false` when a no-op patch hit the skip-write guard. */
|
|
316
|
+
written: boolean;
|
|
315
317
|
}
|
|
316
318
|
|
|
317
319
|
export interface UpdateConfigOptions {
|
|
@@ -21,6 +21,28 @@ import { enforceDestructiveIntent, intentFromConfig } from "./intent-policy.ts";
|
|
|
21
21
|
import { paramRequired } from "./param-error.ts";
|
|
22
22
|
import { RUN_NOT_FOUND_HINT } from "./run-not-found.ts";
|
|
23
23
|
|
|
24
|
+
/** Retryable terminal statuses (a task in one of these can be re-queued). */
|
|
25
|
+
const RETRYABLE_STATUSES: ReadonlySet<string> = new Set(["failed", "cancelled"]);
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Pure pre-lock decision for `action='retry'`: a run whose manifest status is
|
|
29
|
+
* "completed" (terminal success) and has no retryable tasks has nothing to
|
|
30
|
+
* retry. Returns true so the caller short-circuits with a clear message
|
|
31
|
+
* BEFORE acquiring the run lock — avoiding a misleading
|
|
32
|
+
* "run.lock is locked by another operation" error from a stale lock file left
|
|
33
|
+
* behind by a completed async run (finding #4, real-test-2026-08-10-full-9-tier).
|
|
34
|
+
*
|
|
35
|
+
* Exported for unit testing (handleRetry itself needs filesystem state).
|
|
36
|
+
*/
|
|
37
|
+
export function retryShortCircuitsCompleted(
|
|
38
|
+
runStatus: string,
|
|
39
|
+
tasks: ReadonlyArray<{ id: string; status: string }>,
|
|
40
|
+
targetTaskId?: string,
|
|
41
|
+
): boolean {
|
|
42
|
+
if (runStatus !== "completed") return false;
|
|
43
|
+
return !tasks.some((task) => (targetTaskId ? task.id === targetTaskId : true) && RETRYABLE_STATUSES.has(task.status));
|
|
44
|
+
}
|
|
45
|
+
|
|
24
46
|
export interface AbortOwnedResult {
|
|
25
47
|
abortedIds: string[];
|
|
26
48
|
missingIds: string[];
|
|
@@ -126,6 +148,18 @@ export async function handleRetry(params: TeamToolParamsValue, ctx: TeamContext,
|
|
|
126
148
|
|
|
127
149
|
const targetTaskId = typeof params.taskId === "string" ? params.taskId : undefined;
|
|
128
150
|
|
|
151
|
+
// Pre-lock terminal-status check: a completed run has nothing to retry.
|
|
152
|
+
// Short-circuit BEFORE acquiring the run lock so a stale lock file left by a
|
|
153
|
+
// completed async run does not surface a misleading "run.lock is locked by
|
|
154
|
+
// another operation" error (finding #4 in real-test-2026-08-10-full-9-tier).
|
|
155
|
+
if (retryShortCircuitsCompleted(loaded.manifest.status, loaded.tasks, targetTaskId)) {
|
|
156
|
+
return result(
|
|
157
|
+
`Run ${loaded.manifest.runId} is already completed; retry only applies to failed/cancelled runs.`,
|
|
158
|
+
{ action: "retry", status: "error", runId: loaded.manifest.runId },
|
|
159
|
+
true,
|
|
160
|
+
);
|
|
161
|
+
}
|
|
162
|
+
|
|
129
163
|
return withRunLockSync(loaded.manifest, () => {
|
|
130
164
|
const retryableStatuses: ReadonlySet<string> = new Set(["failed", "cancelled"]);
|
|
131
165
|
|
|
@@ -130,10 +130,13 @@ export async function handleManageDomain(params: TeamToolParamsValue, ctx: TeamC
|
|
|
130
130
|
unsetPaths,
|
|
131
131
|
});
|
|
132
132
|
return result(
|
|
133
|
-
[
|
|
134
|
-
"
|
|
135
|
-
|
|
136
|
-
|
|
133
|
+
[
|
|
134
|
+
saved.written ? "Updated pi-crew config." : "Config unchanged (no effective changes).",
|
|
135
|
+
`Path: ${saved.path}`,
|
|
136
|
+
"Effective config:",
|
|
137
|
+
JSON.stringify(saved.config, null, 2),
|
|
138
|
+
].join("\n"),
|
|
139
|
+
{ action: "config", status: "ok", written: saved.written },
|
|
137
140
|
);
|
|
138
141
|
} catch (error) {
|
|
139
142
|
const message = error instanceof Error ? error.message : String(error);
|
|
@@ -2,6 +2,7 @@ import * as fs from "node:fs";
|
|
|
2
2
|
import * as path from "node:path";
|
|
3
3
|
import { loadRunManifestById } from "../../state/stores/state-store.ts";
|
|
4
4
|
import type { TeamRunManifest, TeamTaskState } from "../../state/types.ts";
|
|
5
|
+
import { locateRunCwd } from "../team-tool.ts";
|
|
5
6
|
import { RUN_NOT_FOUND_HINT } from "./run-not-found.ts";
|
|
6
7
|
|
|
7
8
|
/**
|
|
@@ -215,7 +216,8 @@ export function handleExplain(
|
|
|
215
216
|
return result("explain requires runId", { action: "explain", status: "error" }, true);
|
|
216
217
|
}
|
|
217
218
|
|
|
218
|
-
const
|
|
219
|
+
const runCwd = locateRunCwd(params.runId, cwd);
|
|
220
|
+
const loaded = runCwd ? loadRunManifestById(runCwd, params.runId) : undefined; // NOTE: no withRunLock - best-effort only; concurrent writes may cause inconsistency
|
|
219
221
|
if (!loaded) {
|
|
220
222
|
return result(`Run '${params.runId}' not found.${RUN_NOT_FOUND_HINT}`, { action: "explain", status: "error" }, true);
|
|
221
223
|
}
|
|
@@ -16,6 +16,7 @@ import { listImportedRuns } from "../import-index.ts";
|
|
|
16
16
|
import { exportRunBundle } from "../run-export.ts";
|
|
17
17
|
import { importRunBundle } from "../run-import.ts";
|
|
18
18
|
import { pruneFinishedRuns } from "../run-maintenance.ts";
|
|
19
|
+
import { locateRunCwd } from "../team-tool.ts";
|
|
19
20
|
import type { PiTeamsToolResult } from "../tool-result.ts";
|
|
20
21
|
import { configRecord, result, type TeamContext } from "./context.ts";
|
|
21
22
|
import { enforceDestructiveIntent, intentFromConfig } from "./intent-policy.ts";
|
|
@@ -29,7 +30,9 @@ export function handleWorktrees(params: TeamToolParamsValue, ctx: TeamContext):
|
|
|
29
30
|
{ action: "worktrees", status: "error" },
|
|
30
31
|
true,
|
|
31
32
|
);
|
|
32
|
-
const
|
|
33
|
+
const runCwd = locateRunCwd(params.runId, ctx.cwd);
|
|
34
|
+
if (!runCwd) return result(`Run '${params.runId}' not found.${RUN_NOT_FOUND_HINT}`, { action: "worktrees", status: "error" }, true);
|
|
35
|
+
const loaded = loadRunManifestById(runCwd, params.runId); // NOTE: no withRunLock - best-effort only; concurrent writes may cause inconsistency
|
|
33
36
|
if (!loaded) return result(`Run '${params.runId}' not found.${RUN_NOT_FOUND_HINT}`, { action: "worktrees", status: "error" }, true);
|
|
34
37
|
const withWorktrees = loaded.tasks.filter((task) => task.worktree);
|
|
35
38
|
const lines = [
|
|
@@ -10,6 +10,8 @@ export interface TeamToolDetails {
|
|
|
10
10
|
resumedIds?: string[];
|
|
11
11
|
retriedTaskIds?: string[];
|
|
12
12
|
mailboxIds?: string[];
|
|
13
|
+
/** Whether a config write actually persisted (false on no-op/skip-write). */
|
|
14
|
+
written?: boolean;
|
|
13
15
|
/** Resource scope affected by the action (e.g. cleanup: "project"). */
|
|
14
16
|
scope?: string;
|
|
15
17
|
/** Run metrics for compact display in TUI tool result rendering. */
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import type { ExtensionAPI } from "@earendil-works/pi-coding-agent";
|
|
2
2
|
import type { MetricRegistry } from "./metric-registry.ts";
|
|
3
|
+
import { getCardinalityEvictions } from "./metrics-primitives.ts";
|
|
3
4
|
|
|
4
5
|
function recordValue(value: unknown): Record<string, unknown> {
|
|
5
6
|
return value && typeof value === "object" && !Array.isArray(value) ? (value as Record<string, unknown>) : {};
|
|
@@ -41,6 +42,19 @@ export function wireEventToMetrics(events: ExtensionAPI["events"] | undefined, r
|
|
|
41
42
|
const deadletterCount = registry.counter("crew.task.deadletter_total", "Deadletter triggers by reason");
|
|
42
43
|
const overflowCount = registry.counter("crew.task.overflow_phase_total", "Overflow recovery phase transitions");
|
|
43
44
|
const supervisorContactCount = registry.counter("crew.task.supervisor_contact_total", "Supervisor contact requests by reason");
|
|
45
|
+
const unboundedConcurrencyCount = registry.counter(
|
|
46
|
+
"crew.limits.unbounded_total",
|
|
47
|
+
"Runs that enabled allowUnboundedConcurrency (advisory; bypasses hard cap)",
|
|
48
|
+
);
|
|
49
|
+
// Gauge reflecting cumulative label-combination evictions across all
|
|
50
|
+
// metrics in this process. Updated lazily on every metric event so the
|
|
51
|
+
// value stays fresh without a separate timer. Non-zero = aggregation
|
|
52
|
+
// for high-cardinality labels is unreliable. See metrics-primitives.ts
|
|
53
|
+
// `enforceLabelCap`.
|
|
54
|
+
const cardinalityEvictedGauge = registry.gauge(
|
|
55
|
+
"crew.metrics.cardinality_evicted",
|
|
56
|
+
"Cumulative label-combination evictions (non-zero = unreliable aggregation)",
|
|
57
|
+
);
|
|
44
58
|
registry.gauge("crew.heartbeat.staleness_ms", "Heartbeat elapsed since last seen, milliseconds");
|
|
45
59
|
const runDuration = registry.histogram(
|
|
46
60
|
"crew.run.duration_ms",
|
|
@@ -146,11 +160,26 @@ export function wireEventToMetrics(events: ExtensionAPI["events"] | undefined, r
|
|
|
146
160
|
});
|
|
147
161
|
},
|
|
148
162
|
],
|
|
163
|
+
[
|
|
164
|
+
"crew.limits.unbounded",
|
|
165
|
+
() => {
|
|
166
|
+
// Fires when a run enables allowUnboundedConcurrency (bypasses the
|
|
167
|
+
// hard cap of 8 and the worker cap). One emit per affected run.
|
|
168
|
+
unboundedConcurrencyCount.inc({});
|
|
169
|
+
},
|
|
170
|
+
],
|
|
149
171
|
];
|
|
150
172
|
|
|
151
173
|
const unsubscribers: Array<() => void> = [];
|
|
152
174
|
for (const [event, handler] of handlers) {
|
|
153
175
|
const unsubscribe = events?.on?.(event, (data: unknown) => {
|
|
176
|
+
// Refresh the cardinality-eviction gauge on every event so the
|
|
177
|
+
// value reflects the latest eviction state without a timer.
|
|
178
|
+
try {
|
|
179
|
+
cardinalityEvictedGauge.set({}, getCardinalityEvictions());
|
|
180
|
+
} catch {
|
|
181
|
+
/* gauge refresh must never break event delivery */
|
|
182
|
+
}
|
|
154
183
|
try {
|
|
155
184
|
handler(data);
|
|
156
185
|
} catch {
|
|
@@ -36,13 +36,51 @@ interface StoredHistogram {
|
|
|
36
36
|
|
|
37
37
|
export const DEFAULT_HISTOGRAM_BUCKETS = [1, 2, 5, 10, 25, 50, 100, 250, 500, 1000, 2500, 5000, 10000] as const;
|
|
38
38
|
|
|
39
|
-
/**
|
|
39
|
+
/**
|
|
40
|
+
* Maximum number of unique label combinations per metric.
|
|
41
|
+
*
|
|
42
|
+
* When this cap is reached, the oldest label combination is silently
|
|
43
|
+
* evicted (FIFO by insertion order, with MRU promotion on use). Read
|
|
44
|
+
* {@link getCardinalityEvictions} to detect when aggregation has become
|
|
45
|
+
* unreliable for high-cardinality labels.
|
|
46
|
+
*/
|
|
40
47
|
const MAX_LABEL_COMBINATIONS = 10_000;
|
|
41
48
|
|
|
42
|
-
|
|
49
|
+
/**
|
|
50
|
+
* Cumulative count of label-combination evictions across all metrics in
|
|
51
|
+
* this process. Incremented every time {@link enforceLabelCap} drops an
|
|
52
|
+
* entry. Exported so observability wiring and OTLP/Prometheus exporters
|
|
53
|
+
* can surface `crew.metrics.cardinality_evicted` without risking the
|
|
54
|
+
* recursion of routing this signal through a `Counter` (a Counter itself
|
|
55
|
+
* consumes a label slot and could self-evict).
|
|
56
|
+
*/
|
|
57
|
+
let cardinalityEvictions = 0;
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Returns the cumulative number of metric label-combination evictions
|
|
61
|
+
* that have occurred in this process. A non-zero value means at least
|
|
62
|
+
* one metric exceeded {@link MAX_LABEL_COMBINATIONS} and silently dropped
|
|
63
|
+
* the oldest label combination — aggregation for high-cardinality labels
|
|
64
|
+
* is unreliable.
|
|
65
|
+
*/
|
|
66
|
+
export function getCardinalityEvictions(): number {
|
|
67
|
+
return cardinalityEvictions;
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** Reset the eviction counter. Intended for tests only. */
|
|
71
|
+
export function _resetCardinalityEvictionsForTests(): void {
|
|
72
|
+
cardinalityEvictions = 0;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function enforceLabelCap(map: Map<string, unknown>, _metricName: string): void {
|
|
43
76
|
while (map.size > MAX_LABEL_COMBINATIONS) {
|
|
44
77
|
const firstKey = map.keys().next().value;
|
|
45
|
-
if (firstKey !== undefined)
|
|
78
|
+
if (firstKey !== undefined) {
|
|
79
|
+
map.delete(firstKey);
|
|
80
|
+
cardinalityEvictions++;
|
|
81
|
+
} else {
|
|
82
|
+
break;
|
|
83
|
+
}
|
|
46
84
|
}
|
|
47
85
|
}
|
|
48
86
|
|
package/src/runtime/README.md
CHANGED
|
@@ -7,7 +7,7 @@ clusters are extracted into subdirectories; the remaining files stay at the root
|
|
|
7
7
|
|
|
8
8
|
| Subdir | Phase | Cluster | Key files |
|
|
9
9
|
|--------|-------|---------|-----------|
|
|
10
|
-
| [`child-pi/`](./child-pi/) | B-1 | child Pi worker spawn/lifecycle/steering/transcript | `child-pi.ts`, `child-pi-spawn.ts`, `child-pi-kill.ts`, `child-pi-constants.ts`, `child-pi-steering.ts`, `child-pi-streams.ts`, `child-pi-transcript.ts
|
|
10
|
+
| [`child-pi/`](./child-pi/) | B-1 | child Pi worker spawn/lifecycle/steering/transcript | `child-pi.ts`, `child-pi-spawn.ts`, `child-pi-kill.ts`, `child-pi-constants.ts`, `child-pi-steering.ts`, `child-pi-streams.ts`, `child-pi-transcript.ts` |
|
|
11
11
|
| [`broker/`](./broker/) | B-2 | crew broker server/client/auth (child-pi ↔ parent IPC) | `crew-broker.ts`, `crew-broker-client.ts`, `crew-broker-child.ts`, `crew-broker-tokens.ts`, `broker-issuer.ts` |
|
|
12
12
|
| [`task-runner/`](./task-runner/) | (pre-existing) | per-task execution (pre-execution, child-executor) | `child-executor.ts`, `pre-execution.ts`, ... |
|
|
13
13
|
| [`live-session/`](./live-session/) | 3 | live agent control/manager, session runtime, IRC, health, extension bridge | `live-session-runtime.ts`, `live-agent-manager.ts`, `live-agent-control.ts`, `live-control-realtime.ts`, `live-irc.ts`, `live-session-health.ts`, `live-extension-bridge.ts`, `intercom-bridge.ts` |
|
|
@@ -357,22 +357,6 @@ export class CrewBroker {
|
|
|
357
357
|
this.resolvedSocketPath = null;
|
|
358
358
|
}
|
|
359
359
|
|
|
360
|
-
/**
|
|
361
|
-
* Non-throwing enqueue entry point for the post-append mailbox observer
|
|
362
|
-
* (Phase 1) or any other in-process producer. Phase 0 accepts `notifyMessage`
|
|
363
|
-
* as a no-op shape so the lifecycle controller can install a single
|
|
364
|
-
* observer regardless of broker state.
|
|
365
|
-
*
|
|
366
|
-
* Fanout goes ONLY to authenticated connections matching the recipient.
|
|
367
|
-
* Phase 0 keeps this as a typed no-op (`not-implemented` would be
|
|
368
|
-
* inappropriate here — the caller is in-process and shouldn't be
|
|
369
|
-
* punished for testing the broker skeleton).
|
|
370
|
-
*/
|
|
371
|
-
notifyMessage(_message: unknown): void {
|
|
372
|
-
// Phase 0: no fanout. Phase 1 replaces this with the single
|
|
373
|
-
// post-durable mailbox observer fanout.
|
|
374
|
-
}
|
|
375
|
-
|
|
376
360
|
// ------------------------------------------------------------------------
|
|
377
361
|
// Connection lifecycle
|
|
378
362
|
// ------------------------------------------------------------------------
|
|
@@ -26,6 +26,24 @@ export function taskHasObservableWorkerActivity(task: TeamTaskState): boolean {
|
|
|
26
26
|
);
|
|
27
27
|
}
|
|
28
28
|
|
|
29
|
+
/**
|
|
30
|
+
* True when the task completed but its result artifact is an EMPTY file
|
|
31
|
+
* (0 bytes). Closes the F4 monitoring gap (real-test-2026-08-10 finding):
|
|
32
|
+
* a child worker absorbed by a rate-limit (429) or a model-not-found
|
|
33
|
+
* failure still emits transcript/usage events, so
|
|
34
|
+
* {@link taskHasObservableWorkerActivity} reports "activity" — but the
|
|
35
|
+
* actual result content is empty. A completed task with an empty result
|
|
36
|
+
* has done no real work and must not count toward run effectiveness.
|
|
37
|
+
*
|
|
38
|
+
* Deliberately only flags `sizeBytes === 0` (a written-but-empty file).
|
|
39
|
+
* A missing resultArtifact stays non-flagging (some tasks legitimately
|
|
40
|
+
* produce no result file), and an undefined sizeBytes stays non-flagging
|
|
41
|
+
* (legacy tasks — do not false-positive on absent metadata).
|
|
42
|
+
*/
|
|
43
|
+
export function taskHasEmptyResult(task: TeamTaskState): boolean {
|
|
44
|
+
return Boolean(task.resultArtifact && task.resultArtifact.sizeBytes === 0);
|
|
45
|
+
}
|
|
46
|
+
|
|
29
47
|
export function resolveEffectivenessGuardMode(
|
|
30
48
|
runtimeConfig: CrewRuntimeConfig | undefined,
|
|
31
49
|
manifest?: TeamRunManifest,
|
|
@@ -43,7 +61,11 @@ export function evaluateRunEffectiveness(input: {
|
|
|
43
61
|
runtimeConfig?: CrewRuntimeConfig;
|
|
44
62
|
}): RunEffectivenessSummary {
|
|
45
63
|
const completedTasks = input.tasks.filter((task) => task.status === "completed");
|
|
46
|
-
|
|
64
|
+
// F4: a completed task with an EMPTY result artifact (0 bytes) is
|
|
65
|
+
// treated exactly like no-observed-work — the worker may have spun up
|
|
66
|
+
// and streamed transcript events, but it produced no real output
|
|
67
|
+
// (e.g. a 429 rate-limit absorbed by the fallback chain).
|
|
68
|
+
const noObservedWorkTasks = completedTasks.filter((task) => !taskHasObservableWorkerActivity(task) || taskHasEmptyResult(task));
|
|
47
69
|
const needsAttentionTasks = input.tasks.filter((task) => task.agentProgress?.activityState === "needs_attention");
|
|
48
70
|
const workerExecution: WorkerExecutionState = input.executeWorkers ? "enabled" : "disabled/scaffold";
|
|
49
71
|
const guardMode = resolveEffectivenessGuardMode(input.runtimeConfig, input.manifest);
|