niceeval 0.10.3-canary.7 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agents/types.d.ts +5 -4
- package/dist/context/turn-errors.d.ts +27 -23
- package/dist/i18n/en.d.ts +14 -0
- package/dist/i18n/en.js +15 -1
- package/dist/i18n/zh-CN.d.ts +15 -1
- package/dist/i18n/zh-CN.js +15 -1
- package/dist/o11y/derive.js +28 -24
- package/dist/o11y/types.d.ts +8 -5
- package/dist/report/components/attempt-detail/UsageTable.js +4 -7
- package/dist/report/components/attempt-detail/compute.d.ts +2 -2
- package/dist/report/components/attempt-detail/compute.js +2 -6
- package/dist/report/components/attempt-detail/faces.js +7 -7
- package/dist/report/components/attempt-detail/index.js +0 -3
- package/dist/report/components/entity-lists/EvalList.js +0 -0
- package/dist/report/components/metric-views/compute.js +1 -1
- package/dist/report/model/types.d.ts +3 -4
- package/dist/results/locator.js +0 -0
- package/dist/results/select.d.ts +6 -0
- package/dist/results/select.js +8 -0
- package/dist/runner/feedback/sink.d.ts +26 -1
- package/dist/runner/fingerprint.d.ts +23 -0
- package/dist/runner/types.d.ts +105 -7
- package/dist/sandbox/errors.d.ts +29 -0
- package/dist/sandbox/resolve.d.ts +9 -0
- package/dist/shared/failure-class.d.ts +91 -0
- package/dist/types.d.ts +1 -0
- package/dist/util.d.ts +3 -2
- package/dist/util.js +31 -5
- package/docs-site/zh/explanation/runner.mdx +35 -0
- package/docs-site/zh/reference/cli.mdx +3 -3
- package/docs-site/zh/reference/events.mdx +4 -4
- package/docs-site/zh/troubleshooting/debugging.mdx +2 -2
- package/docs-site/zh/tutorials/viewing-results.mdx +4 -5
- package/package.json +4 -12
- package/src/agents/ai-sdk.test.ts +26 -0
- package/src/agents/ai-sdk.ts +7 -4
- package/src/agents/index.ts +5 -3
- package/src/agents/langgraph.test.ts +30 -0
- package/src/agents/langgraph.ts +5 -2
- package/src/agents/openai-compat.test.ts +35 -0
- package/src/agents/openai-compat.ts +16 -4
- package/src/agents/sdk-streams.test.ts +66 -0
- package/src/agents/sdk-streams.ts +3 -1
- package/src/agents/types.ts +5 -4
- package/src/cli.ts +55 -14
- package/src/context/context.test.ts +34 -0
- package/src/context/context.ts +14 -5
- package/src/context/send-retry.test.ts +86 -0
- package/src/context/send-retry.ts +37 -12
- package/src/context/session.ts +24 -0
- package/src/context/turn-errors.test.ts +124 -17
- package/src/context/turn-errors.ts +60 -50
- package/src/define.ts +5 -0
- package/src/i18n/en.ts +20 -1
- package/src/i18n/zh-CN.ts +20 -1
- package/src/index.ts +8 -0
- package/src/o11y/cost.ts +4 -2
- package/src/o11y/derive.test.ts +40 -0
- package/src/o11y/derive.ts +28 -22
- package/src/o11y/otlp/sandbox-receiver.test.ts +201 -0
- package/src/o11y/otlp/sandbox-receiver.ts +73 -27
- package/src/o11y/parsers/bub.test.ts +30 -0
- package/src/o11y/parsers/bub.ts +5 -2
- package/src/o11y/parsers/codex.test.ts +19 -0
- package/src/o11y/parsers/codex.ts +5 -2
- package/src/o11y/types.ts +8 -5
- package/src/report/components/attempt-detail/UsageTable.tsx +4 -6
- package/src/report/components/attempt-detail/attempt-components.test.tsx +5 -8
- package/src/report/components/attempt-detail/compute.ts +2 -7
- package/src/report/components/attempt-detail/faces.ts +7 -7
- package/src/report/components/attempt-detail/index.tsx +0 -3
- package/src/report/components/entity-lists/EvalList.tsx +0 -0
- package/src/report/components/metric-views/compute.ts +1 -1
- package/src/report/model/types.ts +3 -4
- package/src/results/format.ts +9 -2
- package/src/results/index.ts +2 -0
- package/src/results/locator.ts +0 -0
- package/src/results/open.ts +132 -21
- package/src/results/select.ts +12 -0
- package/src/results/skipped-notice.ts +0 -0
- package/src/runner/attempt.test.ts +116 -0
- package/src/runner/attempt.ts +89 -8
- package/src/runner/discover.ts +21 -3
- package/src/runner/feedback/coordinator.ts +24 -2
- package/src/runner/feedback/eval-conclusions.ts +6 -3
- package/src/runner/feedback/human.test.ts +300 -6
- package/src/runner/feedback/human.ts +132 -30
- package/src/runner/feedback/json.test.ts +127 -2
- package/src/runner/feedback/json.ts +51 -3
- package/src/runner/feedback/reducer.test.ts +328 -29
- package/src/runner/feedback/reducer.ts +84 -9
- package/src/runner/feedback/sink.ts +39 -1
- package/src/runner/fingerprint.ts +49 -19
- package/src/runner/gate-lease.test.ts +510 -0
- package/src/runner/gate-lease.ts +350 -0
- package/src/runner/lock.test.ts +454 -0
- package/src/runner/lock.ts +288 -0
- package/src/runner/report.test.ts +1 -0
- package/src/runner/run.test.ts +2044 -9
- package/src/runner/run.ts +825 -61
- package/src/runner/teardown-registry.ts +20 -78
- package/src/runner/types.ts +103 -7
- package/src/sandbox/errors.test.ts +72 -0
- package/src/sandbox/errors.ts +83 -0
- package/src/sandbox/keep-registry.ts +22 -46
- package/src/sandbox/resolve.test.ts +100 -0
- package/src/sandbox/resolve.ts +84 -27
- package/src/shared/entry-file-store.test.ts +149 -0
- package/src/shared/entry-file-store.ts +117 -0
- package/src/shared/failure-class.test.ts +137 -0
- package/src/shared/failure-class.ts +175 -0
- package/src/show/index.ts +5 -6
- package/src/show/render.test.ts +116 -15
- package/src/show/render.ts +124 -35
- package/src/types.ts +9 -0
- package/src/util.ts +31 -4
- package/src/view/app/App.tsx +5 -1
- package/src/view/client-dist/app.js +1 -1
- package/src/view/data.ts +5 -9
- package/src/view/shared/types.ts +7 -1
- package/src/view/view-report.test.ts +54 -0
|
@@ -5,10 +5,10 @@
|
|
|
5
5
|
// boxed 能力下产生可识别的框线字符与正确的面板顺序/分隔,plain/非 TTY 下不产生任何框字符。
|
|
6
6
|
|
|
7
7
|
import { afterEach, describe, expect, it } from "vitest";
|
|
8
|
-
import { createHumanRenderer, renderDurableLines } from "./human.ts";
|
|
8
|
+
import { createHumanRenderer, renderDurableLines, renderHumanDryPlan } from "./human.ts";
|
|
9
9
|
import { createFakeFeedbackIO } from "./testing.ts";
|
|
10
|
-
import { createInitialRunFeedbackState } from "./reducer.ts";
|
|
11
|
-
import { encodeAttemptKey } from "../types.ts";
|
|
10
|
+
import { createInitialRunFeedbackState, reduceRunFeedback } from "./reducer.ts";
|
|
11
|
+
import { encodeAttemptKey, HALT_DIAGNOSTIC_CODE } from "../types.ts";
|
|
12
12
|
import { stringWidth } from "../../report/model/text-layout.ts";
|
|
13
13
|
import type { DurableFeedbackEvent, InvocationCompletion, InvocationSummary, RunFeedbackPlan, RunFeedbackState } from "../types.ts";
|
|
14
14
|
import type { AttemptLocator } from "../../results/locator.ts";
|
|
@@ -182,7 +182,8 @@ describe("live dashboard — 接线到 panel.ts", () => {
|
|
|
182
182
|
reused: 6,
|
|
183
183
|
running: 19,
|
|
184
184
|
queued: 12,
|
|
185
|
-
|
|
185
|
+
passed: 6,
|
|
186
|
+
failed: 2,
|
|
186
187
|
elapsedMs: 134_000,
|
|
187
188
|
estimatedCostUSD: 0.84,
|
|
188
189
|
active: new Map([[key, { identity, who: "compare/bub-e2b", phase: "eval.run", phaseStartedAt: 0 }]]),
|
|
@@ -260,7 +261,8 @@ describe("live dashboard — 宽终端下 ACTIVE 行与身份列分配", () => {
|
|
|
260
261
|
reused: 6,
|
|
261
262
|
running: 19,
|
|
262
263
|
queued: 12,
|
|
263
|
-
|
|
264
|
+
passed: 6,
|
|
265
|
+
failed: 2,
|
|
264
266
|
elapsedMs: 134_000,
|
|
265
267
|
active: new Map([
|
|
266
268
|
[key, { identity, who: "compare/bub-e2b", phase: "eval.run", phaseStartedAt: 0, detail: longDetail }],
|
|
@@ -358,7 +360,7 @@ describe("live dashboard — 宽终端下 ACTIVE 行与身份列分配", () => {
|
|
|
358
360
|
...createInitialRunFeedbackState(),
|
|
359
361
|
total: 2,
|
|
360
362
|
running: 1,
|
|
361
|
-
|
|
363
|
+
passed: 1,
|
|
362
364
|
active: new Map([[shortKey, { identity: shortIdentity, who, phase: "eval.run", phaseStartedAt: 1 }]]),
|
|
363
365
|
};
|
|
364
366
|
renderer.onLifecycle?.(
|
|
@@ -418,3 +420,295 @@ describe("live dashboard — 宽终端下 ACTIVE 行与身份列分配", () => {
|
|
|
418
420
|
for (const l of framedLines) expect(stringWidth(l)).toBe(100);
|
|
419
421
|
});
|
|
420
422
|
});
|
|
423
|
+
|
|
424
|
+
// cases: docs/engineering/testing/unit/experiments-runner.md「用例锁与并发 Invocation」——
|
|
425
|
+
// 字节级精确渲染归 E2E · CLI「反馈输出格式」;这里只做与 precheck/experiment-hook 同等级别
|
|
426
|
+
// 的最小 smoke 断言(行是否出现、关键子串是否存在),不断言列宽算术。
|
|
427
|
+
describe("用例锁等待(elsewhere)的显示", () => {
|
|
428
|
+
it("TTY:等待期间显示运行级行,面板首行的 elsewhere 计数非零", () => {
|
|
429
|
+
const { io, stderr } = createFakeFeedbackIO({ stderr: { isTTY: true, columns: 100, rows: 30 } });
|
|
430
|
+
const renderer = createHumanRenderer({ io, command: "niceeval exp compare/codex" });
|
|
431
|
+
const state: RunFeedbackState = {
|
|
432
|
+
...createInitialRunFeedbackState(),
|
|
433
|
+
total: 3,
|
|
434
|
+
elsewhere: 2,
|
|
435
|
+
queued: 1,
|
|
436
|
+
lockWaits: new Map([
|
|
437
|
+
[
|
|
438
|
+
"compare/codex",
|
|
439
|
+
{
|
|
440
|
+
experimentId: "compare/codex",
|
|
441
|
+
waiting: new Map([
|
|
442
|
+
["memory/a", { startedAt: 0, holderPid: 41267, holderHost: "mba.local" }],
|
|
443
|
+
["memory/b", { startedAt: 5, holderPid: 41267, holderHost: "mba.local" }],
|
|
444
|
+
]),
|
|
445
|
+
resolvedCarried: 0,
|
|
446
|
+
resolvedDispatched: 0,
|
|
447
|
+
},
|
|
448
|
+
],
|
|
449
|
+
]),
|
|
450
|
+
};
|
|
451
|
+
renderer.redrawDynamic?.(state);
|
|
452
|
+
|
|
453
|
+
const plain = stripAnsi(stderr.writes.join(""));
|
|
454
|
+
expect(plain).toMatch(/├─ ACTIVE ─+┤/);
|
|
455
|
+
expect(plain).toContain("waiting on another run");
|
|
456
|
+
expect(plain).toContain("compare/codex");
|
|
457
|
+
expect(plain).toContain("2 evals");
|
|
458
|
+
expect(plain).toContain("pid 41267");
|
|
459
|
+
expect(plain).toContain("2 elsewhere");
|
|
460
|
+
});
|
|
461
|
+
|
|
462
|
+
it("TTY appendDurable 对 lock-wait 直接返回,不写 scrollback 永久行(运行级行由 state.lockWaits 驱动)", () => {
|
|
463
|
+
const { io, stdout, stderr } = createFakeFeedbackIO({ stderr: { isTTY: true, columns: 100, rows: 30 } });
|
|
464
|
+
const renderer = createHumanRenderer({ io, command: "niceeval exp compare/codex" });
|
|
465
|
+
const state: RunFeedbackState = { ...createInitialRunFeedbackState(), total: 1, elsewhere: 1 };
|
|
466
|
+
renderer.appendDurable(
|
|
467
|
+
{ type: "lock-wait", at: 0, experimentId: "compare/codex", evalId: "memory/a", status: "started", attempts: 1, holderPid: 1, holderHost: "h" },
|
|
468
|
+
state,
|
|
469
|
+
);
|
|
470
|
+
expect(stdout.writes.join("") + stderr.writes.join("")).toBe("");
|
|
471
|
+
});
|
|
472
|
+
|
|
473
|
+
it("非 TTY:started 只在窗口第一次打开(唯一等待用例)时追加一行,中途加入的用例不逐条刷屏", () => {
|
|
474
|
+
const { io, stdout } = createFakeFeedbackIO({ stderr: { isTTY: false } });
|
|
475
|
+
const renderer = createHumanRenderer({ io, command: "niceeval exp compare/codex" });
|
|
476
|
+
const firstState: RunFeedbackState = {
|
|
477
|
+
...createInitialRunFeedbackState(),
|
|
478
|
+
total: 2,
|
|
479
|
+
elsewhere: 1,
|
|
480
|
+
lockWaits: new Map([
|
|
481
|
+
[
|
|
482
|
+
"compare/codex",
|
|
483
|
+
{
|
|
484
|
+
experimentId: "compare/codex",
|
|
485
|
+
waiting: new Map([["memory/a", { startedAt: 0, holderPid: 41267 }]]),
|
|
486
|
+
resolvedCarried: 0,
|
|
487
|
+
resolvedDispatched: 0,
|
|
488
|
+
},
|
|
489
|
+
],
|
|
490
|
+
]),
|
|
491
|
+
};
|
|
492
|
+
renderer.appendDurable(
|
|
493
|
+
{ type: "lock-wait", at: 0, experimentId: "compare/codex", evalId: "memory/a", status: "started", attempts: 1, holderPid: 41267, holderHost: "h" },
|
|
494
|
+
firstState,
|
|
495
|
+
);
|
|
496
|
+
const secondState: RunFeedbackState = {
|
|
497
|
+
...firstState,
|
|
498
|
+
elsewhere: 2,
|
|
499
|
+
lockWaits: new Map([
|
|
500
|
+
[
|
|
501
|
+
"compare/codex",
|
|
502
|
+
{
|
|
503
|
+
...firstState.lockWaits.get("compare/codex")!,
|
|
504
|
+
waiting: new Map([
|
|
505
|
+
["memory/a", { startedAt: 0, holderPid: 41267 }],
|
|
506
|
+
["memory/b", { startedAt: 1, holderPid: 41267 }],
|
|
507
|
+
]),
|
|
508
|
+
},
|
|
509
|
+
],
|
|
510
|
+
]),
|
|
511
|
+
};
|
|
512
|
+
renderer.appendDurable(
|
|
513
|
+
{ type: "lock-wait", at: 1, experimentId: "compare/codex", evalId: "memory/b", status: "started", attempts: 1, holderPid: 41267, holderHost: "h" },
|
|
514
|
+
secondState,
|
|
515
|
+
);
|
|
516
|
+
|
|
517
|
+
const out = stdout.writes.join("");
|
|
518
|
+
expect(out).toContain("waiting on another run · compare/codex");
|
|
519
|
+
// 只出现一次:第二条(memory/b 加入)是同一窗口内的非首条,静默不刷屏。
|
|
520
|
+
expect(out.split("waiting on another run").length - 1).toBe(1);
|
|
521
|
+
});
|
|
522
|
+
|
|
523
|
+
it("非 TTY:resolved 只在窗口最后一次关闭(全部等待用例都已解决)时追加聚合收尾行", () => {
|
|
524
|
+
const { io, stdout } = createFakeFeedbackIO({ stderr: { isTTY: false } });
|
|
525
|
+
const renderer = createHumanRenderer({ io, command: "niceeval exp compare/codex" });
|
|
526
|
+
// 还剩一个用例没解决:窗口未关闭,静默。
|
|
527
|
+
const stillWaitingState: RunFeedbackState = {
|
|
528
|
+
...createInitialRunFeedbackState(),
|
|
529
|
+
lockWaits: new Map([
|
|
530
|
+
[
|
|
531
|
+
"compare/codex",
|
|
532
|
+
{
|
|
533
|
+
experimentId: "compare/codex",
|
|
534
|
+
waiting: new Map([["memory/b", { startedAt: 1, holderPid: 1 }]]),
|
|
535
|
+
resolvedCarried: 2,
|
|
536
|
+
resolvedDispatched: 0,
|
|
537
|
+
},
|
|
538
|
+
],
|
|
539
|
+
]),
|
|
540
|
+
};
|
|
541
|
+
renderer.appendDurable(
|
|
542
|
+
{ type: "lock-wait", at: 5, experimentId: "compare/codex", evalId: "memory/a", status: "resolved", carried: 2, dispatched: 0, waitedMs: 5_000 },
|
|
543
|
+
stillWaitingState,
|
|
544
|
+
);
|
|
545
|
+
expect(stdout.writes.join("")).toBe("");
|
|
546
|
+
|
|
547
|
+
// 最后一个也解决了:窗口关闭,打印聚合收尾行(carried + dispatched 混合的措辞两面都要覆盖)。
|
|
548
|
+
const closedState: RunFeedbackState = {
|
|
549
|
+
...createInitialRunFeedbackState(),
|
|
550
|
+
lockWaits: new Map([
|
|
551
|
+
[
|
|
552
|
+
"compare/codex",
|
|
553
|
+
{
|
|
554
|
+
experimentId: "compare/codex",
|
|
555
|
+
waiting: new Map(),
|
|
556
|
+
resolvedCarried: 2,
|
|
557
|
+
resolvedDispatched: 1,
|
|
558
|
+
},
|
|
559
|
+
],
|
|
560
|
+
]),
|
|
561
|
+
};
|
|
562
|
+
renderer.appendDurable(
|
|
563
|
+
{ type: "lock-wait", at: 94_000, experimentId: "compare/codex", evalId: "memory/b", status: "resolved", carried: 0, dispatched: 1, waitedMs: 94_000 },
|
|
564
|
+
closedState,
|
|
565
|
+
);
|
|
566
|
+
const out = stdout.writes.join("");
|
|
567
|
+
expect(out).toContain("lock wait resolved · compare/codex");
|
|
568
|
+
expect(out).toContain("2 carried");
|
|
569
|
+
expect(out).toContain("1 to run");
|
|
570
|
+
});
|
|
571
|
+
});
|
|
572
|
+
|
|
573
|
+
describe("诊断行:标题是「阶段标签 · code」,止损闸落闸是一行 error 级通知", () => {
|
|
574
|
+
/** 把同一条诊断喂 N 次(emitter 刷新 data.unstarted 时就是这个形状),返回每一次的渲染行。 */
|
|
575
|
+
function replayDiagnostic(event: DurableFeedbackEvent & { type: "diagnostic" }, times: number): string[][] {
|
|
576
|
+
let state = createInitialRunFeedbackState();
|
|
577
|
+
const out: string[][] = [];
|
|
578
|
+
for (let i = 0; i < times; i++) {
|
|
579
|
+
state = reduceRunFeedback(state, event);
|
|
580
|
+
out.push(renderDurableLines(event, state, { mode: "plain", width: 100 }));
|
|
581
|
+
}
|
|
582
|
+
return out;
|
|
583
|
+
}
|
|
584
|
+
|
|
585
|
+
it("普通诊断的标题用 code,不把编了身份的去重 key 甩进人读的一行", () => {
|
|
586
|
+
const [lines] = replayDiagnostic(
|
|
587
|
+
{
|
|
588
|
+
type: "diagnostic",
|
|
589
|
+
at: 0,
|
|
590
|
+
key: "lock-taken-over:compare/codex|memory/retention",
|
|
591
|
+
code: "lock-taken-over",
|
|
592
|
+
severity: "warning",
|
|
593
|
+
message: "took over a stale lock from pid 41267",
|
|
594
|
+
},
|
|
595
|
+
1,
|
|
596
|
+
);
|
|
597
|
+
expect(lines![0]).toBe("! lock-taken-over");
|
|
598
|
+
expect(lines![1]).toContain("took over a stale lock");
|
|
599
|
+
});
|
|
600
|
+
|
|
601
|
+
it("同一 key 再次出现时标题带 ×N 折叠计数(非止损闸诊断保持既有形态)", () => {
|
|
602
|
+
const rounds = replayDiagnostic(
|
|
603
|
+
{ type: "diagnostic", at: 0, key: "memory-warmup-degraded", severity: "warning", message: "cold index" },
|
|
604
|
+
3,
|
|
605
|
+
);
|
|
606
|
+
expect(rounds[0]![0]).toBe("! memory-warmup-degraded");
|
|
607
|
+
expect(rounds[2]![0]).toBe("! memory-warmup-degraded (3 attempts)");
|
|
608
|
+
});
|
|
609
|
+
|
|
610
|
+
it("attempt 级诊断的标题是「阶段标签 · code」,阶段标签与失败行同一投影", () => {
|
|
611
|
+
// 失败行已有的 ` · <阶段标签>` 段是同一个投影的既有出口:拿它当参照,断言诊断行没有
|
|
612
|
+
// 另写一份阶段词表(不硬编码语言,en/zh-CN 两种 locale 下都成立)。
|
|
613
|
+
const failureLines = renderDurableLines(
|
|
614
|
+
{
|
|
615
|
+
type: "failure",
|
|
616
|
+
at: 0,
|
|
617
|
+
locator: locator("@1bwcxxiy"),
|
|
618
|
+
identity: { experimentId: "compare/codex", evalId: "memory/x", attempt: 1 },
|
|
619
|
+
who: "codex",
|
|
620
|
+
verdict: "errored",
|
|
621
|
+
reason: "boom",
|
|
622
|
+
phase: "sandbox.setup",
|
|
623
|
+
},
|
|
624
|
+
createInitialRunFeedbackState(),
|
|
625
|
+
{ mode: "plain", width: 100 },
|
|
626
|
+
);
|
|
627
|
+
const label = failureLines[0]!.split("\n")[0]!.split(" · ")[1]!;
|
|
628
|
+
|
|
629
|
+
const rounds = replayDiagnostic(
|
|
630
|
+
{
|
|
631
|
+
type: "diagnostic",
|
|
632
|
+
at: 0,
|
|
633
|
+
// 作者没传 dedupeKey 时 attempt.ts 折出来的 key —— 编了身份,不能出现在人读标题里。
|
|
634
|
+
key: "memory-warmup-degraded:compare/codex|memory/x|1",
|
|
635
|
+
code: "memory-warmup-degraded",
|
|
636
|
+
severity: "warning",
|
|
637
|
+
message: "Memory warmup failed; continuing with a cold index",
|
|
638
|
+
identity: { experimentId: "compare/codex", evalId: "memory/x", attempt: 1 },
|
|
639
|
+
data: { phase: "sandbox.setup" },
|
|
640
|
+
},
|
|
641
|
+
12,
|
|
642
|
+
);
|
|
643
|
+
expect(rounds[0]![0]).toBe(`! ${label} · memory-warmup-degraded`);
|
|
644
|
+
expect(rounds[11]![0]).toBe(`! ${label} · memory-warmup-degraded (12 attempts)`);
|
|
645
|
+
expect(rounds[11]![1]).toBe(" Memory warmup failed; continuing with a cold index");
|
|
646
|
+
});
|
|
647
|
+
|
|
648
|
+
it("运行级诊断没有 phase:标题只有 code,不留空的 · 分隔符", () => {
|
|
649
|
+
const [lines] = replayDiagnostic(
|
|
650
|
+
{
|
|
651
|
+
type: "diagnostic",
|
|
652
|
+
at: 0,
|
|
653
|
+
key: "budget-unenforceable:compare/codex",
|
|
654
|
+
code: "budget-unenforceable",
|
|
655
|
+
severity: "warning",
|
|
656
|
+
message: "no cost data; budget cannot be enforced",
|
|
657
|
+
data: { experimentId: "compare/codex" },
|
|
658
|
+
},
|
|
659
|
+
1,
|
|
660
|
+
);
|
|
661
|
+
expect(lines![0]).toBe("! budget-unenforceable");
|
|
662
|
+
});
|
|
663
|
+
|
|
664
|
+
it("实验闸落闸:一行 error 级通知,文案就是契约字面,不再多一行标题", () => {
|
|
665
|
+
const [lines] = replayDiagnostic(
|
|
666
|
+
{
|
|
667
|
+
type: "diagnostic",
|
|
668
|
+
at: 0,
|
|
669
|
+
key: "dispatch-halted:experiment:compare/codex",
|
|
670
|
+
code: HALT_DIAGNOSTIC_CODE,
|
|
671
|
+
severity: "error",
|
|
672
|
+
message: "experiment halted (dispatch-halted): shared service is down; restart the tunnel",
|
|
673
|
+
data: { experimentId: "compare/codex", scope: "experiment", phase: "eval.run", unstarted: 0 },
|
|
674
|
+
},
|
|
675
|
+
1,
|
|
676
|
+
);
|
|
677
|
+
expect(lines).toEqual(["✗ experiment halted (dispatch-halted): shared service is down; restart the tunnel"]);
|
|
678
|
+
});
|
|
679
|
+
|
|
680
|
+
it("每条未派发 attempt 刷一次的后续声明零输出:被中止的等待集不逐条刷屏,数量归完成状态的 unstarted", () => {
|
|
681
|
+
const rounds = replayDiagnostic(
|
|
682
|
+
{
|
|
683
|
+
type: "diagnostic",
|
|
684
|
+
at: 0,
|
|
685
|
+
key: "dispatch-halted:eval:compare/codex|memory/retention",
|
|
686
|
+
code: HALT_DIAGNOSTIC_CODE,
|
|
687
|
+
severity: "error",
|
|
688
|
+
message: "eval halted: fixture db is empty; run scripts/seed.ts",
|
|
689
|
+
data: { experimentId: "compare/codex", scope: "eval", evalId: "memory/retention", phase: "eval.run", unstarted: 4 },
|
|
690
|
+
},
|
|
691
|
+
5,
|
|
692
|
+
);
|
|
693
|
+
expect(rounds[0]).toEqual(["✗ eval halted: fixture db is empty; run scripts/seed.ts"]);
|
|
694
|
+
expect(rounds.slice(1).flat()).toEqual([]);
|
|
695
|
+
});
|
|
696
|
+
});
|
|
697
|
+
|
|
698
|
+
describe("renderHumanDryPlan: locked 标注", () => {
|
|
699
|
+
it("locked 为 true 的行尾标注 locked;false/省略的行不受影响", () => {
|
|
700
|
+
const text = renderHumanDryPlan({
|
|
701
|
+
totalAttempts: 2,
|
|
702
|
+
evals: 2,
|
|
703
|
+
configs: 1,
|
|
704
|
+
runs: 1,
|
|
705
|
+
rows: [
|
|
706
|
+
{ experimentId: "compare/codex", evalId: "memory/a", locked: true },
|
|
707
|
+
{ experimentId: "compare/codex", evalId: "memory/b" },
|
|
708
|
+
],
|
|
709
|
+
});
|
|
710
|
+
const lines = text.trim().split("\n");
|
|
711
|
+
expect(lines.find((l) => l.includes("memory/a"))).toContain("locked");
|
|
712
|
+
expect(lines.find((l) => l.includes("memory/b"))).not.toContain("locked");
|
|
713
|
+
});
|
|
714
|
+
});
|
|
@@ -20,7 +20,7 @@ import { t } from "../../i18n/index.ts";
|
|
|
20
20
|
import { verdictSymbol } from "../reporters/shared.ts";
|
|
21
21
|
import { formatCost } from "../../shared/format.ts";
|
|
22
22
|
import { assertionSummaryLines } from "../../scoring/display.ts";
|
|
23
|
-
import { encodeAttemptKey } from "../types.ts";
|
|
23
|
+
import { encodeAttemptKey, HALT_DIAGNOSTIC_CODE } from "../types.ts";
|
|
24
24
|
import {
|
|
25
25
|
panelCapabilityOf as panelCapability,
|
|
26
26
|
panelContentWidth,
|
|
@@ -32,6 +32,7 @@ import { stringWidth } from "../../report/model/text-layout.ts";
|
|
|
32
32
|
import type {
|
|
33
33
|
ActiveAttempt,
|
|
34
34
|
ActiveExperimentHook,
|
|
35
|
+
ActiveLockWait,
|
|
35
36
|
ActivePrecheck,
|
|
36
37
|
AttemptKey,
|
|
37
38
|
ExperimentHookName,
|
|
@@ -140,6 +141,40 @@ export function renderDurableLines(
|
|
|
140
141
|
const duration = event.durationMs !== undefined ? ` (${formatElapsed(event.durationMs)})` : "";
|
|
141
142
|
return [`${t("feedback.human.precheckJudgeDone")}${duration}`];
|
|
142
143
|
}
|
|
144
|
+
case "lock-wait": {
|
|
145
|
+
// 只服务非 TTY 退化流(TTY dashboard 的 appendDurable 对这个事件直接返回,运行级行由
|
|
146
|
+
// state.lockWaits 驱动,不进 scrollback,见 cli.md「等待并发 run 的显示」)。按实验聚合
|
|
147
|
+
// ——同一实验可能有多个用例先后撞锁,只在这个「有等待用例」窗口第一次打开(这是当前
|
|
148
|
+
// 唯一一条等待中的用例)与最后一次关闭(等待全部解决)各打印一行,中途加入/解决的用例
|
|
149
|
+
// 不逐条刷屏,与诊断按 dedupeKey 折叠同一种克制(state 已经是这条事件 reduce 之后的
|
|
150
|
+
// 快照,size 天然反映"这条事件之后"的计数)。
|
|
151
|
+
const agg = state.lockWaits.get(event.experimentId);
|
|
152
|
+
if (event.status === "started") {
|
|
153
|
+
if (!agg || agg.waiting.size !== 1) return []; // 不是这个窗口的第一条,静默
|
|
154
|
+
const holder = agg.waiting.get(event.evalId);
|
|
155
|
+
return [
|
|
156
|
+
t("feedback.human.lockWaitStarted", {
|
|
157
|
+
experimentId: event.experimentId,
|
|
158
|
+
count: agg.waiting.size,
|
|
159
|
+
pid: holder?.holderPid ?? "?",
|
|
160
|
+
}),
|
|
161
|
+
];
|
|
162
|
+
}
|
|
163
|
+
if (!agg || agg.waiting.size !== 0) return []; // 窗口还没关闭,静默
|
|
164
|
+
const parts: string[] = [];
|
|
165
|
+
if (agg.resolvedCarried > 0) parts.push(t("feedback.human.lockWaitCarried", { count: agg.resolvedCarried }));
|
|
166
|
+
if (agg.resolvedDispatched > 0) {
|
|
167
|
+
parts.push(t("feedback.human.lockWaitDispatched", { count: agg.resolvedDispatched }));
|
|
168
|
+
}
|
|
169
|
+
const summary = parts.length > 0 ? parts.join(" · ") : t("feedback.human.lockWaitCarried", { count: 0 });
|
|
170
|
+
return [
|
|
171
|
+
t("feedback.human.lockWaitResolved", {
|
|
172
|
+
experimentId: event.experimentId,
|
|
173
|
+
summary,
|
|
174
|
+
elapsed: formatElapsed(event.waitedMs ?? 0),
|
|
175
|
+
}),
|
|
176
|
+
];
|
|
177
|
+
}
|
|
143
178
|
case "summary":
|
|
144
179
|
return buildSummaryLines(event, state, panel);
|
|
145
180
|
case "saved":
|
|
@@ -208,8 +243,27 @@ function buildDiagnosticLines(event: DurableFeedbackEvent & { type: "diagnostic"
|
|
|
208
243
|
// count 从 state.diagnostics 读(reducer 已经按 key 去重累加),不在这里自己维护第二份计数。
|
|
209
244
|
const count = state.diagnostics.find((d) => d.key === event.key)?.count ?? 1;
|
|
210
245
|
const sym = event.severity === "error" ? "✗" : "!";
|
|
246
|
+
if (event.code === HALT_DIAGNOSTIC_CODE) {
|
|
247
|
+
// 止损闸落闸:一行 error 级通知,文案已经是完整的一句话(`experiment halted
|
|
248
|
+
// (dispatch-halted): <message>` / `eval halted: <message>`,见 docs/feature/
|
|
249
|
+
// error-classification/architecture.md「观察面」),不再加标题行、也不加 ×N 后缀——
|
|
250
|
+
// emitter 对每条未派发 attempt 都刷一次这条诊断以更新 data.unstarted,逐次打印就是
|
|
251
|
+
// 同一页文档明令禁止的「被中止的等待集 attempt 逐条刷屏」;未派发的数量由完成状态的
|
|
252
|
+
// `unstarted` 回答,不在这行重复。因此只在第一次出现时落一行(与 json profile 的
|
|
253
|
+
// isFirstOccurrence 同一条去重纪律)。
|
|
254
|
+
if (count > 1) return [];
|
|
255
|
+
return [`${sym} ${event.message}`];
|
|
256
|
+
}
|
|
257
|
+
// 标题用稳定词法(`code`),不是把折叠身份一起编进去的去重 key —— 人读的一行要能一眼认出
|
|
258
|
+
// 「这是哪一类诊断」,`compare/codex|memory/x` 那串身份属于 message 与机器面的具名字段。
|
|
211
259
|
const suffix = count > 1 ? ` (${count} attempts)` : "";
|
|
212
|
-
|
|
260
|
+
// 阶段标签走与失败行(`buildFailureLine`)同一个 `phaseLabel()` 投影:「在哪一步降级的」是
|
|
261
|
+
// 读者的第一个问题,message 里未必答得上。attempt 级诊断的 phase 由运行器写进 `data`
|
|
262
|
+
// (见 attempt.ts 的 recordDiagnostic);运行级诊断(止损闸、锁接管、budget)不属于任何
|
|
263
|
+
// 单条 attempt,天然没有 phase,标题退化成只有 code 一段。
|
|
264
|
+
const phase = typeof event.data?.phase === "string" ? (event.data.phase as LifecyclePhase) : undefined;
|
|
265
|
+
const heading = phase !== undefined ? `${phaseLabel(phase)} · ${event.code ?? event.key}` : (event.code ?? event.key);
|
|
266
|
+
return [`${sym} ${heading}${suffix}`, ` ${event.message}`];
|
|
213
267
|
}
|
|
214
268
|
|
|
215
269
|
/** 结束结论(`FAILED`/`PASSED`/…)+ `FAILURES`(有失败才出现)+ `KEPT SANDBOXES`(有留存才
|
|
@@ -418,14 +472,37 @@ function experimentHookLabel(hook: ExperimentHookName): string {
|
|
|
418
472
|
return hook === "setup" ? t("feedback.phase.experimentSetup") : t("feedback.phase.experimentTeardown");
|
|
419
473
|
}
|
|
420
474
|
|
|
475
|
+
/** 首行守恒计数的文案(见 cli.md「运行中的 live 面板」)。四项结局恒显示、零值不省略——
|
|
476
|
+
* 「0 errored」是一句有价值的肯定;`elsewhere` 只在非零时出现,没有并发 run 的场景少一项。
|
|
477
|
+
* 行长不是压缩它的理由:首行跟随终端全宽,九项写满仍是一行。两个调用点(live dashboard
|
|
478
|
+
* 首行、非 TTY heartbeat)共用这一份,不各自维护一份键选择逻辑。 */
|
|
479
|
+
function countsText(state: RunFeedbackState): string {
|
|
480
|
+
const outcomes = {
|
|
481
|
+
passed: state.passed,
|
|
482
|
+
failed: state.failed,
|
|
483
|
+
errored: state.errored,
|
|
484
|
+
skipped: state.skipped,
|
|
485
|
+
};
|
|
486
|
+
return state.elsewhere > 0
|
|
487
|
+
? t("feedback.human.countsWithElsewhere", {
|
|
488
|
+
total: state.total,
|
|
489
|
+
reused: state.reused,
|
|
490
|
+
running: state.running,
|
|
491
|
+
elsewhere: state.elsewhere,
|
|
492
|
+
queued: state.queued,
|
|
493
|
+
...outcomes,
|
|
494
|
+
})
|
|
495
|
+
: t("feedback.human.counts", {
|
|
496
|
+
total: state.total,
|
|
497
|
+
reused: state.reused,
|
|
498
|
+
running: state.running,
|
|
499
|
+
queued: state.queued,
|
|
500
|
+
...outcomes,
|
|
501
|
+
});
|
|
502
|
+
}
|
|
503
|
+
|
|
421
504
|
function formatCounts(state: RunFeedbackState): string {
|
|
422
|
-
const counts =
|
|
423
|
-
total: state.total,
|
|
424
|
-
reused: state.reused,
|
|
425
|
-
running: state.running,
|
|
426
|
-
queued: state.queued,
|
|
427
|
-
completed: state.completed,
|
|
428
|
-
});
|
|
505
|
+
const counts = countsText(state);
|
|
429
506
|
if (state.estimatedCostUSD === undefined || state.estimatedCostUSD <= 0) return counts;
|
|
430
507
|
return `${counts} ${formatCost(state.estimatedCostUSD)}`;
|
|
431
508
|
}
|
|
@@ -479,30 +556,23 @@ function createDashboardRenderer(io: FeedbackIO, command: string): FeedbackRende
|
|
|
479
556
|
// renderPanel 内部按另一个宽度钳制,行尾会被框吃掉——这正是 memory/
|
|
480
557
|
// live-dashboard-active-row-width-clamp-mismatch.md 的根因类别。
|
|
481
558
|
const contentWidth = panelContentWidth(capability.width, capability.mode, false);
|
|
482
|
-
const rows: PanelRow[] = [
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
reused: state.reused,
|
|
488
|
-
running: state.running,
|
|
489
|
-
queued: state.queued,
|
|
490
|
-
completed: state.completed,
|
|
491
|
-
}),
|
|
492
|
-
},
|
|
493
|
-
];
|
|
494
|
-
// 运行级行(judge 预检 + 实验钩子)排在 attempt 行前面(见 cli.md「judge 预检的显示」/
|
|
495
|
-
// 「实验级钩子的显示」):它们解释了为什么后面的 attempt 还停在 queued。预检又排在实验
|
|
496
|
-
// 钩子前面——它发生在最前(任何 attempt 派发之前)。Map 按插入序迭代,天然满足稳定 slot。
|
|
559
|
+
const rows: PanelRow[] = [{ kind: "line", text: countsText(state) }];
|
|
560
|
+
// 运行级行(judge 预检 + 实验钩子 + 用例锁等待)排在 attempt 行前面(见 cli.md「judge 预检
|
|
561
|
+
// 的显示」/「实验级钩子的显示」/「等待并发 run 的显示」):它们解释了为什么后面的 attempt
|
|
562
|
+
// 还停在 queued。预检排最前(发生在任何 attempt 派发之前),其次实验钩子,再是锁等待
|
|
563
|
+
// (排在实验钩子行之后、attempt 行之前)。Map 按插入序迭代,天然满足稳定 slot。
|
|
497
564
|
const precheck = state.activePrecheck;
|
|
498
565
|
const hookRows = [...state.experimentHooks.values()];
|
|
499
|
-
|
|
566
|
+
// 只有仍在等待(waiting 非空)的实验才占运行级行;窗口已关闭(全部 resolved)的条目只是
|
|
567
|
+
// 给非 TTY 聚合收尾行留的历史计数,TTY 不展示。
|
|
568
|
+
const lockWaitRows = [...state.lockWaits.values()].filter((w) => w.waiting.size > 0);
|
|
569
|
+
if (activeOrder.length > 0 || hookRows.length > 0 || lockWaitRows.length > 0 || precheck) {
|
|
500
570
|
rows.push({ kind: "divider", title: t("feedback.human.active") });
|
|
501
571
|
// 固定开销:上边框 + counts 行 + ACTIVE 横隔 + 下边框(boxed);plain 时同样按 4 行估算,
|
|
502
572
|
// 差一两行不影响「窄/矮终端先减 active slots」这条大方向。
|
|
503
573
|
const rowBudget = Math.max(0, io.stderr.rows - 4 - DASHBOARD_ROW_RESERVE);
|
|
504
574
|
const precheckCount = precheck ? 1 : 0;
|
|
505
|
-
const total = precheckCount + hookRows.length + activeOrder.length;
|
|
575
|
+
const total = precheckCount + hookRows.length + lockWaitRows.length + activeOrder.length;
|
|
506
576
|
// 窄/矮终端先减 active slots(减少行数),而不是先压缩单行内容 ——
|
|
507
577
|
// 单行内容的截断在 formatActiveRow 里按 contentWidth 单独处理。
|
|
508
578
|
const showCount = total <= rowBudget ? total : Math.max(0, rowBudget - 1);
|
|
@@ -512,7 +582,11 @@ function createDashboardRenderer(io: FeedbackIO, command: string): FeedbackRende
|
|
|
512
582
|
// 的行又观测到更长的值再推宽,导致同一帧内本该对齐的列错位。
|
|
513
583
|
const shownPrecheck = precheck && showCount > 0 ? precheck : undefined;
|
|
514
584
|
const shownHooks = hookRows.slice(0, Math.max(0, showCount - (shownPrecheck ? 1 : 0)));
|
|
515
|
-
const
|
|
585
|
+
const shownLockWaits = lockWaitRows.slice(
|
|
586
|
+
0,
|
|
587
|
+
Math.max(0, showCount - (shownPrecheck ? 1 : 0) - shownHooks.length),
|
|
588
|
+
);
|
|
589
|
+
const shownRunLevel = (shownPrecheck ? 1 : 0) + shownHooks.length + shownLockWaits.length;
|
|
516
590
|
const shownActive: ActiveAttempt[] = [];
|
|
517
591
|
for (const key of activeOrder) {
|
|
518
592
|
if (shownRunLevel + shownActive.length >= showCount) break;
|
|
@@ -533,6 +607,9 @@ function createDashboardRenderer(io: FeedbackIO, command: string): FeedbackRende
|
|
|
533
607
|
for (const hookRow of shownHooks) {
|
|
534
608
|
activeLines.push(formatExperimentHookRow(hookRow, io, contentWidth, evalWidth, whoWidth));
|
|
535
609
|
}
|
|
610
|
+
for (const lockWaitRow of shownLockWaits) {
|
|
611
|
+
activeLines.push(formatLockWaitRow(lockWaitRow, io, contentWidth));
|
|
612
|
+
}
|
|
536
613
|
for (const active of shownActive) {
|
|
537
614
|
activeLines.push(formatActiveRow(active, io, contentWidth, evalWidth, whoWidth));
|
|
538
615
|
}
|
|
@@ -584,8 +661,9 @@ function createDashboardRenderer(io: FeedbackIO, command: string): FeedbackRende
|
|
|
584
661
|
// 更新,coordinator 紧接着的 redrawDynamic 会画出来);成功钩子不写 scrollback 永久行
|
|
585
662
|
// (见 cli.md「实验级钩子的显示」)。非 TTY 退化流才逐行追加(见 renderDurableLines)。
|
|
586
663
|
// judge 预检同理:TTY 下只驱动 state.activePrecheck 的运行级 active 行(coordinator 紧接着的
|
|
587
|
-
// redrawDynamic 会画出来),不写 scrollback 永久行(见 cli.md「judge 预检的显示」)
|
|
588
|
-
|
|
664
|
+
// redrawDynamic 会画出来),不写 scrollback 永久行(见 cli.md「judge 预检的显示」)。用例锁
|
|
665
|
+
// 等待同理:TTY 下由 state.lockWaits 驱动运行级 active 行(见 cli.md「等待并发 run 的显示」)。
|
|
666
|
+
if (event.type === "experiment-hook" || event.type === "precheck" || event.type === "lock-wait") return;
|
|
589
667
|
writeDurable(io, event, state, false);
|
|
590
668
|
},
|
|
591
669
|
activity(text) {
|
|
@@ -682,6 +760,26 @@ function formatExperimentHookRow(
|
|
|
682
760
|
return prefix + (hook.detail ?? "").slice(0, budget);
|
|
683
761
|
}
|
|
684
762
|
|
|
763
|
+
/** 用例锁等待的运行级行:`● waiting on another run · <exp> <elapsed> <n> evals · pid <pid>`
|
|
764
|
+
* (cli.md「等待并发 run 的显示」)。elapsed 从最早一条等待的 startedAt 算(存活性证明,
|
|
765
|
+
* 与其它运行级行同一约定);pid 取最早一条等待对应的持有方——一个实验可能同时撞上多把不同
|
|
766
|
+
* 持有方的锁,这里选一个稳定的代表值展示,不逐条列出(与 earlyExit 代表 attempt 的选法同一
|
|
767
|
+
* 种「挑一个确定性代表」思路)。不吃 evalWidth/whoWidth 身份列宽约束——与 `formatPrecheckRow`
|
|
768
|
+
* 同理:选中用例全在等锁时,本实验没有派发中的 attempt,那两个宽度还压在初始值 0,会把
|
|
769
|
+
* label 截成 "w…"(与 memory/live-dashboard-active-row-width-clamp-mismatch.md 同一根因类别,
|
|
770
|
+
* 只是发生在锁等待场景);label 直接用整行宽度。 */
|
|
771
|
+
function formatLockWaitRow(wait: ActiveLockWait, io: FeedbackIO, columns: number): string {
|
|
772
|
+
const entries = [...wait.waiting.values()].sort((a, b) => a.startedAt - b.startedAt);
|
|
773
|
+
const earliest = entries[0]!;
|
|
774
|
+
const elapsed = formatElapsed(io.clock.now() - earliest.startedAt).padStart(6);
|
|
775
|
+
const sym = "● ";
|
|
776
|
+
const label = `${t("feedback.human.waitingOnAnotherRun")} · ${wait.experimentId}`;
|
|
777
|
+
const prefix = `${sym}${label} ${elapsed} `;
|
|
778
|
+
const budget = Math.max(0, columns - prefix.length);
|
|
779
|
+
const detail = t("feedback.human.lockWaitDetail", { count: entries.length, pid: earliest.holderPid ?? "?" });
|
|
780
|
+
return padTrunc(prefix + detail.slice(0, budget), columns);
|
|
781
|
+
}
|
|
782
|
+
|
|
685
783
|
// ───────────────────────── 非 TTY:human 文案的纯追加流 ─────────────────────────
|
|
686
784
|
//
|
|
687
785
|
// 单一 stdout 有序流(见 memory/exp-output-two-forms-ruling.md 的补充裁决):从 start 到结束
|
|
@@ -730,6 +828,9 @@ function createPlainRenderer(io: FeedbackIO): FeedbackRenderer {
|
|
|
730
828
|
export interface HumanDryPlanRow {
|
|
731
829
|
experimentId: string;
|
|
732
830
|
evalId: string;
|
|
831
|
+
/** 该用例正被另一条并行 Invocation 持锁运行(见 docs/feature/experiments/architecture.md
|
|
832
|
+
* 「并发 Invocation:用例锁」);计划行尾如实标注,`--dry` 本身不取锁、不等待。 */
|
|
833
|
+
locked?: boolean;
|
|
733
834
|
}
|
|
734
835
|
|
|
735
836
|
export interface HumanDryPlanInput {
|
|
@@ -772,7 +873,8 @@ export function renderHumanDryPlan(input: HumanDryPlanInput): string {
|
|
|
772
873
|
}
|
|
773
874
|
const idWidth = Math.max(0, ...input.rows.map((row) => stringWidth(row.experimentId)));
|
|
774
875
|
for (const row of input.rows) {
|
|
775
|
-
|
|
876
|
+
const base = `${row.experimentId}${" ".repeat(idWidth - stringWidth(row.experimentId) + 2)}${row.evalId}`;
|
|
877
|
+
lines.push(row.locked ? `${base} ${t("feedback.human.lockedRowSuffix")}` : base);
|
|
776
878
|
}
|
|
777
879
|
return `${lines.join("\n")}\n`;
|
|
778
880
|
}
|