@tea-agent/loop-agent 0.13.0-alpha.0 → 0.13.0-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +155 -153
- package/CHANGELOG.md +326 -301
- package/README.md +345 -326
- package/bin/agent-worker.js +22 -22
- package/bin/loop-agent.js +21 -21
- package/dist/application/dag/generate-task-dag.js +28 -58
- package/dist/application/evaluation/candidate-hash.js +75 -0
- package/dist/application/evaluation/candidate.js +52 -0
- package/dist/application/evaluation/replay.js +289 -0
- package/dist/application/evaluation/types.js +130 -0
- package/dist/cli/command-definitions.js +17 -4
- package/dist/cli/program.js +8 -4
- package/dist/commands/cursor-prompt.js +6 -6
- package/dist/commands/eval.js +235 -0
- package/dist/commands/init.js +544 -506
- package/dist/commands/loop-benchmark.js +11 -11
- package/dist/commands/pi-reuse-benchmark.js +16 -16
- package/dist/executors/pi-sdk-executor.js +38 -24
- package/dist/executors/shell-executor.js +34 -2
- package/dist/executors/shell-presets.js +20 -0
- package/dist/executors/shell-verification.js +7 -0
- package/dist/governance/manifest-types.js +1 -0
- package/dist/infrastructure/evaluation/candidate-store.js +435 -0
- package/dist/infrastructure/evaluation/store.js +40 -0
- package/dist/sidecars/cursor-prompt/executor.js +1 -1
- package/dist/task/config-types.js +23 -0
- package/dist/task/runtime.js +27 -27
- package/dist/worker/observe/routes.js +18 -3
- package/dist/worker/observe/spec-evidence.js +1 -1
- package/dist/worker/observe/static/api.js +46 -46
- package/dist/worker/observe/static/app.js +150 -150
- package/dist/worker/observe/static/constants.js +148 -148
- package/dist/worker/observe/static/copy.js +67 -67
- package/dist/worker/observe/static/dag-helpers.js +172 -172
- package/dist/worker/observe/static/dag-layout.d.ts +31 -31
- package/dist/worker/observe/static/dag-layout.js +83 -83
- package/dist/worker/observe/static/dag-model.js +72 -72
- package/dist/worker/observe/static/dom.js +61 -61
- package/dist/worker/observe/static/format-pool.js +67 -67
- package/dist/worker/observe/static/format.js +292 -292
- package/dist/worker/observe/static/index.html +308 -308
- package/dist/worker/observe/static/kpi.js +94 -94
- package/dist/worker/observe/static/relations.js +133 -133
- package/dist/worker/observe/static/router.js +93 -93
- package/dist/worker/observe/static/run-processing.js +148 -148
- package/dist/worker/observe/static/shell-chrome.js +68 -68
- package/dist/worker/observe/static/state.js +253 -253
- package/dist/worker/observe/static/styles.css +1902 -1902
- package/dist/worker/observe/static/views/batch.js +227 -227
- package/dist/worker/observe/static/views/dag-graph.js +172 -172
- package/dist/worker/observe/static/views/dag-inspector.js +607 -596
- package/dist/worker/observe/static/views/dag.js +362 -362
- package/dist/worker/observe/static/views/dashboard.js +445 -445
- package/dist/worker/observe/static/views/failures.js +143 -143
- package/dist/worker/observe/static/views/feature.js +492 -492
- package/dist/worker/observe/static/views/pool.js +350 -350
- package/dist/worker/observe/static/views/run.js +453 -453
- package/dist/worker/observe/static/views/session-timeline.js +205 -205
- package/dist/worker/observe/static/views/shell.js +7 -7
- package/dist/worker/observe/static/views/task.js +314 -314
- package/dist/worker/observe/static/views/timeline.js +163 -163
- package/dist/workflows/dag/backend-test-analysis-contract.js +120 -0
- package/dist/workflows/dag/canvas-observer.js +275 -275
- package/dist/workflows/dag/dynamic-runtime/map.js +90 -2
- package/dist/workflows/dag/init-hybrid.js +1415 -200
- package/dist/workflows/dag/node-execution.js +9 -0
- package/dist/workflows/dag/prompt.js +9 -0
- package/dist/workflows/dag/report.js +35 -1
- package/dist/workflows/dag/runner.js +28 -2
- package/dist/workflows/dag/task-demand-routing.js +383 -0
- package/dist/workflows/dag/types.js +50 -13
- package/dist/workflows/dag/upstream-artifacts.js +1 -0
- package/dist/workflows/dag/validate.js +59 -1
- package/docs/README.md +106 -104
- package/docs/agent-dag-recovery-playbook.md +195 -193
- package/docs/agent-dag-runner.md +67 -67
- package/docs/architecture/README.md +26 -26
- package/docs/architecture/dag-execution.md +140 -140
- package/docs/architecture/evolution.md +54 -54
- package/docs/architecture/facts-and-state.md +71 -71
- package/docs/architecture/runtime-boundaries.md +191 -191
- package/docs/architecture/system-overview.md +93 -93
- package/docs/architecture/worker-and-feature.md +85 -85
- package/docs/cursor-prompt-sidecar.md +36 -36
- package/docs/decisions/README.md +18 -18
- package/docs/design/README.md +167 -85
- package/docs/development-principles.md +73 -73
- package/docs/exec-plans/README.md +6 -6
- package/docs/exec-plans/active/README.md +15 -11
- package/docs/exec-plans/completed/README.md +85 -74
- package/docs/feature-workflow.md +389 -339
- package/docs/harness-methodology-debugging.md +153 -153
- package/docs/harness-methodology-tdd.md +130 -130
- package/docs/harness-methodology-verification.md +27 -27
- package/docs/init-surface.manifest.json +289 -280
- package/docs/loop-agent-harness.md +142 -141
- package/docs/production-readiness.md +96 -96
- package/docs/progress/README.md +64 -58
- package/docs/reports/README.md +117 -100
- package/docs/skills/README.md +7 -7
- package/docs/skills/vetted-skill-registry.md +29 -27
- package/docs/templates/adr.md +60 -60
- package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
- package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
- package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
- package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
- package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
- package/docs/templates/agent-dag-report.schema.json +473 -473
- package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
- package/docs/templates/agent-dag.base.json +190 -190
- package/docs/templates/agent-dag.final-verification.json +185 -185
- package/docs/templates/agent-dag.schema.json +411 -383
- package/docs/templates/agent-dag.supervised-implementation.json +501 -501
- package/docs/templates/backend-test-analysis.schema.json +44 -0
- package/docs/templates/backend-test-dag.generate-pytest.prompt.md +202 -139
- package/docs/templates/backend-test-dag.json +311 -288
- package/docs/templates/backend-test-dag.retrospect.prompt.md +125 -125
- package/docs/templates/backend-test-dag.review-cases.prompt.md +81 -81
- package/docs/templates/exec-plan.md +64 -64
- package/docs/templates/feature-spec.md +53 -53
- package/docs/templates/frontend-design-contract.md +42 -33
- package/docs/templates/frontend-task-constraints.md +35 -25
- package/docs/templates/frontend-task-requirement.md +70 -61
- package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -0
- package/docs/templates/frontend-test-dag.json +23 -0
- package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -0
- package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -0
- package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -0
- package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -0
- package/docs/templates/harness.schema.json +221 -221
- package/docs/templates/hybrid-dag.json +188 -188
- package/docs/templates/init-evolution-review.md +35 -35
- package/docs/templates/interactive-ui-round2-experiment.md +66 -66
- package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
- package/docs/templates/knowledge-sync-dag.json +178 -177
- package/docs/templates/knowledge-sync-draft.schema.json +71 -71
- package/docs/templates/product-line/AGENTS.md +8 -8
- package/docs/templates/product-line/README.md +9 -9
- package/docs/templates/product-line/acceptance.yaml +14 -14
- package/docs/templates/product-line/closeout.yaml +9 -9
- package/docs/templates/product-line/design.md +13 -13
- package/docs/templates/product-line/links.md +10 -10
- package/docs/templates/product-line/requirement.md +17 -17
- package/docs/templates/product-line/task-graph.yaml +15 -15
- package/docs/templates/product-line/task.yaml +64 -64
- package/docs/templates/product-line/test-plan.md +7 -7
- package/docs/templates/production-readiness-checklist.md +57 -57
- package/docs/templates/progress-log.md +17 -17
- package/docs/templates/project-start-checklist.md +9 -9
- package/docs/templates/qa-report.md +48 -48
- package/docs/templates/sprint-contract.md +29 -29
- package/docs/templates/worker-dogfood-evidence.md +80 -80
- package/docs/templates/worker-dogfood-setup.md +68 -68
- package/docs/verification-matrix.md +70 -67
- package/examples/decision-gate-agent-dag.json +177 -177
- package/examples/example-dag.json +46 -46
- package/examples/hybrid-loop-agent-dag.json +189 -189
- package/harness.json +66 -66
- package/package.json +88 -52
- package/scripts/check-product-line-docs.sh +29 -29
- package/scripts/check-task-pool-root.sh +32 -32
- package/scripts/kb-bootstrap-init-skeleton.sh +240 -239
- package/scripts/kb-graph-incremental-prepare.mjs +386 -372
- package/scripts/kb-graph-incremental-prepare.sh +5 -5
- package/scripts/kb-graph-materialize.mjs +105 -105
- package/scripts/kb-graph-materialize.sh +4 -4
- package/scripts/kb-graph-promote.mjs +164 -153
- package/scripts/kb-graph-promote.sh +4 -4
- package/scripts/kb-query.mjs +554 -554
- package/scripts/kb-query.sh +5 -5
- package/skills/agent-worker/SKILL.md +39 -39
- package/skills/agent-worker/references/agent-worker-operator.md +60 -60
- package/skills/ai-engineering-context/SKILL.md +48 -48
- package/skills/analyze-product-dependencies/SKILL.md +67 -0
- package/skills/analyze-product-dependencies/agents/openai.yaml +4 -0
- package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -0
- package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -0
- package/skills/analyze-product-dependencies/references/example.md +76 -0
- package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -0
- package/skills/analyze-product-dependencies/references/input-contract.md +11 -0
- package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -0
- package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -0
- package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -0
- package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -0
- package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -0
- package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -0
- package/skills/analyze-product-requirements/SKILL.md +90 -0
- package/skills/analyze-product-requirements/agents/openai.yaml +4 -0
- package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -0
- package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -0
- package/skills/analyze-product-requirements/references/example.md +86 -0
- package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -0
- package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -0
- package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -0
- package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -0
- package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -0
- package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -0
- package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -0
- package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -0
- package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -0
- package/skills/code-review-core/SKILL.md +20 -20
- package/skills/codebase-scout/SKILL.md +19 -19
- package/skills/frontend-design-review/SKILL.md +66 -61
- package/skills/frontend-design-review/references/review-checklist.md +58 -37
- package/skills/frontend-implementation/SKILL.md +45 -52
- package/skills/frontend-implementation/references/code-standards.md +32 -34
- package/skills/frontend-implementation/references/design-spec.md +46 -46
- package/skills/frontend-implementation/references/node-contracts.md +76 -63
- package/skills/frontend-review/SKILL.md +59 -53
- package/skills/frontend-review/references/review-findings.md +47 -42
- package/skills/frontend-verification/SKILL.md +53 -40
- package/skills/frontend-verification/references/verification-checklist.md +68 -56
- package/skills/grill-me/SKILL.md +10 -10
- package/skills/grill-with-docs/SKILL.md +88 -88
- package/skills/grill-with-docs/adr-format.md +47 -47
- package/skills/grill-with-docs/context-format.md +60 -60
- package/skills/init-capability-evolution/SKILL.md +70 -70
- package/skills/loop-agent/SKILL.md +151 -151
- package/skills/loop-agent/references/README.md +67 -67
- package/skills/loop-agent/references/command-reference.md +505 -453
- package/skills/loop-agent/references/docs-converge.md +126 -126
- package/skills/loop-agent/references/harness-policy.md +263 -263
- package/skills/loop-agent/references/hybrid-dag.md +238 -233
- package/skills/loop-agent/references/learned/README.md +21 -21
- package/skills/loop-agent/references/long-running-loop.md +57 -57
- package/skills/loop-agent/references/model-routing.md +36 -36
- package/skills/loop-agent/references/multi-worktree.md +54 -54
- package/skills/loop-agent/references/one-shot-runs.md +85 -85
- package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
- package/skills/loop-agent/references/pi-prompt.md +23 -23
- package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
- package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
- package/skills/loop-agent/references/task-workflow.md +89 -89
- package/skills/loop-agent/references/verification-and-failure-handling.md +139 -139
- package/skills/playwright-cli/SKILL.md +420 -0
- package/skills/playwright-cli/references/element-attributes.md +23 -0
- package/skills/playwright-cli/references/playwright-tests.md +39 -0
- package/skills/playwright-cli/references/request-mocking.md +87 -0
- package/skills/playwright-cli/references/running-code.md +241 -0
- package/skills/playwright-cli/references/session-management.md +225 -0
- package/skills/playwright-cli/references/storage-state.md +275 -0
- package/skills/playwright-cli/references/test-generation.md +433 -0
- package/skills/playwright-cli/references/tracing.md +139 -0
- package/skills/playwright-cli/references/video-recording.md +143 -0
- package/skills/playwright-cli-case-generator/SKILL.md +74 -0
- package/skills/requesting-code-review/SKILL.md +101 -101
- package/skills/requesting-code-review/code-reviewer.md +168 -168
- package/skills/systematic-debugging/CREATION-LOG.md +119 -119
- package/skills/systematic-debugging/SKILL.md +296 -296
- package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
- package/skills/systematic-debugging/condition-based-waiting.md +115 -115
- package/skills/systematic-debugging/defense-in-depth.md +122 -122
- package/skills/systematic-debugging/find-polluter.sh +63 -63
- package/skills/systematic-debugging/root-cause-tracing.md +169 -169
- package/skills/systematic-debugging/test-academic.md +14 -14
- package/skills/systematic-debugging/test-pressure-1.md +58 -58
- package/skills/systematic-debugging/test-pressure-2.md +68 -68
- package/skills/systematic-debugging/test-pressure-3.md +69 -69
- package/skills/test-driven-development/SKILL.md +20 -20
- package/skills/using-git-worktrees/SKILL.md +215 -215
- package/skills/verification-before-completion/SKILL.md +154 -154
- package/skills/webapp-testing/SKILL.md +19 -19
package/bin/agent-worker.js
CHANGED
|
@@ -1,22 +1,22 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
import { existsSync } from "node:fs";
|
|
3
|
-
import { dirname, join } from "node:path";
|
|
4
|
-
import { fileURLToPath, pathToFileURL } from "node:url";
|
|
5
|
-
|
|
6
|
-
const packageRoot = dirname(dirname(fileURLToPath(import.meta.url)));
|
|
7
|
-
const cliEntry = join(packageRoot, "dist", "worker", "cli.js");
|
|
8
|
-
|
|
9
|
-
if (!existsSync(cliEntry)) {
|
|
10
|
-
console.error(
|
|
11
|
-
`agent-worker: cannot find built CLI at ${cliEntry}. Run \`npm run build\` before using the package bin.`,
|
|
12
|
-
);
|
|
13
|
-
process.exit(1);
|
|
14
|
-
}
|
|
15
|
-
|
|
16
|
-
try {
|
|
17
|
-
const cli = await import(pathToFileURL(cliEntry).href);
|
|
18
|
-
await cli.main(process.argv);
|
|
19
|
-
} catch (error) {
|
|
20
|
-
console.error(error instanceof Error ? error.message : String(error));
|
|
21
|
-
process.exit(1);
|
|
22
|
-
}
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { existsSync } from "node:fs";
|
|
3
|
+
import { dirname, join } from "node:path";
|
|
4
|
+
import { fileURLToPath, pathToFileURL } from "node:url";
|
|
5
|
+
|
|
6
|
+
const packageRoot = dirname(dirname(fileURLToPath(import.meta.url)));
|
|
7
|
+
const cliEntry = join(packageRoot, "dist", "worker", "cli.js");
|
|
8
|
+
|
|
9
|
+
if (!existsSync(cliEntry)) {
|
|
10
|
+
console.error(
|
|
11
|
+
`agent-worker: cannot find built CLI at ${cliEntry}. Run \`npm run build\` before using the package bin.`,
|
|
12
|
+
);
|
|
13
|
+
process.exit(1);
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
try {
|
|
17
|
+
const cli = await import(pathToFileURL(cliEntry).href);
|
|
18
|
+
await cli.main(process.argv);
|
|
19
|
+
} catch (error) {
|
|
20
|
+
console.error(error instanceof Error ? error.message : String(error));
|
|
21
|
+
process.exit(1);
|
|
22
|
+
}
|
package/bin/loop-agent.js
CHANGED
|
@@ -1,21 +1,21 @@
|
|
|
1
|
-
#!/usr/bin/env node
|
|
2
|
-
import { existsSync } from "node:fs";
|
|
3
|
-
import { dirname, join } from "node:path";
|
|
4
|
-
import { fileURLToPath, pathToFileURL } from "node:url";
|
|
5
|
-
|
|
6
|
-
const packageRoot = dirname(dirname(fileURLToPath(import.meta.url)));
|
|
7
|
-
const cliEntry = join(packageRoot, "dist", "cli.js");
|
|
8
|
-
|
|
9
|
-
if (!existsSync(cliEntry)) {
|
|
10
|
-
console.error(
|
|
11
|
-
`loop-agent: cannot find built CLI at ${cliEntry}. Run \`npm run build\` before using the package bin.`,
|
|
12
|
-
);
|
|
13
|
-
process.exit(1);
|
|
14
|
-
}
|
|
15
|
-
|
|
16
|
-
try {
|
|
17
|
-
await import(pathToFileURL(cliEntry).href);
|
|
18
|
-
} catch (error) {
|
|
19
|
-
console.error(error instanceof Error ? error.message : String(error));
|
|
20
|
-
process.exit(1);
|
|
21
|
-
}
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import { existsSync } from "node:fs";
|
|
3
|
+
import { dirname, join } from "node:path";
|
|
4
|
+
import { fileURLToPath, pathToFileURL } from "node:url";
|
|
5
|
+
|
|
6
|
+
const packageRoot = dirname(dirname(fileURLToPath(import.meta.url)));
|
|
7
|
+
const cliEntry = join(packageRoot, "dist", "cli.js");
|
|
8
|
+
|
|
9
|
+
if (!existsSync(cliEntry)) {
|
|
10
|
+
console.error(
|
|
11
|
+
`loop-agent: cannot find built CLI at ${cliEntry}. Run \`npm run build\` before using the package bin.`,
|
|
12
|
+
);
|
|
13
|
+
process.exit(1);
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
try {
|
|
17
|
+
await import(pathToFileURL(cliEntry).href);
|
|
18
|
+
} catch (error) {
|
|
19
|
+
console.error(error instanceof Error ? error.message : String(error));
|
|
20
|
+
process.exit(1);
|
|
21
|
+
}
|
|
@@ -1,4 +1,6 @@
|
|
|
1
|
-
import { readFile } from "node:fs/promises";
|
|
1
|
+
import { readFile, mkdir } from "node:fs/promises";
|
|
2
|
+
import { randomUUID } from "node:crypto";
|
|
3
|
+
import path from "node:path";
|
|
2
4
|
import { resolveAutoRoutingProfile, requiresSupervisedQualityGate, } from "../../workflows/dag/governance-profile.js";
|
|
3
5
|
import { resolveShellCommands } from "../../executors/shell-executor.js";
|
|
4
6
|
import { parseDagSpec } from "../../workflows/dag/types.js";
|
|
@@ -6,7 +8,6 @@ import { pathMatchesPattern } from "../../shared/git-progress.js";
|
|
|
6
8
|
import { loadHarnessManifest } from "../../governance/harness.js";
|
|
7
9
|
import { assertExecPlanIndexConsistent } from "../../governance/exec-plans.js";
|
|
8
10
|
import { defaultHybridDagOutputPath, initHybridDagFromTask, } from "../../workflows/dag/init-hybrid.js";
|
|
9
|
-
import { loadTaskConfig } from "../../task/runtime.js";
|
|
10
11
|
import { validateDagUseCase } from "./validate-dag.js";
|
|
11
12
|
import { runDagUseCase } from "./run-dag.js";
|
|
12
13
|
const PLACEHOLDER_WRITESET_MARKER = "REPLACE/WITH";
|
|
@@ -202,11 +203,6 @@ export async function generateTaskDagUseCase(input) {
|
|
|
202
203
|
// before any expensive DAG generation or execution. Empty/consistent
|
|
203
204
|
// repos stay compatible so the default DAG flow is unblocked.
|
|
204
205
|
await assertExecPlanIndexConsistent(repoRoot);
|
|
205
|
-
const taskConfig = await loadTaskConfig(repoRoot, parsed.taskId);
|
|
206
|
-
const isFrontendImplementationTask = taskConfig.taskKind === "frontend-implementation";
|
|
207
|
-
const isBackendTestTask = taskConfig.taskKind === "backend-test";
|
|
208
|
-
const isKnowledgeSyncTask = taskConfig.taskKind === "knowledge-sync";
|
|
209
|
-
const isKnowledgeGraphBootstrapTask = taskConfig.taskKind === "knowledge-graph-bootstrap";
|
|
210
206
|
const candidateResult = await initHybridDagFromTask(repoRoot, parsed.taskId, {
|
|
211
207
|
outputPath: parsed.outputPath,
|
|
212
208
|
template: "standard-dag",
|
|
@@ -219,12 +215,16 @@ export async function generateTaskDagUseCase(input) {
|
|
|
219
215
|
codeChange: [],
|
|
220
216
|
reasons: ["dag run-task validate did not report governanceProfile"],
|
|
221
217
|
});
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
profileRouting.
|
|
225
|
-
profileRouting.
|
|
226
|
-
|
|
227
|
-
|
|
218
|
+
const hasExplicitSpecializedTaskKind = candidateResult.templateSelection.source === "taskKind";
|
|
219
|
+
const hasSafeAutomaticTaskSourceRoute = candidateResult.templateSelection.source === "task-source" &&
|
|
220
|
+
profileRouting.selectedByProfile !== "supervised" &&
|
|
221
|
+
profileRouting.selectedTemplate !== "supervised-implementation";
|
|
222
|
+
if (hasExplicitSpecializedTaskKind || hasSafeAutomaticTaskSourceRoute) {
|
|
223
|
+
profileRouting.selectedTemplate = candidateResult.template;
|
|
224
|
+
profileRouting.source = hasExplicitSpecializedTaskKind
|
|
225
|
+
? "taskKind"
|
|
226
|
+
: "task-source";
|
|
227
|
+
profileRouting.routingReasons = candidateResult.templateSelection.reasons;
|
|
228
228
|
if (parsed.profile === "auto") {
|
|
229
229
|
profileRouting.selectedByProfile =
|
|
230
230
|
resolveAutoRoutingProfile(profileRouting.candidateProfile);
|
|
@@ -233,56 +233,14 @@ export async function generateTaskDagUseCase(input) {
|
|
|
233
233
|
profileRouting.selectedByProfile = parsed.profile;
|
|
234
234
|
}
|
|
235
235
|
}
|
|
236
|
-
|
|
237
|
-
profileRouting.selectedTemplate = "backend-test-dag";
|
|
238
|
-
profileRouting.source = "taskKind";
|
|
239
|
-
profileRouting.routingReasons = [
|
|
240
|
-
'taskKind "backend-test" selects the dedicated backend test DAG template',
|
|
241
|
-
];
|
|
242
|
-
if (parsed.profile === "auto") {
|
|
243
|
-
profileRouting.selectedByProfile =
|
|
244
|
-
resolveAutoRoutingProfile(profileRouting.candidateProfile);
|
|
245
|
-
}
|
|
246
|
-
else if (parsed.profileExplicit) {
|
|
247
|
-
profileRouting.selectedByProfile = parsed.profile;
|
|
248
|
-
}
|
|
249
|
-
}
|
|
250
|
-
if (isKnowledgeSyncTask) {
|
|
251
|
-
profileRouting.selectedTemplate = "knowledge-sync-dag";
|
|
252
|
-
profileRouting.source = "taskKind";
|
|
253
|
-
profileRouting.routingReasons = [
|
|
254
|
-
'taskKind "knowledge-sync" selects the dedicated knowledge-sync DAG template',
|
|
255
|
-
];
|
|
256
|
-
if (parsed.profile === "auto") {
|
|
257
|
-
profileRouting.selectedByProfile =
|
|
258
|
-
resolveAutoRoutingProfile(profileRouting.candidateProfile);
|
|
259
|
-
}
|
|
260
|
-
else if (parsed.profileExplicit) {
|
|
261
|
-
profileRouting.selectedByProfile = parsed.profile;
|
|
262
|
-
}
|
|
263
|
-
}
|
|
264
|
-
if (isKnowledgeGraphBootstrapTask) {
|
|
265
|
-
profileRouting.selectedTemplate = "knowledge-graph-bootstrap-dag";
|
|
266
|
-
profileRouting.source = "taskKind";
|
|
267
|
-
profileRouting.routingReasons = [
|
|
268
|
-
'taskKind "knowledge-graph-bootstrap" selects the dedicated knowledge-graph bootstrap DAG template',
|
|
269
|
-
];
|
|
270
|
-
if (parsed.profile === "auto") {
|
|
271
|
-
profileRouting.selectedByProfile =
|
|
272
|
-
resolveAutoRoutingProfile(profileRouting.candidateProfile);
|
|
273
|
-
}
|
|
274
|
-
else if (parsed.profileExplicit) {
|
|
275
|
-
profileRouting.selectedByProfile = parsed.profile;
|
|
276
|
-
}
|
|
277
|
-
}
|
|
278
|
-
const initResult = profileRouting.selectedTemplate === "standard-dag"
|
|
236
|
+
const initResult = profileRouting.selectedTemplate === candidateResult.template
|
|
279
237
|
? candidateResult
|
|
280
238
|
: await initHybridDagFromTask(repoRoot, parsed.taskId, {
|
|
281
239
|
outputPath: parsed.outputPath,
|
|
282
240
|
template: profileRouting.selectedTemplate,
|
|
283
241
|
});
|
|
284
242
|
const outputPath = initResult.outputPath;
|
|
285
|
-
const validateSummary = profileRouting.selectedTemplate ===
|
|
243
|
+
const validateSummary = profileRouting.selectedTemplate === candidateResult.template
|
|
286
244
|
? candidateValidateSummary
|
|
287
245
|
: await validateDagUseCase(buildValidateInput(repoRoot, outputPath, parsed));
|
|
288
246
|
const governanceProfile = validateSummary.governanceProfile ?? {
|
|
@@ -317,6 +275,17 @@ export async function generateTaskDagUseCase(input) {
|
|
|
317
275
|
};
|
|
318
276
|
}
|
|
319
277
|
await assertSafeForExecution(outputPath);
|
|
278
|
+
// Mirror worker run-task layout so Observe can find dag-events.jsonl under
|
|
279
|
+
// .harness/task-pool/observability/runs/<runId>/ (CLI path, not Task Pool state).
|
|
280
|
+
const runId = parsed.runId ?? `dag-${Date.now()}-${randomUUID().slice(0, 8)}`;
|
|
281
|
+
const eventsJsonlPath = path.join(parsed.cwd, ".harness", "task-pool", "observability", "runs", runId, "dag-events.jsonl");
|
|
282
|
+
try {
|
|
283
|
+
await mkdir(path.dirname(eventsJsonlPath), { recursive: true });
|
|
284
|
+
}
|
|
285
|
+
catch (error) {
|
|
286
|
+
const message = error instanceof Error ? error.message : String(error);
|
|
287
|
+
throw new Error(`failed to create dag events directory for observe: ${path.dirname(eventsJsonlPath)}: ${message}`);
|
|
288
|
+
}
|
|
320
289
|
const runSummary = await runDagUseCase({
|
|
321
290
|
repoRoot,
|
|
322
291
|
dagPath: outputPath,
|
|
@@ -324,7 +293,8 @@ export async function generateTaskDagUseCase(input) {
|
|
|
324
293
|
initOnly: parsed.initOnly,
|
|
325
294
|
dryRun: parsed.dryRun,
|
|
326
295
|
maxConcurrent: parsed.maxConcurrent,
|
|
327
|
-
runId
|
|
296
|
+
runId,
|
|
297
|
+
eventsJsonlPath,
|
|
328
298
|
canvasPath: parsed.canvasPath,
|
|
329
299
|
canvasName: parsed.canvasName,
|
|
330
300
|
canvasesDir: parsed.canvasesDir,
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
/** Deterministic JSON stringify: object keys sorted at every level. */
|
|
3
|
+
export function stableStringify(value) {
|
|
4
|
+
return JSON.stringify(canonicalize(value));
|
|
5
|
+
}
|
|
6
|
+
function canonicalize(value) {
|
|
7
|
+
if (value === null || typeof value !== "object") {
|
|
8
|
+
return value;
|
|
9
|
+
}
|
|
10
|
+
if (Array.isArray(value)) {
|
|
11
|
+
return value.map((item) => canonicalize(item));
|
|
12
|
+
}
|
|
13
|
+
const obj = value;
|
|
14
|
+
const keys = Object.keys(obj).sort();
|
|
15
|
+
const out = {};
|
|
16
|
+
for (const key of keys) {
|
|
17
|
+
out[key] = canonicalize(obj[key]);
|
|
18
|
+
}
|
|
19
|
+
return out;
|
|
20
|
+
}
|
|
21
|
+
export function normalizeContentSha(value) {
|
|
22
|
+
const trimmed = value.trim();
|
|
23
|
+
const bare = trimmed.startsWith("sha256:")
|
|
24
|
+
? trimmed.slice("sha256:".length)
|
|
25
|
+
: trimmed;
|
|
26
|
+
if (!/^[a-f0-9]{64}$/i.test(bare)) {
|
|
27
|
+
throw new Error(`invalid content sha256: ${value}`);
|
|
28
|
+
}
|
|
29
|
+
return bare.toLowerCase();
|
|
30
|
+
}
|
|
31
|
+
export function formatContentSha(hex) {
|
|
32
|
+
return `sha256:${normalizeContentSha(hex)}`;
|
|
33
|
+
}
|
|
34
|
+
export function normalizeContentRefs(refs) {
|
|
35
|
+
const normalized = refs.map((ref) => ({
|
|
36
|
+
path: ref.path.replace(/\\/g, "/"),
|
|
37
|
+
sha256: normalizeContentSha(ref.sha256),
|
|
38
|
+
}));
|
|
39
|
+
normalized.sort((a, b) => (a.path < b.path ? -1 : a.path > b.path ? 1 : 0));
|
|
40
|
+
const seen = new Set();
|
|
41
|
+
for (const ref of normalized) {
|
|
42
|
+
if (seen.has(ref.path)) {
|
|
43
|
+
throw new Error(`duplicate content ref path: ${ref.path}`);
|
|
44
|
+
}
|
|
45
|
+
seen.add(ref.path);
|
|
46
|
+
}
|
|
47
|
+
return normalized;
|
|
48
|
+
}
|
|
49
|
+
/**
|
|
50
|
+
* Canonical payload for bundleHash.
|
|
51
|
+
* Excludes candidateId, createdAt, description, bundleHash, absolute paths,
|
|
52
|
+
* registry location, and any mutable lifecycle state.
|
|
53
|
+
*/
|
|
54
|
+
export function buildCanonicalBundlePayload(manifest) {
|
|
55
|
+
return {
|
|
56
|
+
schemaVersion: 1,
|
|
57
|
+
candidateKind: manifest.candidateKind,
|
|
58
|
+
parentCandidateId: manifest.parentCandidateId ?? null,
|
|
59
|
+
contentRefs: normalizeContentRefs(manifest.contentRefs),
|
|
60
|
+
};
|
|
61
|
+
}
|
|
62
|
+
export function computeBundleHash(manifest) {
|
|
63
|
+
const payload = buildCanonicalBundlePayload(manifest);
|
|
64
|
+
const digest = createHash("sha256")
|
|
65
|
+
.update(stableStringify(payload))
|
|
66
|
+
.digest("hex");
|
|
67
|
+
return formatContentSha(digest);
|
|
68
|
+
}
|
|
69
|
+
export function sha256Hex(content) {
|
|
70
|
+
return createHash("sha256").update(content).digest("hex");
|
|
71
|
+
}
|
|
72
|
+
export function eventHashHex(payload) {
|
|
73
|
+
return createHash("sha256").update(stableStringify(payload)).digest("hex");
|
|
74
|
+
}
|
|
75
|
+
export const LIFECYCLE_GENESIS_HASH = "0".repeat(64);
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
import { loadManifestInputFromPath, listCandidateIds, materializeManifest, readCandidateRecord, registerCandidateManifest, transitionCandidateLifecycle, } from "../../infrastructure/evaluation/candidate-store.js";
|
|
2
|
+
export async function registerCandidate(input) {
|
|
3
|
+
const raw = await loadManifestInputFromPath(input.repoRoot, input.manifestPath);
|
|
4
|
+
const manifest = await materializeManifest(input.repoRoot, raw);
|
|
5
|
+
return registerCandidateManifest({
|
|
6
|
+
repoRoot: input.repoRoot,
|
|
7
|
+
manifest,
|
|
8
|
+
now: input.now,
|
|
9
|
+
});
|
|
10
|
+
}
|
|
11
|
+
export async function showCandidate(input) {
|
|
12
|
+
return readCandidateRecord(input.repoRoot, input.candidateId);
|
|
13
|
+
}
|
|
14
|
+
export async function listCandidates(input) {
|
|
15
|
+
const ids = await listCandidateIds(input.repoRoot);
|
|
16
|
+
const rows = [];
|
|
17
|
+
for (const candidateId of ids) {
|
|
18
|
+
const record = await readCandidateRecord(input.repoRoot, candidateId);
|
|
19
|
+
rows.push({
|
|
20
|
+
candidateId: record.manifest.candidateId,
|
|
21
|
+
bundleHash: record.manifest.bundleHash,
|
|
22
|
+
candidateKind: record.manifest.candidateKind,
|
|
23
|
+
status: record.status,
|
|
24
|
+
promotionApplied: false,
|
|
25
|
+
});
|
|
26
|
+
}
|
|
27
|
+
return rows;
|
|
28
|
+
}
|
|
29
|
+
export async function transitionCandidate(input) {
|
|
30
|
+
return transitionCandidateLifecycle(input);
|
|
31
|
+
}
|
|
32
|
+
export function formatCandidateMarkdown(record) {
|
|
33
|
+
const lines = [
|
|
34
|
+
`# Candidate: ${record.manifest.candidateId}`,
|
|
35
|
+
"",
|
|
36
|
+
`- status: \`${record.status}\``,
|
|
37
|
+
`- kind: \`${record.manifest.candidateKind}\``,
|
|
38
|
+
`- bundleHash: \`${record.manifest.bundleHash}\``,
|
|
39
|
+
`- parent: \`${record.manifest.parentCandidateId ?? "null"}\``,
|
|
40
|
+
`- promotionApplied: \`false\` (registry lifecycle only; no alias/incumbent)`,
|
|
41
|
+
"",
|
|
42
|
+
"## Content refs",
|
|
43
|
+
"",
|
|
44
|
+
...record.manifest.contentRefs.map((ref) => `- \`${ref.path}\` — \`${ref.sha256}\``),
|
|
45
|
+
"",
|
|
46
|
+
"## Lifecycle",
|
|
47
|
+
"",
|
|
48
|
+
...record.events.map((event) => `- #${event.seq} ${event.from ?? "∅"} → ${event.to}: ${event.reason} (${event.at})`),
|
|
49
|
+
"",
|
|
50
|
+
];
|
|
51
|
+
return `${lines.join("\n")}\n`;
|
|
52
|
+
}
|
|
@@ -0,0 +1,289 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
import { readFile } from "node:fs/promises";
|
|
3
|
+
import path from "node:path";
|
|
4
|
+
import { reportDagUseCase } from "../dag/report-dag.js";
|
|
5
|
+
import { readReplaySpec, writeReplayArtifacts, } from "../../infrastructure/evaluation/store.js";
|
|
6
|
+
function sha256(content) {
|
|
7
|
+
return createHash("sha256").update(content).digest("hex");
|
|
8
|
+
}
|
|
9
|
+
function pairKey(input) {
|
|
10
|
+
return `${input.split}\u0000${input.taskRef}\u0000${input.seed}`;
|
|
11
|
+
}
|
|
12
|
+
async function verifyEvidenceHash(input) {
|
|
13
|
+
const content = await readFile(input.filePath);
|
|
14
|
+
const actual = sha256(content);
|
|
15
|
+
if (actual !== input.expected) {
|
|
16
|
+
throw new Error(`${input.label} hash mismatch: expected ${input.expected}, got ${actual}`);
|
|
17
|
+
}
|
|
18
|
+
}
|
|
19
|
+
function sumOptional(values) {
|
|
20
|
+
if (values.some((value) => value === undefined)) {
|
|
21
|
+
return { value: null, missing: true };
|
|
22
|
+
}
|
|
23
|
+
return {
|
|
24
|
+
value: values.reduce((sum, value) => sum + (value ?? 0), 0),
|
|
25
|
+
missing: false,
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
function verifyPassed(input) {
|
|
29
|
+
if (input.status !== "finished")
|
|
30
|
+
return false;
|
|
31
|
+
const verificationNodes = input.nodes.filter((node) => node.executor === "shell" ||
|
|
32
|
+
node.nodeId.includes("verify") ||
|
|
33
|
+
node.nodeId.includes("gate"));
|
|
34
|
+
return (verificationNodes.length > 0 &&
|
|
35
|
+
verificationNodes.every((node) => node.status === "FINISHED" &&
|
|
36
|
+
(!node.failureCategory || node.failureCategory === "success")));
|
|
37
|
+
}
|
|
38
|
+
function metricsForRun(run) {
|
|
39
|
+
const tokens = sumOptional(run.nodes.map((node) => node.tokensUsed));
|
|
40
|
+
const duration = sumOptional(run.nodes.map((node) => node.durationMs));
|
|
41
|
+
const missingFields = [];
|
|
42
|
+
if (tokens.missing)
|
|
43
|
+
missingFields.push("tokens");
|
|
44
|
+
if (duration.missing)
|
|
45
|
+
missingFields.push("durationMs");
|
|
46
|
+
return {
|
|
47
|
+
metrics: {
|
|
48
|
+
verifyPassed: verifyPassed(run),
|
|
49
|
+
tokens: tokens.value,
|
|
50
|
+
durationMs: duration.value,
|
|
51
|
+
executorCalls: run.nodes.length,
|
|
52
|
+
repairPasses: run.convergence?.currentPass ?? 0,
|
|
53
|
+
},
|
|
54
|
+
missingFields,
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
function compareRows(incumbent, challenger) {
|
|
58
|
+
const reasons = [];
|
|
59
|
+
let verdict;
|
|
60
|
+
if (incumbent.metrics.verifyPassed !== challenger.metrics.verifyPassed) {
|
|
61
|
+
verdict = challenger.metrics.verifyPassed
|
|
62
|
+
? "challenger_win"
|
|
63
|
+
: "incumbent_win";
|
|
64
|
+
reasons.push("verification_outcome");
|
|
65
|
+
}
|
|
66
|
+
else if (!incumbent.metrics.verifyPassed) {
|
|
67
|
+
verdict = "tie";
|
|
68
|
+
reasons.push("both_failed_verification");
|
|
69
|
+
}
|
|
70
|
+
else if (incumbent.metrics.tokens === null ||
|
|
71
|
+
challenger.metrics.tokens === null ||
|
|
72
|
+
incumbent.metrics.durationMs === null ||
|
|
73
|
+
challenger.metrics.durationMs === null) {
|
|
74
|
+
verdict = "incomparable";
|
|
75
|
+
reasons.push("missing_cost_metrics");
|
|
76
|
+
}
|
|
77
|
+
else {
|
|
78
|
+
const challengerNoWorse = challenger.metrics.tokens <= incumbent.metrics.tokens &&
|
|
79
|
+
challenger.metrics.durationMs <= incumbent.metrics.durationMs;
|
|
80
|
+
const incumbentNoWorse = incumbent.metrics.tokens <= challenger.metrics.tokens &&
|
|
81
|
+
incumbent.metrics.durationMs <= challenger.metrics.durationMs;
|
|
82
|
+
if (challengerNoWorse && !incumbentNoWorse) {
|
|
83
|
+
verdict = "challenger_win";
|
|
84
|
+
reasons.push("lower_cost");
|
|
85
|
+
}
|
|
86
|
+
else if (incumbentNoWorse && !challengerNoWorse) {
|
|
87
|
+
verdict = "incumbent_win";
|
|
88
|
+
reasons.push("lower_cost");
|
|
89
|
+
}
|
|
90
|
+
else if (challengerNoWorse && incumbentNoWorse) {
|
|
91
|
+
verdict = "tie";
|
|
92
|
+
reasons.push("equal_metrics");
|
|
93
|
+
}
|
|
94
|
+
else {
|
|
95
|
+
verdict = "incomparable";
|
|
96
|
+
reasons.push("cost_tradeoff");
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
return {
|
|
100
|
+
taskRef: incumbent.taskRef,
|
|
101
|
+
seed: incumbent.seed,
|
|
102
|
+
split: incumbent.split,
|
|
103
|
+
incumbentRunId: incumbent.runId,
|
|
104
|
+
challengerRunId: challenger.runId,
|
|
105
|
+
verdict,
|
|
106
|
+
reasons,
|
|
107
|
+
};
|
|
108
|
+
}
|
|
109
|
+
function buildScorecard(input) {
|
|
110
|
+
const rows = [...input.rows].sort((left, right) => [
|
|
111
|
+
left.split,
|
|
112
|
+
left.taskRef,
|
|
113
|
+
String(left.seed).padStart(12, "0"),
|
|
114
|
+
left.candidateId,
|
|
115
|
+
left.runId,
|
|
116
|
+
]
|
|
117
|
+
.join("\u0000")
|
|
118
|
+
.localeCompare([
|
|
119
|
+
right.split,
|
|
120
|
+
right.taskRef,
|
|
121
|
+
String(right.seed).padStart(12, "0"),
|
|
122
|
+
right.candidateId,
|
|
123
|
+
right.runId,
|
|
124
|
+
].join("\u0000")));
|
|
125
|
+
const incumbentByKey = new Map(rows
|
|
126
|
+
.filter((row) => row.candidateId === input.incumbentCandidateId)
|
|
127
|
+
.map((row) => [pairKey(row), row]));
|
|
128
|
+
const challengerByKey = new Map(rows
|
|
129
|
+
.filter((row) => row.candidateId === input.challengerCandidateId)
|
|
130
|
+
.map((row) => [pairKey(row), row]));
|
|
131
|
+
const keys = [
|
|
132
|
+
...new Set([...incumbentByKey.keys(), ...challengerByKey.keys()]),
|
|
133
|
+
].sort();
|
|
134
|
+
const comparisons = [];
|
|
135
|
+
let unpairedEvidenceCount = 0;
|
|
136
|
+
for (const key of keys) {
|
|
137
|
+
const incumbent = incumbentByKey.get(key);
|
|
138
|
+
const challenger = challengerByKey.get(key);
|
|
139
|
+
if (!incumbent || !challenger) {
|
|
140
|
+
unpairedEvidenceCount +=
|
|
141
|
+
Number(Boolean(incumbent)) + Number(Boolean(challenger));
|
|
142
|
+
continue;
|
|
143
|
+
}
|
|
144
|
+
comparisons.push(compareRows(incumbent, challenger));
|
|
145
|
+
}
|
|
146
|
+
const reasons = ["replay_only"];
|
|
147
|
+
if (comparisons.length === 0)
|
|
148
|
+
reasons.push("insufficient_samples");
|
|
149
|
+
if (unpairedEvidenceCount > 0)
|
|
150
|
+
reasons.push("unpaired_evidence");
|
|
151
|
+
return {
|
|
152
|
+
schemaVersion: 1,
|
|
153
|
+
replayId: input.replayId,
|
|
154
|
+
incumbentCandidateId: input.incumbentCandidateId,
|
|
155
|
+
challengerCandidateId: input.challengerCandidateId,
|
|
156
|
+
rows,
|
|
157
|
+
comparisons,
|
|
158
|
+
aggregate: {
|
|
159
|
+
pairedSampleCount: comparisons.length,
|
|
160
|
+
incumbentWins: comparisons.filter((item) => item.verdict === "incumbent_win").length,
|
|
161
|
+
challengerWins: comparisons.filter((item) => item.verdict === "challenger_win").length,
|
|
162
|
+
ties: comparisons.filter((item) => item.verdict === "tie").length,
|
|
163
|
+
incomparable: comparisons.filter((item) => item.verdict === "incomparable").length,
|
|
164
|
+
unpairedEvidenceCount,
|
|
165
|
+
promotionEligible: false,
|
|
166
|
+
reasons,
|
|
167
|
+
},
|
|
168
|
+
};
|
|
169
|
+
}
|
|
170
|
+
export function formatReplayMarkdown(scorecard) {
|
|
171
|
+
const lines = [
|
|
172
|
+
`# Eval Replay Scorecard: ${scorecard.replayId}`,
|
|
173
|
+
"",
|
|
174
|
+
"> Replay-only evidence. This report never authorizes promotion or executes Pi/DAG work.",
|
|
175
|
+
"",
|
|
176
|
+
`- incumbent: ${scorecard.incumbentCandidateId}`,
|
|
177
|
+
`- challenger: ${scorecard.challengerCandidateId}`,
|
|
178
|
+
`- promotionEligible: ${scorecard.aggregate.promotionEligible}`,
|
|
179
|
+
`- reasons: ${scorecard.aggregate.reasons.join(", ")}`,
|
|
180
|
+
"",
|
|
181
|
+
"## Evidence",
|
|
182
|
+
"",
|
|
183
|
+
"| split | task | seed | candidate | run | verified | tokens | durationMs | calls | repairPasses | missing |",
|
|
184
|
+
"|---|---|---:|---|---|---|---:|---:|---:|---:|---|",
|
|
185
|
+
];
|
|
186
|
+
for (const row of scorecard.rows) {
|
|
187
|
+
lines.push(`| ${row.split} | ${row.taskRef} | ${row.seed} | ${row.candidateId} | ${row.runId} | ${row.metrics.verifyPassed} | ${row.metrics.tokens ?? "n/a"} | ${row.metrics.durationMs ?? "n/a"} | ${row.metrics.executorCalls} | ${row.metrics.repairPasses} | ${row.missingFields.join(", ") || "none"} |`);
|
|
188
|
+
}
|
|
189
|
+
lines.push("", "## Paired Comparisons", "", "| split | task | seed | incumbent run | challenger run | verdict | reasons |", "|---|---|---:|---|---|---|---|");
|
|
190
|
+
for (const comparison of scorecard.comparisons) {
|
|
191
|
+
lines.push(`| ${comparison.split} | ${comparison.taskRef} | ${comparison.seed} | ${comparison.incumbentRunId} | ${comparison.challengerRunId} | ${comparison.verdict} | ${comparison.reasons.join(", ")} |`);
|
|
192
|
+
}
|
|
193
|
+
if (scorecard.comparisons.length === 0) {
|
|
194
|
+
lines.push("| - | - | - | - | - | - | no paired evidence |");
|
|
195
|
+
}
|
|
196
|
+
return `${lines.join("\n")}\n`;
|
|
197
|
+
}
|
|
198
|
+
export async function replayEvaluation(input) {
|
|
199
|
+
const spec = await readReplaySpec(input.repoRoot, input.specPath);
|
|
200
|
+
const rows = [];
|
|
201
|
+
for (const evidence of spec.evidence) {
|
|
202
|
+
const runDir = path.join(input.repoRoot, ".harness", "dag-runs", "completed", evidence.runId);
|
|
203
|
+
const statePath = path.join(runDir, "state.json");
|
|
204
|
+
const runPath = path.join(runDir, "run.json");
|
|
205
|
+
const stateRefPath = path
|
|
206
|
+
.relative(input.repoRoot, statePath)
|
|
207
|
+
.split(path.sep)
|
|
208
|
+
.join("/");
|
|
209
|
+
const runRefPath = path
|
|
210
|
+
.relative(input.repoRoot, runPath)
|
|
211
|
+
.split(path.sep)
|
|
212
|
+
.join("/");
|
|
213
|
+
await verifyEvidenceHash({
|
|
214
|
+
filePath: statePath,
|
|
215
|
+
expected: evidence.stateSha256,
|
|
216
|
+
label: "state.json",
|
|
217
|
+
});
|
|
218
|
+
await verifyEvidenceHash({
|
|
219
|
+
filePath: runPath,
|
|
220
|
+
expected: evidence.runSha256,
|
|
221
|
+
label: "run.json",
|
|
222
|
+
});
|
|
223
|
+
const report = await reportDagUseCase({
|
|
224
|
+
repoRoot: input.repoRoot,
|
|
225
|
+
runId: evidence.runId,
|
|
226
|
+
lifecycle: "completed",
|
|
227
|
+
failedOnly: false,
|
|
228
|
+
latest: false,
|
|
229
|
+
});
|
|
230
|
+
const run = report.runs[0];
|
|
231
|
+
if (!run) {
|
|
232
|
+
throw new Error(`completed DAG run not found: ${evidence.runId}`);
|
|
233
|
+
}
|
|
234
|
+
if (run.evaluationAssociation.status === "present") {
|
|
235
|
+
if (run.evaluationAssociation.candidateId !== evidence.candidateId) {
|
|
236
|
+
throw new Error(`replay evidence candidateId conflict for ${evidence.runId}: evidence=${evidence.candidateId}, run=${run.evaluationAssociation.candidateId}`);
|
|
237
|
+
}
|
|
238
|
+
if (run.evaluationAssociation.seed !== evidence.seed) {
|
|
239
|
+
throw new Error(`replay evidence seed conflict for ${evidence.runId}: evidence=${evidence.seed}, run=${run.evaluationAssociation.seed}`);
|
|
240
|
+
}
|
|
241
|
+
if (run.evaluationAssociation.split &&
|
|
242
|
+
run.evaluationAssociation.split !== evidence.split) {
|
|
243
|
+
throw new Error(`replay evidence split conflict for ${evidence.runId}: evidence=${evidence.split}, run=${run.evaluationAssociation.split}`);
|
|
244
|
+
}
|
|
245
|
+
if (run.evaluationAssociation.taskRef &&
|
|
246
|
+
run.evaluationAssociation.taskRef !== evidence.taskRef) {
|
|
247
|
+
throw new Error(`replay evidence taskRef conflict for ${evidence.runId}: evidence=${evidence.taskRef}, run=${run.evaluationAssociation.taskRef}`);
|
|
248
|
+
}
|
|
249
|
+
}
|
|
250
|
+
if (![
|
|
251
|
+
"finished",
|
|
252
|
+
"failed",
|
|
253
|
+
"partial_failed",
|
|
254
|
+
"superseded",
|
|
255
|
+
"abandoned",
|
|
256
|
+
].includes(run.status)) {
|
|
257
|
+
throw new Error(`completed DAG run ${evidence.runId} is not terminal (status=${run.status})`);
|
|
258
|
+
}
|
|
259
|
+
const { metrics, missingFields } = metricsForRun(run);
|
|
260
|
+
rows.push({
|
|
261
|
+
...evidence,
|
|
262
|
+
lifecycle: "completed",
|
|
263
|
+
runStatus: run.status,
|
|
264
|
+
metrics,
|
|
265
|
+
missingFields,
|
|
266
|
+
evidenceRefs: [
|
|
267
|
+
{ path: stateRefPath, sha256: evidence.stateSha256 },
|
|
268
|
+
{ path: runRefPath, sha256: evidence.runSha256 },
|
|
269
|
+
],
|
|
270
|
+
});
|
|
271
|
+
}
|
|
272
|
+
const scorecard = buildScorecard({
|
|
273
|
+
replayId: spec.replayId,
|
|
274
|
+
incumbentCandidateId: spec.incumbentCandidateId,
|
|
275
|
+
challengerCandidateId: spec.challengerCandidateId,
|
|
276
|
+
rows,
|
|
277
|
+
});
|
|
278
|
+
const markdown = formatReplayMarkdown(scorecard);
|
|
279
|
+
if (input.writeArtifacts === false) {
|
|
280
|
+
return { scorecard, markdown };
|
|
281
|
+
}
|
|
282
|
+
const paths = await writeReplayArtifacts({
|
|
283
|
+
repoRoot: input.repoRoot,
|
|
284
|
+
replayId: spec.replayId,
|
|
285
|
+
scorecard,
|
|
286
|
+
markdown,
|
|
287
|
+
});
|
|
288
|
+
return { scorecard, markdown, ...paths };
|
|
289
|
+
}
|