@tea-agent/loop-agent 0.13.0-alpha.0 → 0.13.0-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +155 -153
- package/CHANGELOG.md +326 -301
- package/README.md +345 -326
- package/bin/agent-worker.js +22 -22
- package/bin/loop-agent.js +21 -21
- package/dist/application/dag/generate-task-dag.js +28 -58
- package/dist/application/evaluation/candidate-hash.js +75 -0
- package/dist/application/evaluation/candidate.js +52 -0
- package/dist/application/evaluation/replay.js +289 -0
- package/dist/application/evaluation/types.js +130 -0
- package/dist/cli/command-definitions.js +17 -4
- package/dist/cli/program.js +8 -4
- package/dist/commands/cursor-prompt.js +6 -6
- package/dist/commands/eval.js +235 -0
- package/dist/commands/init.js +544 -506
- package/dist/commands/loop-benchmark.js +11 -11
- package/dist/commands/pi-reuse-benchmark.js +16 -16
- package/dist/executors/pi-sdk-executor.js +38 -24
- package/dist/executors/shell-executor.js +34 -2
- package/dist/executors/shell-presets.js +20 -0
- package/dist/executors/shell-verification.js +7 -0
- package/dist/governance/manifest-types.js +1 -0
- package/dist/infrastructure/evaluation/candidate-store.js +435 -0
- package/dist/infrastructure/evaluation/store.js +40 -0
- package/dist/sidecars/cursor-prompt/executor.js +1 -1
- package/dist/task/config-types.js +23 -0
- package/dist/task/runtime.js +27 -27
- package/dist/worker/observe/routes.js +18 -3
- package/dist/worker/observe/spec-evidence.js +1 -1
- package/dist/worker/observe/static/api.js +46 -46
- package/dist/worker/observe/static/app.js +150 -150
- package/dist/worker/observe/static/constants.js +148 -148
- package/dist/worker/observe/static/copy.js +67 -67
- package/dist/worker/observe/static/dag-helpers.js +172 -172
- package/dist/worker/observe/static/dag-layout.d.ts +31 -31
- package/dist/worker/observe/static/dag-layout.js +83 -83
- package/dist/worker/observe/static/dag-model.js +72 -72
- package/dist/worker/observe/static/dom.js +61 -61
- package/dist/worker/observe/static/format-pool.js +67 -67
- package/dist/worker/observe/static/format.js +292 -292
- package/dist/worker/observe/static/index.html +308 -308
- package/dist/worker/observe/static/kpi.js +94 -94
- package/dist/worker/observe/static/relations.js +133 -133
- package/dist/worker/observe/static/router.js +93 -93
- package/dist/worker/observe/static/run-processing.js +148 -148
- package/dist/worker/observe/static/shell-chrome.js +68 -68
- package/dist/worker/observe/static/state.js +253 -253
- package/dist/worker/observe/static/styles.css +1902 -1902
- package/dist/worker/observe/static/views/batch.js +227 -227
- package/dist/worker/observe/static/views/dag-graph.js +172 -172
- package/dist/worker/observe/static/views/dag-inspector.js +607 -596
- package/dist/worker/observe/static/views/dag.js +362 -362
- package/dist/worker/observe/static/views/dashboard.js +445 -445
- package/dist/worker/observe/static/views/failures.js +143 -143
- package/dist/worker/observe/static/views/feature.js +492 -492
- package/dist/worker/observe/static/views/pool.js +350 -350
- package/dist/worker/observe/static/views/run.js +453 -453
- package/dist/worker/observe/static/views/session-timeline.js +205 -205
- package/dist/worker/observe/static/views/shell.js +7 -7
- package/dist/worker/observe/static/views/task.js +314 -314
- package/dist/worker/observe/static/views/timeline.js +163 -163
- package/dist/workflows/dag/backend-test-analysis-contract.js +120 -0
- package/dist/workflows/dag/canvas-observer.js +275 -275
- package/dist/workflows/dag/dynamic-runtime/map.js +90 -2
- package/dist/workflows/dag/init-hybrid.js +1415 -200
- package/dist/workflows/dag/node-execution.js +9 -0
- package/dist/workflows/dag/prompt.js +9 -0
- package/dist/workflows/dag/report.js +35 -1
- package/dist/workflows/dag/runner.js +28 -2
- package/dist/workflows/dag/task-demand-routing.js +383 -0
- package/dist/workflows/dag/types.js +50 -13
- package/dist/workflows/dag/upstream-artifacts.js +1 -0
- package/dist/workflows/dag/validate.js +59 -1
- package/docs/README.md +106 -104
- package/docs/agent-dag-recovery-playbook.md +195 -193
- package/docs/agent-dag-runner.md +67 -67
- package/docs/architecture/README.md +26 -26
- package/docs/architecture/dag-execution.md +140 -140
- package/docs/architecture/evolution.md +54 -54
- package/docs/architecture/facts-and-state.md +71 -71
- package/docs/architecture/runtime-boundaries.md +191 -191
- package/docs/architecture/system-overview.md +93 -93
- package/docs/architecture/worker-and-feature.md +85 -85
- package/docs/cursor-prompt-sidecar.md +36 -36
- package/docs/decisions/README.md +18 -18
- package/docs/design/README.md +167 -85
- package/docs/development-principles.md +73 -73
- package/docs/exec-plans/README.md +6 -6
- package/docs/exec-plans/active/README.md +15 -11
- package/docs/exec-plans/completed/README.md +85 -74
- package/docs/feature-workflow.md +389 -339
- package/docs/harness-methodology-debugging.md +153 -153
- package/docs/harness-methodology-tdd.md +130 -130
- package/docs/harness-methodology-verification.md +27 -27
- package/docs/init-surface.manifest.json +289 -280
- package/docs/loop-agent-harness.md +142 -141
- package/docs/production-readiness.md +96 -96
- package/docs/progress/README.md +64 -58
- package/docs/reports/README.md +117 -100
- package/docs/skills/README.md +7 -7
- package/docs/skills/vetted-skill-registry.md +29 -27
- package/docs/templates/adr.md +60 -60
- package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
- package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
- package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
- package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
- package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
- package/docs/templates/agent-dag-report.schema.json +473 -473
- package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
- package/docs/templates/agent-dag.base.json +190 -190
- package/docs/templates/agent-dag.final-verification.json +185 -185
- package/docs/templates/agent-dag.schema.json +411 -383
- package/docs/templates/agent-dag.supervised-implementation.json +501 -501
- package/docs/templates/backend-test-analysis.schema.json +44 -0
- package/docs/templates/backend-test-dag.generate-pytest.prompt.md +202 -139
- package/docs/templates/backend-test-dag.json +311 -288
- package/docs/templates/backend-test-dag.retrospect.prompt.md +125 -125
- package/docs/templates/backend-test-dag.review-cases.prompt.md +81 -81
- package/docs/templates/exec-plan.md +64 -64
- package/docs/templates/feature-spec.md +53 -53
- package/docs/templates/frontend-design-contract.md +42 -33
- package/docs/templates/frontend-task-constraints.md +35 -25
- package/docs/templates/frontend-task-requirement.md +70 -61
- package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -0
- package/docs/templates/frontend-test-dag.json +23 -0
- package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -0
- package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -0
- package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -0
- package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -0
- package/docs/templates/harness.schema.json +221 -221
- package/docs/templates/hybrid-dag.json +188 -188
- package/docs/templates/init-evolution-review.md +35 -35
- package/docs/templates/interactive-ui-round2-experiment.md +66 -66
- package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -118
- package/docs/templates/knowledge-sync-dag.json +178 -177
- package/docs/templates/knowledge-sync-draft.schema.json +71 -71
- package/docs/templates/product-line/AGENTS.md +8 -8
- package/docs/templates/product-line/README.md +9 -9
- package/docs/templates/product-line/acceptance.yaml +14 -14
- package/docs/templates/product-line/closeout.yaml +9 -9
- package/docs/templates/product-line/design.md +13 -13
- package/docs/templates/product-line/links.md +10 -10
- package/docs/templates/product-line/requirement.md +17 -17
- package/docs/templates/product-line/task-graph.yaml +15 -15
- package/docs/templates/product-line/task.yaml +64 -64
- package/docs/templates/product-line/test-plan.md +7 -7
- package/docs/templates/production-readiness-checklist.md +57 -57
- package/docs/templates/progress-log.md +17 -17
- package/docs/templates/project-start-checklist.md +9 -9
- package/docs/templates/qa-report.md +48 -48
- package/docs/templates/sprint-contract.md +29 -29
- package/docs/templates/worker-dogfood-evidence.md +80 -80
- package/docs/templates/worker-dogfood-setup.md +68 -68
- package/docs/verification-matrix.md +70 -67
- package/examples/decision-gate-agent-dag.json +177 -177
- package/examples/example-dag.json +46 -46
- package/examples/hybrid-loop-agent-dag.json +189 -189
- package/harness.json +66 -66
- package/package.json +88 -52
- package/scripts/check-product-line-docs.sh +29 -29
- package/scripts/check-task-pool-root.sh +32 -32
- package/scripts/kb-bootstrap-init-skeleton.sh +240 -239
- package/scripts/kb-graph-incremental-prepare.mjs +386 -372
- package/scripts/kb-graph-incremental-prepare.sh +5 -5
- package/scripts/kb-graph-materialize.mjs +105 -105
- package/scripts/kb-graph-materialize.sh +4 -4
- package/scripts/kb-graph-promote.mjs +164 -153
- package/scripts/kb-graph-promote.sh +4 -4
- package/scripts/kb-query.mjs +554 -554
- package/scripts/kb-query.sh +5 -5
- package/skills/agent-worker/SKILL.md +39 -39
- package/skills/agent-worker/references/agent-worker-operator.md +60 -60
- package/skills/ai-engineering-context/SKILL.md +48 -48
- package/skills/analyze-product-dependencies/SKILL.md +67 -0
- package/skills/analyze-product-dependencies/agents/openai.yaml +4 -0
- package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -0
- package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -0
- package/skills/analyze-product-dependencies/references/example.md +76 -0
- package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -0
- package/skills/analyze-product-dependencies/references/input-contract.md +11 -0
- package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -0
- package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -0
- package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -0
- package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -0
- package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -0
- package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -0
- package/skills/analyze-product-requirements/SKILL.md +90 -0
- package/skills/analyze-product-requirements/agents/openai.yaml +4 -0
- package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -0
- package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -0
- package/skills/analyze-product-requirements/references/example.md +86 -0
- package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -0
- package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -0
- package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -0
- package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -0
- package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -0
- package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -0
- package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -0
- package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -0
- package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -0
- package/skills/code-review-core/SKILL.md +20 -20
- package/skills/codebase-scout/SKILL.md +19 -19
- package/skills/frontend-design-review/SKILL.md +66 -61
- package/skills/frontend-design-review/references/review-checklist.md +58 -37
- package/skills/frontend-implementation/SKILL.md +45 -52
- package/skills/frontend-implementation/references/code-standards.md +32 -34
- package/skills/frontend-implementation/references/design-spec.md +46 -46
- package/skills/frontend-implementation/references/node-contracts.md +76 -63
- package/skills/frontend-review/SKILL.md +59 -53
- package/skills/frontend-review/references/review-findings.md +47 -42
- package/skills/frontend-verification/SKILL.md +53 -40
- package/skills/frontend-verification/references/verification-checklist.md +68 -56
- package/skills/grill-me/SKILL.md +10 -10
- package/skills/grill-with-docs/SKILL.md +88 -88
- package/skills/grill-with-docs/adr-format.md +47 -47
- package/skills/grill-with-docs/context-format.md +60 -60
- package/skills/init-capability-evolution/SKILL.md +70 -70
- package/skills/loop-agent/SKILL.md +151 -151
- package/skills/loop-agent/references/README.md +67 -67
- package/skills/loop-agent/references/command-reference.md +505 -453
- package/skills/loop-agent/references/docs-converge.md +126 -126
- package/skills/loop-agent/references/harness-policy.md +263 -263
- package/skills/loop-agent/references/hybrid-dag.md +238 -233
- package/skills/loop-agent/references/learned/README.md +21 -21
- package/skills/loop-agent/references/long-running-loop.md +57 -57
- package/skills/loop-agent/references/model-routing.md +36 -36
- package/skills/loop-agent/references/multi-worktree.md +54 -54
- package/skills/loop-agent/references/one-shot-runs.md +85 -85
- package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
- package/skills/loop-agent/references/pi-prompt.md +23 -23
- package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
- package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
- package/skills/loop-agent/references/task-workflow.md +89 -89
- package/skills/loop-agent/references/verification-and-failure-handling.md +139 -139
- package/skills/playwright-cli/SKILL.md +420 -0
- package/skills/playwright-cli/references/element-attributes.md +23 -0
- package/skills/playwright-cli/references/playwright-tests.md +39 -0
- package/skills/playwright-cli/references/request-mocking.md +87 -0
- package/skills/playwright-cli/references/running-code.md +241 -0
- package/skills/playwright-cli/references/session-management.md +225 -0
- package/skills/playwright-cli/references/storage-state.md +275 -0
- package/skills/playwright-cli/references/test-generation.md +433 -0
- package/skills/playwright-cli/references/tracing.md +139 -0
- package/skills/playwright-cli/references/video-recording.md +143 -0
- package/skills/playwright-cli-case-generator/SKILL.md +74 -0
- package/skills/requesting-code-review/SKILL.md +101 -101
- package/skills/requesting-code-review/code-reviewer.md +168 -168
- package/skills/systematic-debugging/CREATION-LOG.md +119 -119
- package/skills/systematic-debugging/SKILL.md +296 -296
- package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
- package/skills/systematic-debugging/condition-based-waiting.md +115 -115
- package/skills/systematic-debugging/defense-in-depth.md +122 -122
- package/skills/systematic-debugging/find-polluter.sh +63 -63
- package/skills/systematic-debugging/root-cause-tracing.md +169 -169
- package/skills/systematic-debugging/test-academic.md +14 -14
- package/skills/systematic-debugging/test-pressure-1.md +58 -58
- package/skills/systematic-debugging/test-pressure-2.md +68 -68
- package/skills/systematic-debugging/test-pressure-3.md +69 -69
- package/skills/test-driven-development/SKILL.md +20 -20
- package/skills/using-git-worktrees/SKILL.md +215 -215
- package/skills/verification-before-completion/SKILL.md +154 -154
- package/skills/webapp-testing/SKILL.md +19 -19
|
@@ -0,0 +1,435 @@
|
|
|
1
|
+
import { access, lstat, mkdir, mkdtemp, readdir, readFile, realpath, rename, rm, writeFile, } from "node:fs/promises";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
import { computeBundleHash, eventHashHex, formatContentSha, LIFECYCLE_GENESIS_HASH, normalizeContentSha, sha256Hex, } from "../../application/evaluation/candidate-hash.js";
|
|
4
|
+
import { candidateManifestInputSchema, candidateManifestSchema, lifecycleEventSchema, } from "../../application/evaluation/types.js";
|
|
5
|
+
import { appendJsonlLineAtomic, writeJsonAtomic, } from "../harness/atomic-write.js";
|
|
6
|
+
import { EVALUATION_ROOT } from "./store.js";
|
|
7
|
+
const SAFE_ID = /^[A-Za-z0-9][A-Za-z0-9._-]*$/;
|
|
8
|
+
const MANIFEST_INTEGRITY_FILE = "manifest.sha256";
|
|
9
|
+
const LIFECYCLE_LOCK_DIR = ".lifecycle.lock";
|
|
10
|
+
const LIFECYCLE_LOCK_RETRIES = 100;
|
|
11
|
+
const LIFECYCLE_LOCK_DELAY_MS = 10;
|
|
12
|
+
const FORBIDDEN_PREFIXES = [
|
|
13
|
+
"src/application/evaluation/",
|
|
14
|
+
"src/workflows/dag/",
|
|
15
|
+
"src/executors/",
|
|
16
|
+
"src/worker/",
|
|
17
|
+
"src/infrastructure/harness/completed-facts-guard.ts",
|
|
18
|
+
".harness/dag-runs/",
|
|
19
|
+
".harness/runs/",
|
|
20
|
+
".harness/evaluation/",
|
|
21
|
+
];
|
|
22
|
+
const FORBIDDEN_SEGMENTS = [
|
|
23
|
+
"private-verifier",
|
|
24
|
+
"private_verifier",
|
|
25
|
+
"held-out-evaluator",
|
|
26
|
+
"held_out_evaluator",
|
|
27
|
+
"held-out",
|
|
28
|
+
"completed-facts",
|
|
29
|
+
];
|
|
30
|
+
export function assertCandidateId(value) {
|
|
31
|
+
if (!SAFE_ID.test(value)) {
|
|
32
|
+
throw new Error(`candidate-id must contain only letters, numbers, dot, underscore, or hyphen: ${value}`);
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
export function candidateDir(repoRoot, candidateId) {
|
|
36
|
+
assertCandidateId(candidateId);
|
|
37
|
+
return path.join(repoRoot, EVALUATION_ROOT, "candidates", candidateId);
|
|
38
|
+
}
|
|
39
|
+
export function candidateManifestPath(repoRoot, candidateId) {
|
|
40
|
+
return path.join(candidateDir(repoRoot, candidateId), "manifest.json");
|
|
41
|
+
}
|
|
42
|
+
export function candidateLifecyclePath(repoRoot, candidateId) {
|
|
43
|
+
return path.join(candidateDir(repoRoot, candidateId), "lifecycle.jsonl");
|
|
44
|
+
}
|
|
45
|
+
export async function resolveRepoRelativeSafe(repoRoot, repoRelativePath) {
|
|
46
|
+
if (path.isAbsolute(repoRelativePath) ||
|
|
47
|
+
repoRelativePath.startsWith("/") ||
|
|
48
|
+
/^[A-Za-z]:[\\/]/.test(repoRelativePath)) {
|
|
49
|
+
throw new Error(`content ref path must be repo-relative: ${repoRelativePath}`);
|
|
50
|
+
}
|
|
51
|
+
const normalized = repoRelativePath.replace(/\\/g, "/");
|
|
52
|
+
if (normalized.split("/").includes("..") ||
|
|
53
|
+
normalized.startsWith("./../") ||
|
|
54
|
+
normalized.includes("/../")) {
|
|
55
|
+
throw new Error(`content ref path escapes repo root: ${repoRelativePath}`);
|
|
56
|
+
}
|
|
57
|
+
const canonicalRoot = await realpath(repoRoot);
|
|
58
|
+
const lexical = path.resolve(repoRoot, normalized);
|
|
59
|
+
let absolute;
|
|
60
|
+
try {
|
|
61
|
+
absolute = await realpath(lexical);
|
|
62
|
+
}
|
|
63
|
+
catch {
|
|
64
|
+
// File may not exist yet for some flows; still check lexical containment.
|
|
65
|
+
const relativeLexical = path.relative(canonicalRoot, lexical);
|
|
66
|
+
if (!relativeLexical ||
|
|
67
|
+
relativeLexical.startsWith("..") ||
|
|
68
|
+
path.isAbsolute(relativeLexical)) {
|
|
69
|
+
throw new Error(`content ref path escapes repo root: ${repoRelativePath}`);
|
|
70
|
+
}
|
|
71
|
+
throw new Error(`content ref missing: ${normalized}`);
|
|
72
|
+
}
|
|
73
|
+
const relative = path.relative(canonicalRoot, absolute).replace(/\\/g, "/");
|
|
74
|
+
if (!relative || relative.startsWith("..") || path.isAbsolute(relative)) {
|
|
75
|
+
throw new Error(`content ref path escapes repo root: ${repoRelativePath}`);
|
|
76
|
+
}
|
|
77
|
+
const st = await lstat(absolute).catch(() => null);
|
|
78
|
+
if (st?.isSymbolicLink()) {
|
|
79
|
+
// realpath already resolved; re-check containment after resolve (done above)
|
|
80
|
+
}
|
|
81
|
+
return { absolute, relative };
|
|
82
|
+
}
|
|
83
|
+
export function assertAllowedContentRefPath(relativePosix) {
|
|
84
|
+
const rel = relativePosix.replace(/\\/g, "/");
|
|
85
|
+
if (rel.startsWith("..") || path.isAbsolute(rel)) {
|
|
86
|
+
throw new Error(`forbidden content ref path: ${rel}`);
|
|
87
|
+
}
|
|
88
|
+
for (const prefix of FORBIDDEN_PREFIXES) {
|
|
89
|
+
if (rel === prefix.replace(/\/$/, "") || rel.startsWith(prefix)) {
|
|
90
|
+
throw new Error(`content ref enters forbidden candidate surface: ${rel}`);
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
const lower = rel.toLowerCase();
|
|
94
|
+
for (const segment of FORBIDDEN_SEGMENTS) {
|
|
95
|
+
if (lower.includes(`/${segment}/`) ||
|
|
96
|
+
lower.endsWith(`/${segment}`) ||
|
|
97
|
+
lower.startsWith(`${segment}/`) ||
|
|
98
|
+
lower === segment) {
|
|
99
|
+
throw new Error(`content ref enters forbidden candidate surface: ${rel}`);
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
async function hashExistingContent(repoRoot, repoRelativePath) {
|
|
104
|
+
const { absolute, relative } = await resolveRepoRelativeSafe(repoRoot, repoRelativePath);
|
|
105
|
+
assertAllowedContentRefPath(relative);
|
|
106
|
+
const content = await readFile(absolute);
|
|
107
|
+
return { relative, sha256: sha256Hex(content) };
|
|
108
|
+
}
|
|
109
|
+
function lifecycleTransitionAllowed(from, to) {
|
|
110
|
+
if (from === null && to === "proposed")
|
|
111
|
+
return true;
|
|
112
|
+
const edges = {
|
|
113
|
+
proposed: ["eligible", "invalid"],
|
|
114
|
+
eligible: ["experimenting", "invalid"],
|
|
115
|
+
experimenting: ["accepted", "rejected", "invalid"],
|
|
116
|
+
accepted: ["retired"],
|
|
117
|
+
rejected: ["retired"],
|
|
118
|
+
invalid: ["retired"],
|
|
119
|
+
retired: [],
|
|
120
|
+
};
|
|
121
|
+
if (from === null)
|
|
122
|
+
return false;
|
|
123
|
+
return edges[from]?.includes(to) ?? false;
|
|
124
|
+
}
|
|
125
|
+
function computeEventHash(event) {
|
|
126
|
+
return eventHashHex({
|
|
127
|
+
schemaVersion: event.schemaVersion,
|
|
128
|
+
seq: event.seq,
|
|
129
|
+
from: event.from,
|
|
130
|
+
to: event.to,
|
|
131
|
+
reason: event.reason,
|
|
132
|
+
at: event.at,
|
|
133
|
+
previousEventHash: event.previousEventHash,
|
|
134
|
+
});
|
|
135
|
+
}
|
|
136
|
+
export async function materializeManifest(repoRoot, raw) {
|
|
137
|
+
const input = candidateManifestInputSchema.parse(raw);
|
|
138
|
+
const verifiedRefs = [];
|
|
139
|
+
for (const ref of input.contentRefs) {
|
|
140
|
+
const expected = normalizeContentSha(ref.sha256);
|
|
141
|
+
const actual = await hashExistingContent(repoRoot, ref.path);
|
|
142
|
+
assertAllowedContentRefPath(actual.relative);
|
|
143
|
+
if (actual.sha256 !== expected) {
|
|
144
|
+
throw new Error(`content ref hash mismatch for ${actual.relative}: expected ${expected}, got ${actual.sha256}`);
|
|
145
|
+
}
|
|
146
|
+
// Persist repo-relative posix path as resolved relative (no abs paths).
|
|
147
|
+
verifiedRefs.push({
|
|
148
|
+
path: actual.relative,
|
|
149
|
+
sha256: formatContentSha(actual.sha256),
|
|
150
|
+
});
|
|
151
|
+
}
|
|
152
|
+
const withoutHash = {
|
|
153
|
+
schemaVersion: 1,
|
|
154
|
+
candidateId: input.candidateId,
|
|
155
|
+
parentCandidateId: input.parentCandidateId ?? null,
|
|
156
|
+
candidateKind: input.candidateKind,
|
|
157
|
+
createdAt: input.createdAt,
|
|
158
|
+
...(input.description !== undefined
|
|
159
|
+
? { description: input.description }
|
|
160
|
+
: {}),
|
|
161
|
+
contentRefs: verifiedRefs,
|
|
162
|
+
};
|
|
163
|
+
const bundleHash = computeBundleHash(withoutHash);
|
|
164
|
+
if (input.bundleHash) {
|
|
165
|
+
const provided = formatContentSha(normalizeContentSha(input.bundleHash));
|
|
166
|
+
if (provided !== bundleHash) {
|
|
167
|
+
throw new Error(`bundleHash mismatch: expected ${bundleHash}, got ${provided}`);
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
const manifest = {
|
|
171
|
+
...withoutHash,
|
|
172
|
+
bundleHash,
|
|
173
|
+
};
|
|
174
|
+
return candidateManifestSchema.parse(manifest);
|
|
175
|
+
}
|
|
176
|
+
function serializeManifest(manifest) {
|
|
177
|
+
return `${JSON.stringify(manifest, null, 2)}\n`;
|
|
178
|
+
}
|
|
179
|
+
function manifestIntegrityPath(repoRoot, candidateId) {
|
|
180
|
+
return path.join(candidateDir(repoRoot, candidateId), MANIFEST_INTEGRITY_FILE);
|
|
181
|
+
}
|
|
182
|
+
async function assertManifestIntegrity(repoRoot, candidateId, rawManifest) {
|
|
183
|
+
const expected = (await readFile(manifestIntegrityPath(repoRoot, candidateId), "utf-8")).trim();
|
|
184
|
+
const actual = sha256Hex(rawManifest);
|
|
185
|
+
if (expected !== actual) {
|
|
186
|
+
throw new Error(`candidate manifest integrity mismatch for ${candidateId}: expected ${expected}, got ${actual}`);
|
|
187
|
+
}
|
|
188
|
+
}
|
|
189
|
+
async function acquireLifecycleLock(repoRoot, candidateId) {
|
|
190
|
+
const lockPath = path.join(candidateDir(repoRoot, candidateId), LIFECYCLE_LOCK_DIR);
|
|
191
|
+
for (let attempt = 0; attempt < LIFECYCLE_LOCK_RETRIES; attempt += 1) {
|
|
192
|
+
try {
|
|
193
|
+
await mkdir(lockPath);
|
|
194
|
+
return async () => {
|
|
195
|
+
await rm(lockPath, { recursive: true, force: true });
|
|
196
|
+
};
|
|
197
|
+
}
|
|
198
|
+
catch (error) {
|
|
199
|
+
if (error.code !== "EEXIST")
|
|
200
|
+
throw error;
|
|
201
|
+
await new Promise((resolve) => setTimeout(resolve, LIFECYCLE_LOCK_DELAY_MS));
|
|
202
|
+
}
|
|
203
|
+
}
|
|
204
|
+
throw new Error(`timed out acquiring lifecycle lock for ${candidateId}`);
|
|
205
|
+
}
|
|
206
|
+
async function pathExists(filePath) {
|
|
207
|
+
try {
|
|
208
|
+
await access(filePath);
|
|
209
|
+
return true;
|
|
210
|
+
}
|
|
211
|
+
catch {
|
|
212
|
+
return false;
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
function manifestsEqual(existing, manifest) {
|
|
216
|
+
return (existing.candidateId === manifest.candidateId &&
|
|
217
|
+
existing.bundleHash === manifest.bundleHash &&
|
|
218
|
+
existing.candidateKind === manifest.candidateKind &&
|
|
219
|
+
existing.createdAt === manifest.createdAt &&
|
|
220
|
+
(existing.description ?? "") === (manifest.description ?? "") &&
|
|
221
|
+
(existing.parentCandidateId ?? null) ===
|
|
222
|
+
(manifest.parentCandidateId ?? null) &&
|
|
223
|
+
JSON.stringify(existing.contentRefs) ===
|
|
224
|
+
JSON.stringify(manifest.contentRefs));
|
|
225
|
+
}
|
|
226
|
+
async function readExistingRegistration(repoRoot, manifest, manifestPath, lifecyclePath) {
|
|
227
|
+
const existingRaw = await readFile(manifestPath, "utf-8");
|
|
228
|
+
await assertManifestIntegrity(repoRoot, manifest.candidateId, existingRaw);
|
|
229
|
+
const existing = candidateManifestSchema.parse(JSON.parse(existingRaw));
|
|
230
|
+
if (!manifestsEqual(existing, manifest)) {
|
|
231
|
+
throw new Error(`candidate already exists with different content: ${manifest.candidateId}`);
|
|
232
|
+
}
|
|
233
|
+
const record = await readCandidateRecord(repoRoot, manifest.candidateId);
|
|
234
|
+
return { record, idempotent: true, manifestPath, lifecyclePath };
|
|
235
|
+
}
|
|
236
|
+
export async function registerCandidateManifest(input) {
|
|
237
|
+
const { repoRoot, manifest } = input;
|
|
238
|
+
assertCandidateId(manifest.candidateId);
|
|
239
|
+
const manifestPath = candidateManifestPath(repoRoot, manifest.candidateId);
|
|
240
|
+
const lifecyclePath = candidateLifecyclePath(repoRoot, manifest.candidateId);
|
|
241
|
+
if (await pathExists(manifestPath)) {
|
|
242
|
+
return readExistingRegistration(repoRoot, manifest, manifestPath, lifecyclePath);
|
|
243
|
+
}
|
|
244
|
+
const at = input.now ?? new Date().toISOString();
|
|
245
|
+
const baseEvent = {
|
|
246
|
+
schemaVersion: 1,
|
|
247
|
+
seq: 1,
|
|
248
|
+
from: null,
|
|
249
|
+
to: "proposed",
|
|
250
|
+
reason: "registered",
|
|
251
|
+
at,
|
|
252
|
+
previousEventHash: LIFECYCLE_GENESIS_HASH,
|
|
253
|
+
};
|
|
254
|
+
const event = {
|
|
255
|
+
...baseEvent,
|
|
256
|
+
eventHash: computeEventHash(baseEvent),
|
|
257
|
+
};
|
|
258
|
+
lifecycleEventSchema.parse(event);
|
|
259
|
+
const candidatesRoot = path.dirname(candidateDir(repoRoot, manifest.candidateId));
|
|
260
|
+
await mkdir(candidatesRoot, { recursive: true });
|
|
261
|
+
const stagingDir = await mkdtemp(path.join(candidatesRoot, `.${manifest.candidateId}.register-`));
|
|
262
|
+
try {
|
|
263
|
+
await writeJsonAtomic(path.join(stagingDir, "manifest.json"), manifest, {
|
|
264
|
+
repoRoot,
|
|
265
|
+
});
|
|
266
|
+
await writeFile(path.join(stagingDir, MANIFEST_INTEGRITY_FILE), `${sha256Hex(serializeManifest(manifest))}\n`, "utf-8");
|
|
267
|
+
await appendJsonlLineAtomic(path.join(stagingDir, "lifecycle.jsonl"), event, {
|
|
268
|
+
repoRoot,
|
|
269
|
+
});
|
|
270
|
+
try {
|
|
271
|
+
await rename(stagingDir, candidateDir(repoRoot, manifest.candidateId));
|
|
272
|
+
}
|
|
273
|
+
catch (error) {
|
|
274
|
+
const code = error.code;
|
|
275
|
+
if (code !== "EEXIST" && code !== "ENOTEMPTY")
|
|
276
|
+
throw error;
|
|
277
|
+
await rm(stagingDir, { recursive: true, force: true });
|
|
278
|
+
return readExistingRegistration(repoRoot, manifest, manifestPath, lifecyclePath);
|
|
279
|
+
}
|
|
280
|
+
}
|
|
281
|
+
catch (error) {
|
|
282
|
+
await rm(stagingDir, { recursive: true, force: true });
|
|
283
|
+
throw error;
|
|
284
|
+
}
|
|
285
|
+
return {
|
|
286
|
+
record: {
|
|
287
|
+
manifest,
|
|
288
|
+
status: "proposed",
|
|
289
|
+
events: [event],
|
|
290
|
+
promotionApplied: false,
|
|
291
|
+
},
|
|
292
|
+
idempotent: false,
|
|
293
|
+
manifestPath,
|
|
294
|
+
lifecyclePath,
|
|
295
|
+
};
|
|
296
|
+
}
|
|
297
|
+
async function readLifecycleEvents(repoRoot, candidateId) {
|
|
298
|
+
const lifecyclePath = candidateLifecyclePath(repoRoot, candidateId);
|
|
299
|
+
const text = await readFile(lifecyclePath, "utf-8");
|
|
300
|
+
const lines = text
|
|
301
|
+
.split("\n")
|
|
302
|
+
.map((line) => line.trim())
|
|
303
|
+
.filter(Boolean);
|
|
304
|
+
const events = [];
|
|
305
|
+
let previous = LIFECYCLE_GENESIS_HASH;
|
|
306
|
+
for (let i = 0; i < lines.length; i += 1) {
|
|
307
|
+
const parsed = lifecycleEventSchema.parse(JSON.parse(lines[i]));
|
|
308
|
+
if (parsed.seq !== i + 1) {
|
|
309
|
+
throw new Error(`lifecycle seq gap for ${candidateId}: expected ${i + 1}, got ${parsed.seq}`);
|
|
310
|
+
}
|
|
311
|
+
if (parsed.previousEventHash !== previous) {
|
|
312
|
+
throw new Error(`lifecycle chain break for ${candidateId} at seq ${parsed.seq}`);
|
|
313
|
+
}
|
|
314
|
+
const expectedHash = computeEventHash({
|
|
315
|
+
schemaVersion: parsed.schemaVersion,
|
|
316
|
+
seq: parsed.seq,
|
|
317
|
+
from: parsed.from,
|
|
318
|
+
to: parsed.to,
|
|
319
|
+
reason: parsed.reason,
|
|
320
|
+
at: parsed.at,
|
|
321
|
+
previousEventHash: parsed.previousEventHash,
|
|
322
|
+
});
|
|
323
|
+
if (parsed.eventHash !== expectedHash) {
|
|
324
|
+
throw new Error(`lifecycle event hash mismatch for ${candidateId} at seq ${parsed.seq}`);
|
|
325
|
+
}
|
|
326
|
+
if (i === 0) {
|
|
327
|
+
if (parsed.from !== null || parsed.to !== "proposed") {
|
|
328
|
+
throw new Error(`lifecycle genesis must be null→proposed for ${candidateId}`);
|
|
329
|
+
}
|
|
330
|
+
}
|
|
331
|
+
else {
|
|
332
|
+
const prev = events[i - 1];
|
|
333
|
+
if (parsed.from !== prev.to) {
|
|
334
|
+
throw new Error(`lifecycle from-state mismatch for ${candidateId} at seq ${parsed.seq}`);
|
|
335
|
+
}
|
|
336
|
+
if (!lifecycleTransitionAllowed(parsed.from, parsed.to)) {
|
|
337
|
+
throw new Error(`illegal lifecycle transition recorded for ${candidateId}: ${parsed.from}→${parsed.to}`);
|
|
338
|
+
}
|
|
339
|
+
}
|
|
340
|
+
events.push(parsed);
|
|
341
|
+
previous = parsed.eventHash;
|
|
342
|
+
}
|
|
343
|
+
if (events.length === 0) {
|
|
344
|
+
throw new Error(`empty lifecycle for candidate ${candidateId}`);
|
|
345
|
+
}
|
|
346
|
+
return events;
|
|
347
|
+
}
|
|
348
|
+
export async function readCandidateRecord(repoRoot, candidateId) {
|
|
349
|
+
assertCandidateId(candidateId);
|
|
350
|
+
const manifestPath = candidateManifestPath(repoRoot, candidateId);
|
|
351
|
+
const rawText = await readFile(manifestPath, "utf-8");
|
|
352
|
+
await assertManifestIntegrity(repoRoot, candidateId, rawText);
|
|
353
|
+
const raw = JSON.parse(rawText);
|
|
354
|
+
const stored = candidateManifestSchema.parse(raw);
|
|
355
|
+
if (stored.candidateId !== candidateId) {
|
|
356
|
+
throw new Error(`candidate manifest identity mismatch: directory=${candidateId}, manifest=${stored.candidateId}`);
|
|
357
|
+
}
|
|
358
|
+
// Re-verify each content ref against workspace bytes.
|
|
359
|
+
for (const ref of stored.contentRefs) {
|
|
360
|
+
const expected = normalizeContentSha(ref.sha256);
|
|
361
|
+
const actual = await hashExistingContent(repoRoot, ref.path);
|
|
362
|
+
if (actual.sha256 !== expected) {
|
|
363
|
+
throw new Error(`content ref hash mismatch for ${actual.relative}: expected ${expected}, got ${actual.sha256}`);
|
|
364
|
+
}
|
|
365
|
+
}
|
|
366
|
+
const recomputed = computeBundleHash(stored);
|
|
367
|
+
if (recomputed !== stored.bundleHash) {
|
|
368
|
+
throw new Error(`bundleHash mismatch for ${candidateId}: expected ${recomputed}, got ${stored.bundleHash}`);
|
|
369
|
+
}
|
|
370
|
+
const events = await readLifecycleEvents(repoRoot, candidateId);
|
|
371
|
+
const status = events[events.length - 1].to;
|
|
372
|
+
return {
|
|
373
|
+
manifest: stored,
|
|
374
|
+
status,
|
|
375
|
+
events,
|
|
376
|
+
promotionApplied: false,
|
|
377
|
+
};
|
|
378
|
+
}
|
|
379
|
+
export async function listCandidateIds(repoRoot) {
|
|
380
|
+
const root = path.join(repoRoot, EVALUATION_ROOT, "candidates");
|
|
381
|
+
try {
|
|
382
|
+
const entries = await readdir(root, { withFileTypes: true });
|
|
383
|
+
return entries
|
|
384
|
+
.filter((entry) => entry.isDirectory() && SAFE_ID.test(entry.name))
|
|
385
|
+
.map((entry) => entry.name)
|
|
386
|
+
.sort();
|
|
387
|
+
}
|
|
388
|
+
catch (error) {
|
|
389
|
+
const code = error.code;
|
|
390
|
+
if (code === "ENOENT")
|
|
391
|
+
return [];
|
|
392
|
+
throw error;
|
|
393
|
+
}
|
|
394
|
+
}
|
|
395
|
+
export async function transitionCandidateLifecycle(input) {
|
|
396
|
+
const reason = input.reason.trim();
|
|
397
|
+
if (!reason) {
|
|
398
|
+
throw new Error("lifecycle transition requires non-empty --reason");
|
|
399
|
+
}
|
|
400
|
+
const releaseLock = await acquireLifecycleLock(input.repoRoot, input.candidateId);
|
|
401
|
+
try {
|
|
402
|
+
const record = await readCandidateRecord(input.repoRoot, input.candidateId);
|
|
403
|
+
const from = record.status;
|
|
404
|
+
if (!lifecycleTransitionAllowed(from, input.to)) {
|
|
405
|
+
throw new Error(`illegal lifecycle transition: ${from} → ${input.to}`);
|
|
406
|
+
}
|
|
407
|
+
const previous = record.events[record.events.length - 1];
|
|
408
|
+
const base = {
|
|
409
|
+
schemaVersion: 1,
|
|
410
|
+
seq: previous.seq + 1,
|
|
411
|
+
from,
|
|
412
|
+
to: input.to,
|
|
413
|
+
reason,
|
|
414
|
+
at: input.now ?? new Date().toISOString(),
|
|
415
|
+
previousEventHash: previous.eventHash,
|
|
416
|
+
};
|
|
417
|
+
const event = {
|
|
418
|
+
...base,
|
|
419
|
+
eventHash: computeEventHash(base),
|
|
420
|
+
};
|
|
421
|
+
lifecycleEventSchema.parse(event);
|
|
422
|
+
await appendJsonlLineAtomic(candidateLifecyclePath(input.repoRoot, input.candidateId), event, { repoRoot: input.repoRoot });
|
|
423
|
+
return readCandidateRecord(input.repoRoot, input.candidateId);
|
|
424
|
+
}
|
|
425
|
+
finally {
|
|
426
|
+
await releaseLock();
|
|
427
|
+
}
|
|
428
|
+
}
|
|
429
|
+
export async function loadManifestInputFromPath(repoRoot, manifestPath) {
|
|
430
|
+
const resolved = path.isAbsolute(manifestPath)
|
|
431
|
+
? manifestPath
|
|
432
|
+
: path.resolve(repoRoot, manifestPath);
|
|
433
|
+
const raw = JSON.parse(await readFile(resolved, "utf-8"));
|
|
434
|
+
return candidateManifestInputSchema.parse(raw);
|
|
435
|
+
}
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
import { readFile } from "node:fs/promises";
|
|
2
|
+
import path from "node:path";
|
|
3
|
+
import { replaySpecSchema, } from "../../application/evaluation/types.js";
|
|
4
|
+
import { writeJsonAtomic, writeTextAtomic } from "../harness/atomic-write.js";
|
|
5
|
+
export const EVALUATION_ROOT = path.join(".harness", "evaluation");
|
|
6
|
+
export function assertReplayId(value) {
|
|
7
|
+
if (!/^[A-Za-z0-9][A-Za-z0-9._-]*$/.test(value)) {
|
|
8
|
+
throw new Error(`replay-id must contain only letters, numbers, dot, underscore, or hyphen: ${value}`);
|
|
9
|
+
}
|
|
10
|
+
}
|
|
11
|
+
export async function readReplaySpec(repoRoot, specPath) {
|
|
12
|
+
const resolved = path.resolve(repoRoot, specPath);
|
|
13
|
+
const raw = JSON.parse(await readFile(resolved, "utf-8"));
|
|
14
|
+
return replaySpecSchema.parse(raw);
|
|
15
|
+
}
|
|
16
|
+
export function replayDir(repoRoot, replayId) {
|
|
17
|
+
assertReplayId(replayId);
|
|
18
|
+
return path.join(repoRoot, EVALUATION_ROOT, "replays", replayId);
|
|
19
|
+
}
|
|
20
|
+
export function replayScorecardPath(repoRoot, replayId) {
|
|
21
|
+
return path.join(replayDir(repoRoot, replayId), "scorecard.json");
|
|
22
|
+
}
|
|
23
|
+
export function replayMarkdownPath(repoRoot, replayId) {
|
|
24
|
+
return path.join(replayDir(repoRoot, replayId), "report.md");
|
|
25
|
+
}
|
|
26
|
+
export async function writeReplayArtifacts(input) {
|
|
27
|
+
const scorecardPath = replayScorecardPath(input.repoRoot, input.replayId);
|
|
28
|
+
const markdownPath = replayMarkdownPath(input.repoRoot, input.replayId);
|
|
29
|
+
await writeJsonAtomic(scorecardPath, input.scorecard, {
|
|
30
|
+
repoRoot: input.repoRoot,
|
|
31
|
+
});
|
|
32
|
+
await writeTextAtomic(markdownPath, input.markdown, {
|
|
33
|
+
repoRoot: input.repoRoot,
|
|
34
|
+
});
|
|
35
|
+
return { scorecardPath, markdownPath };
|
|
36
|
+
}
|
|
37
|
+
export async function readReplayScorecard(repoRoot, replayId) {
|
|
38
|
+
const raw = await readFile(replayScorecardPath(repoRoot, replayId), "utf-8");
|
|
39
|
+
return JSON.parse(raw);
|
|
40
|
+
}
|
|
@@ -29,7 +29,7 @@ export function resolveArtifactWriteDir(options) {
|
|
|
29
29
|
}
|
|
30
30
|
export function buildArtifactPathPrompt(writeDir) {
|
|
31
31
|
if (!writeDir)
|
|
32
|
-
return `After changes, write artifacts/修改记录.md and artifacts/验证结果.md with verification evidence.
|
|
32
|
+
return `After changes, write artifacts/修改记录.md and artifacts/验证结果.md with verification evidence.
|
|
33
33
|
${ARTIFACT_INSTRUCTIONS}`;
|
|
34
34
|
return [
|
|
35
35
|
`After changes, write the following files:`,
|
|
@@ -11,6 +11,7 @@ export const taskKindSchema = z.enum([
|
|
|
11
11
|
"standard",
|
|
12
12
|
"feature-study",
|
|
13
13
|
"frontend-implementation",
|
|
14
|
+
"frontend-test",
|
|
14
15
|
"backend-test",
|
|
15
16
|
"knowledge-sync",
|
|
16
17
|
"knowledge-graph-bootstrap",
|
|
@@ -53,6 +54,24 @@ export const loopAutoExecutionPolicySchema = z.enum([
|
|
|
53
54
|
]);
|
|
54
55
|
export const LOOP_AUTO_WRITE_POLICY_REMOVED_ERROR = "task field loopAutoWritePolicy is no longer supported; use loopAutoExecutionPolicy (off | approval-required | enabled)";
|
|
55
56
|
export const CURSOR_TASK_FIELD_REMOVED_ERROR = 'task fields "executor" and "cursorModel" are no longer supported; governed runtime is Pi-only';
|
|
57
|
+
export const frontendMockPolicySchema = z.enum(["auto", "required", "disabled"]);
|
|
58
|
+
export const frontendMockVerifyCommandSchema = z.object({
|
|
59
|
+
label: z.string().min(1),
|
|
60
|
+
command: z.string().min(1),
|
|
61
|
+
timeoutMs: z.number().int().positive().optional(),
|
|
62
|
+
});
|
|
63
|
+
export const frontendMockConfigSchema = z.object({
|
|
64
|
+
policy: frontendMockPolicySchema.optional().default("auto"),
|
|
65
|
+
serviceRoot: z.string().min(1).optional(),
|
|
66
|
+
verifyCommands: z.array(frontendMockVerifyCommandSchema).optional().default([]),
|
|
67
|
+
});
|
|
68
|
+
/** Batch limits for the browser-driven frontend test DAG. These are post-case
|
|
69
|
+
* stop thresholds, not model-provider hard token caps. */
|
|
70
|
+
export const frontendTestConfigSchema = z.object({
|
|
71
|
+
maxCasesPerBatch: z.number().int().min(1).max(50).optional().default(20),
|
|
72
|
+
maxTokensPerCase: z.number().int().positive().optional(),
|
|
73
|
+
maxTotalTokens: z.number().int().positive().optional(),
|
|
74
|
+
});
|
|
56
75
|
export const convergenceConfigSchema = z.object({
|
|
57
76
|
enabled: z.boolean().optional().default(false),
|
|
58
77
|
maxPasses: z.number().int().positive().optional().default(3),
|
|
@@ -111,6 +130,10 @@ const taskConfigObjectSchema = z.object({
|
|
|
111
130
|
maxGoalContinuationsPerRun: z.number().int().positive().optional().default(5),
|
|
112
131
|
/** Pi subagent assisted mode: 'off' (default), 'analyze-plan', or 'full' */
|
|
113
132
|
piSubagentMode: piSubagentModeSchema.optional().default("off"),
|
|
133
|
+
/** Frontend mock data workflow policy, service root hint, and deterministic mock verification commands. */
|
|
134
|
+
frontendMock: frontendMockConfigSchema.optional(),
|
|
135
|
+
/** Frontend browser-test batch and post-case token-stop configuration. */
|
|
136
|
+
frontendTest: frontendTestConfigSchema.optional(),
|
|
114
137
|
notes: z.string().optional().default(""),
|
|
115
138
|
});
|
|
116
139
|
export const taskConfigSchema = z.preprocess((raw) => {
|
package/dist/task/runtime.js
CHANGED
|
@@ -236,35 +236,35 @@ export function shouldInjectSubagentGuidance(step, mode) {
|
|
|
236
236
|
return false;
|
|
237
237
|
}
|
|
238
238
|
/** Advisory guidance for `analyze-plan` mode. */
|
|
239
|
-
export const SUBAGENT_GUIDANCE_STANDARD = `<subagent_guidance>
|
|
240
|
-
You have access to the \`subagent\` tool for lightweight delegation within this step.
|
|
241
|
-
Use it only for read-only tasks:
|
|
242
|
-
- Parallel scout: dispatch multiple subagents to search/read different areas simultaneously
|
|
243
|
-
- Chain: scout -> planner (one subagent scouts, another plans based on findings)
|
|
244
|
-
- Reviewer: have a subagent review your analysis/plan before finalizing
|
|
245
|
-
Do NOT use subagent for writing, editing, or executing commands.
|
|
246
|
-
Subagent output is advisory only; always verify and incorporate findings into your own output.
|
|
247
|
-
Do NOT treat subagent results as authoritative state or artifact sources.
|
|
239
|
+
export const SUBAGENT_GUIDANCE_STANDARD = `<subagent_guidance>
|
|
240
|
+
You have access to the \`subagent\` tool for lightweight delegation within this step.
|
|
241
|
+
Use it only for read-only tasks:
|
|
242
|
+
- Parallel scout: dispatch multiple subagents to search/read different areas simultaneously
|
|
243
|
+
- Chain: scout -> planner (one subagent scouts, another plans based on findings)
|
|
244
|
+
- Reviewer: have a subagent review your analysis/plan before finalizing
|
|
245
|
+
Do NOT use subagent for writing, editing, or executing commands.
|
|
246
|
+
Subagent output is advisory only; always verify and incorporate findings into your own output.
|
|
247
|
+
Do NOT treat subagent results as authoritative state or artifact sources.
|
|
248
248
|
</subagent_guidance>`;
|
|
249
249
|
/** Strong guidance for `full` mode — prescriptive when to delegate. */
|
|
250
|
-
export const SUBAGENT_GUIDANCE_STRONG = `<subagent_guidance>
|
|
251
|
-
You have access to the \`subagent\` tool for lightweight delegation within this step.
|
|
252
|
-
|
|
253
|
-
You SHOULD delegate to subagent scouts when:
|
|
254
|
-
- The task requires scanning 3+ directories or comparing implementations across modules
|
|
255
|
-
- You would otherwise need 5+ sequential read/grep calls to gather context
|
|
256
|
-
- A reviewer subagent can independently catch scope drift before you finalize your output
|
|
257
|
-
|
|
258
|
-
Delegation saves context tokens and produces better results.
|
|
259
|
-
|
|
260
|
-
Allowed patterns:
|
|
261
|
-
- Parallel scout: dispatch 2-3 subagents simultaneously to cover different file trees
|
|
262
|
-
- Chain: scout -> planner (one subagent scouts, another plans based on findings)
|
|
263
|
-
- Reviewer: have a subagent review your analysis/plan before finalizing
|
|
264
|
-
|
|
265
|
-
Do NOT use subagent for writing, editing, or executing commands.
|
|
266
|
-
Subagent output is advisory only; always verify and incorporate findings into your own output.
|
|
267
|
-
Do NOT treat subagent results as authoritative state or artifact sources.
|
|
250
|
+
export const SUBAGENT_GUIDANCE_STRONG = `<subagent_guidance>
|
|
251
|
+
You have access to the \`subagent\` tool for lightweight delegation within this step.
|
|
252
|
+
|
|
253
|
+
You SHOULD delegate to subagent scouts when:
|
|
254
|
+
- The task requires scanning 3+ directories or comparing implementations across modules
|
|
255
|
+
- You would otherwise need 5+ sequential read/grep calls to gather context
|
|
256
|
+
- A reviewer subagent can independently catch scope drift before you finalize your output
|
|
257
|
+
|
|
258
|
+
Delegation saves context tokens and produces better results.
|
|
259
|
+
|
|
260
|
+
Allowed patterns:
|
|
261
|
+
- Parallel scout: dispatch 2-3 subagents simultaneously to cover different file trees
|
|
262
|
+
- Chain: scout -> planner (one subagent scouts, another plans based on findings)
|
|
263
|
+
- Reviewer: have a subagent review your analysis/plan before finalizing
|
|
264
|
+
|
|
265
|
+
Do NOT use subagent for writing, editing, or executing commands.
|
|
266
|
+
Subagent output is advisory only; always verify and incorporate findings into your own output.
|
|
267
|
+
Do NOT treat subagent results as authoritative state or artifact sources.
|
|
268
268
|
</subagent_guidance>`;
|
|
269
269
|
/** Preserved for backward compatibility (alias of STANDARD). */
|
|
270
270
|
export const SUBAGENT_GUIDANCE = SUBAGENT_GUIDANCE_STANDARD;
|
|
@@ -6,6 +6,7 @@ import { parseWorkerEventLine } from "../observability/events.js";
|
|
|
6
6
|
import { isSafeObservabilityIdentifier } from "../observability/event-store.js";
|
|
7
7
|
import { clampEventHistoryLimit, listBatchEventHistory, listPoolEventHistory, } from "../observability/event-history.js";
|
|
8
8
|
import { buildGlobalSnapshot, clampTaskRunHistoryLimit, listTaskRunHistory, resolveLegacyTask, } from "../observability/read-model.js";
|
|
9
|
+
import { dagSourceBindingSchema } from "../../workflows/dag/types.js";
|
|
9
10
|
import { getTaskPoolRoot } from "../pool/run-store.js";
|
|
10
11
|
import { isAllowedArtifactTextPath, resolveArtifactPath, toRepoRelativeArtifactPath, } from "./paths.js";
|
|
11
12
|
import { extractSpecEvidence, } from "./spec-evidence.js";
|
|
@@ -714,12 +715,16 @@ async function handleDagNodeSpecEvidence(_req, res, match, ctx) {
|
|
|
714
715
|
}
|
|
715
716
|
// Extract skill injection info from the DAG run spec (run.json)
|
|
716
717
|
const skillInjection = { skills: [], references: [] };
|
|
718
|
+
let sourceBinding;
|
|
717
719
|
const dagRunsRoot = path.resolve(ctx.repoRoot, ".harness", "dag-runs");
|
|
718
720
|
for (const lifecycle of ["active", "completed", "paused"]) {
|
|
719
721
|
const runJsonPath = path.join(dagRunsRoot, lifecycle, dagRunId, "run.json");
|
|
720
722
|
try {
|
|
721
723
|
const runRaw = await readFile(runJsonPath, "utf-8");
|
|
722
724
|
const runSpec = JSON.parse(runRaw);
|
|
725
|
+
const parsedSourceBinding = dagSourceBindingSchema.safeParse(runSpec.sourceBinding);
|
|
726
|
+
if (parsedSourceBinding.success)
|
|
727
|
+
sourceBinding = parsedSourceBinding.data;
|
|
723
728
|
const task = runSpec.tasks?.find((t) => t.id === nodeId);
|
|
724
729
|
if (task?.skills && Array.isArray(task.skills)) {
|
|
725
730
|
skillInjection.skills = task.skills;
|
|
@@ -732,26 +737,36 @@ async function handleDagNodeSpecEvidence(_req, res, match, ctx) {
|
|
|
732
737
|
}
|
|
733
738
|
const evidence = await extractSpecEvidence(ctx.repoRoot, dagRunId, nodeId);
|
|
734
739
|
if (!evidence) {
|
|
740
|
+
const hasSourceBinding = Boolean(sourceBinding);
|
|
735
741
|
sendJson(res, 200, {
|
|
736
742
|
dagRunId,
|
|
737
743
|
nodeId,
|
|
738
|
-
status: "no-evidence",
|
|
739
|
-
skillInjection
|
|
744
|
+
status: hasSourceBinding ? "source-bound" : "no-evidence",
|
|
745
|
+
skillInjection,
|
|
746
|
+
sourceBinding,
|
|
740
747
|
specReads: [],
|
|
741
748
|
specSearches: [],
|
|
742
749
|
knowledgeBaseQueries: [],
|
|
743
|
-
summary:
|
|
750
|
+
summary: hasSourceBinding
|
|
751
|
+
? `已绑定 ${sourceBinding.sources.length} 个任务源文件和 ${sourceBinding.requirementIds.length} 个显式需求编号;未观察到 read 工具调用。`
|
|
752
|
+
: "未找到该节点的 session events 记录。",
|
|
744
753
|
});
|
|
745
754
|
return;
|
|
746
755
|
}
|
|
747
756
|
evidence.skillInjection = skillInjection;
|
|
757
|
+
evidence.sourceBinding = sourceBinding;
|
|
748
758
|
// Recompute status considering skill injection
|
|
749
759
|
if (evidence.specReads.length === 0 && evidence.knowledgeBaseQueries.length === 0) {
|
|
750
760
|
if (evidence.specSearches.length > 0) {
|
|
751
761
|
evidence.status = "search-only";
|
|
752
762
|
}
|
|
763
|
+
else if (sourceBinding) {
|
|
764
|
+
evidence.status = "source-bound";
|
|
765
|
+
evidence.summary = `已绑定 ${sourceBinding.sources.length} 个任务源文件和 ${sourceBinding.requirementIds.length} 个显式需求编号;未观察到 read 工具调用。`;
|
|
766
|
+
}
|
|
753
767
|
else if (skillInjection.skills.length > 0) {
|
|
754
768
|
evidence.status = "spec-injected";
|
|
769
|
+
evidence.summary = "规范 skill 已注入但未观察到规范文件读取或知识库查询。";
|
|
755
770
|
}
|
|
756
771
|
}
|
|
757
772
|
sendJson(res, 200, evidence);
|
|
@@ -216,7 +216,7 @@ export async function extractSpecEvidence(repoRoot, dagRunId, nodeId) {
|
|
|
216
216
|
path: p,
|
|
217
217
|
timestamp: t || undefined,
|
|
218
218
|
}));
|
|
219
|
-
const hasInjection =
|
|
219
|
+
const hasInjection = false; // Injection is authoritative only after run.json is inspected by the route.
|
|
220
220
|
if (specReads.length > 0) {
|
|
221
221
|
status = "spec-read";
|
|
222
222
|
}
|