@tea-agent/loop-agent 0.12.0 → 0.13.0-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +155 -153
- package/CHANGELOG.md +338 -265
- package/README.md +345 -298
- package/bin/agent-worker.js +22 -22
- package/bin/loop-agent.js +21 -21
- package/dist/application/dag/generate-task-dag.js +28 -28
- package/dist/application/evaluation/candidate-hash.js +75 -0
- package/dist/application/evaluation/candidate.js +52 -0
- package/dist/application/evaluation/replay.js +289 -0
- package/dist/application/evaluation/types.js +130 -0
- package/dist/cli/command-definitions.js +27 -7
- package/dist/cli/program.js +8 -4
- package/dist/commands/cursor-prompt.js +6 -6
- package/dist/commands/eval.js +235 -0
- package/dist/commands/init.js +544 -506
- package/dist/commands/knowledge.js +129 -31
- package/dist/commands/loop-benchmark.js +11 -11
- package/dist/commands/pi-reuse-benchmark.js +16 -16
- package/dist/executors/pi-sdk-executor.js +38 -24
- package/dist/executors/shell-executor.js +34 -2
- package/dist/executors/shell-presets.js +20 -0
- package/dist/executors/shell-verification.js +7 -0
- package/dist/governance/manifest-types.js +4 -0
- package/dist/infrastructure/evaluation/candidate-store.js +435 -0
- package/dist/infrastructure/evaluation/store.js +40 -0
- package/dist/sidecars/cursor-prompt/executor.js +1 -1
- package/dist/task/config-types.js +28 -1
- package/dist/task/runtime.js +27 -27
- package/dist/worker/cli.js +96 -1
- package/dist/worker/delivery/package.js +3 -3
- package/dist/worker/feature/decision-loader.js +37 -6
- package/dist/worker/feature/next-action.js +10 -2
- package/dist/worker/feature/ready-plan-projection.js +81 -0
- package/dist/worker/feature/reducer.js +2 -1
- package/dist/worker/feature/review.js +19 -2
- package/dist/worker/feature/run.js +27 -2
- package/dist/worker/follow-up/approve.js +5 -2
- package/dist/worker/follow-up/factory.js +1 -1
- package/dist/worker/observability/read-model.js +246 -41
- package/dist/worker/observe/routes.js +173 -15
- package/dist/worker/observe/spec-evidence.js +281 -0
- package/dist/worker/observe/static/api.js +46 -27
- package/dist/worker/observe/static/app.js +150 -150
- package/dist/worker/observe/static/constants.js +148 -148
- package/dist/worker/observe/static/copy.js +67 -67
- package/dist/worker/observe/static/dag-helpers.js +172 -172
- package/dist/worker/observe/static/dag-layout.d.ts +31 -31
- package/dist/worker/observe/static/dag-layout.js +83 -83
- package/dist/worker/observe/static/dag-model.js +72 -72
- package/dist/worker/observe/static/dom.js +61 -61
- package/dist/worker/observe/static/format-pool.js +67 -67
- package/dist/worker/observe/static/format.js +292 -292
- package/dist/worker/observe/static/index.html +308 -308
- package/dist/worker/observe/static/kpi.js +94 -94
- package/dist/worker/observe/static/relations.js +133 -128
- package/dist/worker/observe/static/router.js +93 -85
- package/dist/worker/observe/static/run-processing.js +148 -148
- package/dist/worker/observe/static/shell-chrome.js +68 -68
- package/dist/worker/observe/static/state.js +253 -253
- package/dist/worker/observe/static/styles.css +1902 -1890
- package/dist/worker/observe/static/views/batch.js +227 -226
- package/dist/worker/observe/static/views/dag-graph.js +172 -172
- package/dist/worker/observe/static/views/dag-inspector.js +607 -477
- package/dist/worker/observe/static/views/dag.js +362 -362
- package/dist/worker/observe/static/views/dashboard.js +445 -442
- package/dist/worker/observe/static/views/failures.js +143 -143
- package/dist/worker/observe/static/views/feature.js +492 -453
- package/dist/worker/observe/static/views/pool.js +350 -347
- package/dist/worker/observe/static/views/run.js +453 -453
- package/dist/worker/observe/static/views/session-timeline.js +205 -205
- package/dist/worker/observe/static/views/shell.js +7 -7
- package/dist/worker/observe/static/views/task.js +314 -260
- package/dist/worker/observe/static/views/timeline.js +163 -163
- package/dist/worker/pool/doctor.js +165 -0
- package/dist/worker/pool/migrate-state.js +303 -0
- package/dist/worker/pool/run-store.js +205 -17
- package/dist/worker/pool/types.js +17 -1
- package/dist/worker/pool/validation.js +100 -15
- package/dist/worker/report/morning-report.js +12 -2
- package/dist/worker/runner/run-ready.js +41 -26
- package/dist/worker/task-graph/ready-planner.js +136 -0
- package/dist/workflows/dag/backend-test-analysis-contract.js +120 -0
- package/dist/workflows/dag/canvas-observer.js +275 -275
- package/dist/workflows/dag/convergence/controller.js +16 -8
- package/dist/workflows/dag/dynamic-runtime/map.js +90 -2
- package/dist/workflows/dag/failure-routing.js +12 -1
- package/dist/workflows/dag/init-hybrid.js +2404 -360
- package/dist/workflows/dag/node-execution.js +9 -0
- package/dist/workflows/dag/prompt.js +9 -0
- package/dist/workflows/dag/report.js +35 -1
- package/dist/workflows/dag/runner.js +28 -2
- package/dist/workflows/dag/task-demand-routing.js +383 -0
- package/dist/workflows/dag/types.js +51 -13
- package/dist/workflows/dag/upstream-artifacts.js +1 -0
- package/dist/workflows/dag/validate.js +59 -1
- package/docs/README.md +106 -104
- package/docs/agent-dag-recovery-playbook.md +195 -184
- package/docs/agent-dag-runner.md +67 -67
- package/docs/architecture/README.md +26 -26
- package/docs/architecture/dag-execution.md +140 -140
- package/docs/architecture/evolution.md +54 -53
- package/docs/architecture/facts-and-state.md +71 -58
- package/docs/architecture/runtime-boundaries.md +191 -191
- package/docs/architecture/system-overview.md +93 -93
- package/docs/architecture/worker-and-feature.md +85 -81
- package/docs/cursor-prompt-sidecar.md +36 -36
- package/docs/decisions/README.md +18 -15
- package/docs/design/README.md +167 -77
- package/docs/development-principles.md +73 -73
- package/docs/exec-plans/README.md +6 -6
- package/docs/exec-plans/active/README.md +15 -9
- package/docs/exec-plans/completed/README.md +85 -73
- package/docs/feature-workflow.md +389 -261
- package/docs/harness-methodology-debugging.md +153 -153
- package/docs/harness-methodology-tdd.md +130 -130
- package/docs/harness-methodology-verification.md +27 -27
- package/docs/init-surface.manifest.json +289 -280
- package/docs/loop-agent-harness.md +142 -130
- package/docs/production-readiness.md +96 -96
- package/docs/progress/README.md +64 -54
- package/docs/reports/README.md +117 -94
- package/docs/skills/README.md +7 -7
- package/docs/skills/vetted-skill-registry.md +29 -27
- package/docs/templates/adr.md +60 -60
- package/docs/templates/agent-dag-authority-surface-audit.prompt.md +94 -94
- package/docs/templates/agent-dag-decision-envelope.schema.json +213 -213
- package/docs/templates/agent-dag-decision-gate-dogfood-report.md +117 -117
- package/docs/templates/agent-dag-decision-gate.prompt.md +246 -246
- package/docs/templates/agent-dag-process-supervisor.prompt.md +98 -98
- package/docs/templates/agent-dag-report.schema.json +473 -473
- package/docs/templates/agent-dag-review-verdict.prompt.md +68 -68
- package/docs/templates/agent-dag.base.json +190 -190
- package/docs/templates/agent-dag.final-verification.json +185 -185
- package/docs/templates/agent-dag.schema.json +411 -383
- package/docs/templates/agent-dag.supervised-implementation.json +501 -501
- package/docs/templates/backend-test-analysis.schema.json +44 -0
- package/docs/templates/backend-test-dag.generate-pytest.prompt.md +202 -139
- package/docs/templates/backend-test-dag.json +311 -276
- package/docs/templates/backend-test-dag.retrospect.prompt.md +125 -125
- package/docs/templates/backend-test-dag.review-cases.prompt.md +81 -81
- package/docs/templates/exec-plan.md +64 -64
- package/docs/templates/feature-spec.md +53 -53
- package/docs/templates/frontend-design-contract.md +42 -33
- package/docs/templates/frontend-task-constraints.md +35 -25
- package/docs/templates/frontend-task-requirement.md +70 -61
- package/docs/templates/frontend-test-dag.generate-cases.prompt.md +5 -0
- package/docs/templates/frontend-test-dag.json +23 -0
- package/docs/templates/frontend-test-dag.retrieve-context.prompt.md +3 -0
- package/docs/templates/frontend-test-dag.retrospect.prompt.md +3 -0
- package/docs/templates/frontend-test-dag.review-cases.prompt.md +3 -0
- package/docs/templates/frontend-test-dag.review-execution.prompt.md +3 -0
- package/docs/templates/harness.schema.json +221 -221
- package/docs/templates/hybrid-dag.json +188 -188
- package/docs/templates/init-evolution-review.md +35 -35
- package/docs/templates/interactive-ui-round2-experiment.md +66 -66
- package/docs/templates/knowledge-graph-bootstrap-dag.json +118 -0
- package/docs/templates/knowledge-sync-dag.json +178 -0
- package/docs/templates/knowledge-sync-draft.schema.json +71 -0
- package/docs/templates/product-line/AGENTS.md +8 -8
- package/docs/templates/product-line/README.md +9 -9
- package/docs/templates/product-line/acceptance.yaml +14 -14
- package/docs/templates/product-line/closeout.yaml +9 -9
- package/docs/templates/product-line/design.md +13 -13
- package/docs/templates/product-line/links.md +10 -10
- package/docs/templates/product-line/requirement.md +17 -17
- package/docs/templates/product-line/task-graph.yaml +15 -15
- package/docs/templates/product-line/task.yaml +64 -64
- package/docs/templates/product-line/test-plan.md +7 -7
- package/docs/templates/production-readiness-checklist.md +57 -57
- package/docs/templates/progress-log.md +17 -17
- package/docs/templates/project-start-checklist.md +9 -9
- package/docs/templates/qa-report.md +48 -48
- package/docs/templates/sprint-contract.md +29 -29
- package/docs/templates/worker-dogfood-evidence.md +80 -80
- package/docs/templates/worker-dogfood-setup.md +68 -68
- package/docs/verification-matrix.md +70 -66
- package/examples/decision-gate-agent-dag.json +177 -177
- package/examples/example-dag.json +46 -46
- package/examples/hybrid-loop-agent-dag.json +189 -189
- package/harness.json +66 -66
- package/package.json +88 -46
- package/scripts/check-product-line-docs.sh +29 -29
- package/scripts/check-task-pool-root.sh +32 -32
- package/scripts/kb-bootstrap-init-skeleton.sh +240 -0
- package/scripts/kb-graph-incremental-prepare.mjs +386 -0
- package/scripts/kb-graph-incremental-prepare.sh +5 -0
- package/scripts/kb-graph-materialize.mjs +105 -0
- package/scripts/kb-graph-materialize.sh +4 -0
- package/scripts/kb-graph-promote.mjs +164 -0
- package/scripts/kb-graph-promote.sh +4 -0
- package/scripts/kb-query.mjs +554 -0
- package/scripts/kb-query.sh +5 -0
- package/skills/agent-worker/SKILL.md +39 -37
- package/skills/agent-worker/references/agent-worker-operator.md +60 -43
- package/skills/ai-engineering-context/SKILL.md +48 -48
- package/skills/analyze-product-dependencies/SKILL.md +67 -0
- package/skills/analyze-product-dependencies/agents/openai.yaml +4 -0
- package/skills/analyze-product-dependencies/references/api-documentation-schema.md +30 -0
- package/skills/analyze-product-dependencies/references/dependency-analysis-schema.md +28 -0
- package/skills/analyze-product-dependencies/references/example.md +76 -0
- package/skills/analyze-product-dependencies/references/forward-test-cases.md +35 -0
- package/skills/analyze-product-dependencies/references/input-contract.md +11 -0
- package/skills/analyze-product-dependencies/references/scouting-rules.md +61 -0
- package/skills/analyze-product-dependencies/scripts/test-validators.mjs +267 -0
- package/skills/analyze-product-dependencies/scripts/validate-api-documentation.mjs +101 -0
- package/skills/analyze-product-dependencies/scripts/validate-dependency-analysis.mjs +142 -0
- package/skills/analyze-product-dependencies/scripts/validate-product-requirement-input.mjs +76 -0
- package/skills/analyze-product-dependencies/scripts/validation-helpers.mjs +146 -0
- package/skills/analyze-product-requirements/SKILL.md +90 -0
- package/skills/analyze-product-requirements/agents/openai.yaml +4 -0
- package/skills/analyze-product-requirements/references/acceptance-criteria.md +91 -0
- package/skills/analyze-product-requirements/references/clarification-and-knowledge.md +56 -0
- package/skills/analyze-product-requirements/references/example.md +86 -0
- package/skills/analyze-product-requirements/references/forward-test-cases.md +66 -0
- package/skills/analyze-product-requirements/references/product-analysis-schema.md +32 -0
- package/skills/analyze-product-requirements/references/product-requirement-schema.md +33 -0
- package/skills/analyze-product-requirements/references/requirement-clarification-schema.md +35 -0
- package/skills/analyze-product-requirements/scripts/test-validators.mjs +193 -0
- package/skills/analyze-product-requirements/scripts/validate-product-analysis.mjs +69 -0
- package/skills/analyze-product-requirements/scripts/validate-product-requirement.mjs +97 -0
- package/skills/analyze-product-requirements/scripts/validate-requirement-clarification.mjs +98 -0
- package/skills/analyze-product-requirements/scripts/validation-helpers.mjs +156 -0
- package/skills/code-review-core/SKILL.md +20 -20
- package/skills/codebase-scout/SKILL.md +19 -19
- package/skills/frontend-design-review/SKILL.md +66 -59
- package/skills/frontend-design-review/references/review-checklist.md +58 -37
- package/skills/frontend-implementation/SKILL.md +47 -51
- package/skills/frontend-implementation/references/code-standards.md +32 -34
- package/skills/frontend-implementation/references/design-spec.md +46 -46
- package/skills/frontend-implementation/references/node-contracts.md +76 -32
- package/skills/frontend-review/SKILL.md +59 -53
- package/skills/frontend-review/references/review-findings.md +47 -42
- package/skills/frontend-verification/SKILL.md +53 -40
- package/skills/frontend-verification/references/verification-checklist.md +68 -56
- package/skills/grill-me/SKILL.md +10 -10
- package/skills/grill-with-docs/SKILL.md +88 -88
- package/skills/grill-with-docs/adr-format.md +47 -47
- package/skills/grill-with-docs/context-format.md +60 -60
- package/skills/init-capability-evolution/SKILL.md +70 -70
- package/skills/loop-agent/SKILL.md +151 -151
- package/skills/loop-agent/references/README.md +67 -67
- package/skills/loop-agent/references/command-reference.md +505 -452
- package/skills/loop-agent/references/docs-converge.md +126 -126
- package/skills/loop-agent/references/harness-policy.md +263 -263
- package/skills/loop-agent/references/hybrid-dag.md +238 -233
- package/skills/loop-agent/references/learned/README.md +21 -21
- package/skills/loop-agent/references/long-running-loop.md +57 -57
- package/skills/loop-agent/references/model-routing.md +36 -36
- package/skills/loop-agent/references/multi-worktree.md +54 -54
- package/skills/loop-agent/references/one-shot-runs.md +85 -85
- package/skills/loop-agent/references/orchestrator-and-interventions.md +169 -169
- package/skills/loop-agent/references/pi-prompt.md +23 -23
- package/skills/loop-agent/references/pi-subagent-assisted-mode.md +84 -84
- package/skills/loop-agent/references/post-implementation-and-patterns.md +44 -44
- package/skills/loop-agent/references/task-workflow.md +89 -89
- package/skills/loop-agent/references/verification-and-failure-handling.md +139 -139
- package/skills/playwright-cli/SKILL.md +420 -0
- package/skills/playwright-cli/references/element-attributes.md +23 -0
- package/skills/playwright-cli/references/playwright-tests.md +39 -0
- package/skills/playwright-cli/references/request-mocking.md +87 -0
- package/skills/playwright-cli/references/running-code.md +241 -0
- package/skills/playwright-cli/references/session-management.md +225 -0
- package/skills/playwright-cli/references/storage-state.md +275 -0
- package/skills/playwright-cli/references/test-generation.md +433 -0
- package/skills/playwright-cli/references/tracing.md +139 -0
- package/skills/playwright-cli/references/video-recording.md +143 -0
- package/skills/playwright-cli-case-generator/SKILL.md +74 -0
- package/skills/requesting-code-review/SKILL.md +101 -101
- package/skills/requesting-code-review/code-reviewer.md +168 -168
- package/skills/systematic-debugging/CREATION-LOG.md +119 -119
- package/skills/systematic-debugging/SKILL.md +296 -296
- package/skills/systematic-debugging/condition-based-waiting-example.ts +158 -158
- package/skills/systematic-debugging/condition-based-waiting.md +115 -115
- package/skills/systematic-debugging/defense-in-depth.md +122 -122
- package/skills/systematic-debugging/find-polluter.sh +63 -63
- package/skills/systematic-debugging/root-cause-tracing.md +169 -169
- package/skills/systematic-debugging/test-academic.md +14 -14
- package/skills/systematic-debugging/test-pressure-1.md +58 -58
- package/skills/systematic-debugging/test-pressure-2.md +68 -68
- package/skills/systematic-debugging/test-pressure-3.md +69 -69
- package/skills/test-driven-development/SKILL.md +20 -20
- package/skills/using-git-worktrees/SKILL.md +215 -215
- package/skills/verification-before-completion/SKILL.md +154 -154
- package/skills/webapp-testing/SKILL.md +19 -19
|
@@ -1,41 +1,104 @@
|
|
|
1
1
|
import path from "node:path";
|
|
2
|
+
import { spawnSync } from "node:child_process";
|
|
3
|
+
import { fileURLToPath } from "node:url";
|
|
4
|
+
import { existsSync } from "node:fs";
|
|
2
5
|
import { curateKnowledgePatterns } from "../workflows/dag/knowledge-curator.js";
|
|
6
|
+
export const KNOWLEDGE_CLI_SUBCOMMANDS = [
|
|
7
|
+
"curate",
|
|
8
|
+
"query",
|
|
9
|
+
"graph-init",
|
|
10
|
+
"graph-materialize",
|
|
11
|
+
"graph-promote",
|
|
12
|
+
"graph-incremental-prepare",
|
|
13
|
+
];
|
|
14
|
+
const USAGE = "usage: knowledge <curate|query|graph-init|graph-materialize|graph-promote|graph-incremental-prepare> ...";
|
|
15
|
+
function packageRoot() {
|
|
16
|
+
// dist/commands/knowledge.js or src/commands/knowledge.ts → package/repo root
|
|
17
|
+
return path.resolve(path.dirname(fileURLToPath(import.meta.url)), "../..");
|
|
18
|
+
}
|
|
19
|
+
function resolveScript(repoRoot, fileName) {
|
|
20
|
+
const candidates = [
|
|
21
|
+
path.join(repoRoot, "scripts", fileName),
|
|
22
|
+
path.join(packageRoot(), "scripts", fileName),
|
|
23
|
+
];
|
|
24
|
+
for (const c of candidates) {
|
|
25
|
+
if (existsSync(c))
|
|
26
|
+
return c;
|
|
27
|
+
}
|
|
28
|
+
throw new Error(`knowledge script not found: ${fileName} (looked under repo scripts/ and package scripts/)`);
|
|
29
|
+
}
|
|
30
|
+
function runProcess(command, args, cwd) {
|
|
31
|
+
const result = spawnSync(command, args, {
|
|
32
|
+
cwd,
|
|
33
|
+
encoding: "utf8",
|
|
34
|
+
stdio: ["inherit", "pipe", "pipe"],
|
|
35
|
+
});
|
|
36
|
+
if (result.stdout)
|
|
37
|
+
process.stdout.write(result.stdout);
|
|
38
|
+
if (result.stderr)
|
|
39
|
+
process.stderr.write(result.stderr);
|
|
40
|
+
if (result.error)
|
|
41
|
+
throw result.error;
|
|
42
|
+
if (result.status !== 0 && result.status !== null) {
|
|
43
|
+
process.exitCode = result.status;
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
function runNodeScript(repoRoot, scriptFile, forwardArgs) {
|
|
47
|
+
const scriptPath = resolveScript(repoRoot, scriptFile);
|
|
48
|
+
runProcess(process.execPath, [scriptPath, ...forwardArgs], repoRoot);
|
|
49
|
+
}
|
|
50
|
+
function runBashScript(repoRoot, scriptFile, forwardArgs) {
|
|
51
|
+
const scriptPath = resolveScript(repoRoot, scriptFile);
|
|
52
|
+
runProcess("bash", [scriptPath, ...forwardArgs], repoRoot);
|
|
53
|
+
}
|
|
54
|
+
function ensureRootFlag(repoRoot, rest) {
|
|
55
|
+
if (rest.includes("--root"))
|
|
56
|
+
return rest;
|
|
57
|
+
return ["--root", repoRoot, ...rest];
|
|
58
|
+
}
|
|
3
59
|
export function parseKnowledgeArgs(args) {
|
|
4
60
|
const [command, ...rest] = args;
|
|
5
|
-
if (command
|
|
6
|
-
|
|
61
|
+
if (!command ||
|
|
62
|
+
!KNOWLEDGE_CLI_SUBCOMMANDS.includes(command)) {
|
|
63
|
+
throw new Error(USAGE);
|
|
7
64
|
}
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
json
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
markdown
|
|
19
|
-
|
|
65
|
+
if (command === "curate") {
|
|
66
|
+
let outputPath;
|
|
67
|
+
let json = false;
|
|
68
|
+
let markdown = false;
|
|
69
|
+
for (let i = 0; i < rest.length; i += 1) {
|
|
70
|
+
const arg = rest[i];
|
|
71
|
+
if (arg === "--json") {
|
|
72
|
+
json = true;
|
|
73
|
+
continue;
|
|
74
|
+
}
|
|
75
|
+
if (arg === "--markdown") {
|
|
76
|
+
markdown = true;
|
|
77
|
+
continue;
|
|
78
|
+
}
|
|
79
|
+
if (arg === "--output") {
|
|
80
|
+
outputPath = rest[++i];
|
|
81
|
+
if (!outputPath) {
|
|
82
|
+
throw new Error("knowledge curate --output requires a path");
|
|
83
|
+
}
|
|
84
|
+
continue;
|
|
85
|
+
}
|
|
86
|
+
if (arg.startsWith("--output=")) {
|
|
87
|
+
outputPath = arg.slice("--output=".length);
|
|
88
|
+
continue;
|
|
89
|
+
}
|
|
90
|
+
throw new Error(`unknown knowledge curate argument: ${arg}`);
|
|
20
91
|
}
|
|
21
|
-
if (
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
throw new Error("knowledge curate --output requires a path");
|
|
25
|
-
continue;
|
|
26
|
-
}
|
|
27
|
-
if (arg.startsWith("--output=")) {
|
|
28
|
-
outputPath = arg.slice("--output=".length);
|
|
29
|
-
continue;
|
|
30
|
-
}
|
|
31
|
-
throw new Error(`unknown knowledge curate argument: ${arg}`);
|
|
92
|
+
if (!json && !markdown)
|
|
93
|
+
json = true;
|
|
94
|
+
return { command: "curate", outputPath, json, markdown };
|
|
32
95
|
}
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
96
|
+
return {
|
|
97
|
+
command: command,
|
|
98
|
+
forward: rest,
|
|
99
|
+
};
|
|
36
100
|
}
|
|
37
|
-
|
|
38
|
-
const parsed = parseKnowledgeArgs(args);
|
|
101
|
+
async function runCurate(repoRoot, parsed) {
|
|
39
102
|
const result = await curateKnowledgePatterns({
|
|
40
103
|
repoRoot,
|
|
41
104
|
outputPath: parsed.outputPath,
|
|
@@ -50,7 +113,9 @@ export async function runKnowledge(repoRoot, args) {
|
|
|
50
113
|
console.log(JSON.stringify({
|
|
51
114
|
ok: result.ok,
|
|
52
115
|
patternsPath: path.relative(repoRoot, result.patternsPath),
|
|
53
|
-
outputPath: result.outputPath
|
|
116
|
+
outputPath: result.outputPath
|
|
117
|
+
? path.relative(repoRoot, result.outputPath)
|
|
118
|
+
: undefined,
|
|
54
119
|
patternCount: result.patternCount,
|
|
55
120
|
safetyFindings: result.safetyFindings,
|
|
56
121
|
message: result.message,
|
|
@@ -62,3 +127,36 @@ export async function runKnowledge(repoRoot, args) {
|
|
|
62
127
|
process.stdout.write(result.proposalMarkdown);
|
|
63
128
|
}
|
|
64
129
|
}
|
|
130
|
+
/**
|
|
131
|
+
* knowledge CLI:
|
|
132
|
+
* - curate: repair/learned guidance proposals (existing)
|
|
133
|
+
* - query / graph-*: thin wrappers over scripts/kb-*.{mjs,sh} for graph KB ops
|
|
134
|
+
*/
|
|
135
|
+
export async function runKnowledge(repoRoot, args) {
|
|
136
|
+
const parsed = parseKnowledgeArgs(args);
|
|
137
|
+
if (parsed.command === "curate") {
|
|
138
|
+
await runCurate(repoRoot, parsed);
|
|
139
|
+
return;
|
|
140
|
+
}
|
|
141
|
+
const forward = ensureRootFlag(repoRoot, parsed.forward);
|
|
142
|
+
switch (parsed.command) {
|
|
143
|
+
case "query":
|
|
144
|
+
runNodeScript(repoRoot, "kb-query.mjs", forward);
|
|
145
|
+
return;
|
|
146
|
+
case "graph-init":
|
|
147
|
+
// B1 skeleton is a bash script (no LLM)
|
|
148
|
+
runBashScript(repoRoot, "kb-bootstrap-init-skeleton.sh", forward);
|
|
149
|
+
return;
|
|
150
|
+
case "graph-materialize":
|
|
151
|
+
runNodeScript(repoRoot, "kb-graph-materialize.mjs", forward);
|
|
152
|
+
return;
|
|
153
|
+
case "graph-promote":
|
|
154
|
+
runNodeScript(repoRoot, "kb-graph-promote.mjs", forward);
|
|
155
|
+
return;
|
|
156
|
+
case "graph-incremental-prepare":
|
|
157
|
+
runNodeScript(repoRoot, "kb-graph-incremental-prepare.mjs", forward);
|
|
158
|
+
return;
|
|
159
|
+
default:
|
|
160
|
+
throw new Error(USAGE);
|
|
161
|
+
}
|
|
162
|
+
}
|
|
@@ -37,17 +37,17 @@ export function parseLoopBenchmarkArgs(args) {
|
|
|
37
37
|
return { json, markdown, outputPath };
|
|
38
38
|
}
|
|
39
39
|
export function printLoopBenchmarkUsage() {
|
|
40
|
-
console.log(`usage: loop-benchmark [options]
|
|
41
|
-
|
|
42
|
-
Deterministic loop-agent loop benchmark baseline (no live Pi/Cursor calls).
|
|
43
|
-
|
|
44
|
-
Options:
|
|
45
|
-
--json Emit JSON (default when no format flag is set)
|
|
46
|
-
--markdown Emit Markdown report
|
|
47
|
-
--output <path> Write Markdown report to a repo-relative or absolute path
|
|
48
|
-
-h, --help Show this help
|
|
49
|
-
|
|
50
|
-
Control groups: single-repair, 3-pass-convergence, 3-pass-convergence+quota.
|
|
40
|
+
console.log(`usage: loop-benchmark [options]
|
|
41
|
+
|
|
42
|
+
Deterministic loop-agent loop benchmark baseline (no live Pi/Cursor calls).
|
|
43
|
+
|
|
44
|
+
Options:
|
|
45
|
+
--json Emit JSON (default when no format flag is set)
|
|
46
|
+
--markdown Emit Markdown report
|
|
47
|
+
--output <path> Write Markdown report to a repo-relative or absolute path
|
|
48
|
+
-h, --help Show this help
|
|
49
|
+
|
|
50
|
+
Control groups: single-repair, 3-pass-convergence, 3-pass-convergence+quota.
|
|
51
51
|
Recommendation never changes convergence.enabled default.`);
|
|
52
52
|
}
|
|
53
53
|
export async function runLoopBenchmark(repoRoot, rawArgs) {
|
|
@@ -106,22 +106,22 @@ export function parsePiReuseBenchmarkArgs(args) {
|
|
|
106
106
|
};
|
|
107
107
|
}
|
|
108
108
|
export function printPiReuseBenchmarkUsage() {
|
|
109
|
-
console.log(`usage: pi-reuse-benchmark [options]
|
|
110
|
-
|
|
111
|
-
Deterministic Pi runtime reuse benchmark/decision summary (no live Pi calls).
|
|
112
|
-
|
|
113
|
-
Options:
|
|
114
|
-
--report <path> Benchmark report markdown (approval status)
|
|
115
|
-
--approval <path> Explicit approval JSON artifact
|
|
116
|
-
--off-executor <path> Baseline executor.jsonl (reuse off)
|
|
117
|
-
--on-executor <path> Treatment executor.jsonl (reuse on)
|
|
118
|
-
--off-task <task-id> Resolve baseline from .harness/tasks/<id>/logs/executor.jsonl
|
|
119
|
-
--on-task <task-id> Resolve treatment from .harness/tasks/<id>/logs/executor.jsonl
|
|
120
|
-
--json Emit JSON (default when no format flag is set)
|
|
121
|
-
--markdown Emit Markdown summary
|
|
122
|
-
-h, --help Show this help
|
|
123
|
-
|
|
124
|
-
Recommendations: defer | maintain-opt-in | eligible-for-human-review
|
|
109
|
+
console.log(`usage: pi-reuse-benchmark [options]
|
|
110
|
+
|
|
111
|
+
Deterministic Pi runtime reuse benchmark/decision summary (no live Pi calls).
|
|
112
|
+
|
|
113
|
+
Options:
|
|
114
|
+
--report <path> Benchmark report markdown (approval status)
|
|
115
|
+
--approval <path> Explicit approval JSON artifact
|
|
116
|
+
--off-executor <path> Baseline executor.jsonl (reuse off)
|
|
117
|
+
--on-executor <path> Treatment executor.jsonl (reuse on)
|
|
118
|
+
--off-task <task-id> Resolve baseline from .harness/tasks/<id>/logs/executor.jsonl
|
|
119
|
+
--on-task <task-id> Resolve treatment from .harness/tasks/<id>/logs/executor.jsonl
|
|
120
|
+
--json Emit JSON (default when no format flag is set)
|
|
121
|
+
--markdown Emit Markdown summary
|
|
122
|
+
-h, --help Show this help
|
|
123
|
+
|
|
124
|
+
Recommendations: defer | maintain-opt-in | eligible-for-human-review
|
|
125
125
|
Never changes CODE_AGENT_PI_REUSE_RUNTIME default (off).`);
|
|
126
126
|
}
|
|
127
127
|
function resolveRepoRelative(repoRoot, filePath) {
|
|
@@ -51,31 +51,38 @@ export function setPiSdkImportOverrideForTests(fn) {
|
|
|
51
51
|
export function setPiSdkModuleOverrideForTests(fn) {
|
|
52
52
|
sdkModuleOverrideForTests = fn;
|
|
53
53
|
}
|
|
54
|
-
/** Check whether the Pi SDK optional dependency
|
|
54
|
+
/** Check whether the Pi SDK optional dependency satisfies the 0.80.10 runtime contract. */
|
|
55
55
|
export async function checkPiSdkAvailability(_repoRoot) {
|
|
56
56
|
if (sdkSessionFactoryOverride) {
|
|
57
57
|
return { ok: true, detail: 'pi SDK session factory override active' };
|
|
58
58
|
}
|
|
59
59
|
try {
|
|
60
|
-
|
|
61
|
-
await sdkImportOverrideForTests()
|
|
60
|
+
const imported = sdkImportOverrideForTests
|
|
61
|
+
? await sdkImportOverrideForTests()
|
|
62
|
+
: await loadPiSdkModule();
|
|
63
|
+
const sdk = imported;
|
|
64
|
+
const ModelRuntime = sdk.ModelRuntime;
|
|
65
|
+
if (typeof sdk.createAgentSession !== 'function'
|
|
66
|
+
|| typeof sdk.getAgentDir !== 'function'
|
|
67
|
+
|| typeof ModelRuntime?.create !== 'function') {
|
|
68
|
+
return {
|
|
69
|
+
ok: false,
|
|
70
|
+
detail: 'pi SDK incompatible: requires createAgentSession, getAgentDir, and ModelRuntime.create (0.80.10 contract)',
|
|
71
|
+
};
|
|
62
72
|
}
|
|
63
|
-
|
|
64
|
-
await import('@earendil-works/pi-coding-agent');
|
|
65
|
-
}
|
|
66
|
-
return { ok: true, detail: 'pi SDK available' };
|
|
73
|
+
return { ok: true, detail: 'pi SDK 0.80.10 contract available' };
|
|
67
74
|
}
|
|
68
75
|
catch (error) {
|
|
69
76
|
const message = error instanceof Error ? error.message : String(error);
|
|
70
77
|
return { ok: false, detail: `pi SDK not available: ${message}` };
|
|
71
78
|
}
|
|
72
79
|
}
|
|
73
|
-
async function
|
|
74
|
-
const
|
|
75
|
-
if (
|
|
76
|
-
return
|
|
80
|
+
async function resolveModel(modelRuntime, provider, modelId) {
|
|
81
|
+
const fromRuntime = modelRuntime.getModel(provider, modelId);
|
|
82
|
+
if (fromRuntime)
|
|
83
|
+
return fromRuntime;
|
|
77
84
|
try {
|
|
78
|
-
const piAi = await import('@earendil-works/pi-ai');
|
|
85
|
+
const piAi = await import('@earendil-works/pi-ai/compat');
|
|
79
86
|
return piAi.getModel?.(provider, modelId);
|
|
80
87
|
}
|
|
81
88
|
catch {
|
|
@@ -93,12 +100,16 @@ async function getOrCreateSharedResources(state) {
|
|
|
93
100
|
return state.resources;
|
|
94
101
|
const sdk = await loadPiSdkModule();
|
|
95
102
|
const getAgentDir = sdk.getAgentDir;
|
|
96
|
-
const
|
|
97
|
-
|
|
103
|
+
const ModelRuntime = sdk.ModelRuntime;
|
|
104
|
+
if (typeof ModelRuntime?.create !== 'function') {
|
|
105
|
+
throw new Error('incompatible pi SDK: ModelRuntime.create is unavailable');
|
|
106
|
+
}
|
|
98
107
|
const agentDir = getAgentDir();
|
|
99
|
-
const
|
|
100
|
-
|
|
101
|
-
|
|
108
|
+
const modelRuntime = await ModelRuntime.create({
|
|
109
|
+
authPath: path.join(agentDir, 'auth.json'),
|
|
110
|
+
modelsPath: path.join(agentDir, 'models.json'),
|
|
111
|
+
});
|
|
112
|
+
state.resources = { agentDir, modelRuntime };
|
|
102
113
|
return state.resources;
|
|
103
114
|
}
|
|
104
115
|
async function createSdkSession(sdk, input, shared) {
|
|
@@ -106,13 +117,17 @@ async function createSdkSession(sdk, input, shared) {
|
|
|
106
117
|
const SessionManager = sdk.SessionManager;
|
|
107
118
|
const DefaultResourceLoader = sdk.DefaultResourceLoader;
|
|
108
119
|
const getAgentDir = sdk.getAgentDir;
|
|
109
|
-
const
|
|
110
|
-
const ModelRegistry = sdk.ModelRegistry;
|
|
120
|
+
const ModelRuntime = sdk.ModelRuntime;
|
|
111
121
|
const agentDir = shared?.agentDir ?? getAgentDir();
|
|
112
|
-
|
|
113
|
-
|
|
122
|
+
if (!shared && typeof ModelRuntime?.create !== 'function') {
|
|
123
|
+
throw new Error('incompatible pi SDK: ModelRuntime.create is unavailable');
|
|
124
|
+
}
|
|
125
|
+
const modelRuntime = shared?.modelRuntime ?? await ModelRuntime.create({
|
|
126
|
+
authPath: path.join(agentDir, 'auth.json'),
|
|
127
|
+
modelsPath: path.join(agentDir, 'models.json'),
|
|
128
|
+
});
|
|
114
129
|
const model = input.provider && input.model
|
|
115
|
-
? await
|
|
130
|
+
? await resolveModel(modelRuntime, input.provider, input.model)
|
|
116
131
|
: undefined;
|
|
117
132
|
const loader = new DefaultResourceLoader({
|
|
118
133
|
cwd: input.cwd,
|
|
@@ -127,8 +142,7 @@ async function createSdkSession(sdk, input, shared) {
|
|
|
127
142
|
sessionManager: SessionManager.inMemory(input.cwd),
|
|
128
143
|
resourceLoader: loader,
|
|
129
144
|
tools: input.toolNames,
|
|
130
|
-
|
|
131
|
-
modelRegistry,
|
|
145
|
+
modelRuntime,
|
|
132
146
|
...(model ? { model } : {}),
|
|
133
147
|
...(input.thinking ? { thinkingLevel: input.thinking } : {}),
|
|
134
148
|
});
|
|
@@ -3,7 +3,8 @@ import { appendFileSync, existsSync, mkdirSync, writeFileSync } from "node:fs";
|
|
|
3
3
|
import path from "node:path";
|
|
4
4
|
import { writeDagNodeTextArtifact } from "../infrastructure/harness/artifact-store.js";
|
|
5
5
|
import { truncateOutput } from "../shared/output-truncation.js";
|
|
6
|
-
import { expandShellPreset, buildVerdictGateShellCommand } from "./shell-presets.js";
|
|
6
|
+
import { buildRequirementCoverageGateShellCommand, expandShellPreset, buildVerdictGateShellCommand } from "./shell-presets.js";
|
|
7
|
+
import { materializeBackendTestAnalysisContract } from "../workflows/dag/backend-test-analysis-contract.js";
|
|
7
8
|
import { pathsChangedDuringRun, snapshotGitStatusPorcelain, validateShellWriteGuard, } from "./shell-write-guard.js";
|
|
8
9
|
import { buildShellProcessEnv } from "./shell-verification.js";
|
|
9
10
|
const DEFAULT_SHELL_TIMEOUT_MS = 300_000;
|
|
@@ -44,7 +45,10 @@ export function resolveShellCommands(shell) {
|
|
|
44
45
|
const fromVerdictGate = shell.verdictGate
|
|
45
46
|
? [buildVerdictGateShellCommand(shell.verdictGate)]
|
|
46
47
|
: [];
|
|
47
|
-
|
|
48
|
+
const fromRequirementCoverageGate = shell.requirementCoverageGate
|
|
49
|
+
? [buildRequirementCoverageGateShellCommand(shell.requirementCoverageGate)]
|
|
50
|
+
: [];
|
|
51
|
+
return [...fromPreset, ...explicit, ...fromVerdictGate, ...fromRequirementCoverageGate];
|
|
48
52
|
}
|
|
49
53
|
async function readGitStatusPorcelain(cwd) {
|
|
50
54
|
return new Promise((resolve, reject) => {
|
|
@@ -279,6 +283,34 @@ async function runShellWriteGuard(input) {
|
|
|
279
283
|
}
|
|
280
284
|
export async function executeDagShellNode(input, meta) {
|
|
281
285
|
const shell = input.task.shell;
|
|
286
|
+
if (shell?.jsonArtifactGate) {
|
|
287
|
+
const started = Date.now();
|
|
288
|
+
try {
|
|
289
|
+
const artifact = await materializeBackendTestAnalysisContract({
|
|
290
|
+
runDir: meta.runDir,
|
|
291
|
+
fromNodeId: shell.jsonArtifactGate.fromNodeId,
|
|
292
|
+
artifactName: shell.jsonArtifactGate.artifactName,
|
|
293
|
+
outputDir: shell.jsonArtifactGate.outputDir,
|
|
294
|
+
sourceBinding: meta.spec.sourceBinding,
|
|
295
|
+
});
|
|
296
|
+
return {
|
|
297
|
+
ok: true,
|
|
298
|
+
stdout: `Structured artifact: ${artifact.path}\nSchema: ${artifact.schemaId}\nSHA-256: ${artifact.sha256}`,
|
|
299
|
+
stderr: "",
|
|
300
|
+
failureCategory: "success",
|
|
301
|
+
durationMs: Date.now() - started,
|
|
302
|
+
};
|
|
303
|
+
}
|
|
304
|
+
catch (error) {
|
|
305
|
+
return {
|
|
306
|
+
ok: false,
|
|
307
|
+
stdout: "",
|
|
308
|
+
stderr: error instanceof Error ? error.message : String(error),
|
|
309
|
+
failureCategory: "invalid-output",
|
|
310
|
+
durationMs: Date.now() - started,
|
|
311
|
+
};
|
|
312
|
+
}
|
|
313
|
+
}
|
|
282
314
|
const commands = shell ? resolveShellCommands(shell) : [];
|
|
283
315
|
if (!shell || commands.length === 0) {
|
|
284
316
|
throw new Error(`shell task ${input.task.id} requires shell.preset, shell.verdictGate, and/or non-empty shell.commands`);
|
|
@@ -45,3 +45,23 @@ export function buildVerdictGateShellCommand(gate) {
|
|
|
45
45
|
.join("|");
|
|
46
46
|
return `${preamble}; case "\${FIRST}" in ${acceptPattern}) exit 0 ;; *) echo "${blockedMessage}" >&2; exit 1 ;; esac`;
|
|
47
47
|
}
|
|
48
|
+
/** Build a deterministic current-run gate over exact REQ-/BR-/AC- ids in node facts. */
|
|
49
|
+
export function buildRequirementCoverageGateShellCommand(gate) {
|
|
50
|
+
const label = gate.label ?? "requirement coverage";
|
|
51
|
+
const requiredJson = escapeShellSingleQuoted(JSON.stringify(gate.requiredIds.map((id) => id.toUpperCase())));
|
|
52
|
+
const fileArgs = gate.fromNodeIds.map((nodeId) => `"\${HARNESS_DAG_RUN_DIR}/${nodeId}.json"`).join(" ");
|
|
53
|
+
const program = [
|
|
54
|
+
'const fs=require("fs");',
|
|
55
|
+
"const required=JSON.parse(process.argv[1]);",
|
|
56
|
+
"let text='';",
|
|
57
|
+
"for(const file of process.argv.slice(2)){",
|
|
58
|
+
" if(!fs.existsSync(file)){console.error('missing requirement coverage node output: '+file);process.exit(1);}",
|
|
59
|
+
" const raw=JSON.parse(fs.readFileSync(file,'utf8')); text+='\\n'+String(raw.assistantText ?? raw.stdout ?? '');",
|
|
60
|
+
"}",
|
|
61
|
+
"const found=new Set((text.toUpperCase().match(/\\b(?:REQ|BR|AC)-[A-Z0-9]+(?:-[A-Z0-9]+)*\\b/g)||[]));",
|
|
62
|
+
"const missing=required.filter((id)=>!found.has(id));",
|
|
63
|
+
`if(missing.length){console.error(${JSON.stringify(`${label} gate blocked: missing `)}+missing.join(', '));process.exit(1);}`,
|
|
64
|
+
`console.log(${JSON.stringify(`${label} gate passed: `)}+required.join(', '));`,
|
|
65
|
+
].join("");
|
|
66
|
+
return `test -n "\${HARNESS_DAG_RUN_DIR:-}" || { echo "missing HARNESS_DAG_RUN_DIR for ${label} gate" >&2; exit 1; }; node -e '${escapeShellSingleQuoted(program)}' '${requiredJson}' ${fileArgs}`;
|
|
67
|
+
}
|
|
@@ -20,6 +20,13 @@ const VERIFY_ENV_ALLOWLIST = [
|
|
|
20
20
|
"COREPACK_ENABLE",
|
|
21
21
|
"COREPACK_ENABLE_AUTO_PIN",
|
|
22
22
|
"COREPACK_HOME",
|
|
23
|
+
// Windows DLL 加载必需
|
|
24
|
+
"SystemRoot",
|
|
25
|
+
"SYSTEMROOT",
|
|
26
|
+
"windir",
|
|
27
|
+
"WINDIR",
|
|
28
|
+
"ComSpec",
|
|
29
|
+
"PATHEXT",
|
|
23
30
|
];
|
|
24
31
|
export function buildVerifyProcessEnv(mode = "clean") {
|
|
25
32
|
if (mode === "inherit") {
|
|
@@ -43,6 +43,10 @@ export const workflowPolicyDagTemplateSchema = z.enum([
|
|
|
43
43
|
"review-gated-dag",
|
|
44
44
|
"supervised-implementation",
|
|
45
45
|
"frontend-implementation",
|
|
46
|
+
"frontend-test-dag",
|
|
47
|
+
"backend-test-dag",
|
|
48
|
+
"knowledge-sync-dag",
|
|
49
|
+
"knowledge-graph-bootstrap-dag",
|
|
46
50
|
]);
|
|
47
51
|
export const workflowPolicyOutputLanguageSchema = z.enum(["zh-CN", "en"]);
|
|
48
52
|
export const workflowPolicySchema = z
|