@mmerterden/multi-agent-pipeline 17.6.0 → 19.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +310 -0
- package/README.md +76 -18
- package/README.tr.md +55 -16
- package/docs/adr/0002-instruction-driven-flag.md +1 -0
- package/docs/adr/0005-lazy-phase-docs.md +11 -1
- package/docs/adr/0008-installer-modularization-and-secret-leak-defense.md +1 -0
- package/docs/adr/0010-own-code-graph.md +1 -0
- package/docs/adr/0011-dormant-ci.md +25 -1
- package/docs/adr/0014-six-phase-consolidation.md +134 -0
- package/docs/adr/README.md +2 -1
- package/docs/architecture.md +37 -38
- package/docs/best-practices.md +1 -1
- package/docs/ecosystem.md +37 -26
- package/docs/engineering.md +1 -1
- package/docs/facts.json +45 -0
- package/docs/features.md +54 -53
- package/docs/performance.md +5 -5
- package/docs/recovery-guide.md +9 -9
- package/docs/server-readiness.md +188 -0
- package/docs/token-budget-history.md +3 -1
- package/index.js +18 -3
- package/install/_codex-agents.mjs +1 -1
- package/install/_common.mjs +42 -17
- package/install/_dev-only-files.mjs +8 -0
- package/install/_unattended-profile.mjs +113 -0
- package/install/index.mjs +48 -0
- package/install/templates/claude-hooks.json +1 -1
- package/install/templates/codex-instructions.md +1 -1
- package/install/templates/copilot-instructions.md +28 -28
- package/manifest.json +1065 -0
- package/package.json +6 -3
- package/pipeline/agents/dev-critic.md +3 -3
- package/pipeline/commands/figma-to-swiftui.md +1 -1
- package/pipeline/commands/multi-agent/SKILL.md +8 -8
- package/pipeline/commands/multi-agent/analysis/SKILL.md +9 -9
- package/pipeline/commands/multi-agent/autopilot/SKILL.md +7 -7
- package/pipeline/commands/multi-agent/channels/SKILL.md +15 -15
- package/pipeline/commands/multi-agent/diff-explain/SKILL.md +6 -6
- package/pipeline/commands/multi-agent/garbage-collect/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/graph/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/help/SKILL.md +62 -62
- package/pipeline/commands/multi-agent/language/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/local/SKILL.md +11 -11
- package/pipeline/commands/multi-agent/local-autopilot/SKILL.md +13 -13
- package/pipeline/commands/multi-agent/log/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/manual-test/SKILL.md +9 -9
- package/pipeline/commands/multi-agent/model/SKILL.md +69 -0
- package/pipeline/commands/multi-agent/refactor/SKILL.md +3 -3
- package/pipeline/commands/multi-agent/resume/SKILL.md +4 -4
- package/pipeline/commands/multi-agent/resume-local/SKILL.md +19 -17
- package/pipeline/commands/multi-agent/review/SKILL.md +1 -1
- package/pipeline/commands/multi-agent/route-off/SKILL.md +36 -0
- package/pipeline/commands/multi-agent/route-on/SKILL.md +74 -0
- package/pipeline/commands/multi-agent/route-status/SKILL.md +56 -0
- package/pipeline/commands/multi-agent/setup/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/status/SKILL.md +54 -23
- package/pipeline/commands/multi-agent/steer/SKILL.md +2 -2
- package/pipeline/commands/multi-agent/sync/SKILL.md +12 -13
- package/pipeline/commands/multi-agent/test/SKILL.md +1 -1
- package/pipeline/lib/_jira-auth.sh +8 -0
- package/pipeline/lib/analysis-jira-write.sh +32 -0
- package/pipeline/lib/ask-choice.sh +13 -2
- package/pipeline/lib/autopilot-state.sh +8 -0
- package/pipeline/lib/credential-inventory.sh +1 -1
- package/pipeline/lib/fatal.mjs +129 -0
- package/pipeline/lib/fetch-fortify.sh +1 -1
- package/pipeline/lib/figma-mcp-refresh.sh +18 -0
- package/pipeline/lib/figma-screenshot.sh +18 -0
- package/pipeline/lib/invoked-directly.mjs +43 -0
- package/pipeline/lib/jira-publish.sh +42 -0
- package/pipeline/lib/md2confluence-v3.py +47 -0
- package/pipeline/lib/model-rung.sh +142 -0
- package/pipeline/lib/outbound-gate.mjs +175 -0
- package/pipeline/lib/phase-schema.mjs +88 -0
- package/pipeline/lib/plan-todos.sh +32 -11
- package/pipeline/lib/post-pr-review.sh +77 -8
- package/pipeline/lib/repo-hygiene.sh +8 -3
- package/pipeline/lib/require-jq.sh +40 -0
- package/pipeline/lib/route-state.sh +161 -0
- package/pipeline/lib/run-paths.sh +335 -0
- package/pipeline/multi-agent-refs/_account-picker.md +1 -1
- package/pipeline/multi-agent-refs/_dev-context.md +1 -1
- package/pipeline/multi-agent-refs/_input-parser.md +1 -1
- package/pipeline/multi-agent-refs/analysis/evidence.md +0 -9
- package/pipeline/multi-agent-refs/analysis/intake.md +1 -1
- package/pipeline/multi-agent-refs/analysis/locked.md +21 -22
- package/pipeline/multi-agent-refs/analysis/render.md +1 -1
- package/pipeline/multi-agent-refs/analysis/synthesis.md +12 -6
- package/pipeline/multi-agent-refs/android-guide.md +1 -1
- package/pipeline/multi-agent-refs/audit-guide.md +13 -13
- package/pipeline/multi-agent-refs/channels/issue-comment.md +2 -2
- package/pipeline/multi-agent-refs/channels/jira.md +3 -3
- package/pipeline/multi-agent-refs/channels/pr.md +4 -4
- package/pipeline/multi-agent-refs/channels/wiki.md +1 -1
- package/pipeline/multi-agent-refs/component-dispatch.md +3 -3
- package/pipeline/multi-agent-refs/cross-cli-contract.md +31 -6
- package/pipeline/multi-agent-refs/features/autopilot-circuit-breaker.md +74 -4
- package/pipeline/multi-agent-refs/features/code-graph.md +5 -5
- package/pipeline/multi-agent-refs/features/cost-analysis.md +93 -0
- package/pipeline/multi-agent-refs/features/design-conformance.md +1 -1
- package/pipeline/multi-agent-refs/features/dev-critic.md +3 -3
- package/pipeline/multi-agent-refs/features/doctor.md +47 -2
- package/pipeline/multi-agent-refs/features/external-context-injection.md +3 -3
- package/pipeline/multi-agent-refs/features/maturity-followup.md +3 -3
- package/pipeline/multi-agent-refs/features/model-fallback.md +5 -5
- package/pipeline/multi-agent-refs/features/plan-todos.md +1 -1
- package/pipeline/multi-agent-refs/features/repo-map.md +1 -1
- package/pipeline/multi-agent-refs/features/review-delta.md +3 -3
- package/pipeline/multi-agent-refs/features/review-multi-repo.md +1 -1
- package/pipeline/multi-agent-refs/features/scope-check.md +4 -4
- package/pipeline/multi-agent-refs/features/skill-conformance.md +2 -2
- package/pipeline/multi-agent-refs/features/stack-skill-routing.md +1 -1
- package/pipeline/multi-agent-refs/features/verify-by-test.md +4 -4
- package/pipeline/multi-agent-refs/features/verify.md +83 -0
- package/pipeline/multi-agent-refs/features/visual-evidence.md +19 -19
- package/pipeline/multi-agent-refs/features/worktree-finalize.md +6 -6
- package/pipeline/multi-agent-refs/issue-jira-triad.md +10 -10
- package/pipeline/multi-agent-refs/knowledge.md +11 -11
- package/pipeline/multi-agent-refs/multi-repo-integration-build.md +13 -13
- package/pipeline/multi-agent-refs/payload-contracts.md +8 -8
- package/pipeline/multi-agent-refs/phases/log-format.md +10 -10
- package/pipeline/multi-agent-refs/phases/modes.md +30 -30
- package/pipeline/multi-agent-refs/phases/operations.md +21 -10
- package/pipeline/multi-agent-refs/phases/phase-0-init.md +25 -25
- package/pipeline/multi-agent-refs/phases/phase-1-plan.md +599 -0
- package/pipeline/multi-agent-refs/phases/{phase-3-dev.md → phase-2-dev.md} +129 -49
- package/pipeline/multi-agent-refs/phases/{phase-4-review.md → phase-3-review.md} +225 -107
- package/pipeline/multi-agent-refs/phases/{phase-6-commit.md → phase-4-commit.md} +23 -23
- package/pipeline/multi-agent-refs/phases/{phase-7-report.md → phase-5-report.md} +29 -29
- package/pipeline/multi-agent-refs/phases.md +44 -48
- package/pipeline/multi-agent-refs/picker-contract.md +1 -1
- package/pipeline/multi-agent-refs/progress-contract.md +6 -6
- package/pipeline/multi-agent-refs/readiness-review.md +1 -1
- package/pipeline/multi-agent-refs/rules.md +7 -7
- package/pipeline/multi-agent-refs/swiftui-guide.md +2 -2
- package/pipeline/multi-agent-refs/tracker-contract.md +31 -32
- package/pipeline/multi-agent-refs/unattended-contract.md +129 -0
- package/pipeline/multi-agent-refs/wiki-capture.md +14 -14
- package/pipeline/preferences-template.json +9 -1
- package/pipeline/rules/outside-the-pipeline.md +1 -1
- package/pipeline/schemas/agent-state.schema.json +50 -50
- package/pipeline/schemas/analysis-output.schema.json +2 -2
- package/pipeline/schemas/autopilot-config.schema.json +1 -1
- package/pipeline/schemas/code-graph.schema.json +1 -1
- package/pipeline/schemas/criteria-manifest.schema.json +1 -1
- package/pipeline/schemas/dev-critic-output.schema.json +1 -1
- package/pipeline/schemas/diff-risk.schema.json +1 -1
- package/pipeline/schemas/migrations/prefs-2.4.0-to-2.5.0.mjs +2 -2
- package/pipeline/schemas/migrations/prefs-2.6.0-to-2.7.0.mjs +31 -0
- package/pipeline/schemas/migrations/state-2.1.0-to-2.2.0.mjs +129 -0
- package/pipeline/schemas/phases.json +105 -0
- package/pipeline/schemas/plan-todos.schema.json +5 -5
- package/pipeline/schemas/planning-output.schema.json +1 -1
- package/pipeline/schemas/prefs.schema.json +100 -56
- package/pipeline/schemas/reviewer-output.schema.json +3 -3
- package/pipeline/schemas/route-config.schema.json +74 -0
- package/pipeline/schemas/scope-check.schema.json +1 -1
- package/pipeline/schemas/test-gap.schema.json +1 -1
- package/pipeline/schemas/token-budget.json +12 -18
- package/pipeline/schemas/triage-output.schema.json +6 -6
- package/pipeline/scripts/README.md +3 -3
- package/pipeline/scripts/_code-graph.mjs +2 -2
- package/pipeline/scripts/_run-paths.mjs +372 -0
- package/pipeline/scripts/_smoke-root.sh +1 -1
- package/pipeline/scripts/aggregate-metrics.mjs +65 -65
- package/pipeline/scripts/autopilot-arming.mjs +2 -1
- package/pipeline/scripts/autopilot-intake.mjs +2 -1
- package/pipeline/scripts/autopilot-runner.mjs +206 -2
- package/pipeline/scripts/build-references.mjs +2 -1
- package/pipeline/scripts/build-stack-plugins.mjs +10 -2
- package/pipeline/scripts/capture-evidence.sh +7 -2
- package/pipeline/scripts/capture-flush.sh +8 -8
- package/pipeline/scripts/capture-resume.sh +3 -3
- package/pipeline/scripts/classify-plan-safety.mjs +3 -2
- package/pipeline/scripts/cost-analyze.mjs +600 -0
- package/pipeline/scripts/cost-budget-check.mjs +4 -12
- package/pipeline/scripts/council-view.mjs +2 -1
- package/pipeline/scripts/crush-json.mjs +2 -1
- package/pipeline/scripts/diff-explain.mjs +7 -10
- package/pipeline/scripts/diff-risk-score.mjs +2 -1
- package/pipeline/scripts/doctor.mjs +140 -6
- package/pipeline/scripts/evidence-gate.mjs +9 -3
- package/pipeline/scripts/feedback-send.mjs +12 -2
- package/pipeline/scripts/gc-abandoned.sh +32 -16
- package/pipeline/scripts/gc-tmp.sh +1 -1
- package/pipeline/scripts/gc-worktrees.sh +12 -5
- package/pipeline/scripts/gen-facts.mjs +175 -0
- package/pipeline/scripts/gen-mode-dispatch.mjs +32 -37
- package/pipeline/scripts/gen-ref-toc.mjs +1 -1
- package/pipeline/scripts/github-ssh-setup.sh +64 -7
- package/pipeline/scripts/graph-mermaid.mjs +4 -2
- package/pipeline/scripts/graph-report.mjs +1 -1
- package/pipeline/scripts/jira-attach.sh +1 -1
- package/pipeline/scripts/keychain-save.sh +101 -30
- package/pipeline/scripts/learn-from-transcripts.mjs +3 -2
- package/pipeline/scripts/learning-curve.mjs +36 -31
- package/pipeline/scripts/log-metric.sh +17 -4
- package/pipeline/scripts/make-manifest.mjs +199 -0
- package/pipeline/scripts/memory-save.sh +1 -1
- package/pipeline/scripts/migrate-prefs.mjs +24 -6
- package/pipeline/scripts/migrate-state.mjs +94 -4
- package/pipeline/scripts/phase-banner.sh +26 -22
- package/pipeline/scripts/phase-tracker.sh +48 -10
- package/pipeline/scripts/plan-coverage-gate.mjs +8 -4
- package/pipeline/scripts/pre-commit-check.sh +7 -0
- package/pipeline/scripts/pre-push-check.sh +7 -0
- package/pipeline/scripts/purge.sh +23 -6
- package/pipeline/scripts/render-agent-log-cost.sh +10 -3
- package/pipeline/scripts/render-cost-summary.sh +9 -2
- package/pipeline/scripts/render-work-summary.sh +14 -7
- package/pipeline/scripts/review-file-filter.mjs +5 -3
- package/pipeline/scripts/review-scope.mjs +2 -1
- package/pipeline/scripts/routine-registry.mjs +2 -1
- package/pipeline/scripts/run-aggregator.mjs +26 -20
- package/pipeline/scripts/run-metrics.mjs +4 -2
- package/pipeline/scripts/runs-index.mjs +353 -0
- package/pipeline/scripts/scorecard-snapshot.mjs +178 -0
- package/pipeline/scripts/search-logs.sh +18 -0
- package/pipeline/scripts/smoke-cross-cli-behavior.sh +6 -6
- package/pipeline/scripts/smoke-schema-validation.sh +26 -7
- package/pipeline/scripts/test-gap-scan.mjs +2 -1
- package/pipeline/scripts/test-integrity-gate.mjs +2 -1
- package/pipeline/scripts/token-budget-report.mjs +13 -2
- package/pipeline/scripts/triage-memory.mjs +2 -2
- package/pipeline/scripts/update-issue-progress.sh +56 -7
- package/pipeline/scripts/usage-report.mjs +12 -1
- package/pipeline/scripts/validate-analysis-doc.mjs +75 -18
- package/pipeline/scripts/validate-code-graph.mjs +6 -3
- package/pipeline/scripts/validate-complaint-doc.mjs +2 -1
- package/pipeline/scripts/validate-diff-risk.mjs +6 -3
- package/pipeline/scripts/validate-planning.mjs +1 -1
- package/pipeline/scripts/validate-reviewer.mjs +1 -1
- package/pipeline/scripts/validate-state.mjs +45 -5
- package/pipeline/scripts/validate-test-gap.mjs +6 -3
- package/pipeline/scripts/validate-triage.mjs +6 -4
- package/pipeline/scripts/verify-citations.mjs +4 -2
- package/pipeline/scripts/verify.mjs +327 -0
- package/pipeline/scripts/worktree-finalize.sh +18 -9
- package/pipeline/scripts/write-state.mjs +154 -15
- package/pipeline/skills/.skill-manifest.json +37 -21
- package/pipeline/skills/.skills-index.json +104 -5
- package/pipeline/skills/shared/README.md +15 -6
- package/pipeline/skills/shared/core/apple-archive-compliance/SKILL.md +2 -2
- package/pipeline/skills/shared/core/google-play-compliance/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent/SKILL.md +69 -71
- package/pipeline/skills/shared/core/multi-agent-autopilot/SKILL.md +3 -3
- package/pipeline/skills/shared/core/multi-agent-channels/SKILL.md +14 -14
- package/pipeline/skills/shared/core/multi-agent-diff-explain/SKILL.md +5 -5
- package/pipeline/skills/shared/core/multi-agent-graph/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-help/SKILL.md +25 -23
- package/pipeline/skills/shared/core/multi-agent-language/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-local/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-local-autopilot/SKILL.md +8 -8
- package/pipeline/skills/shared/core/multi-agent-manual-test/SKILL.md +6 -6
- package/pipeline/skills/shared/core/multi-agent-model/SKILL.md +71 -0
- package/pipeline/skills/shared/core/multi-agent-refactor/SKILL.md +3 -3
- package/pipeline/skills/shared/core/multi-agent-resume/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-resume-local/SKILL.md +7 -7
- package/pipeline/skills/shared/core/multi-agent-route-off/SKILL.md +39 -0
- package/pipeline/skills/shared/core/multi-agent-route-on/SKILL.md +76 -0
- package/pipeline/skills/shared/core/multi-agent-route-status/SKILL.md +59 -0
- package/pipeline/skills/shared/core/multi-agent-setup/SKILL.md +1 -1
- package/pipeline/skills/shared/core/multi-agent-status/SKILL.md +35 -11
- package/pipeline/skills/shared/core/multi-agent-steer/SKILL.md +2 -2
- package/pipeline/skills/shared/core/multi-agent-sync/SKILL.md +6 -5
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/package_app.sh +4 -1
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/setup_dev_signing.sh +4 -1
- package/pipeline/skills/shared/external/macos-spm-app-packaging/assets/templates/sign-and-notarize.sh +2 -1
- package/pipeline/skills/skills-index.md +13 -4
- package/pipeline/multi-agent-refs/phases/phase-1-analysis.md +0 -263
- package/pipeline/multi-agent-refs/phases/phase-2-planning.md +0 -344
- package/pipeline/multi-agent-refs/phases/phase-5-test.md +0 -182
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
// by --since=<ISO date> / --task-id=<id> / --phase=<N>.
|
|
7
7
|
//
|
|
8
8
|
// Output is plain-text by default; pass --json for machine-readable output
|
|
9
|
-
// (used by Phase
|
|
9
|
+
// (used by Phase 5 to embed a metrics block in the run report).
|
|
10
10
|
//
|
|
11
11
|
// Aggregations:
|
|
12
12
|
// - Tasks completed
|
|
@@ -221,12 +221,13 @@ const summary = {
|
|
|
221
221
|
events_by_type: byEvent,
|
|
222
222
|
};
|
|
223
223
|
|
|
224
|
+
// An if/else chain rather than three blocks each ending in process.exit(0).
|
|
225
|
+
// stdout to a pipe is asynchronous and process.exit() discards what has not
|
|
226
|
+
// drained, so `aggregate-metrics.mjs --json | jq` read a payload cut at a
|
|
227
|
+
// buffer boundary. Choosing the renderer with `else` needs no exit at all.
|
|
224
228
|
if (opts.json) {
|
|
225
229
|
console.log(JSON.stringify(summary, null, 2));
|
|
226
|
-
|
|
227
|
-
}
|
|
228
|
-
|
|
229
|
-
if (opts.markdown) {
|
|
230
|
+
} else if (opts.markdown) {
|
|
230
231
|
const lines = [];
|
|
231
232
|
lines.push(`# Pipeline Metrics Summary`);
|
|
232
233
|
lines.push(``);
|
|
@@ -288,74 +289,73 @@ if (opts.markdown) {
|
|
|
288
289
|
);
|
|
289
290
|
}
|
|
290
291
|
console.log(lines.join("\n"));
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
console.log(`
|
|
297
|
-
console.log(`
|
|
298
|
-
console.log(
|
|
299
|
-
console.log(`
|
|
300
|
-
console.log(
|
|
301
|
-
console.log(`
|
|
302
|
-
console.log(`
|
|
303
|
-
console.log(` cycles per task avg: ${summary.review_cycles.avg}`);
|
|
304
|
-
console.log(` cycles per task p95: ${summary.review_cycles.p95}`);
|
|
305
|
-
console.log(``);
|
|
306
|
-
console.log(`Triage classification (sum across all reviews):`);
|
|
307
|
-
console.log(` raw findings: ${summary.triage_totals.raw}`);
|
|
308
|
-
console.log(
|
|
309
|
-
` accepted: ${summary.triage_totals.accepted} (rate ${fmt(summary.triage_totals.accept_rate)})`,
|
|
310
|
-
);
|
|
311
|
-
console.log(` deferred: ${summary.triage_totals.deferred}`);
|
|
312
|
-
console.log(
|
|
313
|
-
` rejected: ${summary.triage_totals.rejected} (rate ${fmt(summary.triage_totals.reject_rate)})`,
|
|
314
|
-
);
|
|
315
|
-
|
|
316
|
-
if (Object.keys(summary.edge_cases).length) {
|
|
292
|
+
} else {
|
|
293
|
+
// Plain-text rendering
|
|
294
|
+
const fmt = (n) => (n === null ? " - " : String(n));
|
|
295
|
+
console.log(`Multi-Agent Pipeline - metrics summary`);
|
|
296
|
+
console.log(`source: ${summary.source}`);
|
|
297
|
+
console.log(`events: ${summary.total_events} (${summary.parse_errors} parse errors)`);
|
|
298
|
+
console.log(`unique tasks: ${summary.unique_tasks}`);
|
|
299
|
+
console.log(``);
|
|
300
|
+
console.log(`Reviews:`);
|
|
301
|
+
console.log(` completed: ${summary.reviews_completed}`);
|
|
302
|
+
console.log(` cycles per task avg: ${summary.review_cycles.avg}`);
|
|
303
|
+
console.log(` cycles per task p95: ${summary.review_cycles.p95}`);
|
|
317
304
|
console.log(``);
|
|
318
|
-
console.log(`Triage
|
|
319
|
-
|
|
320
|
-
|
|
305
|
+
console.log(`Triage classification (sum across all reviews):`);
|
|
306
|
+
console.log(` raw findings: ${summary.triage_totals.raw}`);
|
|
307
|
+
console.log(
|
|
308
|
+
` accepted: ${summary.triage_totals.accepted} (rate ${fmt(summary.triage_totals.accept_rate)})`,
|
|
309
|
+
);
|
|
310
|
+
console.log(` deferred: ${summary.triage_totals.deferred}`);
|
|
311
|
+
console.log(
|
|
312
|
+
` rejected: ${summary.triage_totals.rejected} (rate ${fmt(summary.triage_totals.reject_rate)})`,
|
|
313
|
+
);
|
|
314
|
+
|
|
315
|
+
if (Object.keys(summary.edge_cases).length) {
|
|
316
|
+
console.log(``);
|
|
317
|
+
console.log(`Triage edge cases:`);
|
|
318
|
+
for (const [k, v] of Object.entries(summary.edge_cases).sort((a, b) => b[1] - a[1])) {
|
|
319
|
+
console.log(` ${k.padEnd(28)} ${v}`);
|
|
320
|
+
}
|
|
321
321
|
}
|
|
322
|
-
}
|
|
323
322
|
|
|
324
|
-
if (Object.keys(summary.rework_iterations).length) {
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
323
|
+
if (Object.keys(summary.rework_iterations).length) {
|
|
324
|
+
console.log(``);
|
|
325
|
+
console.log(`Phase 3 rework iterations:`);
|
|
326
|
+
for (const [it, n] of Object.entries(summary.rework_iterations).sort(
|
|
327
|
+
(a, b) => Number(a[0]) - Number(b[0]),
|
|
328
|
+
)) {
|
|
329
|
+
console.log(` iteration ${it.padEnd(5)} ${n}`);
|
|
330
|
+
}
|
|
331
331
|
}
|
|
332
|
-
}
|
|
333
332
|
|
|
334
|
-
if (Object.keys(summary.language_preference).length) {
|
|
335
|
-
|
|
336
|
-
|
|
337
|
-
|
|
338
|
-
|
|
333
|
+
if (Object.keys(summary.language_preference).length) {
|
|
334
|
+
console.log(``);
|
|
335
|
+
console.log(`Language preference (events tagged with lang=):`);
|
|
336
|
+
for (const [lang, n] of Object.entries(summary.language_preference)) {
|
|
337
|
+
console.log(` ${lang.padEnd(8)} ${n}`);
|
|
338
|
+
}
|
|
339
339
|
}
|
|
340
|
-
}
|
|
341
340
|
|
|
342
|
-
if (Object.keys(summary.cost_per_model).length) {
|
|
343
|
-
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
341
|
+
if (Object.keys(summary.cost_per_model).length) {
|
|
342
|
+
console.log(``);
|
|
343
|
+
console.log(`Cost / token telemetry (per model):`);
|
|
344
|
+
console.log(
|
|
345
|
+
` ${"model".padEnd(20)} ${"calls".padStart(6)} ${"duration_ms".padStart(13)} ${"tokens_in".padStart(11)} ${"tokens_out".padStart(11)} ${"cached".padStart(11)} ${"cache%".padStart(7)}`,
|
|
346
|
+
);
|
|
347
|
+
for (const [m, t] of Object.entries(summary.cost_per_model).sort(
|
|
348
|
+
(a, b) => b[1].calls - a[1].calls,
|
|
349
|
+
)) {
|
|
350
|
+
const pct = t.cache_ratio === null ? " - " : `${Math.round(t.cache_ratio * 100)}%`;
|
|
351
|
+
console.log(
|
|
352
|
+
` ${m.padEnd(20)} ${String(t.calls).padStart(6)} ${String(t.duration_ms).padStart(13)} ${String(t.tokens_in).padStart(11)} ${String(t.tokens_out).padStart(11)} ${String(t.tokens_cached).padStart(11)} ${pct.padStart(7)}`,
|
|
353
|
+
);
|
|
354
|
+
}
|
|
355
|
+
const oc = summary.cache_reuse;
|
|
356
|
+
const ocPct = oc.cache_ratio === null ? " - " : `${Math.round(oc.cache_ratio * 100)}%`;
|
|
352
357
|
console.log(
|
|
353
|
-
`
|
|
358
|
+
` overall prompt-cache reuse: ${oc.tokens_cached} / ${oc.tokens_in + oc.tokens_cached} input tokens = ${ocPct}`,
|
|
354
359
|
);
|
|
355
360
|
}
|
|
356
|
-
const oc = summary.cache_reuse;
|
|
357
|
-
const ocPct = oc.cache_ratio === null ? " - " : `${Math.round(oc.cache_ratio * 100)}%`;
|
|
358
|
-
console.log(
|
|
359
|
-
` overall prompt-cache reuse: ${oc.tokens_cached} / ${oc.tokens_in + oc.tokens_cached} input tokens = ${ocPct}`,
|
|
360
|
-
);
|
|
361
361
|
}
|
|
@@ -32,6 +32,7 @@
|
|
|
32
32
|
import { existsSync, readFileSync } from "node:fs";
|
|
33
33
|
import { join } from "node:path";
|
|
34
34
|
import { homedir } from "node:os";
|
|
35
|
+
import { invokedDirectly } from "../lib/invoked-directly.mjs";
|
|
35
36
|
|
|
36
37
|
const ROOT = process.env.MA_AUTOPILOT_ROOT || join(homedir(), ".claude", "autopilot");
|
|
37
38
|
const PREFS =
|
|
@@ -142,6 +143,6 @@ function main(argv) {
|
|
|
142
143
|
return 0;
|
|
143
144
|
}
|
|
144
145
|
|
|
145
|
-
if (import.meta.url
|
|
146
|
+
if (invokedDirectly(import.meta.url)) {
|
|
146
147
|
process.exit(main(process.argv));
|
|
147
148
|
}
|
|
@@ -39,6 +39,7 @@ import { execFileSync } from "node:child_process";
|
|
|
39
39
|
import { existsSync, readFileSync, mkdirSync, writeFileSync, renameSync, chmodSync } from "node:fs";
|
|
40
40
|
import { join, dirname } from "node:path";
|
|
41
41
|
import { homedir } from "node:os";
|
|
42
|
+
import { invokedDirectly } from "../lib/invoked-directly.mjs";
|
|
42
43
|
|
|
43
44
|
const ROOT = process.env.MA_AUTOPILOT_ROOT || join(homedir(), ".claude", "autopilot");
|
|
44
45
|
|
|
@@ -382,6 +383,6 @@ function main(argv) {
|
|
|
382
383
|
return 0;
|
|
383
384
|
}
|
|
384
385
|
|
|
385
|
-
if (import.meta.url
|
|
386
|
+
if (invokedDirectly(import.meta.url)) {
|
|
386
387
|
process.exit(main(process.argv));
|
|
387
388
|
}
|
|
@@ -51,10 +51,16 @@ import {
|
|
|
51
51
|
chmodSync,
|
|
52
52
|
readdirSync,
|
|
53
53
|
statSync,
|
|
54
|
+
openSync,
|
|
55
|
+
readSync,
|
|
56
|
+
closeSync,
|
|
57
|
+
truncateSync,
|
|
54
58
|
} from "node:fs";
|
|
55
59
|
import { join } from "node:path";
|
|
56
60
|
import { homedir } from "node:os";
|
|
57
61
|
import { randomUUID } from "node:crypto";
|
|
62
|
+
import { runMain } from "../lib/fatal.mjs";
|
|
63
|
+
import { invokedDirectly } from "../lib/invoked-directly.mjs";
|
|
58
64
|
|
|
59
65
|
const ROOT = process.env.MA_AUTOPILOT_ROOT || join(homedir(), ".claude", "autopilot");
|
|
60
66
|
// Siblings resolve from THIS file's own directory, not from ~/.claude/scripts.
|
|
@@ -76,6 +82,25 @@ const REGISTER_GRACE_MS = Number(process.env.MA_AP_GRACE_MS || 60000);
|
|
|
76
82
|
// A session in one of these is still working. Anything else - and a row with
|
|
77
83
|
// no status at all is not "anything else" - ends supervision.
|
|
78
84
|
const LIVE_STATUSES = new Set(["busy", "running", "waiting"]);
|
|
85
|
+
// launchd appends this process's stdout to runner.log forever. On a laptop that
|
|
86
|
+
// is a file nobody ever looks at; on a machine that ticks every few minutes for
|
|
87
|
+
// months it is the thing that fills the disk, and a full disk stops the runs it
|
|
88
|
+
// was logging.
|
|
89
|
+
const LOG_MAX_BYTES = Number(process.env.MA_AP_LOG_MAX_BYTES || 5 * 1024 * 1024);
|
|
90
|
+
// Consecutive failed attempts before the runner stops taking NEW work. The
|
|
91
|
+
// failure this guards against is a machine-level one - an expired token, a full
|
|
92
|
+
// disk, a `claude` that no longer launches - where every item fails the same
|
|
93
|
+
// way and the queue is consumed one worthless run at a time. Zero disables it.
|
|
94
|
+
const BREAKER_LIMIT = Number(process.env.MA_AP_BREAKER_LIMIT || 3);
|
|
95
|
+
// How long an open breaker waits before letting ONE attempt through. Long
|
|
96
|
+
// enough that a broken machine is not burning the queue (default 30 min, so at
|
|
97
|
+
// a 15-minute tick it is every other tick at most), short enough that a machine
|
|
98
|
+
// fixed at 3am is working again by morning without anyone touching it.
|
|
99
|
+
const BREAKER_COOLDOWN_SEC = Number(process.env.MA_AP_BREAKER_COOLDOWN_SEC || 1800);
|
|
100
|
+
// Outcomes that mean the attempt produced nothing. `needs-input` is NOT here:
|
|
101
|
+
// a run parked on a question did work and is waiting for a person, which is the
|
|
102
|
+
// system behaving correctly.
|
|
103
|
+
const FAILED_OUTCOMES = new Set(["launch-failed", "timed-out", "no-state", "failed", "unknown"]);
|
|
79
104
|
|
|
80
105
|
const log = (m) => process.stdout.write(`autopilot-runner: ${m}\n`);
|
|
81
106
|
|
|
@@ -120,6 +145,111 @@ function record(entry) {
|
|
|
120
145
|
chmodSync(p, 0o600);
|
|
121
146
|
}
|
|
122
147
|
|
|
148
|
+
/**
|
|
149
|
+
* Keep runner.log bounded without breaking the writer.
|
|
150
|
+
*
|
|
151
|
+
* launchd holds the file open with O_APPEND, so RENAMING the log leaves that fd
|
|
152
|
+
* writing to the renamed inode and the new file stays empty forever - the
|
|
153
|
+
* rotation that looks right is the one that silently stops logging. Truncating
|
|
154
|
+
* the SAME inode is what works: the open fd keeps writing, at offset zero.
|
|
155
|
+
* The tail is copied aside first, because the last few hundred lines are the
|
|
156
|
+
* ones somebody is about to need.
|
|
157
|
+
*/
|
|
158
|
+
function rotateLog() {
|
|
159
|
+
const p = join(ROOT, "runner.log");
|
|
160
|
+
let size;
|
|
161
|
+
try {
|
|
162
|
+
size = statSync(p).size;
|
|
163
|
+
} catch {
|
|
164
|
+
// No log yet (a runner that has never been started by launchd), or it is
|
|
165
|
+
// unreadable. Nothing to rotate either way.
|
|
166
|
+
return;
|
|
167
|
+
}
|
|
168
|
+
if (size <= LOG_MAX_BYTES) return;
|
|
169
|
+
try {
|
|
170
|
+
const keep = Math.min(size, 256 * 1024);
|
|
171
|
+
const fd = openSync(p, "r");
|
|
172
|
+
const buf = Buffer.alloc(keep);
|
|
173
|
+
readSync(fd, buf, 0, keep, size - keep);
|
|
174
|
+
closeSync(fd);
|
|
175
|
+
writeFileSync(join(ROOT, "runner.log.1"), buf, { mode: 0o600 });
|
|
176
|
+
// Truncate in place. Not unlink, not rename.
|
|
177
|
+
truncateSync(p, 0);
|
|
178
|
+
log(
|
|
179
|
+
`runner.log reached ${Math.round(size / 1048576)}MB - tail kept in runner.log.1, log truncated`,
|
|
180
|
+
);
|
|
181
|
+
} catch {
|
|
182
|
+
// A log that cannot be rotated is not a reason to skip the tick.
|
|
183
|
+
}
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
/**
|
|
187
|
+
* One line per tick, machine-readable, next to the human log. The human log
|
|
188
|
+
* answers "what happened just now"; this answers "how has it been behaving for
|
|
189
|
+
* a week", which is the question a server actually raises and the one prose
|
|
190
|
+
* cannot be asked.
|
|
191
|
+
*/
|
|
192
|
+
function tick(entry) {
|
|
193
|
+
try {
|
|
194
|
+
ensureRoot();
|
|
195
|
+
const p = join(ROOT, "ticks.jsonl");
|
|
196
|
+
appendFileSync(p, JSON.stringify({ at: new Date().toISOString(), ...entry }) + "\n", {
|
|
197
|
+
mode: 0o600,
|
|
198
|
+
});
|
|
199
|
+
} catch {
|
|
200
|
+
// Telemetry never fails a tick.
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* How many attempts in a row produced nothing.
|
|
206
|
+
*
|
|
207
|
+
* Read from attempted.jsonl rather than kept as a counter, because a counter is
|
|
208
|
+
* a second piece of state that can disagree with the ledger - and the ledger is
|
|
209
|
+
* what a person reads when they ask why the runner stopped.
|
|
210
|
+
*
|
|
211
|
+
* @returns {{count:number, last:string|null}}
|
|
212
|
+
*/
|
|
213
|
+
export function consecutiveFailures(lines) {
|
|
214
|
+
let count = 0;
|
|
215
|
+
let last = null;
|
|
216
|
+
for (let i = lines.length - 1; i >= 0; i--) {
|
|
217
|
+
const row = lines[i];
|
|
218
|
+
if (!row || typeof row.outcome !== "string") continue;
|
|
219
|
+
// A blocked item never ran: arming refused it, which says nothing about
|
|
220
|
+
// whether a run would have worked.
|
|
221
|
+
if (row.outcome.startsWith("blocked-")) continue;
|
|
222
|
+
if (FAILED_OUTCOMES.has(row.outcome)) {
|
|
223
|
+
count++;
|
|
224
|
+
if (!last) last = row.outcome;
|
|
225
|
+
continue;
|
|
226
|
+
}
|
|
227
|
+
break;
|
|
228
|
+
}
|
|
229
|
+
return { count, last };
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
function readAttempted() {
|
|
233
|
+
const p = join(ROOT, "attempted.jsonl");
|
|
234
|
+
if (!existsSync(p)) return [];
|
|
235
|
+
try {
|
|
236
|
+
return readFileSync(p, "utf-8")
|
|
237
|
+
.split("\n")
|
|
238
|
+
.filter(Boolean)
|
|
239
|
+
.slice(-50)
|
|
240
|
+
.map((l) => {
|
|
241
|
+
try {
|
|
242
|
+
return JSON.parse(l);
|
|
243
|
+
} catch {
|
|
244
|
+
return null;
|
|
245
|
+
}
|
|
246
|
+
})
|
|
247
|
+
.filter(Boolean);
|
|
248
|
+
} catch {
|
|
249
|
+
return [];
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
|
|
123
253
|
/** Seconds since the epoch at which this machine booted. */
|
|
124
254
|
export function bootTime() {
|
|
125
255
|
const out = run("sysctl", ["-n", "kern.boottime"]);
|
|
@@ -446,6 +576,61 @@ function main() {
|
|
|
446
576
|
return 0;
|
|
447
577
|
}
|
|
448
578
|
|
|
579
|
+
rotateLog();
|
|
580
|
+
|
|
581
|
+
// ---- 1b. BREAKER -------------------------------------------------------
|
|
582
|
+
// Three failures in a row is not three unlucky items, it is one broken
|
|
583
|
+
// machine: an expired token, a `claude` that no longer launches, a full disk.
|
|
584
|
+
// Left alone the runner eats the whole queue at a few minutes an item and
|
|
585
|
+
// records a failure for each, which destroys the evidence of WHICH item was
|
|
586
|
+
// first and leaves nothing to resume. Stopping keeps the queue intact.
|
|
587
|
+
const attempts = readAttempted();
|
|
588
|
+
const breaker = consecutiveFailures(attempts);
|
|
589
|
+
if (BREAKER_LIMIT > 0 && breaker.count >= BREAKER_LIMIT) {
|
|
590
|
+
// HALF-OPEN, not latched. The first version of this returned here on every
|
|
591
|
+
// tick, and the only writer of attempted.jsonl is downstream of this
|
|
592
|
+
// return - so once it opened, no new attempt could ever be recorded, the
|
|
593
|
+
// consecutive count could never fall, and the runner was stopped for good.
|
|
594
|
+
// The log line said "until an attempt succeeds" and the feature doc said
|
|
595
|
+
// "one successful attempt clears it": both described a state the code made
|
|
596
|
+
// unreachable.
|
|
597
|
+
//
|
|
598
|
+
// A breaker that cannot re-close is not a breaker, it is an off switch with
|
|
599
|
+
// a misleading label. So after a cooldown one probe is let through: if the
|
|
600
|
+
// machine recovered it succeeds and the count resets on its own; if it did
|
|
601
|
+
// not, that probe fails, becomes the new most recent attempt, and the
|
|
602
|
+
// breaker closes again for another cooldown. One wasted run per cooldown is
|
|
603
|
+
// the price of not needing a human to notice.
|
|
604
|
+
const lastAt = attempts.length ? Number(attempts[attempts.length - 1].at || 0) : 0;
|
|
605
|
+
const sinceSec = lastAt ? Math.floor(Date.now() / 1000) - lastAt : Infinity;
|
|
606
|
+
const probeDue = sinceSec >= BREAKER_COOLDOWN_SEC;
|
|
607
|
+
if (!probeDue) {
|
|
608
|
+
const why = `${breaker.count} attempts in a row produced nothing (last: ${breaker.last})`;
|
|
609
|
+
const waitMin = Math.ceil((BREAKER_COOLDOWN_SEC - sinceSec) / 60);
|
|
610
|
+
log(`circuit breaker open - ${why}`);
|
|
611
|
+
log(
|
|
612
|
+
` one probe run is allowed in ${waitMin} min; check the token, disk and \`claude\` binary`,
|
|
613
|
+
);
|
|
614
|
+
log(" to override for one tick: MA_AP_BREAKER_LIMIT=0");
|
|
615
|
+
if (!DRY) {
|
|
616
|
+
writeState("queue.json", {
|
|
617
|
+
...readJson(join(ROOT, "queue.json"), { queued: [], running: [] }),
|
|
618
|
+
blockedReason: `circuit breaker: ${why} - probe in ${waitMin} min`,
|
|
619
|
+
});
|
|
620
|
+
refreshStatus();
|
|
621
|
+
}
|
|
622
|
+
tick({
|
|
623
|
+
action: "breaker-open",
|
|
624
|
+
failures: breaker.count,
|
|
625
|
+
lastOutcome: breaker.last,
|
|
626
|
+
probeInSec: BREAKER_COOLDOWN_SEC - sinceSec,
|
|
627
|
+
});
|
|
628
|
+
return 0;
|
|
629
|
+
}
|
|
630
|
+
log(`circuit breaker half-open - ${breaker.count} failures, letting one probe run through`);
|
|
631
|
+
tick({ action: "breaker-probe", failures: breaker.count, lastOutcome: breaker.last });
|
|
632
|
+
}
|
|
633
|
+
|
|
449
634
|
// ---- 2. INTAKE ---------------------------------------------------------
|
|
450
635
|
const intake = join(SCRIPTS, "autopilot-intake.mjs");
|
|
451
636
|
if (existsSync(intake)) run(process.execPath, [intake]);
|
|
@@ -455,6 +640,7 @@ function main() {
|
|
|
455
640
|
const slots = Number(config.slots ?? 1);
|
|
456
641
|
if (running.length >= slots) {
|
|
457
642
|
log(`${running.length}/${slots} slots in use - nothing to take`);
|
|
643
|
+
tick({ action: "slots-full", running: running.length, slots });
|
|
458
644
|
return 0;
|
|
459
645
|
}
|
|
460
646
|
|
|
@@ -466,6 +652,11 @@ function main() {
|
|
|
466
652
|
const next = (queue.queued || []).find((i) => !busyRepos.has(i.repo));
|
|
467
653
|
if (!next) {
|
|
468
654
|
log(queue.emptyReason || "every queued item belongs to a repo already in flight");
|
|
655
|
+
tick({
|
|
656
|
+
action: "nothing-to-take",
|
|
657
|
+
queued: (queue.queued || []).length,
|
|
658
|
+
running: running.length,
|
|
659
|
+
});
|
|
469
660
|
return 0;
|
|
470
661
|
}
|
|
471
662
|
|
|
@@ -593,9 +784,22 @@ function main() {
|
|
|
593
784
|
log(`${next.id}: claim kept for the next tick (${supervised.reason})`);
|
|
594
785
|
}
|
|
595
786
|
log(`${next.id}: ${outcome}${prUrl ? ` ${prUrl}` : ""} (${supervised.waitedSec}s)`);
|
|
787
|
+
tick({
|
|
788
|
+
action: "ran",
|
|
789
|
+
source: next.source,
|
|
790
|
+
id: next.id,
|
|
791
|
+
taskId,
|
|
792
|
+
outcome,
|
|
793
|
+
reason: supervised.reason,
|
|
794
|
+
waitedSec: supervised.waitedSec,
|
|
795
|
+
prOpened: Boolean(prUrl),
|
|
796
|
+
consecutiveFailuresBefore: breaker.count,
|
|
797
|
+
});
|
|
596
798
|
return 0;
|
|
597
799
|
}
|
|
598
800
|
|
|
599
|
-
if (import.meta.url
|
|
600
|
-
|
|
801
|
+
if (invokedDirectly(import.meta.url)) {
|
|
802
|
+
runMain("autopilot-runner", () => {
|
|
803
|
+
process.exit(main());
|
|
804
|
+
});
|
|
601
805
|
}
|
|
@@ -22,6 +22,7 @@
|
|
|
22
22
|
// Exit codes: 0 ok, 1 coverage failure, 2 usage / parse error.
|
|
23
23
|
|
|
24
24
|
import { readFileSync } from "node:fs";
|
|
25
|
+
import { runMain } from "../lib/fatal.mjs";
|
|
25
26
|
|
|
26
27
|
const HEADERS = {
|
|
27
28
|
tr: ["Tür", "Kaynak", "URL / Yol", "Sürüm / Ref", "Rol", "Erişim", "Notlar"],
|
|
@@ -385,4 +386,4 @@ function main() {
|
|
|
385
386
|
process.stdout.write(`${renderTable(rows, lang)}\n`);
|
|
386
387
|
}
|
|
387
388
|
|
|
388
|
-
main
|
|
389
|
+
runMain("build-references", main);
|
|
@@ -33,7 +33,8 @@ import {
|
|
|
33
33
|
mkdirSync,
|
|
34
34
|
statSync,
|
|
35
35
|
} from "node:fs";
|
|
36
|
-
import { join } from "node:path";
|
|
36
|
+
import { dirname, join } from "node:path";
|
|
37
|
+
import { fileURLToPath } from "node:url";
|
|
37
38
|
import { spawnSync } from "node:child_process";
|
|
38
39
|
import { createHash } from "node:crypto";
|
|
39
40
|
|
|
@@ -65,7 +66,14 @@ if (!HOME) {
|
|
|
65
66
|
process.exit(2);
|
|
66
67
|
}
|
|
67
68
|
const PLUGINS_REPO = getArg("--plugins-repo", join(HOME, "multi-agent-plugins"));
|
|
68
|
-
|
|
69
|
+
// The pipeline root is THIS file's repo, found by walking up from the script,
|
|
70
|
+
// not a guess at where the checkout lives. It defaulted to
|
|
71
|
+
// $HOME/multi-agent-pipeline, which is true on the maintainer's machine and
|
|
72
|
+
// nowhere else: on a CI runner the checkout is under the workspace directory,
|
|
73
|
+
// so --check-routing reported "source not found" and three tests failed on
|
|
74
|
+
// every clean host while passing locally. That is the whole class of bug CI
|
|
75
|
+
// exists to catch, and it survived because CI was asleep.
|
|
76
|
+
const PIPE_ROOT = getArg("--pipeline", join(dirname(fileURLToPath(import.meta.url)), "..", ".."));
|
|
69
77
|
const EXTERNAL = join(PIPE_ROOT, "pipeline/skills/shared/external");
|
|
70
78
|
const DRY = args.includes("--dry-run");
|
|
71
79
|
|
|
@@ -207,7 +207,10 @@ case "$MODE" in
|
|
|
207
207
|
# stale video from yesterday's run attached as today's evidence is worse
|
|
208
208
|
# than no video, because nobody re-checks an artefact that is present.
|
|
209
209
|
SRC=""
|
|
210
|
-
for cand in $(find
|
|
210
|
+
# `while read`, not `for cand in $(find ...)`: word splitting turns one
|
|
211
|
+
# path with a space into two candidates, both of which fail `[ -f ]`, and
|
|
212
|
+
# the evidence goes missing without a word said.
|
|
213
|
+
while IFS= read -r cand; do
|
|
211
214
|
[ -f "$cand" ] || continue
|
|
212
215
|
MT=$(date -r "$cand" +%s 2>/dev/null || echo 0)
|
|
213
216
|
[ "$MT" -ge "$SINCE" ] 2>/dev/null || continue
|
|
@@ -215,7 +218,9 @@ case "$MODE" in
|
|
|
215
218
|
PREV=$(date -r "$SRC" +%s 2>/dev/null || echo 0)
|
|
216
219
|
[ "$MT" -gt "$PREV" ] 2>/dev/null && SRC="$cand"
|
|
217
220
|
fi
|
|
218
|
-
done
|
|
221
|
+
done <<EOF
|
|
222
|
+
$(find test-results cypress/videos -type f \( -name '*.webm' -o -name '*.mp4' \) 2>/dev/null)
|
|
223
|
+
EOF
|
|
219
224
|
[ -n "$SRC" ] || { echo "capture-evidence: the suite recorded no video after the marker (is video enabled in the project's runner config?)" >&2; exit 4; }
|
|
220
225
|
case "$SRC" in
|
|
221
226
|
*.webm)
|
|
@@ -5,14 +5,14 @@
|
|
|
5
5
|
#
|
|
6
6
|
# WHY THIS EXISTS
|
|
7
7
|
#
|
|
8
|
-
# Every persistent write used to live in Phase
|
|
9
|
-
# ledger distill, the knowledge-base append, the code-graph refresh. And Phase
|
|
8
|
+
# Every persistent write used to live in Phase 5: triage ingest, the learnings
|
|
9
|
+
# ledger distill, the knowledge-base append, the code-graph refresh. And Phase 5
|
|
10
10
|
# is, by the pipeline's own admission in features/code-graph.md, the phase a run
|
|
11
11
|
# is LEAST likely to reach. A run killed in Phase 3, a session that hits its
|
|
12
12
|
# context ceiling, a crash after review - each one threw away everything it had
|
|
13
13
|
# established, and the next run on the same repo rediscovered it from scratch.
|
|
14
14
|
#
|
|
15
|
-
# So the writes move here, and Phase
|
|
15
|
+
# So the writes move here, and Phase 5 becomes the LAST flush rather than the
|
|
16
16
|
# only one. Phase boundaries call this, and so does SessionEnd. Nothing about
|
|
17
17
|
# the trigger depends on a model noticing that a moment qualifies: it hangs on
|
|
18
18
|
# a phase transition and on process exit, both objectively visible without any
|
|
@@ -20,15 +20,15 @@
|
|
|
20
20
|
#
|
|
21
21
|
# What it does NOT do: call a model. Everything here is derived from artefacts
|
|
22
22
|
# already on disk (triage-output.json) plus agent-state.json. The parts of
|
|
23
|
-
# Phase
|
|
24
|
-
# per-repo memory synthesis - stay in Phase
|
|
23
|
+
# Phase 5 that genuinely need a model - the knowledge-base extraction, the
|
|
24
|
+
# per-repo memory synthesis - stay in Phase 5, because a hook cannot think.
|
|
25
25
|
#
|
|
26
26
|
# Usage:
|
|
27
27
|
# ./capture-flush.sh [--state <agent-state.json>] [--if-stale] [--json] [--quiet]
|
|
28
28
|
#
|
|
29
29
|
# --state the run to flush. Default: resolved from the newest task dir
|
|
30
30
|
# under $HOME/.claude/logs/multi-agent (see resolve_state).
|
|
31
|
-
# --if-stale flush only when the run did NOT complete Phase
|
|
31
|
+
# --if-stale flush only when the run did NOT complete Phase 5 - the
|
|
32
32
|
# SessionEnd case. A finished run has already flushed.
|
|
33
33
|
# --json machine-readable result for a caller that wants to count rows.
|
|
34
34
|
# --quiet no stdout. Exit status still distinguishes the outcomes.
|
|
@@ -101,13 +101,13 @@ WORKTREE=$(jq -r '.worktreePath // empty' "$STATE" 2>/dev/null)
|
|
|
101
101
|
PHASE7=$(jq -r '[.phases[]? | select((.id // "") == "7") | .status] | first // ""' "$STATE" 2>/dev/null)
|
|
102
102
|
|
|
103
103
|
if [ "$IF_STALE" -eq 1 ] && [ "$PHASE7" = "completed" ]; then
|
|
104
|
-
say "capture-flush: ${TASK_ID:-run} already completed Phase
|
|
104
|
+
say "capture-flush: ${TASK_ID:-run} already completed Phase 5 - nothing stale"
|
|
105
105
|
[ "$JSON" -eq 1 ] && printf '{"status":"noop","reason":"already-flushed","taskId":"%s"}\n' "$TASK_ID"
|
|
106
106
|
exit 0
|
|
107
107
|
fi
|
|
108
108
|
|
|
109
109
|
# The triage artefact is the only input either store needs, and it has two homes:
|
|
110
|
-
# Phase
|
|
110
|
+
# Phase 4 removes the worktree once the PR is open, so the salvaged copy under
|
|
111
111
|
# artifactsPath is tried FIRST. Reading the worktree path first would degrade
|
|
112
112
|
# silently for exactly the runs this script exists to rescue.
|
|
113
113
|
TRIAGE=""
|
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
# the user learns to skip, which costs more than it saves.
|
|
12
12
|
#
|
|
13
13
|
# It answers two questions:
|
|
14
|
-
# - Is there a pipeline run that stopped before Phase
|
|
14
|
+
# - Is there a pipeline run that stopped before Phase 5? Name it and the resume
|
|
15
15
|
# command, because that run's work is recoverable and its findings are not
|
|
16
16
|
# yet in the durable stores until it flushes.
|
|
17
17
|
# - Is the pipeline's own observation queue stale? (>= REVIEW_DAYS since the
|
|
@@ -32,7 +32,7 @@ REVIEW_DAYS=7
|
|
|
32
32
|
# Three buckets, not one line. Measured on the development machine: 22 runs read
|
|
33
33
|
# `in_progress` and every one was over a day old, but they are not one failure.
|
|
34
34
|
# 13 stopped at Phase 0, which is almost entirely questions - those runs never
|
|
35
|
-
# started. 3 had their PR already open and were waiting at Phase
|
|
35
|
+
# started. 3 had their PR already open and were waiting at Phase 4/7, where the
|
|
36
36
|
# pipeline pauses ON PURPOSE for channel selection. 4 died mid-development.
|
|
37
37
|
#
|
|
38
38
|
# Reporting the newest one of those as "stopped at Phase N" told the truth about
|
|
@@ -71,7 +71,7 @@ if [ -d "$LOGS" ] && command -v jq >/dev/null 2>&1; then
|
|
|
71
71
|
PR=$(jq -r 'if (.pr|type)=="string" then .pr elif (.pr|type)=="object" then (.pr.url // .pr.number // "") else "" end | tostring' "$f" 2>/dev/null)
|
|
72
72
|
|
|
73
73
|
# Waiting for you, not broken: an open PR means the work landed, and
|
|
74
|
-
# Phase
|
|
74
|
+
# Phase 5 pauses for channel selection by design (modes.md).
|
|
75
75
|
if [ "$STATUS" = "awaiting_input" ] || [ -n "$PR" ] ||
|
|
76
76
|
[ "$PHASE" = "6" ] || [ "$PHASE" = "7" ]; then
|
|
77
77
|
AWAITING_N=$((AWAITING_N + 1))
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
// classify-plan-safety.mjs - v7.0.G
|
|
3
3
|
//
|
|
4
|
-
// Heuristic safety classifier for a Phase
|
|
4
|
+
// Heuristic safety classifier for a Phase 1 plan. Autopilot's "zero user
|
|
5
5
|
// interaction" contract is fine for small, predictable tasks but dangerous
|
|
6
6
|
// when a plan touches the security path, deletes files without paired
|
|
7
7
|
// tests, or sprawls across many files. This script inspects the plan and
|
|
@@ -197,5 +197,6 @@ if (recommendPause) {
|
|
|
197
197
|
summary = `Low-risk plan - autopilot safe`;
|
|
198
198
|
}
|
|
199
199
|
|
|
200
|
+
// See run-metrics.mjs: exiting here would truncate this write when stdout is a
|
|
201
|
+
// pipe, and 0 is the default code anyway.
|
|
200
202
|
process.stdout.write(JSON.stringify({ score, recommendPause, reasons, summary }, null, 2) + "\n");
|
|
201
|
-
process.exit(0);
|