forge-workflow 0.1.0-beta.3 → 0.1.0-beta.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +14 -7
- package/CHANGELOG.md +43 -1
- package/README.md +6 -2
- package/bin/forge-cmd.js +21 -1
- package/bin/forge.js +16 -369
- package/docs/INDEX.md +1 -1
- package/docs/guides/BEADS_GITHUB_SYNC.md +2 -31
- package/docs/guides/MIGRATION.md +4 -4
- package/docs/guides/SETUP.md +16 -16
- package/docs/reference/COMMANDS.md +9 -4
- package/docs/reference/INSIGHTS_RECAP.md +9 -20
- package/docs/reference/RELEASE.md +5 -3
- package/docs/reference/TOOLCHAIN.md +8 -0
- package/docs/reference/protected-state-surfaces.md +4 -4
- package/docs/reference/shepherd.md +117 -17
- package/lefthook.yml +12 -0
- package/lib/activation/ensure-forge-home.js +33 -15
- package/lib/adapters/greptile-review-adapter.js +1 -1
- package/lib/adapters/pr-state-adapter.js +397 -100
- package/lib/agents-config.js +5 -0
- package/lib/audit-evidence.js +71 -110
- package/lib/capped-jsonl-log.js +236 -0
- package/lib/commands/_issue.js +31 -46
- package/lib/commands/_manifest.js +1 -1
- package/lib/commands/_registry.js +2 -2
- package/lib/commands/_resolve-command-opts.js +36 -29
- package/lib/commands/claim.js +2 -4
- package/lib/commands/clean.js +196 -32
- package/lib/commands/dev.js +4 -33
- package/lib/commands/hooks.js +358 -13
- package/lib/commands/insights.js +8 -3
- package/lib/commands/merge.js +600 -40
- package/lib/commands/plan.js +23 -115
- package/lib/commands/pr.js +1 -1
- package/lib/commands/preflight.js +11 -2
- package/lib/commands/prime.js +23 -3
- package/lib/commands/push.js +41 -51
- package/lib/commands/recall.js +60 -16
- package/lib/commands/recap.js +6 -1
- package/lib/commands/release.js +18 -4
- package/lib/commands/serve.js +5 -2
- package/lib/commands/setup.js +191 -95
- package/lib/commands/shepherd.js +49 -4
- package/lib/commands/ship.js +22 -23
- package/lib/commands/skill.js +383 -0
- package/lib/commands/status.js +54 -33
- package/lib/commands/test.js +56 -34
- package/lib/commands/worktree.js +247 -43
- package/lib/core/runtime-graph.js +89 -15
- package/lib/doc-assertions.js +297 -0
- package/lib/existing-tdd-gate.js +253 -0
- package/lib/forge-context.js +1 -4
- package/lib/forge-issues.js +64 -491
- package/lib/git-defaults.js +56 -0
- package/lib/harness-capability-matrix.js +5 -5
- package/lib/hook-renderer.js +147 -16
- package/lib/insights.js +96 -80
- package/lib/issue-backend.js +42 -3
- package/lib/kernel/backing-issue.js +14 -2
- package/lib/kernel/broker.js +44 -0
- package/lib/kernel/cli-broker-factory.js +12 -1
- package/lib/kernel/close-on-merge.js +154 -0
- package/lib/kernel/fs-class.js +42 -25
- package/lib/kernel/migrations.js +30 -2
- package/lib/kernel/schema.js +35 -0
- package/lib/kernel/sqlite-driver.js +292 -18
- package/lib/lefthook-wiring.js +21 -1
- package/lib/memory/router.js +16 -1
- package/lib/memory-digest.js +47 -15
- package/lib/memory-recall-events.js +145 -0
- package/lib/memory-recall.js +212 -0
- package/lib/merge-rules.js +8 -4
- package/lib/npm-publish-workflow.js +272 -0
- package/lib/orientation.js +371 -49
- package/lib/plugin-catalog.js +14 -4
- package/lib/pr-bundle.js +9 -6
- package/lib/pr-monitor/journal.js +18 -2
- package/lib/pr-monitor/reconcile-executor.js +842 -0
- package/lib/pr-monitor/reconcile-tick.js +138 -0
- package/lib/pr-monitor/reconcile.js +0 -0
- package/lib/pr-monitor/render-summary.js +196 -0
- package/lib/pr-monitor/shepherd-lease.js +252 -0
- package/lib/pr-monitor/watch-lifecycle.js +14 -2
- package/lib/pr-pull.js +98 -24
- package/lib/pr-shepherd.js +34 -8
- package/lib/preflight/gates.js +65 -18
- package/lib/preflight/runner.js +5 -0
- package/lib/project-memory.js +40 -0
- package/lib/protected-state-authority.js +305 -0
- package/lib/protected-state-surfaces.js +64 -44
- package/lib/release-readiness.js +51 -4
- package/lib/rules-sync.js +4 -0
- package/lib/runtime-health.js +15 -46
- package/lib/shell-utils.js +1 -1
- package/lib/skill-eval.js +750 -0
- package/lib/skills-sync.js +6 -3
- package/lib/smart-merge.js +28 -4
- package/lib/status/identity.js +46 -0
- package/lib/status/presenter.js +0 -35
- package/lib/status/snapshot.js +11 -16
- package/lib/symlink-utils.js +74 -26
- package/lib/upgrade-safety.js +47 -9
- package/lib/using-forge.js +328 -0
- package/lib/workflow/enforce-stage.js +5 -5
- package/lib/workflow/state-manager.js +23 -23
- package/package.json +6 -7
- package/rules/using-forge.md +24 -0
- package/scripts/doc-asserting-tests.js +158 -0
- package/scripts/forge-team/index.sh +0 -5
- package/scripts/forge-team/tests/dispatcher.test.sh +1 -1
- package/scripts/forge-team/tests/workflow-integration.test.sh +0 -1
- package/scripts/lib/behavioral-eval-runner.js +310 -0
- package/scripts/lib/behavioral-eval-runtime.js +456 -0
- package/scripts/lib/eval-evidence.js +328 -0
- package/scripts/lib/eval-runner.js +81 -41
- package/scripts/lib/immutable-eval-corpus.js +309 -0
- package/scripts/lib/promotion-evidence-loader.js +94 -0
- package/scripts/lib/promotion-scorecard.js +314 -0
- package/scripts/npm-release-receipt.js +134 -0
- package/scripts/process-tree.js +761 -0
- package/scripts/protected-state-check.js +47 -22
- package/scripts/run-command-eval.js +29 -1
- package/scripts/sync-d20-audit.js +172 -0
- package/scripts/test-full-suite.js +249 -37
- package/scripts/test.js +184 -44
- package/skills/claim-safety/SKILL.md +4 -0
- package/skills/claim-safety/evals/scorecard.json +41 -0
- package/skills/coverage.json +83 -0
- package/skills/dev/SKILL.md +4 -0
- package/skills/dev/evals/scorecard.json +41 -0
- package/skills/gates/SKILL.md +80 -0
- package/skills/gates/evals/evals.json +38 -0
- package/skills/gates/evals/scorecard.json +41 -0
- package/skills/hermes-forge/SKILL.md +1 -0
- package/skills/hermes-forge/evals/scorecard.json +41 -0
- package/skills/issue-basics/SKILL.md +1 -0
- package/skills/issue-basics/evals/scorecard.json +41 -0
- package/skills/kernel/SKILL.md +38 -0
- package/skills/kernel/evals/scorecard.json +41 -0
- package/skills/memory/SKILL.md +16 -1
- package/skills/memory/evals/scorecard.json +41 -0
- package/skills/parallel-deep-research/SKILL.md +1 -0
- package/skills/parallel-deep-research/evals/scorecard.json +41 -0
- package/skills/plan/SKILL.md +6 -0
- package/skills/plan/evals/scorecard.json +41 -0
- package/skills/portability/SKILL.md +47 -0
- package/skills/portability/evals/evals.json +34 -0
- package/skills/portability/evals/scorecard.json +41 -0
- package/skills/research/SKILL.md +1 -0
- package/skills/research/evals/scorecard.json +41 -0
- package/skills/review/SKILL.md +10 -11
- package/skills/review/evals/scorecard.json +41 -0
- package/skills/rollback/SKILL.md +5 -11
- package/skills/rollback/evals/scorecard.json +41 -0
- package/skills/setup/SKILL.md +91 -0
- package/skills/setup/evals/evals.json +42 -0
- package/skills/setup/evals/scorecard.json +41 -0
- package/skills/shepherd/SKILL.md +84 -38
- package/skills/shepherd/evals/evals.json +21 -9
- package/skills/shepherd/evals/scorecard.json +41 -0
- package/skills/ship/SKILL.md +10 -12
- package/skills/ship/evals/scorecard.json +41 -0
- package/skills/smith/SKILL.md +8 -0
- package/skills/smith/evals/scorecard.json +41 -0
- package/skills/sonarcloud/SKILL.md +1 -0
- package/skills/sonarcloud/evals/scorecard.json +41 -0
- package/skills/sonarcloud-analysis/SKILL.md +1 -0
- package/skills/sonarcloud-analysis/evals/scorecard.json +41 -0
- package/skills/status/SKILL.md +3 -0
- package/skills/status/evals/scorecard.json +41 -0
- package/skills/triage-ready/SKILL.md +2 -0
- package/skills/triage-ready/evals/scorecard.json +41 -0
- package/skills/using-forge/SKILL.md +104 -0
- package/skills/using-forge/evals/scorecard.json +41 -0
- package/skills/validate/SKILL.md +4 -0
- package/skills/validate/evals/scorecard.json +41 -0
- package/skills/verify/SKILL.md +4 -0
- package/skills/verify/evals/scorecard.json +41 -0
- package/skills/worktree/SKILL.md +92 -0
- package/skills/worktree/evals/evals.json +38 -0
- package/skills/worktree/evals/scorecard.json +41 -0
- package/lib/adapters/beads-issue-adapter.js +0 -127
- package/lib/beads-nudge.js +0 -91
- package/lib/beads-setup.js +0 -538
- package/lib/beads-sync-scaffold.js +0 -189
- package/lib/commands/board.js +0 -64
- package/lib/pat-setup.js +0 -207
- package/lib/pr-monitor/render-sticky.js +0 -192
- package/lib/pr-monitor/upsert-sticky.js +0 -169
- package/lib/status/beads-snapshot.js +0 -145
- package/scripts/beads-context.sh +0 -577
- package/scripts/beads-migrate-to-dolt.sh +0 -7
- package/scripts/beads-upgrade-smoke.sh +0 -284
- package/scripts/forge-team/lib/dashboard.sh +0 -316
- package/scripts/forge-team/tests/dashboard.test.sh +0 -155
- package/scripts/lib/beads-migrate-to-dolt.mjs +0 -503
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
'use strict';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Runs (or lists) the test suites that assert on changed markdown.
|
|
6
|
+
*
|
|
7
|
+
* A markdown-only PR does not match the `paths:` filter in
|
|
8
|
+
* `.github/workflows/test.yml`, so the full matrix never runs and
|
|
9
|
+
* `required-checks-bypass.yml` reports the required "Test Suite" check green.
|
|
10
|
+
* Several suites nevertheless assert on markdown CONTENT, so such a PR merged
|
|
11
|
+
* green and then broke master for the next unrelated code PR — twice (kernel
|
|
12
|
+
* issue 63556816: the README size badge in #307/#310, and the AGENTS.md
|
|
13
|
+
* convention test after #325).
|
|
14
|
+
*
|
|
15
|
+
* This script closes that gap by running exactly the doc-asserting suites for the
|
|
16
|
+
* changed markdown. It shares `lib/doc-assertions.js` with the local push lane
|
|
17
|
+
* (`lib/commands/test.js`), so CI and local selection cannot drift apart.
|
|
18
|
+
*
|
|
19
|
+
* Usage:
|
|
20
|
+
* node scripts/doc-asserting-tests.js [--base <ref>] [--list]
|
|
21
|
+
*
|
|
22
|
+
* --base <ref> Compare against <ref> (default: origin/<default-branch> or HEAD~1).
|
|
23
|
+
* --list Print the selected suites instead of running them.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
const { execFileSync, spawnSync } = require('node:child_process');
|
|
27
|
+
const fs = require('node:fs');
|
|
28
|
+
const path = require('node:path');
|
|
29
|
+
|
|
30
|
+
const { selectDocAssertingTests } = require('../lib/doc-assertions');
|
|
31
|
+
|
|
32
|
+
const REPO_ROOT = path.resolve(__dirname, '..');
|
|
33
|
+
|
|
34
|
+
/** Wall-clock ceiling for the doc-asserting lane; it is a small, fast subset. */
|
|
35
|
+
const LANE_TIMEOUT_MS = 5 * 60 * 1000;
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* Parses the supported command-line flags.
|
|
39
|
+
*
|
|
40
|
+
* @param {string[]} argv Raw arguments (without node/script).
|
|
41
|
+
* @returns {{base: string|null, list: boolean}} Parsed options.
|
|
42
|
+
*/
|
|
43
|
+
function parseArgs(argv) {
|
|
44
|
+
const baseIndex = argv.indexOf('--base');
|
|
45
|
+
return {
|
|
46
|
+
base: baseIndex !== -1 ? argv[baseIndex + 1] || null : null,
|
|
47
|
+
list: argv.includes('--list'),
|
|
48
|
+
};
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* Resolves the ref the PR should be compared against.
|
|
53
|
+
*
|
|
54
|
+
* @param {string|null} explicitBase Base supplied with `--base`.
|
|
55
|
+
* @returns {string} A git ref usable in `git diff <ref>...HEAD`.
|
|
56
|
+
*/
|
|
57
|
+
function resolveBase(explicitBase) {
|
|
58
|
+
if (explicitBase) return explicitBase;
|
|
59
|
+
for (const ref of ['origin/master', 'origin/main']) {
|
|
60
|
+
try {
|
|
61
|
+
execFileSync('git', ['rev-parse', '--verify', ref], { cwd: REPO_ROOT, stdio: 'pipe', timeout: 5000 });
|
|
62
|
+
return ref;
|
|
63
|
+
} catch (_e) { /* intentional: ref not present in this checkout, try next */ } // NOSONAR S2486
|
|
64
|
+
}
|
|
65
|
+
return 'HEAD~1';
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Parses `git diff --name-only` output into repository-relative paths.
|
|
70
|
+
*
|
|
71
|
+
* Split out so the parsing contract (trimmed, no blanks) is testable without a live
|
|
72
|
+
* repository: CI checks out shallow, so a test that reaches for `HEAD~1` passes locally
|
|
73
|
+
* and fails on the runner.
|
|
74
|
+
*
|
|
75
|
+
* @param {string} output Raw `git diff --name-only` stdout.
|
|
76
|
+
* @returns {string[]} Changed file paths.
|
|
77
|
+
*/
|
|
78
|
+
function parseChangedFiles(output) {
|
|
79
|
+
return output.split('\n').map((line) => line.trim()).filter(Boolean);
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
/**
|
|
83
|
+
* Lists repository-relative paths changed against the base ref.
|
|
84
|
+
*
|
|
85
|
+
* @param {string} base Base git ref.
|
|
86
|
+
* @returns {string[]} Changed file paths.
|
|
87
|
+
*/
|
|
88
|
+
function changedFilesSince(base) {
|
|
89
|
+
const output = execFileSync('git', ['diff', '--name-only', `${base}...HEAD`], {
|
|
90
|
+
cwd: REPO_ROOT,
|
|
91
|
+
encoding: 'utf8',
|
|
92
|
+
timeout: 15000,
|
|
93
|
+
});
|
|
94
|
+
return parseChangedFiles(output);
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* Turns a `spawnSync` result into a process exit status.
|
|
99
|
+
*
|
|
100
|
+
* A spawn that never produced a status (`error`, or killed by a signal so `status`
|
|
101
|
+
* is null) MUST report FAILURE. Reporting 0 there would re-create the bug this lane
|
|
102
|
+
* exists to remove: a required check reporting green without running the tests.
|
|
103
|
+
*
|
|
104
|
+
* @param {{status: number|null, error?: Error}} result Result of `spawnSync`.
|
|
105
|
+
* @returns {number} Exit status; non-zero whenever the run did not demonstrably pass.
|
|
106
|
+
*/
|
|
107
|
+
function resolveSpawnStatus(result) {
|
|
108
|
+
if (!result || result.error) return 1;
|
|
109
|
+
return result.status ?? 1;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function main() {
|
|
113
|
+
const { base: explicitBase, list } = parseArgs(process.argv.slice(2));
|
|
114
|
+
const base = resolveBase(explicitBase);
|
|
115
|
+
|
|
116
|
+
let changedFiles;
|
|
117
|
+
try {
|
|
118
|
+
changedFiles = changedFilesSince(base);
|
|
119
|
+
} catch (error) {
|
|
120
|
+
// Fail closed: if the change set cannot be determined we cannot prove the
|
|
121
|
+
// doc-asserting suites are unaffected.
|
|
122
|
+
console.error(`Could not compute changed files against ${base}: ${error.message}`);
|
|
123
|
+
return 1;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
const suites = selectDocAssertingTests(changedFiles, REPO_ROOT, fs);
|
|
127
|
+
|
|
128
|
+
if (suites.length === 0) {
|
|
129
|
+
console.log('No changed markdown asserts on by any test suite — nothing to run.');
|
|
130
|
+
return 0;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
if (list) {
|
|
134
|
+
console.log(suites.join('\n'));
|
|
135
|
+
return 0;
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
console.log(`Running ${suites.length} doc-asserting suite${suites.length === 1 ? '' : 's'} for changed markdown:`);
|
|
139
|
+
for (const suite of suites) console.log(` ${suite}`);
|
|
140
|
+
|
|
141
|
+
const result = spawnSync('bun', ['test', ...suites], {
|
|
142
|
+
cwd: REPO_ROOT,
|
|
143
|
+
stdio: 'inherit',
|
|
144
|
+
timeout: LANE_TIMEOUT_MS,
|
|
145
|
+
shell: process.platform === 'win32',
|
|
146
|
+
});
|
|
147
|
+
|
|
148
|
+
if (result.error) {
|
|
149
|
+
console.error(`Failed to run doc-asserting suites: ${result.error.message}`);
|
|
150
|
+
}
|
|
151
|
+
return resolveSpawnStatus(result);
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
if (require.main === module) {
|
|
155
|
+
process.exit(main());
|
|
156
|
+
}
|
|
157
|
+
|
|
158
|
+
module.exports = { changedFilesSince, parseArgs, parseChangedFiles, resolveBase, resolveSpawnStatus };
|
|
@@ -4,7 +4,6 @@
|
|
|
4
4
|
# Subcommands:
|
|
5
5
|
# workload Show team workload by developer
|
|
6
6
|
# epic Epic progress rollup
|
|
7
|
-
# dashboard Team health dashboard
|
|
8
7
|
# add Add developer to team map
|
|
9
8
|
# verify Check 1:1 Beads<>GitHub enforcement
|
|
10
9
|
# sync Manual GitHub<>Beads sync
|
|
@@ -22,7 +21,6 @@ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
|
22
21
|
[[ -f "$SCRIPT_DIR/lib/identity.sh" ]] && source "$SCRIPT_DIR/lib/identity.sh"
|
|
23
22
|
[[ -f "$SCRIPT_DIR/lib/workload.sh" ]] && source "$SCRIPT_DIR/lib/workload.sh"
|
|
24
23
|
[[ -f "$SCRIPT_DIR/lib/epic.sh" ]] && source "$SCRIPT_DIR/lib/epic.sh"
|
|
25
|
-
[[ -f "$SCRIPT_DIR/lib/dashboard.sh" ]] && source "$SCRIPT_DIR/lib/dashboard.sh"
|
|
26
24
|
[[ -f "$SCRIPT_DIR/lib/hooks.sh" ]] && source "$SCRIPT_DIR/lib/hooks.sh"
|
|
27
25
|
[[ -f "$SCRIPT_DIR/lib/verify.sh" ]] && source "$SCRIPT_DIR/lib/verify.sh"
|
|
28
26
|
[[ -f "$SCRIPT_DIR/lib/claim.sh" ]] && source "$SCRIPT_DIR/lib/claim.sh"
|
|
@@ -34,7 +32,6 @@ FORGE_SCRIPTS="$(cd "$SCRIPT_DIR/.." && pwd)"
|
|
|
34
32
|
# ── Stub implementations ──
|
|
35
33
|
# cmd_workload provided by lib/workload.sh
|
|
36
34
|
# cmd_epic provided by lib/epic.sh
|
|
37
|
-
# cmd_dashboard provided by lib/dashboard.sh
|
|
38
35
|
cmd_add() { auto_detect_identity "$@"; }
|
|
39
36
|
# cmd_verify provided by lib/verify.sh
|
|
40
37
|
cmd_sync() { forge_team_sync "$@"; }
|
|
@@ -47,7 +44,6 @@ Usage: forge team <subcommand> [args...]
|
|
|
47
44
|
Subcommands:
|
|
48
45
|
workload [--developer=<user>] [--me] Show team workload
|
|
49
46
|
epic <issue-id> Epic progress rollup
|
|
50
|
-
dashboard Team health dashboard
|
|
51
47
|
add [--github=<user>] Add developer to team map
|
|
52
48
|
verify Check 1:1 Beads<>GitHub sync
|
|
53
49
|
sync Manual GitHub<>Beads sync
|
|
@@ -69,7 +65,6 @@ main() {
|
|
|
69
65
|
case "$subcommand" in
|
|
70
66
|
workload) cmd_workload "$@" ;;
|
|
71
67
|
epic) cmd_epic "$@" ;;
|
|
72
|
-
dashboard) cmd_dashboard "$@" ;;
|
|
73
68
|
add) cmd_add "$@" ;;
|
|
74
69
|
verify) cmd_verify "$@" ;;
|
|
75
70
|
sync) cmd_sync "$@" ;;
|
|
@@ -60,7 +60,7 @@ assert_output_contains "unknown subcommand shows error" "unknown subcommand" bas
|
|
|
60
60
|
# Test 4: Each stub subcommand is reachable (exits 0)
|
|
61
61
|
echo ""
|
|
62
62
|
echo "-- stub subcommands reachable --"
|
|
63
|
-
for cmd in workload epic
|
|
63
|
+
for cmd in workload epic add verify sync claim; do
|
|
64
64
|
assert_exit "$cmd exits 0" 0 bash "$DISPATCHER" "$cmd"
|
|
65
65
|
assert_output_contains "$cmd outputs stub message" "not implemented" bash "$DISPATCHER" "$cmd"
|
|
66
66
|
done
|
|
@@ -17,7 +17,6 @@ assert_contains_file() {
|
|
|
17
17
|
echo "── /status integration ──"
|
|
18
18
|
STATUS="$SCRIPT_DIR/.claude/commands/status.md"
|
|
19
19
|
assert_contains_file "status calls workload --me" "forge-team.*workload.*--me" "$STATUS"
|
|
20
|
-
assert_contains_file "status calls dashboard" "forge-team.*dashboard" "$STATUS"
|
|
21
20
|
|
|
22
21
|
echo ""
|
|
23
22
|
echo "── /plan integration ──"
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
const { loadTier, evaluateCase } = require('./immutable-eval-corpus');
|
|
4
|
+
const { createEvalEvidence, appendEvalEvidence } = require('./eval-evidence');
|
|
5
|
+
|
|
6
|
+
const RESULT_FIELDS = Object.freeze(['evidence', 'attribution']);
|
|
7
|
+
const ATTRIBUTION_FIELDS = Object.freeze([
|
|
8
|
+
'model', 'effort', 'role', 'hashes', 'startedAt', 'endedAt', 'activeMs',
|
|
9
|
+
'passiveMs', 'tokens', 'retries', 'compactions',
|
|
10
|
+
]);
|
|
11
|
+
const ATTRIBUTION_HASH_FIELDS = Object.freeze(['prompt', 'skill', 'tool']);
|
|
12
|
+
const BINDING_FIELDS = Object.freeze(['repoSha', 'configHash', 'budgetHash']);
|
|
13
|
+
const ARM_FIELDS = Object.freeze(['id', 'model', 'config', 'budget']);
|
|
14
|
+
const STRUCTURAL_FAILURE_PREFIXES = Object.freeze([
|
|
15
|
+
'evidence.', 'binding.', 'case_id.', 'packet.', 'split.', 'trial.', 'metrics.',
|
|
16
|
+
'manifest.', 'observation.',
|
|
17
|
+
]);
|
|
18
|
+
const SAFE_RUNTIME_FAILURES = new Set([
|
|
19
|
+
'runtime.execution_failed', 'runtime.token_budget_exceeded', 'runtime.usage_unparseable',
|
|
20
|
+
]);
|
|
21
|
+
|
|
22
|
+
function hasExactFields(value, fields) {
|
|
23
|
+
if (!value || typeof value !== 'object' || Array.isArray(value)) return false;
|
|
24
|
+
const keys = Object.keys(value);
|
|
25
|
+
return keys.length === fields.length && fields.every((field) => Object.hasOwn(value, field));
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
function validExecutorResult(result) {
|
|
29
|
+
return hasExactFields(result, RESULT_FIELDS) &&
|
|
30
|
+
hasExactFields(result.attribution, ATTRIBUTION_FIELDS) &&
|
|
31
|
+
hasExactFields(result.attribution.hashes, ATTRIBUTION_HASH_FIELDS);
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
function isStructuralFailure(failure) {
|
|
35
|
+
return STRUCTURAL_FAILURE_PREFIXES.some((prefix) => failure.startsWith(prefix));
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
function snapshotBinding(binding) {
|
|
39
|
+
if (!hasExactFields(binding, BINDING_FIELDS)) return null;
|
|
40
|
+
if (!/^[0-9a-f]{40}$/.test(binding.repoSha)) return null;
|
|
41
|
+
if (!/^[0-9a-f]{64}$/.test(binding.configHash)) return null;
|
|
42
|
+
if (!/^[0-9a-f]{64}$/.test(binding.budgetHash)) return null;
|
|
43
|
+
return Object.freeze({
|
|
44
|
+
repoSha: binding.repoSha,
|
|
45
|
+
configHash: binding.configHash,
|
|
46
|
+
budgetHash: binding.budgetHash,
|
|
47
|
+
});
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
function snapshotArms(arms) {
|
|
51
|
+
if (!Array.isArray(arms) || arms.length !== 4) return null;
|
|
52
|
+
const snapshots = [];
|
|
53
|
+
for (const arm of arms) {
|
|
54
|
+
if (!hasExactFields(arm, ARM_FIELDS)) return null;
|
|
55
|
+
if ([arm.id, arm.model, arm.config, arm.budget]
|
|
56
|
+
.some((value) => typeof value !== 'string' || value.length === 0)) return null;
|
|
57
|
+
if (!['current', 'bounded'].includes(arm.config)) return null;
|
|
58
|
+
snapshots.push(Object.freeze({ ...arm }));
|
|
59
|
+
}
|
|
60
|
+
if (new Set(snapshots.map((arm) => arm.id)).size !== 4) return null;
|
|
61
|
+
if (new Set(snapshots.map((arm) => arm.model)).size !== 2) return null;
|
|
62
|
+
const matrix = new Set(snapshots.map((arm) => `${arm.model}\0${arm.config}`));
|
|
63
|
+
if (matrix.size !== 4) return null;
|
|
64
|
+
return Object.freeze(snapshots);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
function explicitHardFailure(evaluation, result) {
|
|
68
|
+
return result.evidence?.observation?.hardFailure === true
|
|
69
|
+
|| evaluation.hardFailure === true;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function buildEnvelope(input, identity, binding, evaluation, result, hardFailure) {
|
|
73
|
+
const attribution = result.attribution;
|
|
74
|
+
return createEvalEvidence({
|
|
75
|
+
issue_id: input.issueId,
|
|
76
|
+
pr: input.pr,
|
|
77
|
+
head_sha: binding.repoSha,
|
|
78
|
+
model: attribution.model,
|
|
79
|
+
effort: attribution.effort,
|
|
80
|
+
role: attribution.role,
|
|
81
|
+
hashes: {
|
|
82
|
+
eval_set: result.evidence.packetHash,
|
|
83
|
+
prompt: attribution.hashes.prompt,
|
|
84
|
+
skill: attribution.hashes.skill,
|
|
85
|
+
tool: attribution.hashes.tool,
|
|
86
|
+
},
|
|
87
|
+
started_at: attribution.startedAt,
|
|
88
|
+
ended_at: attribution.endedAt,
|
|
89
|
+
active_ms: attribution.activeMs,
|
|
90
|
+
passive_ms: attribution.passiveMs,
|
|
91
|
+
tokens: {
|
|
92
|
+
input: attribution.tokens.input,
|
|
93
|
+
output: attribution.tokens.output,
|
|
94
|
+
cached: attribution.tokens.cached,
|
|
95
|
+
},
|
|
96
|
+
retries: attribution.retries,
|
|
97
|
+
compactions: attribution.compactions,
|
|
98
|
+
gates: [{ name: 'behavioral-case', passed: evaluation.passed }],
|
|
99
|
+
run_identity: {
|
|
100
|
+
arm_id: identity.armId,
|
|
101
|
+
case_id: identity.caseId,
|
|
102
|
+
risk: identity.risk,
|
|
103
|
+
split: identity.split,
|
|
104
|
+
model: identity.model,
|
|
105
|
+
config: identity.config,
|
|
106
|
+
budget: identity.budget,
|
|
107
|
+
tier: input.tier,
|
|
108
|
+
trial_index: identity.trialIndex,
|
|
109
|
+
config_hash: binding.configHash,
|
|
110
|
+
budget_hash: binding.budgetHash,
|
|
111
|
+
},
|
|
112
|
+
case_result: {
|
|
113
|
+
status: evaluation.passed ? 'PASS' : 'FAIL',
|
|
114
|
+
hard_failure: hardFailure,
|
|
115
|
+
latency_ms: result.evidence.metrics.durationMs,
|
|
116
|
+
tokens: result.evidence.metrics.tokensUsed,
|
|
117
|
+
},
|
|
118
|
+
});
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function finding(identity, status, failures, evidence, hardFailure = false) {
|
|
122
|
+
return {
|
|
123
|
+
caseId: identity.caseId,
|
|
124
|
+
risk: identity.risk,
|
|
125
|
+
split: identity.split,
|
|
126
|
+
model: identity.model,
|
|
127
|
+
config: identity.config,
|
|
128
|
+
budget: identity.budget,
|
|
129
|
+
trialIndex: identity.trialIndex,
|
|
130
|
+
status,
|
|
131
|
+
hardFailure,
|
|
132
|
+
latencyMs: Number.isFinite(evidence?.metrics?.durationMs) ? evidence.metrics.durationMs : 0,
|
|
133
|
+
tokens: Number.isFinite(evidence?.metrics?.tokensUsed) ? evidence.metrics.tokensUsed : 0,
|
|
134
|
+
failures,
|
|
135
|
+
};
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
function incompleteResult(tier, arms, expectedRuns, reason) {
|
|
139
|
+
return {
|
|
140
|
+
status: 'INCOMPLETE',
|
|
141
|
+
tier,
|
|
142
|
+
arms,
|
|
143
|
+
expectedRuns,
|
|
144
|
+
completedRuns: 0,
|
|
145
|
+
passedRuns: 0,
|
|
146
|
+
failedRuns: 0,
|
|
147
|
+
incompleteRuns: expectedRuns,
|
|
148
|
+
findings: [{ status: 'INCOMPLETE', failures: [reason] }],
|
|
149
|
+
};
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
function identityFor(packet, trialIndex, arm) {
|
|
153
|
+
return {
|
|
154
|
+
armId: arm.id,
|
|
155
|
+
caseId: packet.caseId,
|
|
156
|
+
risk: packet.risk,
|
|
157
|
+
split: packet.split,
|
|
158
|
+
model: arm.model,
|
|
159
|
+
config: arm.config,
|
|
160
|
+
budget: arm.budget,
|
|
161
|
+
trialIndex,
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
async function invokeExecutor(input, packet, trialIndex, arm, binding) {
|
|
166
|
+
try {
|
|
167
|
+
return await input.executor(Object.freeze({
|
|
168
|
+
armId: arm.id,
|
|
169
|
+
model: arm.model,
|
|
170
|
+
config: arm.config,
|
|
171
|
+
budget: arm.budget,
|
|
172
|
+
packet,
|
|
173
|
+
trialIndex,
|
|
174
|
+
binding,
|
|
175
|
+
skillName: input.skillName,
|
|
176
|
+
}));
|
|
177
|
+
} catch (error) {
|
|
178
|
+
const failure = error?.message;
|
|
179
|
+
return SAFE_RUNTIME_FAILURES.has(failure) ? { runtimeFailure: failure } : null;
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
async function persistEvidence(input, append, identity, binding, evaluation, result, hardFailure) {
|
|
184
|
+
try {
|
|
185
|
+
const envelope = buildEnvelope(input, identity, binding, evaluation, result, hardFailure);
|
|
186
|
+
const appended = await append(input.projectRoot, envelope, input.appendOptions || {});
|
|
187
|
+
if (appended?.conflict) return 'evidence.conflict';
|
|
188
|
+
if (!appended || appended.ok !== true) return 'evidence.append_failed';
|
|
189
|
+
return null;
|
|
190
|
+
} catch {
|
|
191
|
+
return 'evidence.append_failed';
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
async function executeArm({ input, corpus, packet, trialIndex, arm, binding, append }) {
|
|
196
|
+
const identity = identityFor(packet, trialIndex, arm);
|
|
197
|
+
const result = await invokeExecutor(input, packet, trialIndex, arm, binding);
|
|
198
|
+
if (result?.runtimeFailure) {
|
|
199
|
+
return {
|
|
200
|
+
status: 'INCOMPLETE',
|
|
201
|
+
finding: finding(identity, 'INCOMPLETE', [result.runtimeFailure]),
|
|
202
|
+
};
|
|
203
|
+
}
|
|
204
|
+
if (!validExecutorResult(result)) {
|
|
205
|
+
return { status: 'INCOMPLETE', finding: finding(identity, 'INCOMPLETE', ['evidence.malformed']) };
|
|
206
|
+
}
|
|
207
|
+
if (result.attribution.model !== arm.model) {
|
|
208
|
+
return {
|
|
209
|
+
status: 'INCOMPLETE',
|
|
210
|
+
finding: finding(identity, 'INCOMPLETE', ['attribution.model_mismatch'], result.evidence),
|
|
211
|
+
};
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
const evaluation = evaluateCase({
|
|
215
|
+
packet,
|
|
216
|
+
allPackets: corpus.allPackets,
|
|
217
|
+
manifest: corpus.manifest,
|
|
218
|
+
evidence: result.evidence,
|
|
219
|
+
expectedBinding: binding,
|
|
220
|
+
});
|
|
221
|
+
const hardFailure = explicitHardFailure(evaluation, result);
|
|
222
|
+
if (evaluation.failures.some(isStructuralFailure)) {
|
|
223
|
+
return {
|
|
224
|
+
status: 'INCOMPLETE',
|
|
225
|
+
finding: finding(identity, 'INCOMPLETE', evaluation.failures, result.evidence, hardFailure),
|
|
226
|
+
};
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
const persistenceFailure = await persistEvidence(
|
|
230
|
+
input, append, identity, binding, evaluation, result, hardFailure,
|
|
231
|
+
);
|
|
232
|
+
if (persistenceFailure) {
|
|
233
|
+
return {
|
|
234
|
+
status: 'INCOMPLETE',
|
|
235
|
+
finding: finding(identity, 'INCOMPLETE', [persistenceFailure], result.evidence, hardFailure),
|
|
236
|
+
};
|
|
237
|
+
}
|
|
238
|
+
const status = evaluation.passed ? 'PASS' : 'FAIL';
|
|
239
|
+
return {
|
|
240
|
+
status,
|
|
241
|
+
finding: finding(
|
|
242
|
+
identity, status, evaluation.passed ? [] : evaluation.failures, result.evidence, hardFailure,
|
|
243
|
+
),
|
|
244
|
+
};
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
function recordOutcome(counts, outcome, findings) {
|
|
248
|
+
findings.push(outcome.finding);
|
|
249
|
+
if (outcome.status === 'INCOMPLETE') {
|
|
250
|
+
counts.incompleteRuns += 1;
|
|
251
|
+
return;
|
|
252
|
+
}
|
|
253
|
+
counts.completedRuns += 1;
|
|
254
|
+
if (outcome.status === 'PASS') counts.passedRuns += 1;
|
|
255
|
+
else counts.failedRuns += 1;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
function finalStatus(counts, expectedRuns) {
|
|
259
|
+
if (counts.incompleteRuns > 0 || counts.completedRuns !== expectedRuns) return 'INCOMPLETE';
|
|
260
|
+
return counts.failedRuns > 0 ? 'FAIL' : 'PASS';
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
/**
|
|
264
|
+
* Execute the frozen corpus through four opaque, matched executor arms.
|
|
265
|
+
* The executor may observe arm ids and immutable packets, but it receives no
|
|
266
|
+
* merge capability. Only a strict, privacy-safe attribution envelope persists.
|
|
267
|
+
*/
|
|
268
|
+
async function runBehavioralEvaluation(input) {
|
|
269
|
+
let corpus;
|
|
270
|
+
try {
|
|
271
|
+
corpus = loadTier(input.tier);
|
|
272
|
+
} catch (error) {
|
|
273
|
+
return incompleteResult(input.tier, input.arms || [], 0, error.message);
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
const suppliedArms = Array.isArray(input.arms) ? input.arms : [];
|
|
277
|
+
const trialIndices = corpus.manifest.trialIndices;
|
|
278
|
+
const expectedRuns = corpus.cases.length * trialIndices.length * 4;
|
|
279
|
+
const arms = snapshotArms(suppliedArms);
|
|
280
|
+
if (!arms) return incompleteResult(input.tier, suppliedArms, expectedRuns, 'arms.invalid');
|
|
281
|
+
if (typeof input.executor !== 'function') {
|
|
282
|
+
return incompleteResult(input.tier, arms, expectedRuns, 'executor.missing');
|
|
283
|
+
}
|
|
284
|
+
const binding = snapshotBinding(input.binding);
|
|
285
|
+
if (!binding) return incompleteResult(input.tier, arms, expectedRuns, 'binding.invalid');
|
|
286
|
+
|
|
287
|
+
const append = input.appendEvidence || appendEvalEvidence;
|
|
288
|
+
const findings = [];
|
|
289
|
+
const counts = { completedRuns: 0, passedRuns: 0, failedRuns: 0, incompleteRuns: 0 };
|
|
290
|
+
|
|
291
|
+
for (const packet of corpus.cases) {
|
|
292
|
+
for (const trialIndex of trialIndices) {
|
|
293
|
+
for (const arm of arms) {
|
|
294
|
+
const outcome = await executeArm({ input, corpus, packet, trialIndex, arm, binding, append });
|
|
295
|
+
recordOutcome(counts, outcome, findings);
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
return {
|
|
301
|
+
status: finalStatus(counts, expectedRuns),
|
|
302
|
+
tier: input.tier,
|
|
303
|
+
arms,
|
|
304
|
+
expectedRuns,
|
|
305
|
+
...counts,
|
|
306
|
+
findings,
|
|
307
|
+
};
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
module.exports = { runBehavioralEvaluation };
|