forge-workflow 0.1.0-beta.4 → 0.1.0-beta.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +18 -7
- package/CHANGELOG.md +79 -1
- package/CLAUDE.md +0 -12
- package/CODING_STANDARDS.md +72 -0
- package/README.md +6 -2
- package/bin/forge-cmd.js +20 -0
- package/bin/forge.js +28 -375
- package/docs/INDEX.md +1 -1
- package/docs/guides/BEADS_GITHUB_SYNC.md +2 -31
- package/docs/guides/MIGRATION.md +4 -4
- package/docs/guides/SETUP.md +16 -16
- package/docs/reference/COMMANDS.md +8 -5
- package/docs/reference/FORGE_KERNEL_STORAGE_MODEL.md +4 -0
- package/docs/reference/INSIGHTS_RECAP.md +9 -20
- package/docs/reference/INSTALL.md +4 -0
- package/docs/reference/LEGACY_CLAIM_REPAIR.md +112 -0
- package/docs/reference/RELEASE.md +5 -3
- package/docs/reference/TOOLCHAIN.md +8 -0
- package/docs/reference/github-accounts.md +134 -0
- package/docs/reference/protected-state-surfaces.md +4 -4
- package/docs/reference/shepherd.md +114 -35
- package/lefthook.yml +12 -0
- package/lib/activation/ensure-forge-home.js +33 -15
- package/lib/adapters/pr-state-adapter.js +359 -144
- package/lib/audit-evidence.js +71 -110
- package/lib/base-remote.js +138 -0
- package/lib/beta5-compatibility-evidence.js +1093 -0
- package/lib/bun-lockfile-proof.js +413 -0
- package/lib/bun-workflow-pins.js +461 -0
- package/lib/capabilities/index.js +9 -0
- package/lib/capabilities/model.js +141 -0
- package/lib/capabilities/probes.js +347 -0
- package/lib/capped-jsonl-log.js +236 -0
- package/lib/codex-skills.js +2 -2
- package/lib/commands/_manifest.js +1 -0
- package/lib/commands/_registry.js +50 -20
- package/lib/commands/clean.js +252 -32
- package/lib/commands/dev.js +4 -33
- package/lib/commands/doctor.js +37 -6
- package/lib/commands/gate.js +197 -27
- package/lib/commands/github.js +215 -0
- package/lib/commands/hooks.js +276 -30
- package/lib/commands/insights.js +8 -3
- package/lib/commands/memory.js +66 -2
- package/lib/commands/merge.js +1265 -58
- package/lib/commands/plan.js +33 -2
- package/lib/commands/pr.js +3 -1
- package/lib/commands/preflight.js +21 -4
- package/lib/commands/prime.js +21 -8
- package/lib/commands/push.js +146 -54
- package/lib/commands/recall.js +127 -49
- package/lib/commands/recap.js +6 -1
- package/lib/commands/release.js +39 -3
- package/lib/commands/remember.js +28 -4
- package/lib/commands/serve.js +26 -9
- package/lib/commands/setup.js +323 -98
- package/lib/commands/shepherd.js +591 -73
- package/lib/commands/ship.js +36 -91
- package/lib/commands/skill.js +127 -11
- package/lib/commands/status.js +17 -1
- package/lib/commands/team.js +47 -8
- package/lib/commands/test.js +187 -38
- package/lib/commands/validate.js +65 -21
- package/lib/commands/worktree.js +359 -45
- package/lib/core/runtime-graph.js +1 -1
- package/lib/doc-assertions.js +297 -0
- package/lib/existing-tdd-gate.js +253 -0
- package/lib/fixtures/beta5-corpus/v1/README.md +9 -0
- package/lib/fixtures/beta5-corpus/v1/contract/command-contract.json +26 -0
- package/lib/fixtures/beta5-corpus/v1/contract/package-contract.json +13 -0
- package/lib/fixtures/beta5-corpus/v1/contract/workflow-stage-matrix.json +8 -0
- package/lib/fixtures/beta5-corpus/v1/manifest.json +25 -0
- package/lib/fixtures/beta5-corpus/v1/state/comments.jsonl +1 -0
- package/lib/fixtures/beta5-corpus/v1/state/config.yaml +6 -0
- package/lib/fixtures/beta5-corpus/v1/state/dependencies.jsonl +1 -0
- package/lib/fixtures/beta5-corpus/v1/state/issues.jsonl +2 -0
- package/lib/fixtures/beta5-corpus/v1/state/kernel.sql +20 -0
- package/lib/forge-context.js +1 -4
- package/lib/forge-issues.js +134 -32
- package/lib/gate-events.js +98 -10
- package/lib/git-defaults.js +56 -0
- package/lib/github-context.js +308 -0
- package/lib/global-flags.js +1 -0
- package/lib/harness-capability-matrix.js +3 -3
- package/lib/hook-renderer.js +122 -5
- package/lib/insights.js +96 -80
- package/lib/issue-render.js +19 -0
- package/lib/kernel/backing-issue.js +14 -2
- package/lib/kernel/broker.js +739 -31
- package/lib/kernel/claim-reconciler.js +238 -0
- package/lib/kernel/cli-broker-factory.js +12 -1
- package/lib/kernel/close-on-merge.js +154 -0
- package/lib/kernel/fs-class.js +42 -25
- package/lib/kernel/lease-enforcer.js +9 -4
- package/lib/kernel/legacy-claim-repair.js +442 -0
- package/lib/kernel/live-claim-projection.js +26 -0
- package/lib/kernel/migrations.js +118 -3
- package/lib/kernel/readiness-model.js +184 -12
- package/lib/kernel/schema.js +49 -1
- package/lib/kernel/sqlite-driver.js +3435 -172
- package/lib/kernel/taxonomy-validator.js +4 -1
- package/lib/kernel/windows-private-acl.js +239 -0
- package/lib/lefthook-wiring.js +21 -1
- package/lib/memory/hygiene.js +191 -0
- package/lib/memory/router.js +110 -28
- package/lib/memory/usage-evidence.js +4 -0
- package/lib/memory-digest.js +106 -15
- package/lib/memory-recall-events.js +145 -0
- package/lib/memory-recall.js +71 -10
- package/lib/merge-rules.js +143 -21
- package/lib/npm-publish-workflow.js +465 -0
- package/lib/orientation.js +68 -43
- package/lib/package-root.js +2 -0
- package/lib/plugin-catalog.js +14 -4
- package/lib/pr-bundle.js +5 -6
- package/lib/pr-monitor/auto-actions.js +169 -28
- package/lib/pr-monitor/differ.js +110 -4
- package/lib/pr-monitor/events.js +0 -0
- package/lib/pr-monitor/flow-monitor.js +1424 -0
- package/lib/pr-monitor/gather.js +251 -44
- package/lib/pr-monitor/journal.js +18 -39
- package/lib/pr-monitor/monitor.js +117 -10
- package/lib/pr-monitor/process-identity.js +117 -0
- package/lib/pr-monitor/reconcile-executor.js +1129 -470
- package/lib/pr-monitor/reconcile.js +0 -0
- package/lib/pr-monitor/render-summary.js +293 -0
- package/lib/pr-monitor/review-preflight.js +269 -0
- package/lib/pr-monitor/shepherd-lease.js +38 -20
- package/lib/pr-monitor/verdict.js +438 -0
- package/lib/pr-monitor/watch-lifecycle.js +145 -27
- package/lib/pr-monitor/watch-owner.js +1414 -0
- package/lib/pr-monitor/watch.js +129 -58
- package/lib/pr-pull.js +33 -14
- package/lib/pr-shepherd.js +51 -11
- package/lib/preflight/gates.js +65 -18
- package/lib/preflight/runner.js +5 -0
- package/lib/project-memory.js +178 -4
- package/lib/protected-state-authority.js +1100 -0
- package/lib/protected-state-surfaces.js +243 -45
- package/lib/release-readiness.js +53 -7
- package/lib/review-adapter.js +65 -0
- package/lib/shell-utils.js +1 -1
- package/lib/skills-sync.js +71 -35
- package/lib/smart-merge.js +28 -4
- package/lib/symlink-utils.js +74 -26
- package/lib/upgrade-safety.js +39 -0
- package/lib/using-forge.js +19 -6
- package/lib/validation/risk-manifest.js +339 -0
- package/lib/workflow/enforce-stage.js +44 -0
- package/lib/workflow/plan-authority.js +225 -0
- package/package.json +12 -9
- package/scripts/commitlint.js +13 -15
- package/scripts/doc-asserting-tests.js +158 -0
- package/scripts/generate-risk-manifest.js +91 -0
- package/scripts/github-context-bridge.sh +10 -0
- package/scripts/legacy-claim-repair.js +145 -0
- package/scripts/lib/behavioral-eval-runner.js +310 -0
- package/scripts/lib/behavioral-eval-runtime.js +457 -0
- package/scripts/lib/eval-evidence.js +328 -0
- package/scripts/lib/eval-runner.js +81 -41
- package/scripts/lib/immutable-eval-corpus.js +309 -0
- package/scripts/lib/promotion-evidence-loader.js +94 -0
- package/scripts/lib/promotion-scorecard.js +314 -0
- package/scripts/npm-release-receipt.js +134 -0
- package/scripts/process-tree.js +773 -0
- package/scripts/protected-state-check.js +479 -31
- package/scripts/run-command-eval.js +29 -1
- package/scripts/sync-agent-skills.js +333 -34
- package/scripts/sync-d20-audit.js +172 -0
- package/scripts/test-full-suite.js +935 -37
- package/scripts/test-profile.js +13 -3
- package/scripts/test.js +271 -57
- package/skills/coverage.json +1 -0
- package/skills/review/SKILL.md +6 -11
- package/skills/review/evals/scorecard.json +4 -4
- package/skills/rollback/SKILL.md +4 -11
- package/skills/rollback/evals/scorecard.json +3 -3
- package/skills/setup/SKILL.md +18 -0
- package/skills/setup/evals/scorecard.json +3 -3
- package/skills/shepherd/SKILL.md +39 -16
- package/skills/shepherd/evals/scorecard.json +4 -4
- package/skills/ship/SKILL.md +4 -12
- package/skills/ship/evals/scorecard.json +3 -3
- package/skills/validate/SKILL.md +3 -0
- package/skills/validate/evals/scorecard.json +1 -1
- package/skills/worktree/SKILL.md +6 -1
- package/skills/worktree/evals/scorecard.json +2 -2
- package/lib/beads-setup.js +0 -538
- package/lib/beads-sync-scaffold.js +0 -189
- package/lib/pat-setup.js +0 -207
- package/lib/pr-monitor/render-sticky.js +0 -206
- package/lib/pr-monitor/upsert-sticky.js +0 -169
- package/scripts/beads-context.sh +0 -577
- package/scripts/beads-migrate-to-dolt.sh +0 -7
- package/scripts/beads-upgrade-smoke.sh +0 -284
- package/scripts/lib/beads-migrate-to-dolt.mjs +0 -503
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
const { loadTier, evaluateCase } = require('./immutable-eval-corpus');
|
|
4
|
+
const { createEvalEvidence, appendEvalEvidence } = require('./eval-evidence');
|
|
5
|
+
|
|
6
|
+
const RESULT_FIELDS = Object.freeze(['evidence', 'attribution']);
|
|
7
|
+
const ATTRIBUTION_FIELDS = Object.freeze([
|
|
8
|
+
'model', 'effort', 'role', 'hashes', 'startedAt', 'endedAt', 'activeMs',
|
|
9
|
+
'passiveMs', 'tokens', 'retries', 'compactions',
|
|
10
|
+
]);
|
|
11
|
+
const ATTRIBUTION_HASH_FIELDS = Object.freeze(['prompt', 'skill', 'tool']);
|
|
12
|
+
const BINDING_FIELDS = Object.freeze(['repoSha', 'configHash', 'budgetHash']);
|
|
13
|
+
const ARM_FIELDS = Object.freeze(['id', 'model', 'config', 'budget']);
|
|
14
|
+
const STRUCTURAL_FAILURE_PREFIXES = Object.freeze([
|
|
15
|
+
'evidence.', 'binding.', 'case_id.', 'packet.', 'split.', 'trial.', 'metrics.',
|
|
16
|
+
'manifest.', 'observation.',
|
|
17
|
+
]);
|
|
18
|
+
const SAFE_RUNTIME_FAILURES = new Set([
|
|
19
|
+
'runtime.execution_failed', 'runtime.token_budget_exceeded', 'runtime.usage_unparseable',
|
|
20
|
+
]);
|
|
21
|
+
|
|
22
|
+
function hasExactFields(value, fields) {
|
|
23
|
+
if (!value || typeof value !== 'object' || Array.isArray(value)) return false;
|
|
24
|
+
const keys = Object.keys(value);
|
|
25
|
+
return keys.length === fields.length && fields.every((field) => Object.hasOwn(value, field));
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
function validExecutorResult(result) {
|
|
29
|
+
return hasExactFields(result, RESULT_FIELDS) &&
|
|
30
|
+
hasExactFields(result.attribution, ATTRIBUTION_FIELDS) &&
|
|
31
|
+
hasExactFields(result.attribution.hashes, ATTRIBUTION_HASH_FIELDS);
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
function isStructuralFailure(failure) {
|
|
35
|
+
return STRUCTURAL_FAILURE_PREFIXES.some((prefix) => failure.startsWith(prefix));
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
function snapshotBinding(binding) {
|
|
39
|
+
if (!hasExactFields(binding, BINDING_FIELDS)) return null;
|
|
40
|
+
if (!/^[0-9a-f]{40}$/.test(binding.repoSha)) return null;
|
|
41
|
+
if (!/^[0-9a-f]{64}$/.test(binding.configHash)) return null;
|
|
42
|
+
if (!/^[0-9a-f]{64}$/.test(binding.budgetHash)) return null;
|
|
43
|
+
return Object.freeze({
|
|
44
|
+
repoSha: binding.repoSha,
|
|
45
|
+
configHash: binding.configHash,
|
|
46
|
+
budgetHash: binding.budgetHash,
|
|
47
|
+
});
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
function snapshotArms(arms) {
|
|
51
|
+
if (!Array.isArray(arms) || arms.length !== 4) return null;
|
|
52
|
+
const snapshots = [];
|
|
53
|
+
for (const arm of arms) {
|
|
54
|
+
if (!hasExactFields(arm, ARM_FIELDS)) return null;
|
|
55
|
+
if ([arm.id, arm.model, arm.config, arm.budget]
|
|
56
|
+
.some((value) => typeof value !== 'string' || value.length === 0)) return null;
|
|
57
|
+
if (!['current', 'bounded'].includes(arm.config)) return null;
|
|
58
|
+
snapshots.push(Object.freeze({ ...arm }));
|
|
59
|
+
}
|
|
60
|
+
if (new Set(snapshots.map((arm) => arm.id)).size !== 4) return null;
|
|
61
|
+
if (new Set(snapshots.map((arm) => arm.model)).size !== 2) return null;
|
|
62
|
+
const matrix = new Set(snapshots.map((arm) => `${arm.model}\0${arm.config}`));
|
|
63
|
+
if (matrix.size !== 4) return null;
|
|
64
|
+
return Object.freeze(snapshots);
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
function explicitHardFailure(evaluation, result) {
|
|
68
|
+
return result.evidence?.observation?.hardFailure === true
|
|
69
|
+
|| evaluation.hardFailure === true;
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
function buildEnvelope(input, identity, binding, evaluation, result, hardFailure) {
|
|
73
|
+
const attribution = result.attribution;
|
|
74
|
+
return createEvalEvidence({
|
|
75
|
+
issue_id: input.issueId,
|
|
76
|
+
pr: input.pr,
|
|
77
|
+
head_sha: binding.repoSha,
|
|
78
|
+
model: attribution.model,
|
|
79
|
+
effort: attribution.effort,
|
|
80
|
+
role: attribution.role,
|
|
81
|
+
hashes: {
|
|
82
|
+
eval_set: result.evidence.packetHash,
|
|
83
|
+
prompt: attribution.hashes.prompt,
|
|
84
|
+
skill: attribution.hashes.skill,
|
|
85
|
+
tool: attribution.hashes.tool,
|
|
86
|
+
},
|
|
87
|
+
started_at: attribution.startedAt,
|
|
88
|
+
ended_at: attribution.endedAt,
|
|
89
|
+
active_ms: attribution.activeMs,
|
|
90
|
+
passive_ms: attribution.passiveMs,
|
|
91
|
+
tokens: {
|
|
92
|
+
input: attribution.tokens.input,
|
|
93
|
+
output: attribution.tokens.output,
|
|
94
|
+
cached: attribution.tokens.cached,
|
|
95
|
+
},
|
|
96
|
+
retries: attribution.retries,
|
|
97
|
+
compactions: attribution.compactions,
|
|
98
|
+
gates: [{ name: 'behavioral-case', passed: evaluation.passed }],
|
|
99
|
+
run_identity: {
|
|
100
|
+
arm_id: identity.armId,
|
|
101
|
+
case_id: identity.caseId,
|
|
102
|
+
risk: identity.risk,
|
|
103
|
+
split: identity.split,
|
|
104
|
+
model: identity.model,
|
|
105
|
+
config: identity.config,
|
|
106
|
+
budget: identity.budget,
|
|
107
|
+
tier: input.tier,
|
|
108
|
+
trial_index: identity.trialIndex,
|
|
109
|
+
config_hash: binding.configHash,
|
|
110
|
+
budget_hash: binding.budgetHash,
|
|
111
|
+
},
|
|
112
|
+
case_result: {
|
|
113
|
+
status: evaluation.passed ? 'PASS' : 'FAIL',
|
|
114
|
+
hard_failure: hardFailure,
|
|
115
|
+
latency_ms: result.evidence.metrics.durationMs,
|
|
116
|
+
tokens: result.evidence.metrics.tokensUsed,
|
|
117
|
+
},
|
|
118
|
+
});
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function finding(identity, status, failures, evidence, hardFailure = false) {
|
|
122
|
+
return {
|
|
123
|
+
caseId: identity.caseId,
|
|
124
|
+
risk: identity.risk,
|
|
125
|
+
split: identity.split,
|
|
126
|
+
model: identity.model,
|
|
127
|
+
config: identity.config,
|
|
128
|
+
budget: identity.budget,
|
|
129
|
+
trialIndex: identity.trialIndex,
|
|
130
|
+
status,
|
|
131
|
+
hardFailure,
|
|
132
|
+
latencyMs: Number.isFinite(evidence?.metrics?.durationMs) ? evidence.metrics.durationMs : 0,
|
|
133
|
+
tokens: Number.isFinite(evidence?.metrics?.tokensUsed) ? evidence.metrics.tokensUsed : 0,
|
|
134
|
+
failures,
|
|
135
|
+
};
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
function incompleteResult(tier, arms, expectedRuns, reason) {
|
|
139
|
+
return {
|
|
140
|
+
status: 'INCOMPLETE',
|
|
141
|
+
tier,
|
|
142
|
+
arms,
|
|
143
|
+
expectedRuns,
|
|
144
|
+
completedRuns: 0,
|
|
145
|
+
passedRuns: 0,
|
|
146
|
+
failedRuns: 0,
|
|
147
|
+
incompleteRuns: expectedRuns,
|
|
148
|
+
findings: [{ status: 'INCOMPLETE', failures: [reason] }],
|
|
149
|
+
};
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
function identityFor(packet, trialIndex, arm) {
|
|
153
|
+
return {
|
|
154
|
+
armId: arm.id,
|
|
155
|
+
caseId: packet.caseId,
|
|
156
|
+
risk: packet.risk,
|
|
157
|
+
split: packet.split,
|
|
158
|
+
model: arm.model,
|
|
159
|
+
config: arm.config,
|
|
160
|
+
budget: arm.budget,
|
|
161
|
+
trialIndex,
|
|
162
|
+
};
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
async function invokeExecutor(input, packet, trialIndex, arm, binding) {
|
|
166
|
+
try {
|
|
167
|
+
return await input.executor(Object.freeze({
|
|
168
|
+
armId: arm.id,
|
|
169
|
+
model: arm.model,
|
|
170
|
+
config: arm.config,
|
|
171
|
+
budget: arm.budget,
|
|
172
|
+
packet,
|
|
173
|
+
trialIndex,
|
|
174
|
+
binding,
|
|
175
|
+
skillName: input.skillName,
|
|
176
|
+
}));
|
|
177
|
+
} catch (error) {
|
|
178
|
+
const failure = error?.message;
|
|
179
|
+
return SAFE_RUNTIME_FAILURES.has(failure) ? { runtimeFailure: failure } : null;
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
async function persistEvidence(input, append, identity, binding, evaluation, result, hardFailure) {
|
|
184
|
+
try {
|
|
185
|
+
const envelope = buildEnvelope(input, identity, binding, evaluation, result, hardFailure);
|
|
186
|
+
const appended = await append(input.projectRoot, envelope, input.appendOptions || {});
|
|
187
|
+
if (appended?.conflict) return 'evidence.conflict';
|
|
188
|
+
if (!appended || appended.ok !== true) return 'evidence.append_failed';
|
|
189
|
+
return null;
|
|
190
|
+
} catch {
|
|
191
|
+
return 'evidence.append_failed';
|
|
192
|
+
}
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
async function executeArm({ input, corpus, packet, trialIndex, arm, binding, append }) {
|
|
196
|
+
const identity = identityFor(packet, trialIndex, arm);
|
|
197
|
+
const result = await invokeExecutor(input, packet, trialIndex, arm, binding);
|
|
198
|
+
if (result?.runtimeFailure) {
|
|
199
|
+
return {
|
|
200
|
+
status: 'INCOMPLETE',
|
|
201
|
+
finding: finding(identity, 'INCOMPLETE', [result.runtimeFailure]),
|
|
202
|
+
};
|
|
203
|
+
}
|
|
204
|
+
if (!validExecutorResult(result)) {
|
|
205
|
+
return { status: 'INCOMPLETE', finding: finding(identity, 'INCOMPLETE', ['evidence.malformed']) };
|
|
206
|
+
}
|
|
207
|
+
if (result.attribution.model !== arm.model) {
|
|
208
|
+
return {
|
|
209
|
+
status: 'INCOMPLETE',
|
|
210
|
+
finding: finding(identity, 'INCOMPLETE', ['attribution.model_mismatch'], result.evidence),
|
|
211
|
+
};
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
const evaluation = evaluateCase({
|
|
215
|
+
packet,
|
|
216
|
+
allPackets: corpus.allPackets,
|
|
217
|
+
manifest: corpus.manifest,
|
|
218
|
+
evidence: result.evidence,
|
|
219
|
+
expectedBinding: binding,
|
|
220
|
+
});
|
|
221
|
+
const hardFailure = explicitHardFailure(evaluation, result);
|
|
222
|
+
if (evaluation.failures.some(isStructuralFailure)) {
|
|
223
|
+
return {
|
|
224
|
+
status: 'INCOMPLETE',
|
|
225
|
+
finding: finding(identity, 'INCOMPLETE', evaluation.failures, result.evidence, hardFailure),
|
|
226
|
+
};
|
|
227
|
+
}
|
|
228
|
+
|
|
229
|
+
const persistenceFailure = await persistEvidence(
|
|
230
|
+
input, append, identity, binding, evaluation, result, hardFailure,
|
|
231
|
+
);
|
|
232
|
+
if (persistenceFailure) {
|
|
233
|
+
return {
|
|
234
|
+
status: 'INCOMPLETE',
|
|
235
|
+
finding: finding(identity, 'INCOMPLETE', [persistenceFailure], result.evidence, hardFailure),
|
|
236
|
+
};
|
|
237
|
+
}
|
|
238
|
+
const status = evaluation.passed ? 'PASS' : 'FAIL';
|
|
239
|
+
return {
|
|
240
|
+
status,
|
|
241
|
+
finding: finding(
|
|
242
|
+
identity, status, evaluation.passed ? [] : evaluation.failures, result.evidence, hardFailure,
|
|
243
|
+
),
|
|
244
|
+
};
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
function recordOutcome(counts, outcome, findings) {
|
|
248
|
+
findings.push(outcome.finding);
|
|
249
|
+
if (outcome.status === 'INCOMPLETE') {
|
|
250
|
+
counts.incompleteRuns += 1;
|
|
251
|
+
return;
|
|
252
|
+
}
|
|
253
|
+
counts.completedRuns += 1;
|
|
254
|
+
if (outcome.status === 'PASS') counts.passedRuns += 1;
|
|
255
|
+
else counts.failedRuns += 1;
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
function finalStatus(counts, expectedRuns) {
|
|
259
|
+
if (counts.incompleteRuns > 0 || counts.completedRuns !== expectedRuns) return 'INCOMPLETE';
|
|
260
|
+
return counts.failedRuns > 0 ? 'FAIL' : 'PASS';
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
/**
|
|
264
|
+
* Execute the frozen corpus through four opaque, matched executor arms.
|
|
265
|
+
* The executor may observe arm ids and immutable packets, but it receives no
|
|
266
|
+
* merge capability. Only a strict, privacy-safe attribution envelope persists.
|
|
267
|
+
*/
|
|
268
|
+
async function runBehavioralEvaluation(input) {
|
|
269
|
+
let corpus;
|
|
270
|
+
try {
|
|
271
|
+
corpus = loadTier(input.tier);
|
|
272
|
+
} catch (error) {
|
|
273
|
+
return incompleteResult(input.tier, input.arms || [], 0, error.message);
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
const suppliedArms = Array.isArray(input.arms) ? input.arms : [];
|
|
277
|
+
const trialIndices = corpus.manifest.trialIndices;
|
|
278
|
+
const expectedRuns = corpus.cases.length * trialIndices.length * 4;
|
|
279
|
+
const arms = snapshotArms(suppliedArms);
|
|
280
|
+
if (!arms) return incompleteResult(input.tier, suppliedArms, expectedRuns, 'arms.invalid');
|
|
281
|
+
if (typeof input.executor !== 'function') {
|
|
282
|
+
return incompleteResult(input.tier, arms, expectedRuns, 'executor.missing');
|
|
283
|
+
}
|
|
284
|
+
const binding = snapshotBinding(input.binding);
|
|
285
|
+
if (!binding) return incompleteResult(input.tier, arms, expectedRuns, 'binding.invalid');
|
|
286
|
+
|
|
287
|
+
const append = input.appendEvidence || appendEvalEvidence;
|
|
288
|
+
const findings = [];
|
|
289
|
+
const counts = { completedRuns: 0, passedRuns: 0, failedRuns: 0, incompleteRuns: 0 };
|
|
290
|
+
|
|
291
|
+
for (const packet of corpus.cases) {
|
|
292
|
+
for (const trialIndex of trialIndices) {
|
|
293
|
+
for (const arm of arms) {
|
|
294
|
+
const outcome = await executeArm({ input, corpus, packet, trialIndex, arm, binding, append });
|
|
295
|
+
recordOutcome(counts, outcome, findings);
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
return {
|
|
301
|
+
status: finalStatus(counts, expectedRuns),
|
|
302
|
+
tier: input.tier,
|
|
303
|
+
arms,
|
|
304
|
+
expectedRuns,
|
|
305
|
+
...counts,
|
|
306
|
+
findings,
|
|
307
|
+
};
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
module.exports = { runBehavioralEvaluation };
|