contextos-agents 2.0.0-beta.3 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/AGENTS.md +53 -33
- package/.agents/adapters/aider/export.js +41 -14
- package/.agents/adapters/claude/export.js +1 -1
- package/.agents/adapters/copilot/export.js +1 -1
- package/.agents/adapters/cursor/export.js +1 -1
- package/.agents/adapters/drift-detector.js +80 -7
- package/.agents/adapters/gemini/export.js +1 -1
- package/.agents/adapters/pure-compiler.js +10 -0
- package/.agents/adapters/shared.js +13 -4
- package/.agents/adapters/zed/export.js +1 -1
- package/.agents/compiled/registry.v2.json +29 -25
- package/.agents/compiled/registry.v2.sha256 +1 -1
- package/.agents/core/skills/context-os/SKILL.md +34 -37
- package/.agents/core/skills/engineering-workflow/SKILL.md +24 -24
- package/.agents/core/skills/gemini-precision/EXAMPLES.md +72 -0
- package/.agents/core/skills/gemini-precision/SKILL.md +2 -1
- package/.agents/core/skills/gemini-precision/TROUBLESHOOTING.md +25 -0
- package/.agents/core/skills/gemini-precision/skill.yaml +2 -0
- package/.agents/core/skills/gstack-roles/SKILL.md +7 -6
- package/.agents/core/skills/security/SKILL.md +44 -16
- package/.agents/core/skills/security/skill.yaml +0 -1
- package/.agents/ctx.js +16 -10
- package/.agents/doctor.js +7 -20
- package/.agents/generated/claude/skills/context-os/SKILL.md +34 -37
- package/.agents/generated/claude/skills/engineering-workflow/SKILL.md +24 -24
- package/.agents/generated/claude/skills/gemini-precision/SKILL.md +102 -1
- package/.agents/generated/claude/skills/gstack-roles/SKILL.md +7 -6
- package/.agents/generated/claude/skills/security/SKILL.md +44 -16
- package/.agents/generated/gemini/skills/context-os/SKILL.md +34 -37
- package/.agents/generated/gemini/skills/engineering-workflow/SKILL.md +24 -24
- package/.agents/generated/gemini/skills/gemini-precision/SKILL.md +105 -1
- package/.agents/generated/gemini/skills/gstack-roles/SKILL.md +7 -6
- package/.agents/generated/gemini/skills/security/SKILL.md +44 -125
- package/.agents/plugins.js +5 -4
- package/.agents/resolver/canonical-resolver.js +7 -7
- package/.agents/validate.js +69 -1
- package/LICENSE +201 -21
- package/NOTICE +4 -0
- package/README.md +87 -34
- package/bin/commands/hook.js +129 -0
- package/bin/commands/scan.js +70 -0
- package/bin/commands/update.js +7 -8
- package/bin/commands.js +39 -1
- package/bin/index.js +144 -33
- package/bin/lib/gate.js +171 -0
- package/bin/lib/git-snapshot.js +187 -0
- package/bin/lib/scan.js +380 -0
- package/package.json +10 -12
- package/.agents/core/skills/security/security.md +0 -106
- package/benchmarks/v2/analysis/statistics.js +0 -140
- package/benchmarks/v2/analysis/stats.js +0 -69
- package/benchmarks/v2/arms/arm-definitions.js +0 -79
- package/benchmarks/v2/dataset.schema.json +0 -34
- package/benchmarks/v2/evaluators/index.js +0 -25
- package/benchmarks/v2/evaluators/verified-success.js +0 -116
- package/benchmarks/v2/harness/runner.js +0 -88
- package/benchmarks/v2/pilot-tasks.json +0 -392
|
@@ -1,69 +0,0 @@
|
|
|
1
|
-
'use strict';
|
|
2
|
-
|
|
3
|
-
/**
|
|
4
|
-
* Calculates 95% Confidence Interval for a proportion using the Wald method.
|
|
5
|
-
* @param {number} p - sample proportion (success rate)
|
|
6
|
-
* @param {number} n - sample size
|
|
7
|
-
* @returns {number} Margin of Error
|
|
8
|
-
*/
|
|
9
|
-
function calculate95CI(p, n) {
|
|
10
|
-
if (n === 0) return 0;
|
|
11
|
-
// Z-value for 95% confidence is 1.96
|
|
12
|
-
const z = 1.96;
|
|
13
|
-
const standardError = Math.sqrt((p * (1 - p)) / n);
|
|
14
|
-
return z * standardError;
|
|
15
|
-
}
|
|
16
|
-
|
|
17
|
-
/**
|
|
18
|
-
* Analyzes the results of a benchmark run.
|
|
19
|
-
* @param {Array} results Array of execution result objects from the runner
|
|
20
|
-
* @returns {Object} Statistical summary
|
|
21
|
-
*/
|
|
22
|
-
function analyzeResults(results) {
|
|
23
|
-
const statsByArm = {};
|
|
24
|
-
|
|
25
|
-
// Group by arm
|
|
26
|
-
for (const res of results) {
|
|
27
|
-
if (!statsByArm[res.armId]) {
|
|
28
|
-
statsByArm[res.armId] = {
|
|
29
|
-
total: 0,
|
|
30
|
-
successes: 0,
|
|
31
|
-
failures: 0,
|
|
32
|
-
totalTokens: 0,
|
|
33
|
-
totalDurationMs: 0
|
|
34
|
-
};
|
|
35
|
-
}
|
|
36
|
-
|
|
37
|
-
const stats = statsByArm[res.armId];
|
|
38
|
-
stats.total++;
|
|
39
|
-
if (res.success) {
|
|
40
|
-
stats.successes++;
|
|
41
|
-
} else {
|
|
42
|
-
stats.failures++;
|
|
43
|
-
}
|
|
44
|
-
stats.totalTokens += res.usage.totalTokens;
|
|
45
|
-
stats.totalDurationMs += res.durationMs;
|
|
46
|
-
}
|
|
47
|
-
|
|
48
|
-
// Calculate rates and CI
|
|
49
|
-
const finalStats = {};
|
|
50
|
-
for (const [armId, stats] of Object.entries(statsByArm)) {
|
|
51
|
-
const successRate = stats.total > 0 ? stats.successes / stats.total : 0;
|
|
52
|
-
const marginOfError = calculate95CI(successRate, stats.total);
|
|
53
|
-
|
|
54
|
-
finalStats[armId] = {
|
|
55
|
-
...stats,
|
|
56
|
-
successRate: parseFloat((successRate * 100).toFixed(2)),
|
|
57
|
-
confidenceInterval95: `±${(marginOfError * 100).toFixed(2)}%`,
|
|
58
|
-
avgTokensPerTask: stats.total > 0 ? Math.round(stats.totalTokens / stats.total) : 0,
|
|
59
|
-
avgDurationMs: stats.total > 0 ? Math.round(stats.totalDurationMs / stats.total) : 0
|
|
60
|
-
};
|
|
61
|
-
}
|
|
62
|
-
|
|
63
|
-
return finalStats;
|
|
64
|
-
}
|
|
65
|
-
|
|
66
|
-
module.exports = {
|
|
67
|
-
analyzeResults,
|
|
68
|
-
calculate95CI
|
|
69
|
-
};
|
|
@@ -1,79 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* benchmarks/v2/arms/arm-definitions.js
|
|
3
|
-
* ContextOS Benchmark v2 — Evaluation Arms Specification
|
|
4
|
-
*
|
|
5
|
-
* Implements Section 24.2 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
|
|
6
|
-
* - Arm A (Vanilla): Neutral baseline system prompt without artificial debuffing
|
|
7
|
-
* - Arm B (Concise Checklist): 10-15 universal engineering rules (~600 tokens) — Primary Comparator
|
|
8
|
-
* - Arm C (ContextOS Core): Dynamic canonical skill resolver without ceremony
|
|
9
|
-
* - Arm D (Full ContextOS): Resolver + risk workflow + isolated runtime verification + review
|
|
10
|
-
*/
|
|
11
|
-
|
|
12
|
-
'use strict';
|
|
13
|
-
|
|
14
|
-
const ARMS = {
|
|
15
|
-
ARM_A_VANILLA: {
|
|
16
|
-
id: 'arm-a-vanilla',
|
|
17
|
-
name: 'Vanilla Baseline',
|
|
18
|
-
description: 'Neutral baseline system prompt without ContextOS rules or checklists.',
|
|
19
|
-
tokenBudgetEstimate: 120,
|
|
20
|
-
buildSystemPrompt: () => {
|
|
21
|
-
return 'You are an expert software engineer. Write clean, complete, working production code that solves the user request.';
|
|
22
|
-
},
|
|
23
|
-
},
|
|
24
|
-
|
|
25
|
-
ARM_B_CONCISE_CHECKLIST: {
|
|
26
|
-
id: 'arm-b-concise-checklist',
|
|
27
|
-
name: 'Concise Checklist',
|
|
28
|
-
description: 'High-density 12-rule engineering checklist (~600 tokens). Primary comparator.',
|
|
29
|
-
tokenBudgetEstimate: 580,
|
|
30
|
-
buildSystemPrompt: () => {
|
|
31
|
-
return [
|
|
32
|
-
'You are a Senior Staff Engineer.',
|
|
33
|
-
'Follow this strict engineering checklist:',
|
|
34
|
-
'1. Inspect existing files before editing.',
|
|
35
|
-
'2. Never use placeholders, stubs, or TODO comments.',
|
|
36
|
-
'3. Maintain existing codebase naming conventions and architectural boundaries.',
|
|
37
|
-
'4. Minimize blast radius — modify only files required for the task.',
|
|
38
|
-
'5. Validate all user input and sanitize data paths.',
|
|
39
|
-
'6. Use parameterized queries for database operations.',
|
|
40
|
-
'7. Handle all asynchronous error boundaries explicitly.',
|
|
41
|
-
'8. Write comprehensive unit and integration test assertions.',
|
|
42
|
-
'9. Ensure clean TypeScript typing without any unsafe casts.',
|
|
43
|
-
'10. Verify backward compatibility with existing public APIs.',
|
|
44
|
-
'11. No secrets or credentials in code or commits.',
|
|
45
|
-
'12. Ensure code compiles and all tests pass.',
|
|
46
|
-
].join('\n');
|
|
47
|
-
},
|
|
48
|
-
},
|
|
49
|
-
|
|
50
|
-
ARM_C_CONTEXTOS_CORE: {
|
|
51
|
-
id: 'arm-c-contextos-core',
|
|
52
|
-
name: 'ContextOS Core (Dynamic Context Selection)',
|
|
53
|
-
description: 'Dynamic canonical resolver selecting exact skills and rules without ceremony.',
|
|
54
|
-
tokenBudgetEstimate: 1400,
|
|
55
|
-
buildSystemPrompt: (resolvedSkills = []) => {
|
|
56
|
-
const skillsHeader = resolvedSkills.length > 0
|
|
57
|
-
? `[ContextOS Resolved Skills: ${resolvedSkills.join(', ')}]`
|
|
58
|
-
: '[ContextOS Core]';
|
|
59
|
-
return `${skillsHeader}\nExecute task adhering to compiled workspace rules and exact skill invariants.`;
|
|
60
|
-
},
|
|
61
|
-
},
|
|
62
|
-
|
|
63
|
-
ARM_D_FULL_CONTEXTOS: {
|
|
64
|
-
id: 'arm-d-full-contextos',
|
|
65
|
-
name: 'Full ContextOS (Core + Runtime Verification)',
|
|
66
|
-
description: 'Dynamic resolver + risk-based workflow + isolated runtime verification + reviewer pipeline.',
|
|
67
|
-
tokenBudgetEstimate: 2200,
|
|
68
|
-
buildSystemPrompt: (resolvedSkills = [], riskLevel = 'STANDARD') => {
|
|
69
|
-
return [
|
|
70
|
-
`[ContextOS Full Runtime] [RISK: ${riskLevel}] [Skills: ${resolvedSkills.join(', ')}]`,
|
|
71
|
-
'Execution gated by isolated worktree and mandatory verification attestations before merge readiness.',
|
|
72
|
-
].join('\n');
|
|
73
|
-
},
|
|
74
|
-
},
|
|
75
|
-
};
|
|
76
|
-
|
|
77
|
-
module.exports = {
|
|
78
|
-
ARMS,
|
|
79
|
-
};
|
|
@@ -1,34 +0,0 @@
|
|
|
1
|
-
{
|
|
2
|
-
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
-
"title": "ContextOS Benchmark v2 Task Schema",
|
|
4
|
-
"type": "object",
|
|
5
|
-
"properties": {
|
|
6
|
-
"id": {
|
|
7
|
-
"type": "string",
|
|
8
|
-
"description": "Unique immutable identifier for the task"
|
|
9
|
-
},
|
|
10
|
-
"description": {
|
|
11
|
-
"type": "string",
|
|
12
|
-
"description": "The actual prompt provided to the LLM"
|
|
13
|
-
},
|
|
14
|
-
"expectedState": {
|
|
15
|
-
"type": "object",
|
|
16
|
-
"description": "The expected state of the filesystem or output after execution",
|
|
17
|
-
"properties": {
|
|
18
|
-
"filesToExist": {
|
|
19
|
-
"type": "array",
|
|
20
|
-
"items": { "type": "string" }
|
|
21
|
-
},
|
|
22
|
-
"filesToContain": {
|
|
23
|
-
"type": "object",
|
|
24
|
-
"additionalProperties": { "type": "string" }
|
|
25
|
-
}
|
|
26
|
-
}
|
|
27
|
-
},
|
|
28
|
-
"hash": {
|
|
29
|
-
"type": "string",
|
|
30
|
-
"description": "SHA-256 hash of the task for immutability verification"
|
|
31
|
-
}
|
|
32
|
-
},
|
|
33
|
-
"required": ["id", "description", "hash"]
|
|
34
|
-
}
|
|
@@ -1,25 +0,0 @@
|
|
|
1
|
-
'use strict';
|
|
2
|
-
|
|
3
|
-
/**
|
|
4
|
-
* Evaluates the result of a task execution against the expected state.
|
|
5
|
-
* Since we are using a Mock Provider, the evaluation logic is simplified
|
|
6
|
-
* to just trust the provider's mocked success status. In a real system,
|
|
7
|
-
* this would run ESLint, execute the generated code in a sandbox,
|
|
8
|
-
* and verify the exact AST or output.
|
|
9
|
-
*
|
|
10
|
-
* @param {Object} task The benchmark task definition
|
|
11
|
-
* @param {Object} llmResult The result from the LLM/Mock Provider
|
|
12
|
-
* @returns {Object} { passed: boolean, error: string|null }
|
|
13
|
-
*/
|
|
14
|
-
function evaluateTask(task, llmResult) {
|
|
15
|
-
if (llmResult.success) {
|
|
16
|
-
return { passed: true, error: null };
|
|
17
|
-
} else {
|
|
18
|
-
return {
|
|
19
|
-
passed: false,
|
|
20
|
-
error: `Failed to meet expected state for task ${task.id}: ${llmResult.mockedOutput}`
|
|
21
|
-
};
|
|
22
|
-
}
|
|
23
|
-
}
|
|
24
|
-
|
|
25
|
-
module.exports = evaluateTask;
|
|
@@ -1,116 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* benchmarks/v2/evaluators/verified-success.js
|
|
3
|
-
* ContextOS Benchmark v2 — Primary Outcome Evaluator
|
|
4
|
-
*
|
|
5
|
-
* Implements Section 24.8 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
|
|
6
|
-
* - Primary metric: independently_verified_success
|
|
7
|
-
* - Evaluates:
|
|
8
|
-
* 1. Patch application / syntax correctness
|
|
9
|
-
* 2. Typecheck / build status
|
|
10
|
-
* 3. Public unit test pass rate
|
|
11
|
-
* 4. Hidden test suite pass rate (isolated oracle)
|
|
12
|
-
* 5. Zero regression on baseline suites
|
|
13
|
-
* 6. Zero P0/P1 security findings or leaked credentials
|
|
14
|
-
* 7. Zero placeholder stubs (TODO, mock placeholders)
|
|
15
|
-
* 8. Budget constraints (tokens, time, turns)
|
|
16
|
-
*/
|
|
17
|
-
|
|
18
|
-
'use strict';
|
|
19
|
-
|
|
20
|
-
const PLACEHOLDER_PATTERNS = [
|
|
21
|
-
/\/\/\s*TODO:\s*implement\b/i,
|
|
22
|
-
/\/\/\s*\.\.\.\s*rest of code\b/i,
|
|
23
|
-
/\bthrow new Error\(["']Not implemented["']\)/i,
|
|
24
|
-
/\bpass\s*#\s*TODO\b/i,
|
|
25
|
-
];
|
|
26
|
-
|
|
27
|
-
class BenchmarkEvaluator {
|
|
28
|
-
/**
|
|
29
|
-
* Evaluates task run evidence against rigorous quality gates.
|
|
30
|
-
*
|
|
31
|
-
* @param {Object} runEvidence
|
|
32
|
-
* @param {boolean} runEvidence.patchApplied
|
|
33
|
-
* @param {boolean} runEvidence.buildPass
|
|
34
|
-
* @param {number} runEvidence.publicTestsTotal
|
|
35
|
-
* @param {number} runEvidence.publicTestsPassed
|
|
36
|
-
* @param {number} runEvidence.hiddenTestsTotal
|
|
37
|
-
* @param {number} runEvidence.hiddenTestsPassed
|
|
38
|
-
* @param {boolean} [runEvidence.regressions=false]
|
|
39
|
-
* @param {Array<string>} [runEvidence.securityFindings=[]]
|
|
40
|
-
* @param {string} [runEvidence.generatedCode='']
|
|
41
|
-
* @param {Object} [runEvidence.budget]
|
|
42
|
-
* @param {number} [runEvidence.budget.tokensUsed=0]
|
|
43
|
-
* @param {number} [runEvidence.budget.tokenLimit=50000]
|
|
44
|
-
* @param {number} [runEvidence.budget.durationMs=0]
|
|
45
|
-
* @param {number} [runEvidence.budget.timeoutMs=60000]
|
|
46
|
-
* @returns {Object} Evaluation report
|
|
47
|
-
*/
|
|
48
|
-
static evaluate(runEvidence) {
|
|
49
|
-
const {
|
|
50
|
-
patchApplied = false,
|
|
51
|
-
buildPass = false,
|
|
52
|
-
publicTestsTotal = 0,
|
|
53
|
-
publicTestsPassed = 0,
|
|
54
|
-
hiddenTestsTotal = 0,
|
|
55
|
-
hiddenTestsPassed = 0,
|
|
56
|
-
regressions = false,
|
|
57
|
-
securityFindings = [],
|
|
58
|
-
generatedCode = '',
|
|
59
|
-
budget = {},
|
|
60
|
-
} = runEvidence;
|
|
61
|
-
|
|
62
|
-
const failures = [];
|
|
63
|
-
|
|
64
|
-
// 1. Patch & build
|
|
65
|
-
if (!patchApplied) failures.push('Patch was not successfully applied');
|
|
66
|
-
if (!buildPass) failures.push('Compilation or typecheck failed');
|
|
67
|
-
|
|
68
|
-
// 2. Tests
|
|
69
|
-
if (publicTestsTotal > 0 && publicTestsPassed < publicTestsTotal) {
|
|
70
|
-
failures.push(`Public tests failed: ${publicTestsPassed}/${publicTestsTotal}`);
|
|
71
|
-
}
|
|
72
|
-
if (hiddenTestsTotal > 0 && hiddenTestsPassed < hiddenTestsTotal) {
|
|
73
|
-
failures.push(`Hidden test oracle failed: ${hiddenTestsPassed}/${hiddenTestsTotal}`);
|
|
74
|
-
}
|
|
75
|
-
if (regressions) {
|
|
76
|
-
failures.push('Regression detected in existing test baseline');
|
|
77
|
-
}
|
|
78
|
-
|
|
79
|
-
// 3. Security
|
|
80
|
-
if (Array.isArray(securityFindings) && securityFindings.length > 0) {
|
|
81
|
-
failures.push(`Security vulnerabilities detected: ${securityFindings.join(', ')}`);
|
|
82
|
-
}
|
|
83
|
-
|
|
84
|
-
// 4. Zero placeholders
|
|
85
|
-
if (generatedCode) {
|
|
86
|
-
for (const pat of PLACEHOLDER_PATTERNS) {
|
|
87
|
-
if (pat.test(generatedCode)) {
|
|
88
|
-
failures.push(`Lazy placeholder detected matching pattern: ${pat.source}`);
|
|
89
|
-
break;
|
|
90
|
-
}
|
|
91
|
-
}
|
|
92
|
-
}
|
|
93
|
-
|
|
94
|
-
// 5. Budget constraints
|
|
95
|
-
if (budget.tokensUsed && budget.tokenLimit && budget.tokensUsed > budget.tokenLimit) {
|
|
96
|
-
failures.push(`Token budget exceeded: ${budget.tokensUsed} > ${budget.tokenLimit}`);
|
|
97
|
-
}
|
|
98
|
-
if (budget.durationMs && budget.timeoutMs && budget.durationMs > budget.timeoutMs) {
|
|
99
|
-
failures.push(`Time budget exceeded: ${budget.durationMs}ms > ${budget.timeoutMs}ms`);
|
|
100
|
-
}
|
|
101
|
-
|
|
102
|
-
const isSuccess = failures.length === 0;
|
|
103
|
-
|
|
104
|
-
return {
|
|
105
|
-
independently_verified_success: isSuccess,
|
|
106
|
-
publicTestRate: publicTestsTotal > 0 ? publicTestsPassed / publicTestsTotal : 1.0,
|
|
107
|
-
hiddenTestRate: hiddenTestsTotal > 0 ? hiddenTestsPassed / hiddenTestsTotal : 1.0,
|
|
108
|
-
failureCount: failures.length,
|
|
109
|
-
failures,
|
|
110
|
-
};
|
|
111
|
-
}
|
|
112
|
-
}
|
|
113
|
-
|
|
114
|
-
module.exports = {
|
|
115
|
-
BenchmarkEvaluator,
|
|
116
|
-
};
|
|
@@ -1,88 +0,0 @@
|
|
|
1
|
-
'use strict';
|
|
2
|
-
|
|
3
|
-
const crypto = require('crypto');
|
|
4
|
-
const { ARMS } = require('../arms/arm-definitions');
|
|
5
|
-
const evaluateTask = require('../evaluators/index');
|
|
6
|
-
|
|
7
|
-
/**
|
|
8
|
-
* Mock LLM Provider used when real API keys are unavailable.
|
|
9
|
-
* Deterministically simulates success/failure rates based on the arm's capability.
|
|
10
|
-
*/
|
|
11
|
-
class MockProvider {
|
|
12
|
-
/**
|
|
13
|
-
* Probability of success for each arm to simulate real-world capability differences.
|
|
14
|
-
*/
|
|
15
|
-
static getSuccessProbability(armId) {
|
|
16
|
-
switch (armId) {
|
|
17
|
-
case ARMS.ARM_A_VANILLA.id: return 0.40; // 40% success
|
|
18
|
-
case ARMS.ARM_B_CONCISE_CHECKLIST.id: return 0.65; // 65% success
|
|
19
|
-
case ARMS.ARM_C_CONTEXTOS_CORE.id: return 0.85; // 85% success
|
|
20
|
-
case ARMS.ARM_D_FULL_CONTEXTOS.id: return 0.98; // 98% success
|
|
21
|
-
default: return 0.0;
|
|
22
|
-
}
|
|
23
|
-
}
|
|
24
|
-
|
|
25
|
-
static async execute(task, arm) {
|
|
26
|
-
const probability = this.getSuccessProbability(arm.id);
|
|
27
|
-
// Use hash to deterministically seed pseudo-randomness for the mock run
|
|
28
|
-
const hashInt = parseInt(task.hash.substring(7, 15), 16);
|
|
29
|
-
const successThreshold = probability * 0xffffffff;
|
|
30
|
-
|
|
31
|
-
// Slight artificial delay to simulate API request
|
|
32
|
-
await new Promise(resolve => setTimeout(resolve, 50));
|
|
33
|
-
|
|
34
|
-
const isSuccess = hashInt <= successThreshold;
|
|
35
|
-
|
|
36
|
-
return {
|
|
37
|
-
success: isSuccess,
|
|
38
|
-
mockedOutput: isSuccess
|
|
39
|
-
? `Successfully generated code for ${task.id}`
|
|
40
|
-
: `Failed or hallucinated output for ${task.id}`,
|
|
41
|
-
usage: {
|
|
42
|
-
promptTokens: arm.tokenBudgetEstimate,
|
|
43
|
-
completionTokens: 300,
|
|
44
|
-
totalTokens: arm.tokenBudgetEstimate + 300
|
|
45
|
-
}
|
|
46
|
-
};
|
|
47
|
-
}
|
|
48
|
-
}
|
|
49
|
-
|
|
50
|
-
/**
|
|
51
|
-
* Executes a single task against a single arm.
|
|
52
|
-
*/
|
|
53
|
-
async function runTask(task, armId) {
|
|
54
|
-
const arm = Object.values(ARMS).find(a => a.id === armId);
|
|
55
|
-
if (!arm) throw new Error(`Unknown arm: ${armId}`);
|
|
56
|
-
|
|
57
|
-
const startTime = Date.now();
|
|
58
|
-
const requestId = crypto.randomUUID();
|
|
59
|
-
|
|
60
|
-
// Execute using Mock Provider (replace with real LLM client when API keys are available)
|
|
61
|
-
const llmResult = await MockProvider.execute(task, arm);
|
|
62
|
-
|
|
63
|
-
const durationMs = Date.now() - startTime;
|
|
64
|
-
|
|
65
|
-
// Evaluate the output
|
|
66
|
-
const evaluation = evaluateTask(task, llmResult);
|
|
67
|
-
|
|
68
|
-
return {
|
|
69
|
-
taskId: task.id,
|
|
70
|
-
armId: arm.id,
|
|
71
|
-
requestId,
|
|
72
|
-
timestamp: new Date().toISOString(),
|
|
73
|
-
durationMs,
|
|
74
|
-
success: evaluation.passed,
|
|
75
|
-
error: evaluation.error || null,
|
|
76
|
-
usage: llmResult.usage,
|
|
77
|
-
environmentProvenance: {
|
|
78
|
-
nodeVersion: process.version,
|
|
79
|
-
platform: process.platform,
|
|
80
|
-
engine: 'mock-provider-v1'
|
|
81
|
-
}
|
|
82
|
-
};
|
|
83
|
-
}
|
|
84
|
-
|
|
85
|
-
module.exports = {
|
|
86
|
-
runTask,
|
|
87
|
-
MockProvider
|
|
88
|
-
};
|