contextos-agents 2.0.0-beta.3 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/.agents/AGENTS.md +53 -33
  2. package/.agents/adapters/aider/export.js +41 -14
  3. package/.agents/adapters/claude/export.js +1 -1
  4. package/.agents/adapters/copilot/export.js +1 -1
  5. package/.agents/adapters/cursor/export.js +1 -1
  6. package/.agents/adapters/drift-detector.js +80 -7
  7. package/.agents/adapters/gemini/export.js +1 -1
  8. package/.agents/adapters/pure-compiler.js +10 -0
  9. package/.agents/adapters/shared.js +13 -4
  10. package/.agents/adapters/zed/export.js +1 -1
  11. package/.agents/compiled/registry.v2.json +29 -25
  12. package/.agents/compiled/registry.v2.sha256 +1 -1
  13. package/.agents/core/skills/context-os/SKILL.md +34 -37
  14. package/.agents/core/skills/engineering-workflow/SKILL.md +24 -24
  15. package/.agents/core/skills/gemini-precision/EXAMPLES.md +72 -0
  16. package/.agents/core/skills/gemini-precision/SKILL.md +2 -1
  17. package/.agents/core/skills/gemini-precision/TROUBLESHOOTING.md +25 -0
  18. package/.agents/core/skills/gemini-precision/skill.yaml +2 -0
  19. package/.agents/core/skills/gstack-roles/SKILL.md +7 -6
  20. package/.agents/core/skills/security/SKILL.md +44 -16
  21. package/.agents/core/skills/security/skill.yaml +0 -1
  22. package/.agents/ctx.js +16 -10
  23. package/.agents/doctor.js +7 -20
  24. package/.agents/generated/claude/skills/context-os/SKILL.md +34 -37
  25. package/.agents/generated/claude/skills/engineering-workflow/SKILL.md +24 -24
  26. package/.agents/generated/claude/skills/gemini-precision/SKILL.md +102 -1
  27. package/.agents/generated/claude/skills/gstack-roles/SKILL.md +7 -6
  28. package/.agents/generated/claude/skills/security/SKILL.md +44 -16
  29. package/.agents/generated/gemini/skills/context-os/SKILL.md +34 -37
  30. package/.agents/generated/gemini/skills/engineering-workflow/SKILL.md +24 -24
  31. package/.agents/generated/gemini/skills/gemini-precision/SKILL.md +105 -1
  32. package/.agents/generated/gemini/skills/gstack-roles/SKILL.md +7 -6
  33. package/.agents/generated/gemini/skills/security/SKILL.md +44 -125
  34. package/.agents/plugins.js +5 -4
  35. package/.agents/resolver/canonical-resolver.js +7 -7
  36. package/.agents/validate.js +69 -1
  37. package/LICENSE +201 -21
  38. package/NOTICE +4 -0
  39. package/README.md +87 -34
  40. package/bin/commands/hook.js +129 -0
  41. package/bin/commands/scan.js +70 -0
  42. package/bin/commands/update.js +7 -8
  43. package/bin/commands.js +39 -1
  44. package/bin/index.js +144 -33
  45. package/bin/lib/gate.js +171 -0
  46. package/bin/lib/git-snapshot.js +187 -0
  47. package/bin/lib/scan.js +380 -0
  48. package/package.json +10 -12
  49. package/.agents/core/skills/security/security.md +0 -106
  50. package/benchmarks/v2/analysis/statistics.js +0 -140
  51. package/benchmarks/v2/analysis/stats.js +0 -69
  52. package/benchmarks/v2/arms/arm-definitions.js +0 -79
  53. package/benchmarks/v2/dataset.schema.json +0 -34
  54. package/benchmarks/v2/evaluators/index.js +0 -25
  55. package/benchmarks/v2/evaluators/verified-success.js +0 -116
  56. package/benchmarks/v2/harness/runner.js +0 -88
  57. package/benchmarks/v2/pilot-tasks.json +0 -392
@@ -1,69 +0,0 @@
1
- 'use strict';
2
-
3
- /**
4
- * Calculates 95% Confidence Interval for a proportion using the Wald method.
5
- * @param {number} p - sample proportion (success rate)
6
- * @param {number} n - sample size
7
- * @returns {number} Margin of Error
8
- */
9
- function calculate95CI(p, n) {
10
- if (n === 0) return 0;
11
- // Z-value for 95% confidence is 1.96
12
- const z = 1.96;
13
- const standardError = Math.sqrt((p * (1 - p)) / n);
14
- return z * standardError;
15
- }
16
-
17
- /**
18
- * Analyzes the results of a benchmark run.
19
- * @param {Array} results Array of execution result objects from the runner
20
- * @returns {Object} Statistical summary
21
- */
22
- function analyzeResults(results) {
23
- const statsByArm = {};
24
-
25
- // Group by arm
26
- for (const res of results) {
27
- if (!statsByArm[res.armId]) {
28
- statsByArm[res.armId] = {
29
- total: 0,
30
- successes: 0,
31
- failures: 0,
32
- totalTokens: 0,
33
- totalDurationMs: 0
34
- };
35
- }
36
-
37
- const stats = statsByArm[res.armId];
38
- stats.total++;
39
- if (res.success) {
40
- stats.successes++;
41
- } else {
42
- stats.failures++;
43
- }
44
- stats.totalTokens += res.usage.totalTokens;
45
- stats.totalDurationMs += res.durationMs;
46
- }
47
-
48
- // Calculate rates and CI
49
- const finalStats = {};
50
- for (const [armId, stats] of Object.entries(statsByArm)) {
51
- const successRate = stats.total > 0 ? stats.successes / stats.total : 0;
52
- const marginOfError = calculate95CI(successRate, stats.total);
53
-
54
- finalStats[armId] = {
55
- ...stats,
56
- successRate: parseFloat((successRate * 100).toFixed(2)),
57
- confidenceInterval95: `±${(marginOfError * 100).toFixed(2)}%`,
58
- avgTokensPerTask: stats.total > 0 ? Math.round(stats.totalTokens / stats.total) : 0,
59
- avgDurationMs: stats.total > 0 ? Math.round(stats.totalDurationMs / stats.total) : 0
60
- };
61
- }
62
-
63
- return finalStats;
64
- }
65
-
66
- module.exports = {
67
- analyzeResults,
68
- calculate95CI
69
- };
@@ -1,79 +0,0 @@
1
- /**
2
- * benchmarks/v2/arms/arm-definitions.js
3
- * ContextOS Benchmark v2 — Evaluation Arms Specification
4
- *
5
- * Implements Section 24.2 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
6
- * - Arm A (Vanilla): Neutral baseline system prompt without artificial debuffing
7
- * - Arm B (Concise Checklist): 10-15 universal engineering rules (~600 tokens) — Primary Comparator
8
- * - Arm C (ContextOS Core): Dynamic canonical skill resolver without ceremony
9
- * - Arm D (Full ContextOS): Resolver + risk workflow + isolated runtime verification + review
10
- */
11
-
12
- 'use strict';
13
-
14
- const ARMS = {
15
- ARM_A_VANILLA: {
16
- id: 'arm-a-vanilla',
17
- name: 'Vanilla Baseline',
18
- description: 'Neutral baseline system prompt without ContextOS rules or checklists.',
19
- tokenBudgetEstimate: 120,
20
- buildSystemPrompt: () => {
21
- return 'You are an expert software engineer. Write clean, complete, working production code that solves the user request.';
22
- },
23
- },
24
-
25
- ARM_B_CONCISE_CHECKLIST: {
26
- id: 'arm-b-concise-checklist',
27
- name: 'Concise Checklist',
28
- description: 'High-density 12-rule engineering checklist (~600 tokens). Primary comparator.',
29
- tokenBudgetEstimate: 580,
30
- buildSystemPrompt: () => {
31
- return [
32
- 'You are a Senior Staff Engineer.',
33
- 'Follow this strict engineering checklist:',
34
- '1. Inspect existing files before editing.',
35
- '2. Never use placeholders, stubs, or TODO comments.',
36
- '3. Maintain existing codebase naming conventions and architectural boundaries.',
37
- '4. Minimize blast radius — modify only files required for the task.',
38
- '5. Validate all user input and sanitize data paths.',
39
- '6. Use parameterized queries for database operations.',
40
- '7. Handle all asynchronous error boundaries explicitly.',
41
- '8. Write comprehensive unit and integration test assertions.',
42
- '9. Ensure clean TypeScript typing without any unsafe casts.',
43
- '10. Verify backward compatibility with existing public APIs.',
44
- '11. No secrets or credentials in code or commits.',
45
- '12. Ensure code compiles and all tests pass.',
46
- ].join('\n');
47
- },
48
- },
49
-
50
- ARM_C_CONTEXTOS_CORE: {
51
- id: 'arm-c-contextos-core',
52
- name: 'ContextOS Core (Dynamic Context Selection)',
53
- description: 'Dynamic canonical resolver selecting exact skills and rules without ceremony.',
54
- tokenBudgetEstimate: 1400,
55
- buildSystemPrompt: (resolvedSkills = []) => {
56
- const skillsHeader = resolvedSkills.length > 0
57
- ? `[ContextOS Resolved Skills: ${resolvedSkills.join(', ')}]`
58
- : '[ContextOS Core]';
59
- return `${skillsHeader}\nExecute task adhering to compiled workspace rules and exact skill invariants.`;
60
- },
61
- },
62
-
63
- ARM_D_FULL_CONTEXTOS: {
64
- id: 'arm-d-full-contextos',
65
- name: 'Full ContextOS (Core + Runtime Verification)',
66
- description: 'Dynamic resolver + risk-based workflow + isolated runtime verification + reviewer pipeline.',
67
- tokenBudgetEstimate: 2200,
68
- buildSystemPrompt: (resolvedSkills = [], riskLevel = 'STANDARD') => {
69
- return [
70
- `[ContextOS Full Runtime] [RISK: ${riskLevel}] [Skills: ${resolvedSkills.join(', ')}]`,
71
- 'Execution gated by isolated worktree and mandatory verification attestations before merge readiness.',
72
- ].join('\n');
73
- },
74
- },
75
- };
76
-
77
- module.exports = {
78
- ARMS,
79
- };
@@ -1,34 +0,0 @@
1
- {
2
- "$schema": "http://json-schema.org/draft-07/schema#",
3
- "title": "ContextOS Benchmark v2 Task Schema",
4
- "type": "object",
5
- "properties": {
6
- "id": {
7
- "type": "string",
8
- "description": "Unique immutable identifier for the task"
9
- },
10
- "description": {
11
- "type": "string",
12
- "description": "The actual prompt provided to the LLM"
13
- },
14
- "expectedState": {
15
- "type": "object",
16
- "description": "The expected state of the filesystem or output after execution",
17
- "properties": {
18
- "filesToExist": {
19
- "type": "array",
20
- "items": { "type": "string" }
21
- },
22
- "filesToContain": {
23
- "type": "object",
24
- "additionalProperties": { "type": "string" }
25
- }
26
- }
27
- },
28
- "hash": {
29
- "type": "string",
30
- "description": "SHA-256 hash of the task for immutability verification"
31
- }
32
- },
33
- "required": ["id", "description", "hash"]
34
- }
@@ -1,25 +0,0 @@
1
- 'use strict';
2
-
3
- /**
4
- * Evaluates the result of a task execution against the expected state.
5
- * Since we are using a Mock Provider, the evaluation logic is simplified
6
- * to just trust the provider's mocked success status. In a real system,
7
- * this would run ESLint, execute the generated code in a sandbox,
8
- * and verify the exact AST or output.
9
- *
10
- * @param {Object} task The benchmark task definition
11
- * @param {Object} llmResult The result from the LLM/Mock Provider
12
- * @returns {Object} { passed: boolean, error: string|null }
13
- */
14
- function evaluateTask(task, llmResult) {
15
- if (llmResult.success) {
16
- return { passed: true, error: null };
17
- } else {
18
- return {
19
- passed: false,
20
- error: `Failed to meet expected state for task ${task.id}: ${llmResult.mockedOutput}`
21
- };
22
- }
23
- }
24
-
25
- module.exports = evaluateTask;
@@ -1,116 +0,0 @@
1
- /**
2
- * benchmarks/v2/evaluators/verified-success.js
3
- * ContextOS Benchmark v2 — Primary Outcome Evaluator
4
- *
5
- * Implements Section 24.8 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
6
- * - Primary metric: independently_verified_success
7
- * - Evaluates:
8
- * 1. Patch application / syntax correctness
9
- * 2. Typecheck / build status
10
- * 3. Public unit test pass rate
11
- * 4. Hidden test suite pass rate (isolated oracle)
12
- * 5. Zero regression on baseline suites
13
- * 6. Zero P0/P1 security findings or leaked credentials
14
- * 7. Zero placeholder stubs (TODO, mock placeholders)
15
- * 8. Budget constraints (tokens, time, turns)
16
- */
17
-
18
- 'use strict';
19
-
20
- const PLACEHOLDER_PATTERNS = [
21
- /\/\/\s*TODO:\s*implement\b/i,
22
- /\/\/\s*\.\.\.\s*rest of code\b/i,
23
- /\bthrow new Error\(["']Not implemented["']\)/i,
24
- /\bpass\s*#\s*TODO\b/i,
25
- ];
26
-
27
- class BenchmarkEvaluator {
28
- /**
29
- * Evaluates task run evidence against rigorous quality gates.
30
- *
31
- * @param {Object} runEvidence
32
- * @param {boolean} runEvidence.patchApplied
33
- * @param {boolean} runEvidence.buildPass
34
- * @param {number} runEvidence.publicTestsTotal
35
- * @param {number} runEvidence.publicTestsPassed
36
- * @param {number} runEvidence.hiddenTestsTotal
37
- * @param {number} runEvidence.hiddenTestsPassed
38
- * @param {boolean} [runEvidence.regressions=false]
39
- * @param {Array<string>} [runEvidence.securityFindings=[]]
40
- * @param {string} [runEvidence.generatedCode='']
41
- * @param {Object} [runEvidence.budget]
42
- * @param {number} [runEvidence.budget.tokensUsed=0]
43
- * @param {number} [runEvidence.budget.tokenLimit=50000]
44
- * @param {number} [runEvidence.budget.durationMs=0]
45
- * @param {number} [runEvidence.budget.timeoutMs=60000]
46
- * @returns {Object} Evaluation report
47
- */
48
- static evaluate(runEvidence) {
49
- const {
50
- patchApplied = false,
51
- buildPass = false,
52
- publicTestsTotal = 0,
53
- publicTestsPassed = 0,
54
- hiddenTestsTotal = 0,
55
- hiddenTestsPassed = 0,
56
- regressions = false,
57
- securityFindings = [],
58
- generatedCode = '',
59
- budget = {},
60
- } = runEvidence;
61
-
62
- const failures = [];
63
-
64
- // 1. Patch & build
65
- if (!patchApplied) failures.push('Patch was not successfully applied');
66
- if (!buildPass) failures.push('Compilation or typecheck failed');
67
-
68
- // 2. Tests
69
- if (publicTestsTotal > 0 && publicTestsPassed < publicTestsTotal) {
70
- failures.push(`Public tests failed: ${publicTestsPassed}/${publicTestsTotal}`);
71
- }
72
- if (hiddenTestsTotal > 0 && hiddenTestsPassed < hiddenTestsTotal) {
73
- failures.push(`Hidden test oracle failed: ${hiddenTestsPassed}/${hiddenTestsTotal}`);
74
- }
75
- if (regressions) {
76
- failures.push('Regression detected in existing test baseline');
77
- }
78
-
79
- // 3. Security
80
- if (Array.isArray(securityFindings) && securityFindings.length > 0) {
81
- failures.push(`Security vulnerabilities detected: ${securityFindings.join(', ')}`);
82
- }
83
-
84
- // 4. Zero placeholders
85
- if (generatedCode) {
86
- for (const pat of PLACEHOLDER_PATTERNS) {
87
- if (pat.test(generatedCode)) {
88
- failures.push(`Lazy placeholder detected matching pattern: ${pat.source}`);
89
- break;
90
- }
91
- }
92
- }
93
-
94
- // 5. Budget constraints
95
- if (budget.tokensUsed && budget.tokenLimit && budget.tokensUsed > budget.tokenLimit) {
96
- failures.push(`Token budget exceeded: ${budget.tokensUsed} > ${budget.tokenLimit}`);
97
- }
98
- if (budget.durationMs && budget.timeoutMs && budget.durationMs > budget.timeoutMs) {
99
- failures.push(`Time budget exceeded: ${budget.durationMs}ms > ${budget.timeoutMs}ms`);
100
- }
101
-
102
- const isSuccess = failures.length === 0;
103
-
104
- return {
105
- independently_verified_success: isSuccess,
106
- publicTestRate: publicTestsTotal > 0 ? publicTestsPassed / publicTestsTotal : 1.0,
107
- hiddenTestRate: hiddenTestsTotal > 0 ? hiddenTestsPassed / hiddenTestsTotal : 1.0,
108
- failureCount: failures.length,
109
- failures,
110
- };
111
- }
112
- }
113
-
114
- module.exports = {
115
- BenchmarkEvaluator,
116
- };
@@ -1,88 +0,0 @@
1
- 'use strict';
2
-
3
- const crypto = require('crypto');
4
- const { ARMS } = require('../arms/arm-definitions');
5
- const evaluateTask = require('../evaluators/index');
6
-
7
- /**
8
- * Mock LLM Provider used when real API keys are unavailable.
9
- * Deterministically simulates success/failure rates based on the arm's capability.
10
- */
11
- class MockProvider {
12
- /**
13
- * Probability of success for each arm to simulate real-world capability differences.
14
- */
15
- static getSuccessProbability(armId) {
16
- switch (armId) {
17
- case ARMS.ARM_A_VANILLA.id: return 0.40; // 40% success
18
- case ARMS.ARM_B_CONCISE_CHECKLIST.id: return 0.65; // 65% success
19
- case ARMS.ARM_C_CONTEXTOS_CORE.id: return 0.85; // 85% success
20
- case ARMS.ARM_D_FULL_CONTEXTOS.id: return 0.98; // 98% success
21
- default: return 0.0;
22
- }
23
- }
24
-
25
- static async execute(task, arm) {
26
- const probability = this.getSuccessProbability(arm.id);
27
- // Use hash to deterministically seed pseudo-randomness for the mock run
28
- const hashInt = parseInt(task.hash.substring(7, 15), 16);
29
- const successThreshold = probability * 0xffffffff;
30
-
31
- // Slight artificial delay to simulate API request
32
- await new Promise(resolve => setTimeout(resolve, 50));
33
-
34
- const isSuccess = hashInt <= successThreshold;
35
-
36
- return {
37
- success: isSuccess,
38
- mockedOutput: isSuccess
39
- ? `Successfully generated code for ${task.id}`
40
- : `Failed or hallucinated output for ${task.id}`,
41
- usage: {
42
- promptTokens: arm.tokenBudgetEstimate,
43
- completionTokens: 300,
44
- totalTokens: arm.tokenBudgetEstimate + 300
45
- }
46
- };
47
- }
48
- }
49
-
50
- /**
51
- * Executes a single task against a single arm.
52
- */
53
- async function runTask(task, armId) {
54
- const arm = Object.values(ARMS).find(a => a.id === armId);
55
- if (!arm) throw new Error(`Unknown arm: ${armId}`);
56
-
57
- const startTime = Date.now();
58
- const requestId = crypto.randomUUID();
59
-
60
- // Execute using Mock Provider (replace with real LLM client when API keys are available)
61
- const llmResult = await MockProvider.execute(task, arm);
62
-
63
- const durationMs = Date.now() - startTime;
64
-
65
- // Evaluate the output
66
- const evaluation = evaluateTask(task, llmResult);
67
-
68
- return {
69
- taskId: task.id,
70
- armId: arm.id,
71
- requestId,
72
- timestamp: new Date().toISOString(),
73
- durationMs,
74
- success: evaluation.passed,
75
- error: evaluation.error || null,
76
- usage: llmResult.usage,
77
- environmentProvenance: {
78
- nodeVersion: process.version,
79
- platform: process.platform,
80
- engine: 'mock-provider-v1'
81
- }
82
- };
83
- }
84
-
85
- module.exports = {
86
- runTask,
87
- MockProvider
88
- };