contextos-agents 2.0.0-beta.3 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/AGENTS.md +53 -33
- package/.agents/adapters/aider/export.js +41 -14
- package/.agents/adapters/claude/export.js +1 -1
- package/.agents/adapters/copilot/export.js +1 -1
- package/.agents/adapters/cursor/export.js +1 -1
- package/.agents/adapters/drift-detector.js +80 -7
- package/.agents/adapters/gemini/export.js +1 -1
- package/.agents/adapters/pure-compiler.js +10 -0
- package/.agents/adapters/shared.js +13 -4
- package/.agents/adapters/zed/export.js +1 -1
- package/.agents/compiled/registry.v2.json +29 -25
- package/.agents/compiled/registry.v2.sha256 +1 -1
- package/.agents/core/skills/context-os/SKILL.md +34 -37
- package/.agents/core/skills/engineering-workflow/SKILL.md +24 -24
- package/.agents/core/skills/gemini-precision/EXAMPLES.md +72 -0
- package/.agents/core/skills/gemini-precision/SKILL.md +2 -1
- package/.agents/core/skills/gemini-precision/TROUBLESHOOTING.md +25 -0
- package/.agents/core/skills/gemini-precision/skill.yaml +2 -0
- package/.agents/core/skills/gstack-roles/SKILL.md +7 -6
- package/.agents/core/skills/security/SKILL.md +44 -16
- package/.agents/core/skills/security/skill.yaml +0 -1
- package/.agents/ctx.js +16 -10
- package/.agents/doctor.js +7 -20
- package/.agents/generated/claude/skills/context-os/SKILL.md +34 -37
- package/.agents/generated/claude/skills/engineering-workflow/SKILL.md +24 -24
- package/.agents/generated/claude/skills/gemini-precision/SKILL.md +102 -1
- package/.agents/generated/claude/skills/gstack-roles/SKILL.md +7 -6
- package/.agents/generated/claude/skills/security/SKILL.md +44 -16
- package/.agents/generated/gemini/skills/context-os/SKILL.md +34 -37
- package/.agents/generated/gemini/skills/engineering-workflow/SKILL.md +24 -24
- package/.agents/generated/gemini/skills/gemini-precision/SKILL.md +105 -1
- package/.agents/generated/gemini/skills/gstack-roles/SKILL.md +7 -6
- package/.agents/generated/gemini/skills/security/SKILL.md +44 -125
- package/.agents/plugins.js +5 -4
- package/.agents/resolver/canonical-resolver.js +7 -7
- package/.agents/validate.js +69 -1
- package/LICENSE +201 -21
- package/NOTICE +4 -0
- package/README.md +87 -34
- package/bin/commands/hook.js +129 -0
- package/bin/commands/scan.js +70 -0
- package/bin/commands/update.js +7 -8
- package/bin/commands.js +39 -1
- package/bin/index.js +144 -33
- package/bin/lib/gate.js +171 -0
- package/bin/lib/git-snapshot.js +187 -0
- package/bin/lib/scan.js +380 -0
- package/package.json +10 -12
- package/.agents/core/skills/security/security.md +0 -106
- package/benchmarks/v2/analysis/statistics.js +0 -140
- package/benchmarks/v2/analysis/stats.js +0 -69
- package/benchmarks/v2/arms/arm-definitions.js +0 -79
- package/benchmarks/v2/dataset.schema.json +0 -34
- package/benchmarks/v2/evaluators/index.js +0 -25
- package/benchmarks/v2/evaluators/verified-success.js +0 -116
- package/benchmarks/v2/harness/runner.js +0 -88
- package/benchmarks/v2/pilot-tasks.json +0 -392
package/bin/lib/scan.js
ADDED
|
@@ -0,0 +1,380 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* bin/lib/scan.js
|
|
3
|
+
* ContextOS Staged Index Scanner & Code Governance Engine
|
|
4
|
+
*
|
|
5
|
+
* Inspects Git staged changes for:
|
|
6
|
+
* 1. Hardcoded secrets, API tokens, private keys, and blocked files
|
|
7
|
+
* 2. Unfinished lazy placeholder stubs in newly added code
|
|
8
|
+
* 3. Scope containment violations against declared task scopes
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
'use strict';
|
|
12
|
+
|
|
13
|
+
const fs = require('fs');
|
|
14
|
+
const path = require('path');
|
|
15
|
+
const {
|
|
16
|
+
findGitRoot,
|
|
17
|
+
getStagedFiles,
|
|
18
|
+
getStagedBlob,
|
|
19
|
+
getStagedAddedLines,
|
|
20
|
+
} = require('./git-snapshot.js');
|
|
21
|
+
|
|
22
|
+
// ── Blocked File Names and Extensions ───────────────────────────────────────
|
|
23
|
+
|
|
24
|
+
const BLOCKED_EXACT_NAMES = new Set([
|
|
25
|
+
'.env',
|
|
26
|
+
'.env.local',
|
|
27
|
+
'.env.production',
|
|
28
|
+
'.env.development',
|
|
29
|
+
'.env.staging',
|
|
30
|
+
'.env.test',
|
|
31
|
+
'.netrc',
|
|
32
|
+
'.git-credentials',
|
|
33
|
+
'.npmrc',
|
|
34
|
+
'credentials.json',
|
|
35
|
+
'service-account.json',
|
|
36
|
+
'serviceaccountkey.json',
|
|
37
|
+
'id_rsa',
|
|
38
|
+
'id_ed25519',
|
|
39
|
+
'mcp_config.json',
|
|
40
|
+
]);
|
|
41
|
+
|
|
42
|
+
const BLOCKED_EXTENSIONS = new Set([
|
|
43
|
+
'.key',
|
|
44
|
+
'.pem',
|
|
45
|
+
'.pfx',
|
|
46
|
+
'.p12',
|
|
47
|
+
'.pkcs12',
|
|
48
|
+
]);
|
|
49
|
+
|
|
50
|
+
// ── Secret Content Patterns ──────────────────────────────────────────────────
|
|
51
|
+
|
|
52
|
+
const SECRET_PATTERNS = [
|
|
53
|
+
{
|
|
54
|
+
ruleId: 'SEC-001',
|
|
55
|
+
name: 'Private Key Header',
|
|
56
|
+
pattern: /-----BEGIN\s+(?:RSA|OPENSSH|EC|DSA|PGP)?\s*PRIVATE\s+KEY-----/,
|
|
57
|
+
},
|
|
58
|
+
{
|
|
59
|
+
ruleId: 'SEC-002',
|
|
60
|
+
name: 'Hardcoded User Home Path Leak',
|
|
61
|
+
pattern: /(?:[a-zA-Z]:[/\\]Users[/\\]|\/(?:home|Users)\/)[a-zA-Z0-9_-]+[/\\](?:Desktop|Documents|Downloads|code|projects|repos)\b/i,
|
|
62
|
+
},
|
|
63
|
+
{
|
|
64
|
+
ruleId: 'SEC-003',
|
|
65
|
+
name: 'GitHub Personal Access Token',
|
|
66
|
+
pattern: /\b(?:ghp|gho)_[A-Za-z0-9]{36}\b/,
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
ruleId: 'SEC-004',
|
|
70
|
+
name: 'GitHub Fine-Grained PAT',
|
|
71
|
+
pattern: /\bgithub_pat_[A-Za-z0-9_]{82}\b/,
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
ruleId: 'SEC-005',
|
|
75
|
+
name: 'Google API Key',
|
|
76
|
+
pattern: /\bAIza[A-Za-z0-9_-]{35}\b/,
|
|
77
|
+
},
|
|
78
|
+
{
|
|
79
|
+
ruleId: 'SEC-006',
|
|
80
|
+
name: 'OpenAI API Key',
|
|
81
|
+
pattern: /\bsk-[A-Za-z0-9]{32,}\b/,
|
|
82
|
+
},
|
|
83
|
+
{
|
|
84
|
+
ruleId: 'SEC-007',
|
|
85
|
+
name: 'Anthropic API Key',
|
|
86
|
+
pattern: /\bsk-ant-[A-Za-z0-9-]{90,}\b/,
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
ruleId: 'SEC-008',
|
|
90
|
+
name: 'OpenRouter API Key',
|
|
91
|
+
pattern: /\bsk-or-v1-[a-f0-9]{64}\b/,
|
|
92
|
+
},
|
|
93
|
+
{
|
|
94
|
+
ruleId: 'SEC-009',
|
|
95
|
+
name: 'Slack Token',
|
|
96
|
+
pattern: /\bxox[baprs]-[A-Za-z0-9_-]{10,48}\b/,
|
|
97
|
+
},
|
|
98
|
+
{
|
|
99
|
+
ruleId: 'SEC-010',
|
|
100
|
+
name: 'npm Access Token',
|
|
101
|
+
pattern: /\bnpm_[A-Za-z0-9]{32,36}\b/,
|
|
102
|
+
},
|
|
103
|
+
{
|
|
104
|
+
ruleId: 'SEC-011',
|
|
105
|
+
name: 'AWS Access Key ID',
|
|
106
|
+
pattern: /\bAKIA[0-9A-Z]{16}\b/,
|
|
107
|
+
},
|
|
108
|
+
];
|
|
109
|
+
|
|
110
|
+
// ── Lazy Placeholder Patterns ───────────────────────────────────────────────
|
|
111
|
+
|
|
112
|
+
const PLACEHOLDER_PATTERNS = [
|
|
113
|
+
{
|
|
114
|
+
ruleId: 'CODE-001',
|
|
115
|
+
name: 'Lazy Stub Comment',
|
|
116
|
+
pattern: /(?:\/\/|\/\*|#)\s*(?:TODO:\s*implement later|\.\.\.\s*rest of code stays here\s*\.\.\.|TODO:\s*implement\b)/i,
|
|
117
|
+
},
|
|
118
|
+
{
|
|
119
|
+
ruleId: 'CODE-002',
|
|
120
|
+
name: 'NotImplementedError Stub',
|
|
121
|
+
pattern: /raise\s+NotImplementedError\s*\(\s*["'].*TODO.*["']\s*\)/i,
|
|
122
|
+
},
|
|
123
|
+
{
|
|
124
|
+
ruleId: 'CODE-003',
|
|
125
|
+
name: 'Python Pass Stub',
|
|
126
|
+
pattern: /^\s*pass\s*#\s*TODO\b/i,
|
|
127
|
+
},
|
|
128
|
+
];
|
|
129
|
+
|
|
130
|
+
// Exempt path substrings (e.g. test fixtures, test files, lockfiles)
|
|
131
|
+
const EXEMPT_PATH_SUBSTRINGS = [
|
|
132
|
+
path.join('tests', ''),
|
|
133
|
+
path.join('.git', ''),
|
|
134
|
+
'node_modules',
|
|
135
|
+
'check-secrets.js',
|
|
136
|
+
'scan.js',
|
|
137
|
+
];
|
|
138
|
+
|
|
139
|
+
function isExempt(filePath) {
|
|
140
|
+
const norm = path.normalize(filePath);
|
|
141
|
+
return EXEMPT_PATH_SUBSTRINGS.some(exempt => norm.includes(exempt));
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
function redact(str) {
|
|
145
|
+
if (str.length <= 8) return '***';
|
|
146
|
+
return str.slice(0, 4) + '...' + str.slice(-4);
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* Checks if a relative path matches a glob pattern or prefix.
|
|
151
|
+
*/
|
|
152
|
+
function matchesScope(filePath, pattern) {
|
|
153
|
+
const normFile = filePath.replace(/\\/g, '/').toLowerCase();
|
|
154
|
+
const normPattern = pattern.replace(/\\/g, '/').toLowerCase();
|
|
155
|
+
|
|
156
|
+
if (normPattern.endsWith('/**')) {
|
|
157
|
+
const prefix = normPattern.slice(0, -3);
|
|
158
|
+
return normFile.startsWith(prefix);
|
|
159
|
+
}
|
|
160
|
+
if (normPattern.endsWith('/*')) {
|
|
161
|
+
const prefix = normPattern.slice(0, -2);
|
|
162
|
+
const rest = normFile.slice(prefix.length + 1);
|
|
163
|
+
return normFile.startsWith(prefix) && !rest.includes('/');
|
|
164
|
+
}
|
|
165
|
+
if (normPattern.includes('*')) {
|
|
166
|
+
const reg = new RegExp('^' + normPattern.replace(/\./g, '\\.').replace(/\*/g, '.*') + '$');
|
|
167
|
+
return reg.test(normFile);
|
|
168
|
+
}
|
|
169
|
+
return normFile === normPattern || normFile.startsWith(normPattern + '/');
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
/**
|
|
173
|
+
* Runs the staged scanner on the given repository root.
|
|
174
|
+
*
|
|
175
|
+
* @param {Object} options - Scan configuration options
|
|
176
|
+
* @returns {Object} Structured scan results with findings and status code
|
|
177
|
+
*/
|
|
178
|
+
function runScan(options = {}) {
|
|
179
|
+
const cwd = path.resolve(options.cwd || process.cwd());
|
|
180
|
+
const gitRoot = findGitRoot(cwd);
|
|
181
|
+
|
|
182
|
+
if (!gitRoot) {
|
|
183
|
+
return {
|
|
184
|
+
ok: false,
|
|
185
|
+
code: 2,
|
|
186
|
+
error: 'Not a Git repository. ContextOS scanner requires Git.',
|
|
187
|
+
findings: [],
|
|
188
|
+
stats: { filesScanned: 0, violations: 0 },
|
|
189
|
+
};
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
const checkSecrets = options.secrets !== false;
|
|
193
|
+
const checkPlaceholders = Boolean(options.placeholders);
|
|
194
|
+
const scopeFile = options.scope || null;
|
|
195
|
+
const enforce = Boolean(options.enforce);
|
|
196
|
+
|
|
197
|
+
let allowedScopePatterns = null;
|
|
198
|
+
if (scopeFile) {
|
|
199
|
+
const resolvedScopePath = path.resolve(gitRoot, scopeFile);
|
|
200
|
+
if (!fs.existsSync(resolvedScopePath)) {
|
|
201
|
+
return {
|
|
202
|
+
ok: false,
|
|
203
|
+
code: 2,
|
|
204
|
+
error: `Scope file not found: ${scopeFile}`,
|
|
205
|
+
findings: [],
|
|
206
|
+
stats: { filesScanned: 0, violations: 0 },
|
|
207
|
+
};
|
|
208
|
+
}
|
|
209
|
+
try {
|
|
210
|
+
const scopeData = JSON.parse(fs.readFileSync(resolvedScopePath, 'utf8'));
|
|
211
|
+
const list = Array.isArray(scopeData) ? scopeData : scopeData.allowedPaths || scopeData.files;
|
|
212
|
+
if (!Array.isArray(list)) {
|
|
213
|
+
throw new Error('Scope declaration must be an array of paths or have an allowedPaths array');
|
|
214
|
+
}
|
|
215
|
+
for (const p of list) {
|
|
216
|
+
if (path.isAbsolute(p) || p.includes('..')) {
|
|
217
|
+
throw new Error(`Scope entry "${p}" is invalid (must be relative and cannot traverse outside repo)`);
|
|
218
|
+
}
|
|
219
|
+
}
|
|
220
|
+
allowedScopePatterns = list;
|
|
221
|
+
} catch (err) {
|
|
222
|
+
return {
|
|
223
|
+
ok: false,
|
|
224
|
+
code: 2,
|
|
225
|
+
error: `Invalid task scope file: ${err.message}`,
|
|
226
|
+
findings: [],
|
|
227
|
+
stats: { filesScanned: 0, violations: 0 },
|
|
228
|
+
};
|
|
229
|
+
}
|
|
230
|
+
}
|
|
231
|
+
|
|
232
|
+
let stagedEntries = [];
|
|
233
|
+
try {
|
|
234
|
+
stagedEntries = getStagedFiles(gitRoot);
|
|
235
|
+
} catch (err) {
|
|
236
|
+
return {
|
|
237
|
+
ok: false,
|
|
238
|
+
code: 2,
|
|
239
|
+
error: `Failed to inspect Git index: ${err.message}`,
|
|
240
|
+
findings: [],
|
|
241
|
+
stats: { filesScanned: 0, violations: 0 },
|
|
242
|
+
};
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
const findings = [];
|
|
246
|
+
const addedLinesMap = checkPlaceholders ? getStagedAddedLines(gitRoot) : new Map();
|
|
247
|
+
|
|
248
|
+
for (const entry of stagedEntries) {
|
|
249
|
+
// Skip deleted files from content and blocked filename checks
|
|
250
|
+
if (entry.status === 'D') continue;
|
|
251
|
+
|
|
252
|
+
const relPath = entry.path;
|
|
253
|
+
const baseName = path.basename(relPath).toLowerCase();
|
|
254
|
+
const extName = path.extname(relPath).toLowerCase();
|
|
255
|
+
|
|
256
|
+
// 1. Scope Containment Check
|
|
257
|
+
if (allowedScopePatterns) {
|
|
258
|
+
const inScope = allowedScopePatterns.some(p => matchesScope(relPath, p));
|
|
259
|
+
if (!inScope) {
|
|
260
|
+
findings.push({
|
|
261
|
+
ruleId: 'SCOPE-001',
|
|
262
|
+
file: relPath,
|
|
263
|
+
type: 'Scope Violation',
|
|
264
|
+
severity: 'error',
|
|
265
|
+
details: `Staged file "${relPath}" is outside allowed task scope`,
|
|
266
|
+
});
|
|
267
|
+
}
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
// 2. Blocked Exact Names Check
|
|
271
|
+
if (checkSecrets && BLOCKED_EXACT_NAMES.has(baseName)) {
|
|
272
|
+
findings.push({
|
|
273
|
+
ruleId: 'SEC-000',
|
|
274
|
+
file: relPath,
|
|
275
|
+
type: 'Blocked Filename',
|
|
276
|
+
severity: 'error',
|
|
277
|
+
details: `Filename "${baseName}" is forbidden from being committed into Git.`,
|
|
278
|
+
});
|
|
279
|
+
continue;
|
|
280
|
+
}
|
|
281
|
+
|
|
282
|
+
// 3. Blocked Extensions Check
|
|
283
|
+
if (checkSecrets && BLOCKED_EXTENSIONS.has(extName)) {
|
|
284
|
+
findings.push({
|
|
285
|
+
ruleId: 'SEC-000',
|
|
286
|
+
file: relPath,
|
|
287
|
+
type: 'Blocked Extension',
|
|
288
|
+
severity: 'error',
|
|
289
|
+
details: `Extension "${extName}" indicates private cryptographic keys or certificates.`,
|
|
290
|
+
});
|
|
291
|
+
continue;
|
|
292
|
+
}
|
|
293
|
+
|
|
294
|
+
// Skip content scan for exempt paths
|
|
295
|
+
if (isExempt(relPath)) continue;
|
|
296
|
+
|
|
297
|
+
// Read blob from index
|
|
298
|
+
const blob = getStagedBlob(relPath, gitRoot);
|
|
299
|
+
if (!blob) continue;
|
|
300
|
+
|
|
301
|
+
// Skip large files (> 2MB)
|
|
302
|
+
if (blob.length > 2 * 1024 * 1024) continue;
|
|
303
|
+
|
|
304
|
+
const content = blob.toString('utf8');
|
|
305
|
+
const lines = content.split(/\r?\n/);
|
|
306
|
+
|
|
307
|
+
// 4. Staged Content Secrets Check
|
|
308
|
+
if (checkSecrets) {
|
|
309
|
+
for (let lineNum = 1; lineNum <= lines.length; lineNum++) {
|
|
310
|
+
const line = lines[lineNum - 1];
|
|
311
|
+
for (const { ruleId, name, pattern } of SECRET_PATTERNS) {
|
|
312
|
+
const match = line.match(pattern);
|
|
313
|
+
if (match) {
|
|
314
|
+
findings.push({
|
|
315
|
+
ruleId,
|
|
316
|
+
file: relPath,
|
|
317
|
+
line: lineNum,
|
|
318
|
+
type: name,
|
|
319
|
+
severity: 'error',
|
|
320
|
+
details: `Detected pattern "${name}": ${redact(match[0])}`,
|
|
321
|
+
});
|
|
322
|
+
}
|
|
323
|
+
}
|
|
324
|
+
}
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
// 5. Newly Added Code Placeholders Check
|
|
328
|
+
if (checkPlaceholders && addedLinesMap.has(relPath)) {
|
|
329
|
+
const addedLines = addedLinesMap.get(relPath);
|
|
330
|
+
// Skip markdown and doc files for placeholder checks
|
|
331
|
+
if (extName !== '.md' && extName !== '.txt') {
|
|
332
|
+
for (const { line, content: addedLine } of addedLines) {
|
|
333
|
+
if (addedLine.includes('contextos:allow-placeholder')) continue;
|
|
334
|
+
|
|
335
|
+
for (const { ruleId, name, pattern } of PLACEHOLDER_PATTERNS) {
|
|
336
|
+
const match = addedLine.match(pattern);
|
|
337
|
+
if (match) {
|
|
338
|
+
findings.push({
|
|
339
|
+
ruleId,
|
|
340
|
+
file: relPath,
|
|
341
|
+
line,
|
|
342
|
+
type: name,
|
|
343
|
+
severity: 'warning',
|
|
344
|
+
details: `Detected un-implemented placeholder stub: "${match[0].trim()}"`,
|
|
345
|
+
});
|
|
346
|
+
}
|
|
347
|
+
}
|
|
348
|
+
}
|
|
349
|
+
}
|
|
350
|
+
}
|
|
351
|
+
}
|
|
352
|
+
|
|
353
|
+
const hasViolations = findings.length > 0;
|
|
354
|
+
// If enforce is true, violations return code 1. If enforce is false, warnings return code 0.
|
|
355
|
+
const exitCode = hasViolations && enforce ? 1 : 0;
|
|
356
|
+
|
|
357
|
+
return {
|
|
358
|
+
ok: exitCode === 0,
|
|
359
|
+
code: exitCode,
|
|
360
|
+
gitRoot,
|
|
361
|
+
enforce,
|
|
362
|
+
findings,
|
|
363
|
+
stats: {
|
|
364
|
+
stagedFilesCount: stagedEntries.length,
|
|
365
|
+
violationsCount: findings.length,
|
|
366
|
+
errorsCount: findings.filter(f => f.severity === 'error').length,
|
|
367
|
+
warningsCount: findings.filter(f => f.severity === 'warning').length,
|
|
368
|
+
},
|
|
369
|
+
};
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
module.exports = {
|
|
373
|
+
runScan,
|
|
374
|
+
BLOCKED_EXACT_NAMES,
|
|
375
|
+
BLOCKED_EXTENSIONS,
|
|
376
|
+
SECRET_PATTERNS,
|
|
377
|
+
PLACEHOLDER_PATTERNS,
|
|
378
|
+
matchesScope,
|
|
379
|
+
redact,
|
|
380
|
+
};
|
package/package.json
CHANGED
|
@@ -1,13 +1,14 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "contextos-agents",
|
|
3
|
-
"version": "2.
|
|
4
|
-
"description": "
|
|
3
|
+
"version": "2.1.0",
|
|
4
|
+
"description": "Deterministic context and policy compiler for supported AI coding agents.",
|
|
5
5
|
"bin": {
|
|
6
6
|
"contextos": "./bin/index.js",
|
|
7
7
|
"contextos-agents": "./bin/index.js",
|
|
8
8
|
"koko-contextos-agents": "./bin/index.js"
|
|
9
9
|
},
|
|
10
10
|
"files": [
|
|
11
|
+
"NOTICE",
|
|
11
12
|
"bin",
|
|
12
13
|
"registry.json",
|
|
13
14
|
"registry.schema.json",
|
|
@@ -35,22 +36,20 @@
|
|
|
35
36
|
".agents/workspace",
|
|
36
37
|
".agents/filesystem",
|
|
37
38
|
".agents/transaction-core",
|
|
38
|
-
".agents/customization-dx.js"
|
|
39
|
-
"benchmarks/v2"
|
|
39
|
+
".agents/customization-dx.js"
|
|
40
40
|
],
|
|
41
41
|
"scripts": {
|
|
42
42
|
"build": "node .agents/ctx.js compile \u0026\u0026 node .agents/ctx.js export all",
|
|
43
43
|
"compile": "node .agents/ctx.js compile",
|
|
44
44
|
"export": "node .agents/ctx.js export gemini",
|
|
45
45
|
"validate": "node .agents/ctx.js validate",
|
|
46
|
-
"
|
|
47
|
-
"
|
|
48
|
-
"benchmark:runtime": "node benchmarks/run-runtime-benchmark.js",
|
|
49
|
-
"lint:md": "markdownlint README.md CONTRIBUTING.md .agents/core/skills/**/*.md --ignore .agents/generated/** --config .markdownlint.json",
|
|
46
|
+
"validate:catalog": "node .agents/ctx.js validate --catalog",
|
|
47
|
+
"lint:md": "markdownlint README.md CONTRIBUTING.md GUIDE.md \".agents/core/skills/**/*.md\" --ignore \".agents/generated/**\" --config .markdownlint.json",
|
|
50
48
|
"check:secrets": "node scripts/check-secrets.js --all",
|
|
51
49
|
"setup:hooks": "node scripts/install-hooks.js",
|
|
52
50
|
"watch": "node .agents/watch.js",
|
|
53
|
-
"test": "node
|
|
51
|
+
"test:skills": "node scripts/verify-skill-examples.js",
|
|
52
|
+
"test": "node --test --test-concurrency=1 tests/install.test.js tests/lockfile.test.js tests/safe-writer.test.js tests/adapter-safety.test.js tests/update.test.js tests/uninstall.test.js tests/plugin-security.test.js tests/export.test.js tests/skills.test.js tests/validate.test.js tests/plugins.test.js tests/benchmark.test.js tests/benchmark-api.test.js tests/profile.test.js tests/resolver.test.js tests/cli-registry.test.js tests/profile-enforcement.test.js tests/doctor-hardening.test.js tests/manifest-compiler.test.js tests/canonical-resolver.test.js tests/workspace-graph.test.js tests/profiles-v2.test.js tests/safe-path.test.js tests/project-lock.test.js tests/journaled-transaction.test.js tests/pure-adapters.test.js tests/prompt-skill-system.test.js tests/runtime-state-machine.test.js tests/verification-reviewer-pipeline.test.js tests/durable-concurrency-locks.test.js tests/transactional-git.test.js tests/sandbox-execution.test.js tests/plugin-supply-chain.test.js tests/platform-hardening.test.js tests/customization-dx.test.js tests/claims-governance.test.js tests/benchmark-v2.test.js tests/init-engine.test.js tests/mutation-paths.test.js tests/tarball-smoke.test.js tests/catalog-export-integration.test.js tests/consumer-gate.test.js tests/adapter-compatibility.test.js tests/consumer-init.test.js tests/skill-examples.test.js tests/scan.test.js tests/hooks.test.js",
|
|
54
53
|
"prepublishOnly": "npm run validate \u0026\u0026 npm run build"
|
|
55
54
|
},
|
|
56
55
|
"keywords": [
|
|
@@ -65,11 +64,10 @@
|
|
|
65
64
|
"github-copilot",
|
|
66
65
|
"gemini",
|
|
67
66
|
"zed",
|
|
68
|
-
"mcp",
|
|
69
67
|
"contextos"
|
|
70
68
|
],
|
|
71
69
|
"author": "koko-o",
|
|
72
|
-
"license": "
|
|
70
|
+
"license": "Apache-2.0",
|
|
73
71
|
"repository": {
|
|
74
72
|
"type": "git",
|
|
75
73
|
"url": "https://github.com/kok-o/contextos-agents.git"
|
|
@@ -79,7 +77,7 @@
|
|
|
79
77
|
"url": "https://github.com/kok-o/contextos-agents/issues"
|
|
80
78
|
},
|
|
81
79
|
"engines": {
|
|
82
|
-
"node": "\u003e=
|
|
80
|
+
"node": "\u003e=22.0.0"
|
|
83
81
|
},
|
|
84
82
|
"devDependencies": {
|
|
85
83
|
"esbuild": "^0.28.2",
|
|
@@ -1,106 +0,0 @@
|
|
|
1
|
-
# Application Security — Best Practices
|
|
2
|
-
|
|
3
|
-
## OWASP Top 10
|
|
4
|
-
|
|
5
|
-
### 1. Injection (SQL, NoSQL, Command)
|
|
6
|
-
|
|
7
|
-
- **Always use parameterized queries** — never concatenate user input into SQL
|
|
8
|
-
- Use ORM (Prisma, SQLAlchemy, TypeORM) — they parameterize by default
|
|
9
|
-
- Validate and sanitize all user input
|
|
10
|
-
|
|
11
|
-
### 2. Broken Authentication
|
|
12
|
-
|
|
13
|
-
- Use bcrypt/argon2 for password hashing (cost factor ≥ 12)
|
|
14
|
-
- JWT: short-lived access tokens (15min), refresh tokens (7 days)
|
|
15
|
-
- Rate limit login attempts
|
|
16
|
-
- Implement account lockout after N failed attempts
|
|
17
|
-
- MFA for sensitive operations
|
|
18
|
-
|
|
19
|
-
### 3. Sensitive Data Exposure
|
|
20
|
-
|
|
21
|
-
- HTTPS everywhere — redirect HTTP to HTTPS
|
|
22
|
-
- Encrypt sensitive data at rest (AES-256)
|
|
23
|
-
- Never log passwords, tokens, or PII
|
|
24
|
-
- Use environment variables for secrets
|
|
25
|
-
|
|
26
|
-
### 4. XML/XXE
|
|
27
|
-
|
|
28
|
-
- Disable external entity processing
|
|
29
|
-
- Use JSON instead of XML where possible
|
|
30
|
-
|
|
31
|
-
### 5. Broken Access Control
|
|
32
|
-
|
|
33
|
-
- Default deny — explicitly grant access
|
|
34
|
-
- RBAC (Role-Based Access Control) or ABAC (Attribute-Based)
|
|
35
|
-
- Check authorization on every request, not just UI
|
|
36
|
-
- Don't rely on client-side validation for security
|
|
37
|
-
|
|
38
|
-
### 6. Security Misconfiguration
|
|
39
|
-
|
|
40
|
-
- Remove default credentials
|
|
41
|
-
- Disable debug mode in production
|
|
42
|
-
- Security headers (see below)
|
|
43
|
-
- Keep dependencies updated
|
|
44
|
-
|
|
45
|
-
### 7. XSS (Cross-Site Scripting)
|
|
46
|
-
|
|
47
|
-
- Escape all output by default
|
|
48
|
-
- Content-Security-Policy header
|
|
49
|
-
- HttpOnly + Secure + SameSite cookies
|
|
50
|
-
- Use framework's built-in XSS protection
|
|
51
|
-
|
|
52
|
-
### 8. Insecure Deserialization
|
|
53
|
-
|
|
54
|
-
- Validate and schema-check all input (Zod, Pydantic, class-validator)
|
|
55
|
-
- Don't deserialize untrusted data
|
|
56
|
-
|
|
57
|
-
### 9. Insufficient Logging
|
|
58
|
-
|
|
59
|
-
- Log all authentication events
|
|
60
|
-
- Log authorization failures
|
|
61
|
-
- Log input validation failures
|
|
62
|
-
- Include request ID for tracing
|
|
63
|
-
|
|
64
|
-
### 10. SSRF (Server-Side Request Forgery)
|
|
65
|
-
|
|
66
|
-
- Validate and allowlist URLs
|
|
67
|
-
- Don't let users control server-side HTTP requests
|
|
68
|
-
|
|
69
|
-
## Security Headers
|
|
70
|
-
|
|
71
|
-
```
|
|
72
|
-
Content-Security-Policy: default-src 'self'
|
|
73
|
-
X-Content-Type-Options: nosniff
|
|
74
|
-
X-Frame-Options: DENY
|
|
75
|
-
Strict-Transport-Security: max-age=31536000; includeSubDomains
|
|
76
|
-
Referrer-Policy: strict-origin-when-cross-origin
|
|
77
|
-
Permissions-Policy: camera=(), microphone=(), geolocation=()
|
|
78
|
-
```
|
|
79
|
-
|
|
80
|
-
## Authentication Patterns
|
|
81
|
-
|
|
82
|
-
### JWT Flow
|
|
83
|
-
|
|
84
|
-
```
|
|
85
|
-
Login → Access Token (15min) + Refresh Token (7d, HttpOnly cookie)
|
|
86
|
-
Request → Authorization: Bearer <access_token>
|
|
87
|
-
Expired → POST /auth/refresh (sends refresh cookie) → new access token
|
|
88
|
-
```
|
|
89
|
-
|
|
90
|
-
### OAuth2 Flow
|
|
91
|
-
|
|
92
|
-
```
|
|
93
|
-
Redirect → Provider (Google, GitHub) → Callback → Create/link user → JWT
|
|
94
|
-
```
|
|
95
|
-
|
|
96
|
-
## Checklist Before Deploy
|
|
97
|
-
|
|
98
|
-
- [ ] All secrets in environment variables
|
|
99
|
-
- [ ] HTTPS enabled
|
|
100
|
-
- [ ] Security headers configured
|
|
101
|
-
- [ ] Input validation on all endpoints
|
|
102
|
-
- [ ] Rate limiting enabled
|
|
103
|
-
- [ ] CORS configured (not `*`)
|
|
104
|
-
- [ ] Error messages don't leak internals
|
|
105
|
-
- [ ] Dependency audit (`npm audit`, `pip audit`)
|
|
106
|
-
- [ ] Logging for security events
|
|
@@ -1,140 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* benchmarks/v2/analysis/statistics.js
|
|
3
|
-
* ContextOS Benchmark v2 — Statistical Analysis Engine
|
|
4
|
-
*
|
|
5
|
-
* Implements Section 24 of CONTEXTOS_IMPLEMENTATION_PLAN.md:
|
|
6
|
-
* - Primary metric: cost per independently verified successful task
|
|
7
|
-
* - Wilson score 95% confidence intervals for binomial success proportions
|
|
8
|
-
* - Pairwise delta comparison between Arm C/D and Arm B (Concise Checklist comparator)
|
|
9
|
-
* - Structured tabular summary generation
|
|
10
|
-
*/
|
|
11
|
-
|
|
12
|
-
'use strict';
|
|
13
|
-
|
|
14
|
-
/**
|
|
15
|
-
* Calculates Wilson score 95% confidence interval for a proportion.
|
|
16
|
-
*
|
|
17
|
-
* @param {number} successes
|
|
18
|
-
* @param {number} total
|
|
19
|
-
* @param {number} [z=1.96] - 95% confidence z-score
|
|
20
|
-
* @returns {[number, number]} [lower, upper] as percentages [0, 100]
|
|
21
|
-
*/
|
|
22
|
-
function calculateWilsonInterval(successes, total, z = 1.96) {
|
|
23
|
-
if (total === 0) return [0, 0];
|
|
24
|
-
const p = successes / total;
|
|
25
|
-
const z2 = z * z;
|
|
26
|
-
const denominator = 1 + z2 / total;
|
|
27
|
-
const center = (p + z2 / (2 * total)) / denominator;
|
|
28
|
-
const margin = (z * Math.sqrt((p * (1 - p)) / total + z2 / (4 * total * total))) / denominator;
|
|
29
|
-
|
|
30
|
-
return [
|
|
31
|
-
Math.max(0, parseFloat(((center - margin) * 100).toFixed(1))),
|
|
32
|
-
Math.min(100, parseFloat(((center + margin) * 100).toFixed(1))),
|
|
33
|
-
];
|
|
34
|
-
}
|
|
35
|
-
|
|
36
|
-
class BenchmarkStatistics {
|
|
37
|
-
/**
|
|
38
|
-
* Analyzes an array of run outcomes across experimental arms.
|
|
39
|
-
*
|
|
40
|
-
* @param {Array<Object>} runs - List of run records: { armId, success, totalCost, durationMs }
|
|
41
|
-
* @returns {Object} Comprehensive statistical summary
|
|
42
|
-
*/
|
|
43
|
-
static analyze(runs) {
|
|
44
|
-
const armsMap = {};
|
|
45
|
-
|
|
46
|
-
for (const run of runs) {
|
|
47
|
-
const { armId, success, totalCost = 0, durationMs = 0 } = run;
|
|
48
|
-
if (!armsMap[armId]) {
|
|
49
|
-
armsMap[armId] = {
|
|
50
|
-
armId,
|
|
51
|
-
totalRuns: 0,
|
|
52
|
-
successfulRuns: 0,
|
|
53
|
-
totalCost: 0,
|
|
54
|
-
totalDurationMs: 0,
|
|
55
|
-
};
|
|
56
|
-
}
|
|
57
|
-
|
|
58
|
-
const item = armsMap[armId];
|
|
59
|
-
item.totalRuns++;
|
|
60
|
-
if (success) item.successfulRuns++;
|
|
61
|
-
item.totalCost += totalCost;
|
|
62
|
-
item.totalDurationMs += durationMs;
|
|
63
|
-
}
|
|
64
|
-
|
|
65
|
-
const armStats = {};
|
|
66
|
-
for (const [armId, d] of Object.entries(armsMap)) {
|
|
67
|
-
const successRate = d.totalRuns > 0 ? (d.successfulRuns / d.totalRuns) * 100 : 0;
|
|
68
|
-
const ci95 = calculateWilsonInterval(d.successfulRuns, d.totalRuns);
|
|
69
|
-
const costPerSuccess = d.successfulRuns > 0 ? d.totalCost / d.successfulRuns : null;
|
|
70
|
-
const avgDurationMs = d.totalRuns > 0 ? d.totalDurationMs / d.totalRuns : 0;
|
|
71
|
-
|
|
72
|
-
armStats[armId] = {
|
|
73
|
-
armId,
|
|
74
|
-
totalRuns: d.totalRuns,
|
|
75
|
-
successfulRuns: d.successfulRuns,
|
|
76
|
-
successRate: parseFloat(successRate.toFixed(1)),
|
|
77
|
-
ci95,
|
|
78
|
-
totalCost: parseFloat(d.totalCost.toFixed(4)),
|
|
79
|
-
costPerVerifiedSuccess: costPerSuccess !== null ? parseFloat(costPerSuccess.toFixed(4)) : null,
|
|
80
|
-
avgDurationMs: Math.round(avgDurationMs),
|
|
81
|
-
};
|
|
82
|
-
}
|
|
83
|
-
|
|
84
|
-
// Pairwise comparisons against Arm B (Concise Checklist)
|
|
85
|
-
const comparator = armStats['arm-b-concise-checklist'];
|
|
86
|
-
const comparisons = {};
|
|
87
|
-
|
|
88
|
-
if (comparator) {
|
|
89
|
-
for (const [armId, stat] of Object.entries(armStats)) {
|
|
90
|
-
if (armId === 'arm-b-concise-checklist') continue;
|
|
91
|
-
|
|
92
|
-
const rateDelta = parseFloat((stat.successRate - comparator.successRate).toFixed(1));
|
|
93
|
-
let costRatio = null;
|
|
94
|
-
if (comparator.costPerVerifiedSuccess && stat.costPerVerifiedSuccess) {
|
|
95
|
-
costRatio = parseFloat((stat.costPerVerifiedSuccess / comparator.costPerVerifiedSuccess).toFixed(2));
|
|
96
|
-
}
|
|
97
|
-
|
|
98
|
-
comparisons[armId] = {
|
|
99
|
-
vsComparator: 'arm-b-concise-checklist',
|
|
100
|
-
successRateDelta: rateDelta,
|
|
101
|
-
costRatio,
|
|
102
|
-
};
|
|
103
|
-
}
|
|
104
|
-
}
|
|
105
|
-
|
|
106
|
-
return {
|
|
107
|
-
timestamp: Date.now(),
|
|
108
|
-
totalRunsAnalyzed: runs.length,
|
|
109
|
-
arms: armStats,
|
|
110
|
-
comparisons,
|
|
111
|
-
};
|
|
112
|
-
}
|
|
113
|
-
|
|
114
|
-
/**
|
|
115
|
-
* Formats statistical analysis into an aligned markdown summary table.
|
|
116
|
-
*
|
|
117
|
-
* @param {Object} analysis
|
|
118
|
-
* @returns {string}
|
|
119
|
-
*/
|
|
120
|
-
static formatTable(analysis) {
|
|
121
|
-
const lines = [];
|
|
122
|
-
lines.push('| Arm ID | Runs | Successes | Success Rate (95% CI) | Cost / Verified Success | Avg Latency |');
|
|
123
|
-
lines.push('|---|---:|---:|---:|---:|---:|');
|
|
124
|
-
|
|
125
|
-
for (const arm of Object.values(analysis.arms)) {
|
|
126
|
-
const ciStr = `[${arm.ci95[0]}%, ${arm.ci95[1]}%]`;
|
|
127
|
-
const costStr = arm.costPerVerifiedSuccess !== null ? `$${arm.costPerVerifiedSuccess}` : 'N/A';
|
|
128
|
-
lines.push(
|
|
129
|
-
`| **${arm.armId}** | ${arm.totalRuns} | ${arm.successfulRuns} | ${arm.successRate}% ${ciStr} | ${costStr} | ${arm.avgDurationMs}ms |`
|
|
130
|
-
);
|
|
131
|
-
}
|
|
132
|
-
|
|
133
|
-
return lines.join('\n');
|
|
134
|
-
}
|
|
135
|
-
}
|
|
136
|
-
|
|
137
|
-
module.exports = {
|
|
138
|
-
calculateWilsonInterval,
|
|
139
|
-
BenchmarkStatistics,
|
|
140
|
-
};
|